-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathinstall-deploy-runner.sh
More file actions
executable file
·347 lines (324 loc) · 16.2 KB
/
Copy pathinstall-deploy-runner.sh
File metadata and controls
executable file
·347 lines (324 loc) · 16.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
#!/usr/bin/env bash
# Install/repair the "honeypot-home" self-hosted GitHub Actions runner --
# docs/CI-CD.md's "Home deployment" section, labels
# self-hosted/linux/x64/honeypot-home, attached to the protected
# production-home environment. Distinct from install-ci-runner.sh in this
# same directory: this runner has real production write access
# (docker.sock, /opt/stacks, /var/dockge/stacks), that one deliberately has
# none.
#
# #1143: before this script existed, docs/CI-CD.md only described what the
# runner's service account needs in prose ("write access to
# /opt/stacks/apiary and permission to run Docker Compose") -- no script
# ever set that up precisely. Re-registering this runner after a Tier 3
# reinstall, an operator improvised `chown -R github-deploy-runner:
# deploy-runner` over every deploy.yml destination= directory, matching its
# own destination list -- but deploy.yml's own rsync into /opt/stacks/apiary
# already excludes state/, dashboard-state/, and logs/ (see the "Preserved
# path" table in docs/CI-CD.md) precisely because those subtrees are
# container-owned, not deploy-runner's. A recursive chown swept them anyway,
# and two real outages followed: Keycloak (state/keycloak/secrets/
# postgres-password needs to stay UID 1000, the container's own internal
# user) and Filebeat (state/filebeat's registry/lock files, similarly
# container-owned). Both fixed live at the time; this script exists so a
# future re-registration has a precise, safe command to run instead of
# reasoning through the exclusion list by hand again.
#
# Usage:
# sudo scripts/github-ci-runner/install-deploy-runner.sh --repo Xore/APIARY [--token TOKEN]
#
# --token is a short-lived (1h) registration token, same convention as
# install-ci-runner.sh -- omit it to have this script fetch one itself via
# `gh api` (needs `gh auth login` for an account with admin on the repo).
#
# --helpers-only (#3312) applies ONLY the root-owned helper scripts and the
# sudoers grant that workflows invoke as root, then exits. It needs no
# --repo, touches no group membership, no DEPLOY_DIRS ownership, no runner
# registration, and never stops or restarts the runner service. Same shape as
# install-ci-runner.sh's --build-only: the narrow mode exists so applying a
# grant a merged PR depends on does not require the full installer.
#
# Why it exists: every capability this repo hands the runner (the
# isolation-audit sudoers trio from #2778, the libvirt group from #3338, the
# source-health helper grant from #3312) reaches the host only when an
# operator re-runs this script -- merging the PR that adds one changes
# nothing on the box, the same way #2908 found for /opt/stacks/apiary. A
# full re-run stops and restarts the runner service (`svc.sh stop` on an
# already-installed unit), so running it while the homeserver is executing a
# job risks killing that job mid-step. That is the whole cost of applying
# the source-health grant, which is the one grant a Diagnostics run needs
# and the one it cannot do without -- and that cost is what left #3312's
# last step pending. The helper and its grant need no restarted process:
# they take effect on the runner's next sudo call.
#
# Safe to re-run: every step below is idempotent (skips what already
# exists/is already correct) and the ownership fix specifically is safe to
# run repeatedly or on a partially-provisioned host -- it only ever touches
# the exact directories deploy.yml itself writes into, never anything under
# a state/, dashboard-state/, or logs/ subtree anywhere in that list.
set -euo pipefail
[[ ${EUID} -eq 0 ]] || { echo "Run as root" >&2; exit 1; }
RUNNER_VERSION=2.336.0
RUNNER_SHA256=04cf0be1aff4c3ec3554466c39124ca250e3effd8873bb7e8d68535aa9505d5d
RUNNER_USER=github-deploy-runner
RUNNER_GROUP=deploy-runner
RUNNER_HOME=/opt/github-deploy-runner
RUNNER_LABELS="self-hosted,linux,x64,honeypot-home"
# The exact set of directories deploy.yml writes into (destination= across
# every job in .github/workflows/deploy.yml, cross-checked against that
# file directly, not re-derived from memory -- if this list ever drifts
# again, re-derive it the same way: grep destination= out of
# .github/workflows/deploy.yml). The honeypot-arcane destination= is
# deliberately NOT listed here, same as every other stack: that job only
# ever `cp`s one compose.yml into an already-`install -d`'d directory,
# never rsyncs a tree into it, so there is no equivalent ownership risk to
# fix there -- adding it would only widen this script's blast radius for
# no real gain. (#2602: this list used to carry six pre-#1502 paths that
# deploy.yml stopped writing to once the Arcane manifest took over; one of
# them, /var/dockge/stacks/honeypot-keycloak, still existed on disk and
# was getting a gratuitous recursive chown every rerun.)
DEPLOY_DIRS=(
/opt/stacks/apiary
)
# Subtree names that are container-owned wherever they appear under any of
# DEPLOY_DIRS above -- exactly what deploy.yml's own --exclude list already
# protects during rsync (see docs/CI-CD.md's "Preserved path" table).
# find -prune stops descending the moment it matches one of these, so nested
# occurrences (e.g. a per-stack state/ several directories deep) are equally
# protected, not just top-level ones.
STATE_SUBTREE_NAMES=(state dashboard-state logs)
repo=""
token=""
name="${HOSTNAME:-homeserver}-home"
helpers_only=""
while [[ $# -gt 0 ]]; do
case "$1" in
--repo) repo="$2"; shift 2 ;;
--token) token="$2"; shift 2 ;;
--name) name="$2"; shift 2 ;;
--helpers-only) helpers_only=1; shift ;;
*) echo "unknown argument: $1" >&2; exit 1 ;;
esac
done
usage() {
echo "Usage: $0 --repo OWNER/NAME [--token TOKEN] [--name RUNNER_NAME]" >&2
echo " $0 --helpers-only # root-owned helpers + sudoers grant only (#3312)" >&2
exit 1
}
# --repo is only needed to register the runner, which --helpers-only never
# reaches; demanding it there would be a flag the operator has to look up
# for a command that does not talk to GitHub at all.
[[ -n "$helpers_only" || -n "$repo" ]] || usage
# --helpers-only registers nothing, so it never needs a registration token
# and must not call out to gh api for one.
if [[ -z "$helpers_only" && -z "$token" && ! -f "$RUNNER_HOME/.runner" ]]; then
command -v gh >/dev/null 2>&1 || { echo "no --token given and gh is not installed to fetch one" >&2; exit 1; }
echo "fetching a fresh registration token via gh api..."
token=$(gh api -X POST "repos/$repo/actions/runners/registration-token" --jq .token)
fi
# The root-owned helper scripts plus the sudoers fragment that lets the
# runner run them. Split out of the main flow because #3312 needs it
# reachable without the rest of this script: a grant a merged PR depends on
# only reaches the host when someone re-runs the installer, and the full run
# stops and restarts the runner service (see --helpers-only in the header).
# Everything in here takes effect on the runner's next sudo call -- nothing
# here needs a restarted process, which is the whole point of the narrow
# mode.
install_root_helpers() {
# #3312: root-owned so the runner cannot rewrite what its sudoers grant runs.
install -d -m 0755 -o root -g root /opt/github-ci-runner-helpers
install -m 0755 -o root -g root \
"$(dirname "$(readlink -f "$0")")/dashboard-source-health.sh" \
/opt/github-ci-runner-helpers/dashboard-source-health.sh
echo "installed /opt/github-ci-runner-helpers/dashboard-source-health.sh"
local sudoers_file=/etc/sudoers.d/isolation-audit-github-deploy-runner
local sudoers_tmp
sudoers_tmp=$(mktemp)
cat > "$sudoers_tmp" <<EOF
# Managed by scripts/github-ci-runner/install-deploy-runner.sh (#2778).
# Read-only commands scripts/isolation-audit.sh needs and cannot reach via
# docker-group membership or libvirt-group membership alone. Do not widen
# past exactly these three invocations.
$RUNNER_USER ALL=(root) NOPASSWD: /usr/sbin/iptables -S FORWARD
$RUNNER_USER ALL=(root) NOPASSWD: /usr/bin/ss -tlnp
$RUNNER_USER ALL=(root) NOPASSWD: /usr/sbin/aa-status
# #3312: Diagnostics' source-health read. Takes no arguments; the helper
# reads the dashboard service token as root and returns only the JSON.
$RUNNER_USER ALL=(root) NOPASSWD: /opt/github-ci-runner-helpers/dashboard-source-health.sh
EOF
if visudo -cf "$sudoers_tmp" >/dev/null 2>&1; then
install -m 0440 -o root -g root "$sudoers_tmp" "$sudoers_file"
echo "installed $sudoers_file"
else
echo "error: generated sudoers fragment failed visudo -cf, not installing $sudoers_file" >&2
visudo -cf "$sudoers_tmp" >&2 || true
rm -f "$sudoers_tmp"
exit 1
fi
rm -f "$sudoers_tmp"
}
# The narrow mode, deliberately ahead of every other side effect: it must not
# create the runner user, change a group, chown a tree or stop the service, or
# "apply the grant without disturbing the runner" is not what an operator
# running it mid-workday is getting. A sudoers entry for a user that does not
# exist yet simply never matches; a fresh host wants the full installer.
if [[ -n "$helpers_only" ]]; then
install_root_helpers
cat <<'EOF'
--helpers-only: applied the Diagnostics helper and the sudoers grant for it.
Both take effect on the runner's next sudo call -- no restart, no job killed.
Deliberately NOT done here:
* group memberships (docker, deploy-runner, libvirt) -- a running runner
keeps the groups it started with, so these only take effect across the
service restart a full run does
* the DEPLOY_DIRS ownership fix
* runner download, registration, and svc.sh stop/start
Run without --helpers-only for those.
EOF
exit 0
fi
# System user + its two groups: RUNNER_GROUP (secondary, deploy-runner) and
# RUNNER_USER-as-group (primary, github-deploy-runner). useradd -g below
# requires the primary group to exist (#2288: only RUNNER_GROUP was being
# groupadd'd, so a fresh host aborted at useradd(8) with
# "useradd: group 'github-deploy-runner' does not exist" and a tier-3
# disaster-rebuild hit it). Both groups are created here on a genuinely
# new host; previously-provisioned hosts get past the getent guards and
# proceed straight to useradd.
if ! getent group "$RUNNER_GROUP" >/dev/null; then
groupadd --system "$RUNNER_GROUP"
fi
if ! getent group "$RUNNER_USER" >/dev/null; then
groupadd --system "$RUNNER_USER"
fi
if ! id "$RUNNER_USER" >/dev/null 2>&1; then
useradd --system --create-home --home-dir "$RUNNER_HOME" --shell /usr/sbin/nologin \
-g "$RUNNER_USER" -G "docker,$RUNNER_GROUP" "$RUNNER_USER"
else
usermod -aG "docker,$RUNNER_GROUP" "$RUNNER_USER"
fi
# #2778: scripts/isolation-audit.sh (deployed under /opt/stacks/apiary,
# run by Diagnostics' "Isolation invariants (#88)" step against this
# runner) needs read access to libvirt (virsh net-info/net-dumpxml,
# nwfilter-dumpxml) and NOPASSWD root for exactly the read-only host
# commands it cannot reach any other way (iptables -S FORWARD, ss -tlnp,
# aa-status). Plain `docker ps`/`docker inspect`/`docker network inspect`
# are already covered by docker-group membership above and the script no
# longer routes those through sudo. Nothing here is broader than the
# audit script's own use, and nothing here touches github-ci-runner --
# that pool's trust boundary is #2780's decision, not this one's.
if getent group libvirt >/dev/null; then
usermod -aG libvirt "$RUNNER_USER"
else
echo "warning: no 'libvirt' group on this host -- isolation-audit.sh's virsh checks will keep reporting permission errors for $RUNNER_USER" >&2
fi
install_root_helpers
# --- #1143: precisely scoped ownership fix, the actual replacement for the
# broad manual chown that caused this issue. ---
for dir in "${DEPLOY_DIRS[@]}"; do
[[ -d "$dir" ]] || { echo "skip (does not exist yet): $dir"; continue; }
prune_args=()
for name_pattern in "${STATE_SUBTREE_NAMES[@]}"; do
prune_args+=(-o -name "$name_pattern" -prune)
done
# The first -false primes the -o chain so every real prune clause is
# genuinely "-o"'d together rather than the first one silently acting as
# the whole expression's start. -print0 only reaches paths that survived
# every -prune above.
find "$dir" \( -false "${prune_args[@]}" \) -o -print0 \
| xargs -0 -r chown "$RUNNER_USER:$RUNNER_GROUP"
echo "ownership fixed (state/dashboard-state/logs excluded): $dir"
done
# --- Runner binary + registration, same shape as install-ci-runner.sh ---
install -d -m 0755 -o "$RUNNER_USER" -g "$RUNNER_USER" "$RUNNER_HOME"
if [[ ! -f "$RUNNER_HOME/run.sh" ]]; then
tmp=$(mktemp -d)
trap 'rm -rf "$tmp"' EXIT
curl -fsSL -o "$tmp/runner.tar.gz" \
"https://github.com/actions/runner/releases/download/v${RUNNER_VERSION}/actions-runner-linux-x64-${RUNNER_VERSION}.tar.gz"
echo "${RUNNER_SHA256} $tmp/runner.tar.gz" | sha256sum -c -
tar -xzf "$tmp/runner.tar.gz" -C "$RUNNER_HOME"
chown "$RUNNER_USER:$RUNNER_USER" "$RUNNER_HOME"
find "$RUNNER_HOME" -maxdepth 1 -mindepth 1 -exec chown -R "$RUNNER_USER:$RUNNER_USER" {} +
echo "extracted actions-runner v${RUNNER_VERSION} to $RUNNER_HOME"
# installdependencies.sh installs the .NET runtime's prerequisites. On a
# RHEL-family host one of them, lttng-ust, lives in CRB, which Rocky ships
# disabled -- so the script fails with "Error: Unable to find a match:
# lttng-ust" followed by "Can't install dotnet core dependencies", and the
# whole install aborts. Hit live on the 2026-09-03 Rocky 10 rebuild. Enable
# CRB first so its own dependency resolution can succeed.
if command -v dnf >/dev/null 2>&1; then
dnf config-manager --set-enabled crb 2>/dev/null || true
fi
"$RUNNER_HOME/bin/installdependencies.sh"
else
echo "runner already extracted at $RUNNER_HOME, skipping download"
fi
if [[ ! -f "$RUNNER_HOME/.runner" ]]; then
sudo -u "$RUNNER_USER" "$RUNNER_HOME/config.sh" \
--url "https://github.com/$repo" \
--token "$token" \
--name "$name" \
--labels "$RUNNER_LABELS" \
--work "_work" \
--unattended \
--replace
else
echo "runner already configured (found $RUNNER_HOME/.runner) -- not re-registering"
echo "to re-register (e.g. after moving hosts), remove $RUNNER_HOME/.runner first"
fi
cd "$RUNNER_HOME"
service_file="/etc/systemd/system/actions.runner.$(tr '/' '-' <<<"$repo").$name.service"
if [[ ! -f "$service_file" ]]; then
./svc.sh install "$RUNNER_USER"
else
./svc.sh stop || true
fi
# #3105: same shared-cache group-write gap as install-ci-runner.sh -- this
# runner also builds against /var/buildx-cache, and its UMask=0022 default
# writes new cache files 0644, unreadable-for-write by the CI pool's users.
unit_name="$(basename "$service_file")"
unit_dropin_dir="/etc/systemd/system/${unit_name}.d"
install -d -m 0755 -o root -g root "$unit_dropin_dir"
cat > "$unit_dropin_dir/buildx-cache-umask.conf" <<EOF
[Service]
UMask=0002
EOF
systemctl daemon-reload
./svc.sh start
echo "done. status:"
./svc.sh status
# #2749: same hardening #2742 added to install-ci-runner.sh, ported here --
# this runner has real production write access, so a silently-never-enabled
# unit (the 2026-08-29 outage shape: `svc.sh install` succeeds, nothing else
# ever confirms it stuck) is at least as bad here as on the CI fleet. Assert
# both the local systemd state and GitHub's own view of the runner, and fail
# loudly rather than leaving either silently unknown.
unit_name="$(basename "$service_file")"
if ! systemctl is-enabled --quiet "$unit_name"; then
echo "FATAL: $unit_name is not enabled after provisioning -- it will not survive a reboot" >&2
exit 1
fi
echo "confirmed enabled: $unit_name"
if command -v gh >/dev/null 2>&1 && gh auth status >/dev/null 2>&1; then
echo "confirming \"$name\" shows online to GitHub (up to 60s)..."
online=""
status=""
for _ in $(seq 1 12); do
status=$(gh api "repos/$repo/actions/runners" --paginate \
--jq ".runners[] | select(.name==\"$name\") | .status" 2>/dev/null | tail -1) || status=""
if [[ "$status" == "online" ]]; then
online="1"
break
fi
sleep 5
done
if [[ -z "$online" ]]; then
echo "FATAL: \"$name\" did not report status=online to the GitHub API within 60s (last seen: '${status:-none}'). The unit is enabled and running locally but GitHub does not consider this runner available -- check $RUNNER_HOME/_diag for the runner's own connection log." >&2
exit 1
fi
echo "confirmed online: $name"
else
echo "WARNING: gh is not authenticated -- could not verify \"$name\" shows online to the GitHub API. Verify manually:" >&2
echo " gh api repos/$repo/actions/runners --jq '.runners[] | select(.name==\"'\"$name\"'\")'" >&2
fi