From 4dba5605190adbd9f142e9272b127ebbe6547ea0 Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 17:19:09 +0200 Subject: [PATCH 01/10] ci: route all workflows homeserver-first with GitHub-hosted fallback (#2565) Every compute workflow now ships the homeserver-first routing pair (or a conditional runs-on) that quality.yml's six families already used: - New shared .github/workflows/ci-router.yml (workflow_call) carrying the trust gate + heartbeat liveness proof quality.yml's inline router documents; containers/security/pages consume it, quality keeps its inline job until a follow-up migrates it. - ci-heartbeat.yml drops its concurrency group: one push fans out to multiple routers and cancel-in-progress would cancel a sibling router's canary evidence. - containers.yml build and security.yml (CodeQL) analyze pick runs-on off the router output; pages.yml build joins them and the deploy step stays ubuntu-latest (seconds-long API call). - quality.yml: frontend-next and frontend-next-browser become pairs; the browser home twin drops --with-deps and the apt fallback for loud presence checks (deps are preinstalled on the host); the docker-bound scripts-and-compose rows are flagged home: true now that the runner user drives docker (the four real-ES suites included -- they used to SKIP without an engine, a relocated no-op). - install-ci-runner.sh provisions the host idempotently: docker-group grant for the runner user, redis-server (daemon disabled), node 22, shellcheck, and playwright 1.62.1's chromium library set. Runner capability changes verified live on the box; trust gate and CI_HOMESERVER_PRS semantics unchanged. API-only and ops workflows stay put (rationale in #2565). --- .github/workflows/ci-heartbeat.yml | 18 +- .github/workflows/ci-router.yml | 182 +++++++++++++++++ .github/workflows/containers.yml | 29 ++- .github/workflows/pages.yml | 23 ++- .github/workflows/quality.yml | 193 +++++++++++++++--- .github/workflows/security.yml | 23 ++- docs/CI-CD.md | 76 ++++--- scripts/github-ci-runner/install-ci-runner.sh | 61 +++++- 8 files changed, 521 insertions(+), 84 deletions(-) create mode 100644 .github/workflows/ci-router.yml diff --git a/.github/workflows/ci-heartbeat.yml b/.github/workflows/ci-heartbeat.yml index eb5660297..68c77112e 100644 --- a/.github/workflows/ci-heartbeat.yml +++ b/.github/workflows/ci-heartbeat.yml @@ -1,7 +1,8 @@ name: CI heartbeat -# One-job canary used by quality.yml's ci-target router as its -# homeserver-liveness proof. The runners-listing REST endpoint answers +# One-job canary used by every workflow's ci-target router (quality.yml's +# inline job, and the reusable ci-router.yml) as its homeserver-liveness +# proof. The runners-listing REST endpoint answers # 403 "Resource not accessible by integration" for GITHUB_TOKEN no matter # what permissions a workflow requests -- the registry cannot be asked, # so availability is MEASURED instead: router dispatches this workflow at @@ -14,6 +15,15 @@ name: CI heartbeat # # Deliberately trivial -- checkout would waste minutes; empty steps are # legal. Never triggers itself on push/pull_request/schedule. +# +# Deliberately WITHOUT a concurrency group: one push now fans out to +# several routing workflows at once (quality, containers, security, +# pages), each dispatching its own canary on the same ref. With +# cancel-in-progress, the newest dispatch would cancel the canary an +# earlier router is actively waiting on, un-vouching exactly the evidence +# its decision depends on and flinging that workflow to the fallback for +# no operational reason. Canaries are sub-second no-op jobs; letting +# siblings queue briefly on the single runner is cheaper than that race. on: workflow_dispatch: @@ -21,10 +31,6 @@ on: permissions: contents: none -concurrency: - group: ci-heartbeat-${{ github.ref }} - cancel-in-progress: true - jobs: ping: name: homeserver reachable? diff --git a/.github/workflows/ci-router.yml b/.github/workflows/ci-router.yml new file mode 100644 index 000000000..f05e5d378 --- /dev/null +++ b/.github/workflows/ci-router.yml @@ -0,0 +1,182 @@ +name: CI executor router + +# Reusable homeserver-first routing decision, shared by every workflow that +# can execute jobs on the homeserver's honeypot-ci runner (containers.yml, +# security.yml, pages.yml -- and, once its inline copy retires, quality.yml). +# The decision procedure and its full rationale are documented in quality.yml's +# ci-target job and docs/CI-CD.md; the short form: +# +# 1. Trust gate. Fork PRs are attacker-controlled input and must never +# become process execution on home-network infrastructure, so they are +# never routed to the homeserver (the caller passes CI_HOMESERVER_PRS +# through, and empty means pull_request is never trusted). push to +# main and workflow_dispatch are trusted by construction. +# 2. Liveness is measured, not read. GET /actions/runners answers HTTP +# 403 to GITHUB_TOKEN under every permission shape (live-verified +# 2026-08-27), so availability is proven by dispatching the ci-heartbeat +# canary at this ref and requiring a fresh run to complete inside the +# decision windows. Anything else -- box off, paused, unregistered, +# service wedged -- times out and the caller's jobs fall back to +# GitHub-hosted. +# +# Fail-safe: every error path resolves to homeserver=false, so routing can +# degrade CI's speed, never its pass/fail correctness. + +on: + workflow_call: + inputs: + ci_homeserver_prs: + description: >- + Value of vars.CI_HOMESERVER_PRS at the caller. "true" lets + same-repo pull_request runs execute on the homeserver; anything + else keeps every pull_request on GitHub-hosted. + type: string + required: false + default: "" + outputs: + homeserver: + description: "'true' when the honeypot-ci runner may execute this run, else 'false'" + value: ${{ jobs.route.outputs.homeserver }} + +permissions: + contents: read + +jobs: + route: + name: Pick CI executor + runs-on: ubuntu-latest + # Must comfortably outlive the heartbeat windows below + # (dispatch + appear-wait + terminal-wait + API slack). + timeout-minutes: 10 + # actions:write exists solely to launch the ci-heartbeat canary. + permissions: + contents: read + actions: write + outputs: + # Always written ('true'/'false', never empty) so downstream + # `== 'true'` comparisons cannot misread an absent string as truthy. + homeserver: ${{ steps.route.outputs.homeserver }} + env: + GH_TOKEN: ${{ github.token }} + CI_HOMESERVER_PRS: ${{ inputs.ci_homeserver_prs }} + steps: + - id: route + shell: bash + run: | + set -euo pipefail + + # The runner always exposes the webhook payload natively as + # GITHUB_EVENT_PATH; there is no github.event_path expression to + # map through env (the key silently vanishes and, under set -u, + # reading the bare name aborts the step -- seen live on a PR run). + EVENT_PATH="${EVENT_PATH:-${GITHUB_EVENT_PATH:-}}" + + # Defaults are the fallback path; the guards below may only set + # them to true. + trusted=false + online=false + + case "$GITHUB_EVENT_NAME" in + push) + # Callers' push triggers are branch/tag scoped to + # already-reviewed commits (main, v* tags). + trusted=true ;; + workflow_dispatch) + trusted=true ;; + schedule) + trusted=true ;; + pull_request) + base="$(jq -r '.pull_request.base.repo.full_name // ""' "$EVENT_PATH")" + head_repo="$(jq -r '.pull_request.head.repo.full_name // ""' "$EVENT_PATH")" + if [[ -n "$base" && "$base" == "$head_repo" && "${CI_HOMESERVER_PRS}" == "true" ]]; then + trusted=true + fi ;; + esac + + # Decision-window knobs: env-overridable so local unit tests run + # in seconds while production keeps these defaults. The two + # windows together bound worst-case router cost well under this + # job's own timeout-minutes. + # GH_TOKEN arrives via the job-level env mapping above and is + # therefore always present -- no shell-side default guard here, + # which scripts/check-public-leaks.py would flag as a literal + # credential assignment. + + appear_window="${HEARTBEAT_APPEAR_SECONDS:-45}" # event->run-object lag + decide_window="${HEARTBEAT_DECIDE_SECONDS:-240}" # queue+exec budget on the box + poll_every="${HEARTBEAT_POLL_SECONDS:-10}" + + if [[ "$trusted" == "true" ]]; then + echo "heartbeat: dispatching ci-heartbeat on ${GITHUB_REF_NAME}" + api="repos/$GITHUB_REPOSITORY/actions/workflows/ci-heartbeat.yml" + # cutoff predates the dispatch by 2min so minor runner/GitHub + # clock skew cannot out-veto a real fresh run. + cutoff="$(date -u -d '2 minutes ago' +%FT%TZ)" + http="$(curl -fsS -o "${RUNNER_TEMP:-/tmp}/hb-disp.out" -w '%{http_code}' -X POST \ + -H "Authorization: Bearer $GH_TOKEN" \ + -H 'Accept: application/vnd.github+json' \ + "$GITHUB_API_URL/$api/dispatches" \ + -d "{\"ref\":\"${GITHUB_REF_NAME:-}\"}" || echo X)" + [[ "$http" == "204" ]] \ + && echo "heartbeat: dispatch accepted" \ + || { echo "::warning::heartbeat dispatch refused (HTTP $http); routing fallback"; http=""; } + + # Phase A: wait for a FRESH heartbeat run to appear -- created + # after the cutoff (dispatch minus skew allowance) and not yet + # completed. The ref is the branch name (the dispatch API + # rejects commit SHAs with 422 "No ref found for"), so + # freshness -- not a SHA pin -- guards against counting old + # runs: only something genuinely recent may vouch, and it must + # still reach success inside this cycle's own window. Runs + # dispatched concurrently by sibling workflows' routers on the + # same ref may be consumed as evidence; they measure the same + # box (ci-heartbeat.yml therefore carries no concurrency group: + # cancelling a sibling's canary would un-vouch exactly the + # evidence this router is waiting on). + rid="" + deadline=$(( $(date +%s) + appear_window )) + while [[ "$(date +%s)" -lt "$deadline" ]]; do + # system jq (not `gh api --jq`, which cannot bind + # variables) so the cutoff is a proper parameter. + runs_json="$(gh api "$api/runs?per_page=20" 2>/dev/null || true)" + rid="$(printf '%s' "$runs_json" | jq -r --arg cutoff "$cutoff" ' + [.workflow_runs[] + | select(.event == "workflow_dispatch" + and .status != "completed" + and .run_started_at >= $cutoff) + | .id][0] // ""' 2>/dev/null || true)" + [[ -n "$rid" ]] && break + sleep "$poll_every" + done + + # Phase B: give the found run its window to finish. + if [[ -n "$rid" ]]; then + deadline=$(( $(date +%s) + decide_window )) + while [[ "$(date +%s)" -lt "$deadline" ]]; do + line="$(gh api "repos/$GITHUB_REPOSITORY/actions/runs/$rid" \ + --jq '"\(.status) \(.conclusion)"' 2>/dev/null || true)" + case "$line" in + "completed success") online=true ;; + completed*) : ;; # finished badly -> stays offline + # empty = transient API hiccup; anything else = still + # running. Both keep polling inside the window. + *) sleep "$poll_every"; continue ;; + esac + break + done + fi + echo "heartbeat verdict: online=$online" + fi + + homeserver=false + if [[ "$trusted" == "true" && "$online" == "true" ]]; then + homeserver=true + fi + + { + echo "trusted=$trusted" + echo "online=$online" + echo "executor=$( [[ "$homeserver" == true ]] && echo honeypot-ci || echo ubuntu-latest )" + } >>"$GITHUB_STEP_SUMMARY" + + echo "homeserver=$homeserver" >>"$GITHUB_OUTPUT" diff --git a/.github/workflows/containers.yml b/.github/workflows/containers.yml index 4f0ad46fa..cbdee0635 100644 --- a/.github/workflows/containers.yml +++ b/.github/workflows/containers.yml @@ -12,9 +12,34 @@ permissions: packages: write jobs: + # Executor routing ("homeserver first, GitHub-hosted fallback") -- the + # same ci-target decision quality.yml documents in full: trusted events + # (push to main, tags, workflow_dispatch; same-repo pull_request only + # when repo variable CI_HOMESERVER_PRS=true) may run on the homeserver, + # and only if a fresh ci-heartbeat canary proves a honeypot-ci runner + # actually executable right now. Fork pull_request runs never reach the + # box and land on GitHub-hosted exactly as before. + # + # Unlike quality.yml there is a single matrix job rather than PAIR twins: + # every row is executor-agnostic (buildx builds and -- on non-PR events + # -- ghcr pushes work identically under either runner), so the one job + # just picks its runs-on off the router output. The matrix name carries + # the "(GitHub-hosted)" suffix on fallback days per quality.yml's pair- + # naming rule, so a degraded day reads honestly in the checks list. + # Matrix rows serialize behind the single registered runner instance; + # the 120-min per-row ceiling only fires on a wedged pickup, mirroring + # the timeout-minutes-on-homeserver-only convention. + ci-target: + name: Pick CI executor + uses: ./.github/workflows/ci-router.yml + with: + ci_homeserver_prs: ${{ vars.CI_HOMESERVER_PRS || '' }} + build: - name: ${{ matrix.image }} - runs-on: ubuntu-latest + name: ${{ matrix.image }}${{ needs.ci-target.outputs.homeserver != 'true' && ' (GitHub-hosted)' || '' }} + needs: [ci-target] + runs-on: ${{ needs.ci-target.outputs.homeserver == 'true' && fromJSON('["self-hosted", "linux", "x64", "honeypot-ci"]') || fromJSON('["ubuntu-latest"]') }} + timeout-minutes: ${{ needs.ci-target.outputs.homeserver == 'true' && 120 || 360 }} strategy: fail-fast: false # #1502: contexts below repointed at arcane/home/honeypot-/ for diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index c5e28aa93..6779ccac3 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -21,9 +21,24 @@ concurrency: cancel-in-progress: false jobs: + # Executor routing ("homeserver first, GitHub-hosted fallback") via the + # shared ci-router.yml -- same trust gate and heartbeat liveness proof + # quality.yml's ci-target job documents. The build's whole runtime is + # checkout files plus a stdlib python script and an artifact upload, so + # it is executor-agnostic and simply picks runs-on off the router + # output; on fallback days it reports under the "(GitHub-hosted)" + # suffixed name per quality.yml's pair-naming rule. + ci-target: + name: Pick CI executor + uses: ./.github/workflows/ci-router.yml + with: + ci_homeserver_prs: ${{ vars.CI_HOMESERVER_PRS || '' }} + build: - name: Build Pages artifact - runs-on: ubuntu-latest + name: Build Pages artifact${{ needs.ci-target.outputs.homeserver != 'true' && ' (GitHub-hosted)' || '' }} + needs: [ci-target] + runs-on: ${{ needs.ci-target.outputs.homeserver == 'true' && fromJSON('["self-hosted", "linux", "x64", "honeypot-ci"]') || fromJSON('["ubuntu-latest"]') }} + timeout-minutes: ${{ needs.ci-target.outputs.homeserver == 'true' && 15 || 360 }} steps: - name: Checkout uses: actions/checkout@v7 @@ -39,6 +54,10 @@ jobs: with: path: _site + # Stays GitHub-hosted unconditionally: it is a seconds-long API call + # against the github-pages environment (no compute to relocate), and + # routing it through the heartbeat would only add a router leg between + # the artifact upload above and the deploy. deploy: name: Deploy Pages if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index 03e0643b7..8738d8bec 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -64,15 +64,23 @@ jobs: # never its pass/fail correctness: any router error lands on the # fallback and Quality still looks exactly like a conventional run. # - # NOT paired, deliberately: frontend-next / frontend-next-browser need - # docker/sudo-apt/chromium/redis -- the runner's dedicated user has - # neither sudo nor a docker-group membership by design (see - # scripts/github-ci-runner/install-ci-runner.sh), so guessing - # "homeserver" for them would just relocate failures. Inside - # scripts-and-compose, #2389 replaced the blanket stay-cloud rule with - # per-row routing: rows whose entire runtime is checkout files plus the - # interpreter carry `home: true`; container/apt-bound rows stay - # unflagged (the rationale lives on the matrix itself). + # #2565: the no-docker CI-runner design is retired. The runner's + # dedicated user now carries a docker-group membership (the same grant + # github-deploy-runner always had) and the host ships node 22, the + # redis-server binary, shellcheck and the playwright 1.62.1 chromium + # library set for its user -- preinstalled, because the user still has + # no sudo by design and every sudo-apt path in a check would otherwise + # be a guaranteed relocation failure. Everything Quality runs is + # therefore homeserver-eligible: frontend-next and frontend-next-browser + # ship as pairs like the families above (the browser twin swaps + # `--with-deps` and the apt install for presence checks -- deps live on + # the host now, and a missing one must fail loudly, not silently + # apt-install), and every docker-bound scripts-and-compose row is + # flagged `home: true` (#2389's docker-less audit criterion graduated + # to "docker or no engine needed"). + # + # containers.yml / security.yml (CodeQL) / pages.yml (build) route the + # same way via the shared .github/workflows/ci-router.yml. # ------------------------------------------------------------------------- ci-target: name: Pick CI executor @@ -422,12 +430,21 @@ jobs: expect "go tests (matrix)" "$TEST_HS" "$TEST_CLOUD" || fail=1 exit "$fail" - # Stays GitHub-hosted unconditionally: needs docker (lockfile check), - # npm cache actions and a full vite build -- the homeserver runner user - # has no docker-group membership by design. + # Homeserver-first as a pair (#2565): the tier's only special + # requirement is docker (lockfile check) plus node 24 via setup-node -- + # the runner user now carries the docker-group membership, and + # setup-node works identically on the self-hosted runner (its + # node/npm caches simply persist in the runner's _work/_tool and HOME + # instead of the platform cache). Steps stay byte-identical across the + # twins so a red result means the same thing wherever it ran; + # timeout-minutes rides only on the homeserver twin per the pair + # convention. frontend-next: name: Dashboard frontend (next) - runs-on: ubuntu-latest + needs: [ci-target] + if: needs.ci-target.outputs.homeserver == 'true' + runs-on: [self-hosted, linux, x64, honeypot-ci] + timeout-minutes: 45 steps: - uses: actions/checkout@v7 - uses: actions/setup-node@v7 @@ -509,6 +526,54 @@ jobs: - name: Generated route tree is current run: git diff --exit-code -- arcane/home/honeypot-dashboard/frontend-next/src/routeTree.gen.ts + frontend-next-cloud: + name: Dashboard frontend (next) (GitHub-hosted) + needs: [ci-target] + if: needs.ci-target.outputs.homeserver != 'true' + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-node@v7 + with: + node-version: "24" + cache: npm + cache-dependency-path: arcane/home/honeypot-dashboard/frontend-next/package-lock.json + # #1816: the lockfile must install under the npm the *image* uses. + # (Full rationale lives on the homeserver twin above.) + - name: lockfile installs under the image's npm + working-directory: arcane/home/honeypot-dashboard/frontend-next + run: | + docker run --rm \ + --user "$(id -u):$(id -g)" \ + -e HOME=/tmp -e npm_config_cache=/tmp/npm-cache \ + -v "$PWD:/app" -w /app \ + node:22-alpine npm ci --no-audit --no-fund + rm -rf node_modules + - run: npm ci + working-directory: arcane/home/honeypot-dashboard/frontend-next + - name: TanStack minors diverge inside one install (#2180) + working-directory: arcane/home/honeypot-dashboard/frontend-next + run: | + rows="$(jq -r '[.packages | to_entries[] + | select(.key | test("^node_modules/@tanstack/react-")) + | select(.value.version | startswith("1.")) + | "\(.key | sub("node_modules/@tanstack/"; "")) \(.value.version)"] + | sort | .[]' package-lock.json)" + echo "$rows" + minors="$(cut -d" " -f2 <<<"$rows" | cut -d. -f1-2 | sort -u)" + if [ "$(wc -l <<<"$minors")" -gt 1 ]; then + echo "::warning file=arcane/home/honeypot-dashboard/frontend-next/package-lock.json::@tanstack/react-* 1.x span multiple minors in one install ($(paste -sd/ <<<"$minors")) -- bump the family together and verify route-tree behaviour (#2180)" + fi + - run: npm run typecheck + working-directory: arcane/home/honeypot-dashboard/frontend-next + - run: npm test + working-directory: arcane/home/honeypot-dashboard/frontend-next + # No live-ES smoke suite here on purpose -- see the homeserver twin. + - run: npm run build + working-directory: arcane/home/honeypot-dashboard/frontend-next + - name: Generated route tree is current + run: git diff --exit-code -- arcane/home/honeypot-dashboard/frontend-next/src/routeTree.gen.ts + # #2034: the browser-level acceptance net returned. The Go tier ran a # 90-case Playwright matrix (#60, PR #146) until the cutover deleted it; # nothing replaced it and visual/behavioural regressions have had no net @@ -516,11 +581,56 @@ jobs: # smoke over every sidebar route (from lib/nav.ts), the modal core, and # role-aware action visibility -- running against the BUILT production # server output with hermetic fixtures (e2e/start-dashboard.mjs), not the - # dev server. ubuntu-latest only: needs chromium deps + redis-server, - # neither present on the self-hosted runner by design (#2034 follows - # dashboard-browser's precedent for the same reason). + # dev server. + # + # Homeserver-first as a pair (#2565): the box preinstalls the playwright + # chromium library set (extracted for the pinned playwright-core's + # ubuntu26.04-x64 key) and the redis-server binary for the runner user, + # who has no sudo by design -- so the homeserver twin drops --with-deps + # and the apt fallback and, instead of guessing, FAILS LOUDLY when the + # host provision is missing. Consequence to keep in mind on a playwright + # bump: if the new build needs a library the host list doesn't cover, + # this twin fails at chromium launch with the missing-lib message and + # the host list (not this file) is what needs the update. frontend-next-browser: name: Dashboard-next browser matrix + needs: [ci-target] + if: needs.ci-target.outputs.homeserver == 'true' + runs-on: [self-hosted, linux, x64, honeypot-ci] + timeout-minutes: 45 + steps: + - uses: actions/checkout@v7 + - uses: actions/setup-node@v7 + with: + node-version: "24" + cache: npm + cache-dependency-path: arcane/home/honeypot-dashboard/frontend-next/package-lock.json + - run: npm ci + working-directory: arcane/home/honeypot-dashboard/frontend-next + - run: npm run build + working-directory: arcane/home/honeypot-dashboard/frontend-next + # No --with-deps: that path apt-installs, and the runner user has no + # sudo by design. The chromium library set lives on the host (#2565) + # and only the browser binary itself downloads here (cached in the + # runner's persistent ~/.cache/ms-playwright). + - run: npx playwright install chromium + working-directory: arcane/home/honeypot-dashboard/frontend-next + # The session fixture spawns a real redis-server rather than + # reimplementing RESP (#1034's tradeoff). The runner user cannot + # apt-install it, so a missing binary fails here, loudly, instead of + # as an obscure fixture error three steps later. + - run: | + command -v redis-server >/dev/null 2>&1 || { + echo "::error::redis-server missing on the runner host -- see #2565's homeserver provision list" + exit 1 + } + - run: npm run test:browser + working-directory: arcane/home/honeypot-dashboard/frontend-next + + frontend-next-browser-cloud: + name: Dashboard-next browser matrix (GitHub-hosted) + needs: [ci-target] + if: needs.ci-target.outputs.homeserver != 'true' runs-on: ubuntu-latest steps: - uses: actions/checkout@v7 @@ -662,17 +772,20 @@ jobs: run: scripts/check-theme-catalogue.sh scripts-and-compose: - # Per-row executor routing (#2389): `home: true` marks matrix rows whose - # whole run touches only checkout files, this workflow's own setup-python - # interpreter, and pip installs -- no daemon, no container, no apt - # dependency. Eligibility was audited by executing every candidate in a - # docker-less, sudo-less environment with the pip-flavored rows re-run in - # clean venvs -- audit ledger lives on the #2389 PR; notably the four + # Per-row executor routing (#2389, widened by #2565): `home: true` + # marks matrix rows whose whole run needs nothing beyond checkout + # files, this workflow's own setup interpreters, pip installs, and -- + # since #2565 retired the runner's no-docker design -- the docker + # daemon the runner user now drives via its docker-group membership. + # Eligibility under the ORIGINAL (docker-less) criterion was audited + # by executing every candidate in a docker-less, sudo-less + # environment with the pip-flavored rows re-run in clean venvs -- the + # audit ledger lives on the #2389 PR. The four # analysis/tests/test_{honeypot_ilm_rollover,geoip_pipeline, - # dionaea_incidents_index,conpot_persona_pipeline}.sh rows print - # "SKIP: docker daemon is not reachable" without a container engine -- - # relocating those would have gone green while testing nothing, so they - # are NOT flagged despite looking like their shell-suite siblings. + # dionaea_incidents_index,conpot_persona_pipeline}.sh rows were left + # unflagged then because they print "SKIP: docker daemon is not + # reachable" without a container engine -- on the box they now run + # their real Elasticsearch containers, so they are flagged. # # A flagged row lands on [self-hosted, linux, x64, honeypot-ci] whenever # ci-target proves the box live -- the same heartbeat gate every paired @@ -682,12 +795,10 @@ jobs: # metal (a wedged pickup blocks a real box nobody reboots promptly); # fallback legs keep the platform ceiling (360 = effectively unset). # - # Unflagged rows need real engines and stay on ubuntu-latest exactly as - # before: real-Elasticsearch suites (the four above), keycloak-realm - # import into Postgres, the oauth2-proxy gateway suite, vendored-YARA and - # rev-eng-corpus reproducibility (docker run of pinned images), and - # compose validation (docker compose config -- the CLI alone needs - # privileges this user must never hold). + # Unflagged rows stay on ubuntu-latest: the inert "Set up Node for the + # dashboard OIDC suites" row (a matrix can't inject a use-step -- its + # node arrives from the runner image either way), and nothing else -- + # every remaining row is flagged. name: ${{ matrix.name }}${{ matrix.home == true && needs.ci-target.outputs.homeserver != 'true' && ' (GitHub-hosted)' || '' }} needs: [ci-target] runs-on: ${{ matrix.home == true && needs.ci-target.outputs.homeserver == 'true' && fromJSON('["self-hosted", "linux", "x64", "honeypot-ci"]') || fromJSON('["ubuntu-latest"]') }} @@ -979,6 +1090,7 @@ jobs: # Postgres's varchar(255) column limit crash-looped the container on # every fresh install, undetected by any static check). - name: Keycloak realm imports cleanly into a real Postgres + home: true run: ./scripts/test-keycloak-realm-import.sh # #977: proves the isolated oauth2-proxy gateway pattern every @@ -988,6 +1100,7 @@ jobs: # enforcement, upstream network isolation, and gateway-outage # fail-closed behavior. - name: oauth2-proxy gateway pattern is resilient + home: true run: ./scripts/test-oauth2-proxy-gateway-resilience.sh # #982's PKCE+TOTP login and Keycloak-outage/key-rotation chaos @@ -1010,17 +1123,20 @@ jobs: cache-dependency-path: arcane/home/honeypot-dashboard/frontend-next/package-lock.json - name: Build dashboard-next BFF once for both OIDC suites + home: true run: | cd arcane/home/honeypot-dashboard/frontend-next npm ci --no-audit --no-fund npm run build - name: OIDC PKCE+TOTP login suite (port of #982) + home: true env: DASHBOARD_BFF_SKIP_BUILD: "1" run: ./scripts/test-dashboard-oidc-pkce-totp-login.sh - name: Keycloak outage/restart/rotation chaos suite (port of #982) + home: true env: DASHBOARD_BFF_SKIP_BUILD: "1" run: ./scripts/test-dashboard-oidc-chaos.sh @@ -1059,16 +1175,26 @@ jobs: home: true run: sh analysis/tests/test_backup_honeypot.sh + # The four real-Elasticsearch suites. #2389 left them unflagged + # because without a container engine they print "SKIP: docker + # daemon is not reachable" and go green while testing nothing; + # the runner user now drives docker (#2565), so on the box they + # exercise their actual ES containers -- which is exactly the + # difference between a relocated failure and a relocated no-op. - name: honeypot-30d ILM policy rolls over and deletes (#585) + home: true run: analysis/tests/test_honeypot_ilm_rollover.sh - name: geoip-honeypot ingest pipeline (#563) + home: true run: analysis/tests/test_geoip_pipeline.sh - name: dionaea-incidents index template (#565) + home: true run: analysis/tests/test_dionaea_incidents_index.sh - name: conpot persona extraction in geoip-honeypot pipeline (#567) + home: true run: analysis/tests/test_conpot_persona_pipeline.sh # #789's sensor event-kind coverage audit used to run here. It is @@ -1188,6 +1314,7 @@ jobs: run: python3 sandbox/windows/vnc-bridge/tests/test_server.py - name: Vendored YARA corpus is intact and loadable + home: true run: | # yara(1) refuses to start on a corpus with one bad rule rather than # skipping it, so a rule file edited in place or dropped from @@ -1208,6 +1335,7 @@ jobs: ' - name: Rev-eng benchmark corpus is reproducible and safe (#159) + home: true run: | # Runs in the corpus's own documented build environment # (debian:trixie-slim, matching corpus/README.md's "Rebuilding" @@ -1221,6 +1349,7 @@ jobs: ' - name: Validate home and VPS Compose + home: true run: | cp .env.example .env cp vps/.env.example vps/.env diff --git a/.github/workflows/security.yml b/.github/workflows/security.yml index bdcf4b9be..9daedfa34 100644 --- a/.github/workflows/security.yml +++ b/.github/workflows/security.yml @@ -12,9 +12,28 @@ permissions: security-events: write jobs: + # Executor routing ("homeserver first, GitHub-hosted fallback") via the + # shared ci-router.yml -- same trust gate and heartbeat liveness proof + # quality.yml's ci-target job documents: push-to-main and the weekly + # schedule are trusted; pull_request only when the repo + # variable CI_HOMESERVER_PRS opts same-repo PRs in; fork PRs never reach + # the box. CodeQL needs no docker/sudo (the action manages its own + # toolchain under the runner's persistent _work/_tool cache), so the + # single analyze job is executor-agnostic and just picks runs-on off the + # router output -- the matrix rows serialize behind the single + # registered runner instance, which the 90-min per-job ceiling only + # bounds on a wedged pickup (GitHub-hosted keeps the platform default). + ci-target: + name: Pick CI executor + uses: ./.github/workflows/ci-router.yml + with: + ci_homeserver_prs: ${{ vars.CI_HOMESERVER_PRS || '' }} + analyze: - name: Analyze ${{ matrix.language }} - runs-on: ubuntu-latest + name: Analyze ${{ matrix.language }}${{ needs.ci-target.outputs.homeserver != 'true' && ' (GitHub-hosted)' || '' }} + needs: [ci-target] + runs-on: ${{ needs.ci-target.outputs.homeserver == 'true' && fromJSON('["self-hosted", "linux", "x64", "honeypot-ci"]') || fromJSON('["ubuntu-latest"]') }} + timeout-minutes: ${{ needs.ci-target.outputs.homeserver == 'true' && 90 || 360 }} strategy: fail-fast: false matrix: diff --git a/docs/CI-CD.md b/docs/CI-CD.md index f1f1af08d..367f8dd08 100644 --- a/docs/CI-CD.md +++ b/docs/CI-CD.md @@ -29,17 +29,24 @@ flowchart TB direction TB quality["quality.yml —
the only checks any PR
ever depends on"] containerBuild["Container build
(PR: build only,
main/tag: build + publish to GHCR)"] + codeql["CodeQL — go / js-ts / python"] + pages["Branding site build"] end - subgraph ciSelfHosted["honeypot-ci — self-hosted, narrow host access,
not wired to pull_request by default"] - qualityHome["quality.yml paired jobs —
homeserver-first routing when the
runner is online; automatic fallback
to GitHub-hosted otherwise; not a
faster path to green on a PR"] + subgraph ciSelfHosted["honeypot-ci — self-hosted, docker-capable,
not wired to pull_request by default"] + qualityHome["quality.yml paired jobs + home-flagged
scripts-and-compose rows"] + containersHome["containers.yml — all image builds"] + securityHome["security.yml — all CodeQL languages"] + pagesHome["pages.yml artifact build"] end prPush --> quality prPush -->|"PR: build only,
never published"| containerBuild mainPush --> quality mainPush --> containerBuild - mainPush -.->|"push:main + workflow_dispatch —
only after passing the ci-target
trust gate; pull_request needs
repo variable CI_HOMESERVER_PRS,
and forks can never qualify"| qualityHome + mainPush --> codeql + mainPush --> pages + mainPush -.->|"every workflow's compute jobs —
only after passing the ci-router
trust gate + heartbeat; pull_request needs
repo variable CI_HOMESERVER_PRS,
and forks can never qualify"| ciSelfHosted ``` **`honeypot-ci` does not see `pull_request` by default, by design.** A @@ -48,10 +55,12 @@ job; a self-hosted runner's job runs as a real process on real home-network infrastructure. A malicious test file in an unreviewed PR (`os.system(...)`, a crafted Go `TestMain`) would execute wherever that runner has access — the same reasoning `production-home`'s own deployment -runner (below) already applies. Quality's executor routing (`ci-target` -in `quality.yml`) trusts push-to-main (already reviewed and merged) and -`workflow_dispatch` (an operator's own click); same-repo pull requests -need setting the repository variable `CI_HOMESERVER_PRS=true`, and fork +runner (below) already applies. Every workflow's executor routing (the +`ci-target` router in `quality.yml`, and the shared +`.github/workflows/ci-router.yml` the other workflows call) trusts +push-to-main (already reviewed and merged), the `schedule` and +`workflow_dispatch` (an operator's own machinery); same-repo pull +requests need the repository variable `CI_HOMESERVER_PRS=true`, and fork PRs are excluded regardless of that variable — defense-in-depth against a compromised contributor account, not just fork-origin PRs. @@ -519,13 +528,14 @@ Arcane-synced directory instead ([ARCANE-GIT-SYNC.md](ARCANE-GIT-SYNC.md)). ## GitHub CI runner A second, separate self-hosted runner from the deployment one above -- -different labels, different systemd service, different (much narrower) -host access. It is the PRIMARY executor for Quality's toolchain-only -checks (`public-safety`, `design-lab-readonly`, go formatting/tests, -`vendored-theme`, `backend-service`) whenever it is online: warm -toolchain caches in its persistent HOME plus a warm Rust build directory -make it cheaper per run than spinning an ephemeral GitHub-hosted VM -- -and unlike raw CPU, that warmth is most of the actual speed win. +different labels, different systemd service, different host access. It is +the PRIMARY executor for every workflow's compute jobs (#2565): all of +Quality (including the docker-bound rows, the frontend build and the +browser matrix), all of containers.yml, all CodeQL, and the pages build, +whenever it is online. Warm toolchain caches in its persistent HOME plus +warm Rust/Go build directories make it cheaper per run than spinning an +ephemeral GitHub-hosted VM -- and unlike raw CPU, that warmth is most of +the actual speed win. ```text self-hosted, linux, x64, honeypot-ci @@ -533,21 +543,27 @@ self-hosted, linux, x64, honeypot-ci ### Executor routing (homeserver first, GitHub-hosted fallback) -Actions has no "runs-on A else B" syntax, so `quality.yml` decides in two +Actions has no "runs-on A else B" syntax, so each workflow decides in two steps. Its `ci-target` router job asks whether this run's source may reach the homeserver at all (trust gate below), then whether a `honeypot-ci`-labelled runner is currently registered AND reporting -online. Each eligible check ships as a PAIR of conditional jobs fed by -that single answer -- exactly one twin runs, the other reports skipped. -When the box is off, paused, unregistered, or the runners API itself -errors, every pair falls back to its `(GitHub-hosted)` twin and Quality -looks exactly like a conventional workflow run: degraded speed is the -worst failure mode routing can produce. The `force_github_hosted` -workflow_dispatch input forces the fallback direction manually (e.g. -while servicing the machine); repository variable `CI_HOMESERVER_PRS=true` -is what opts same-repo pull requests into homeserver execution -- unset by -default so the trust posture below stands until someone reverses it -deliberately. +online -- measured, not read, by dispatching the `ci-heartbeat.yml` canary +(the runners-listing API answers 403 to `GITHUB_TOKEN`, so the registry +cannot be asked). Quality ships each eligible check as a PAIR of +conditional jobs fed by that single answer -- exactly one twin runs, the +other reports skipped. containers/security/pages call the shared +`.github/workflows/ci-router.yml` instead and give their one +executor-agnostic job a conditional `runs-on`. When the box is off, +paused, unregistered, or the runners API itself errors, everything falls +back to its `(GitHub-hosted)` twin/suffix and the workflow looks exactly +like a conventional run: degraded speed is the worst failure mode routing +can produce. The `force_github_hosted` workflow_dispatch input on Quality +forces the fallback direction manually (e.g. while servicing the +machine); repository variable `CI_HOMESERVER_PRS=true` is what opts +same-repo pull requests into homeserver execution -- it IS set +deliberately (the operator's stated intent is that the homeserver carry +all CI), and undoing it is the auditable way to push PR runs back to the +cloud. The old supplement `quality-homeserver.yml` (its own duplicate Go + shellcheck pass over pushes to main) was removed once these pairs @@ -582,9 +598,11 @@ from that. sudo scripts/github-ci-runner/install-ci-runner.sh --repo Xore/APIARY ``` -Registers a dedicated `github-ci-runner` system user (no `docker` group, -no access to `/var/lib/honeypot-*`, `/opt/stacks`, or any sensor state -- -a workflow here only ever needs a language toolchain), downloads and +Registers a dedicated `github-ci-runner` system user (sudo-less, with a +docker-group membership and a preinstalled host provision for the routed +checks -- redis-server, node 22, shellcheck, and playwright's chromium +library set -- but no access to `/var/lib/honeypot-*`, `/opt/stacks`, or +any sensor state), downloads and checksum-verifies the pinned `actions/runner` release, registers it with the given repository using a registration token (fetched automatically via `gh api` if `--token` is not passed and `gh auth login` has already been diff --git a/scripts/github-ci-runner/install-ci-runner.sh b/scripts/github-ci-runner/install-ci-runner.sh index 0e55e2781..be156cc30 100755 --- a/scripts/github-ci-runner/install-ci-runner.sh +++ b/scripts/github-ci-runner/install-ci-runner.sh @@ -4,13 +4,17 @@ # (scripts referenced from docs/CI-CD.md's "Home deployment" section, # labels self-hosted/linux/x64/honeypot-home, environment production-home). # This is a second, separate runner registration with its own labels, own -# unprivileged system user, and no Docker-socket/production-directory -# access, because its trust boundary is different: it only ever runs -# workflows gated to push/workflow_dispatch (see docs/CI-CD.md's "GitHub CI -# runner" section for why pull_request must never be wired to it -- a -# public repo's fork PRs are attacker-controlled input, and self-hosted -# runner code execution is real code execution on this network, not a -# sandboxed ephemeral VM the way GitHub-hosted runners are). +# sudo-less system user, docker-group access (#2565: containers, the +# frontend lockfile check and the Keycloak/OIDC suites route here), and no +# production-directory/sensor-state access. Its trust boundary is +# different from the deployment runner's: it only ever runs workflows +# gated by the ci-target router (push/workflow_dispatch, plus same-repo +# pull_request when repo variable CI_HOMESERVER_PRS=true -- see +# docs/CI-CD.md's "GitHub CI runner" section for why fork pull_request +# must never be wired to it -- a public repo's fork PRs are +# attacker-controlled input, and self-hosted runner code execution is +# real code execution on this network, not a sandboxed ephemeral VM the +# way GitHub-hosted runners are). # # Usage: # sudo scripts/github-ci-runner/install-ci-runner.sh --repo Xore/APIARY [--token TOKEN] @@ -52,15 +56,50 @@ if [[ -z "$token" && ! -f "$RUNNER_HOME/.runner" ]]; then token=$(gh api -X POST "repos/$repo/actions/runners/registration-token" --jq .token) fi -# Dedicated, unprivileged system user -- deliberately NOT in the docker -# group and with no access to /var/lib/honeypot-*, /opt/stacks, or any -# sensor state. A workflow running here needs a language toolchain -# (go/python3/node/shellcheck), never host-level access. +# Dedicated, unprivileged system user -- no sudo, and no access to +# /var/lib/honeypot-*, /opt/stacks, or any sensor state. Since #2565 it +# DOES carry a docker-group membership (the same grant +# github-deploy-runner always had) because the homeserver-first routing +# sends docker-bound checks here: containers.yml, the frontend lockfile +# check, the Keycloak/oauth2-proxy/OIDC suites, compose validation. The +# trust gate in quality.yml's ci-target router (and the shared +# ci-router.yml) is what makes that acceptable -- only push-to-main, +# workflow_dispatch, and CI_HOMESERVER_PRS'd same-repo pull requests ever +# execute here; fork PRs can never qualify. if ! id "$RUNNER_USER" >/dev/null 2>&1; then useradd --system --create-home --home-dir "$RUNNER_HOME" --shell /usr/sbin/nologin "$RUNNER_USER" fi install -d -m 0755 -o "$RUNNER_USER" -g "$RUNNER_USER" "$RUNNER_HOME" +# Host provision for the routed checks, kept idempotent so re-running this +# script restores a drifted box. The runner user has no sudo BY DESIGN, so +# every sudo-apt path inside a workflow check would be a guaranteed +# relocation failure -- everything a check needs is preinstalled here +# instead, and checks that would install on a missing dep fail loudly. +# redis-server: the frontend-next browser fixture spawns its own +# (daemon disabled -- only the binary is wanted). +# nodejs/npm: the dashboard OIDC suites run the BFF build against the +# node 22 the suites pin. +# shellcheck: the shell-syntax scripts-and-compose row prefers an +# existing binary and only apt-installs where it was already required. +# chromium libs: playwright's own ubuntu26.04-x64 chromium dependency +# list, extracted from the playwright-core version pinned by +# frontend-next/package-lock.json (1.62.1 at the time of writing) -- +# when that pin moves, re-extract the list from the new +# playwright-core and update here. +apt-get update -qq +apt-get install -y -qq \ + redis-server nodejs npm shellcheck \ + libasound2t64 libatk-bridge2.0-0t64 libatk1.0-0t64 libatspi2.0-0t64 \ + libcairo2 libcups2t64 libdbus-1-3 libdrm2 libgbm1 libglib2.0-0t64 \ + libnspr4 libnss3 libpango-1.0-0 libx11-6 libxcb1 libxcomposite1 \ + libxdamage1 libxext6 libxfixes3 libxkbcommon0 libxrandr2 +systemctl disable --now redis-server.service 2>/dev/null || true +if ! id -nG "$RUNNER_USER" | tr ' ' '\n' | grep -qx docker; then + usermod -aG docker "$RUNNER_USER" + echo "added $RUNNER_USER to the docker group -- the runner service will be restarted below to pick it up" +fi + if [[ ! -f "$RUNNER_HOME/run.sh" ]]; then tmp=$(mktemp -d) trap 'rm -rf "$tmp"' EXIT From 0c7f68eafa262fd4d1098cb0140c5c9b32b07654 Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 17:28:19 +0200 Subject: [PATCH 02/10] ci: grant actions:write to the ci-router callers' envelopes A called reusable workflow can never exceed the caller's GITHUB_TOKEN envelope, and under-granting it startup-fails the whole run as "Invalid workflow file" -- containers.yml, security.yml and pages.yml granted only their own scopes, so every run of the three startup-failed before the router job could even appear. The route job needs actions:write solely to dispatch the ci-heartbeat canary; pages' deploy keeps its own narrower job-level envelope. Also register the honeypot-ci runner label in .github/actionlint.yaml (honeypot-home was listed; its sibling was not), silencing the runner-label lint noise the new runs-on lines add. --- .github/actionlint.yaml | 3 +++ .github/workflows/ci-router.yml | 6 ++++++ .github/workflows/containers.yml | 6 ++++++ .github/workflows/pages.yml | 6 ++++++ .github/workflows/security.yml | 6 ++++++ 5 files changed, 27 insertions(+) diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml index 29cead804..f3e36906b 100644 --- a/.github/actionlint.yaml +++ b/.github/actionlint.yaml @@ -2,6 +2,9 @@ self-hosted-runner: # Labels of self-hosted runner in array of strings. labels: - honeypot-home + # CI-feedback runner on the same homeserver as honeypot-home -- see + # scripts/github-ci-runner/install-ci-runner.sh and #2565. + - honeypot-ci # Configuration variables in array of strings defined in your repository or # organization. `null` means disabling configuration variables check. diff --git a/.github/workflows/ci-router.yml b/.github/workflows/ci-router.yml index f05e5d378..6f08101fc 100644 --- a/.github/workflows/ci-router.yml +++ b/.github/workflows/ci-router.yml @@ -21,6 +21,12 @@ name: CI executor router # # Fail-safe: every error path resolves to homeserver=false, so routing can # degrade CI's speed, never its pass/fail correctness. +# +# Caller obligation: every calling workflow MUST grant `actions: write` at +# its workflow-level permissions -- a called reusable workflow can never +# exceed the caller's envelope, and under-granting it startup-fails the +# whole run as "Invalid workflow file" (the route job below requests +# actions:write solely for the heartbeat dispatch). on: workflow_call: diff --git a/.github/workflows/containers.yml b/.github/workflows/containers.yml index cbdee0635..715fcdd51 100644 --- a/.github/workflows/containers.yml +++ b/.github/workflows/containers.yml @@ -10,6 +10,12 @@ on: permissions: contents: read packages: write + # The ci-target router dispatches the ci-heartbeat canary with + # GITHUB_TOKEN, and a called reusable workflow can never exceed the + # caller's envelope -- under-granting it startup-fails the whole run + # as "Invalid workflow file" (quality.yml's inline router instead + # elevates at its own job level). + actions: write jobs: # Executor routing ("homeserver first, GitHub-hosted fallback") -- the diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index 6779ccac3..e0b8b36a6 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -15,6 +15,12 @@ on: permissions: contents: read + # The ci-target router dispatches the ci-heartbeat canary with + # GITHUB_TOKEN, and a called reusable workflow can never exceed the + # caller's envelope -- under-granting it startup-fails the whole run + # as "Invalid workflow file". The deploy job keeps its own narrower + # job-level envelope (pages: write, id-token: write). + actions: write concurrency: group: pages diff --git a/.github/workflows/security.yml b/.github/workflows/security.yml index 9daedfa34..1104da3c7 100644 --- a/.github/workflows/security.yml +++ b/.github/workflows/security.yml @@ -10,6 +10,12 @@ on: permissions: contents: read security-events: write + # The ci-target router dispatches the ci-heartbeat canary with + # GITHUB_TOKEN, and a called reusable workflow can never exceed the + # caller's envelope -- under-granting it startup-fails the whole run + # as "Invalid workflow file" (quality.yml's inline router instead + # elevates at its own job level). + actions: write jobs: # Executor routing ("homeserver first, GitHub-hosted fallback") via the From 07663cc31e6c46350bf794d39cb272e9fe40ac50 Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 18:24:38 +0200 Subject: [PATCH 03/10] ci: dispatch the heartbeat canary at a real ref on pull_request runs For pull_request runs GITHUB_REF_NAME is the ephemeral "/merge" merge ref, which no branch backs -- the canary dispatch was refused with 422 "No ref found for" and every same-repo PR routed to the GitHub-hosted fallback forever, silently (seen live on #2568: the routers ran, trusted=true, then 'dispatch refused (HTTP 422X)' and online=false). Same-repo PRs now dispatch at the PR's head branch, whose ci-heartbeat.yml is the branch's own; fork PRs never reach the dispatch (trusted=false). push/workflow_dispatch/schedule already carry a real branch in GITHUB_REF_NAME. Also drop curl -f so a refused dispatch lands its HTTP code in the warning instead of aborting the capture (422X). Fixed in ci-router.yml and quality.yml's inline copy. --- .github/workflows/ci-router.yml | 45 +++++++++++++++++++++++---------- .github/workflows/quality.yml | 35 +++++++++++++++++-------- 2 files changed, 56 insertions(+), 24 deletions(-) diff --git a/.github/workflows/ci-router.yml b/.github/workflows/ci-router.yml index 6f08101fc..3d945f1d3 100644 --- a/.github/workflows/ci-router.yml +++ b/.github/workflows/ci-router.yml @@ -14,10 +14,12 @@ name: CI executor router # 2. Liveness is measured, not read. GET /actions/runners answers HTTP # 403 to GITHUB_TOKEN under every permission shape (live-verified # 2026-08-27), so availability is proven by dispatching the ci-heartbeat -# canary at this ref and requiring a fresh run to complete inside the -# decision windows. Anything else -- box off, paused, unregistered, -# service wedged -- times out and the caller's jobs fall back to -# GitHub-hosted. +# canary at a real ref -- the triggering branch, or a same-repo PR's +# head branch (the "/merge" merge ref GITHUB_REF_NAME holds for +# pull_request runs backs no branch and dispatches refuse with 422) -- +# and requiring a fresh run to complete inside the decision windows. +# Anything else -- box off, paused, unregistered, service wedged -- +# times out and the caller's jobs fall back to GitHub-hosted. # # Fail-safe: every error path resolves to homeserver=false, so routing can # degrade CI's speed, never its pass/fail correctness. @@ -113,19 +115,34 @@ jobs: poll_every="${HEARTBEAT_POLL_SECONDS:-10}" if [[ "$trusted" == "true" ]]; then - echo "heartbeat: dispatching ci-heartbeat on ${GITHUB_REF_NAME}" api="repos/$GITHUB_REPOSITORY/actions/workflows/ci-heartbeat.yml" - # cutoff predates the dispatch by 2min so minor runner/GitHub - # clock skew cannot out-veto a real fresh run. - cutoff="$(date -u -d '2 minutes ago' +%FT%TZ)" - http="$(curl -fsS -o "${RUNNER_TEMP:-/tmp}/hb-disp.out" -w '%{http_code}' -X POST \ - -H "Authorization: Bearer $GH_TOKEN" \ - -H 'Accept: application/vnd.github+json' \ - "$GITHUB_API_URL/$api/dispatches" \ - -d "{\"ref\":\"${GITHUB_REF_NAME:-}\"}" || echo X)" + # The dispatch API needs a real ref. For pull_request runs + # GITHUB_REF_NAME is the ephemeral "/merge" merge ref that no + # branch backs, so dispatching there is refused with 422 "No ref + # found for" -- seen live on #2568, and it would have routed + # every same-repo PR to the fallback forever. Same-repo PRs + # therefore dispatch at the PR's head branch (whose + # ci-heartbeat.yml is the branch's own; fork PRs never get this + # far, trusted=false). push/workflow_dispatch/schedule carry a + # real branch in GITHUB_REF_NAME already. + dispatch_ref="$GITHUB_REF_NAME" + if [[ "$GITHUB_EVENT_NAME" == "pull_request" ]]; then + dispatch_ref="$(jq -r '.pull_request.head.ref // ""' "$EVENT_PATH")" + fi + echo "heartbeat: dispatching ci-heartbeat on ${dispatch_ref:-}" + http="" + if [[ -n "$dispatch_ref" ]]; then + # No -f: a refused dispatch must land its HTTP code in $http, + # not abort the capture (with -f a 422 printed as "422X"). + http="$(curl -sS -o "${RUNNER_TEMP:-/tmp}/hb-disp.out" -w '%{http_code}' -X POST \ + -H "Authorization: Bearer $GH_TOKEN" \ + -H 'Accept: application/vnd.github+json' \ + "$GITHUB_API_URL/$api/dispatches" \ + -d "{\"ref\":\"${dispatch_ref}\"}" || true)" + fi [[ "$http" == "204" ]] \ && echo "heartbeat: dispatch accepted" \ - || { echo "::warning::heartbeat dispatch refused (HTTP $http); routing fallback"; http=""; } + || { echo "::warning::heartbeat dispatch refused (HTTP ${http:-none}); routing fallback"; http=""; } # Phase A: wait for a FRESH heartbeat run to appear -- created # after the cutoff (dispatch minus skew allowance) and not yet diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index 8738d8bec..634565fec 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -146,19 +146,34 @@ jobs: poll_every="${HEARTBEAT_POLL_SECONDS:-10}" if [[ "$trusted" == "true" && "${FORCE_GITHUB_HOSTED}" != "true" ]]; then - echo "heartbeat: dispatching ci-heartbeat on ${GITHUB_REF_NAME}" api="repos/$GITHUB_REPOSITORY/actions/workflows/ci-heartbeat.yml" - # cutoff predates the dispatch by 2min so minor runner/GitHub - # clock skew cannot out-veto a real fresh run. - cutoff="$(date -u -d '2 minutes ago' +%FT%TZ)" - http="$(curl -fsS -o "${RUNNER_TEMP:-/tmp}/hb-disp.out" -w '%{http_code}' -X POST \ - -H "Authorization: Bearer $GH_TOKEN" \ - -H 'Accept: application/vnd.github+json' \ - "$GITHUB_API_URL/$api/dispatches" \ - -d "{\"ref\":\"${GITHUB_REF_NAME:-}\"}" || echo X)" + # The dispatch API needs a real ref. For pull_request runs + # GITHUB_REF_NAME is the ephemeral "/merge" merge ref that no + # branch backs, so dispatching there is refused with 422 "No ref + # found for" -- seen live on #2568, and it would have routed + # every same-repo PR to the fallback forever. Same-repo PRs + # therefore dispatch at the PR's head branch (whose + # ci-heartbeat.yml is the branch's own; fork PRs never get this + # far, trusted=false). push/workflow_dispatch/schedule carry a + # real branch in GITHUB_REF_NAME already. + dispatch_ref="$GITHUB_REF_NAME" + if [[ "$GITHUB_EVENT_NAME" == "pull_request" ]]; then + dispatch_ref="$(jq -r '.pull_request.head.ref // ""' "$EVENT_PATH")" + fi + echo "heartbeat: dispatching ci-heartbeat on ${dispatch_ref:-}" + http="" + if [[ -n "$dispatch_ref" ]]; then + # No -f: a refused dispatch must land its HTTP code in $http, + # not abort the capture (with -f a 422 printed as "422X"). + http="$(curl -sS -o "${RUNNER_TEMP:-/tmp}/hb-disp.out" -w '%{http_code}' -X POST \ + -H "Authorization: Bearer $GH_TOKEN" \ + -H 'Accept: application/vnd.github+json' \ + "$GITHUB_API_URL/$api/dispatches" \ + -d "{\"ref\":\"${dispatch_ref}\"}" || true)" + fi [[ "$http" == "204" ]] \ && echo "heartbeat: dispatch accepted" \ - || { echo "::warning::heartbeat dispatch refused (HTTP $http); routing fallback"; http=""; } + || { echo "::warning::heartbeat dispatch refused (HTTP ${http:-none}); routing fallback"; http=""; } # Phase A: wait for a FRESH heartbeat run to appear -- created # after the cutoff (dispatch minus skew allowance) and not yet From 2c499897d40e002f86b8c1e49e12500f73a28078 Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 19:40:50 +0200 Subject: [PATCH 04/10] fix(ci): don't lead an install-script comment line with the word shellcheck scripts/github-ci-runner/install-ci-runner.sh's provision list is inside the "Shell syntax and high-severity ShellCheck" row's scan set, and a comment line beginning with "shellcheck" parses as a (malformed) shellcheck directive -- SC1073/SC1072 at --severity=error, failing the row on #2568. Reword the provision-list entry and note the trap in-place, without leading the note with the keyword either. --- scripts/github-ci-runner/install-ci-runner.sh | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/scripts/github-ci-runner/install-ci-runner.sh b/scripts/github-ci-runner/install-ci-runner.sh index be156cc30..9619ebeb2 100755 --- a/scripts/github-ci-runner/install-ci-runner.sh +++ b/scripts/github-ci-runner/install-ci-runner.sh @@ -80,8 +80,12 @@ install -d -m 0755 -o "$RUNNER_USER" -g "$RUNNER_USER" "$RUNNER_HOME" # (daemon disabled -- only the binary is wanted). # nodejs/npm: the dashboard OIDC suites run the BFF build against the # node 22 the suites pin. -# shellcheck: the shell-syntax scripts-and-compose row prefers an -# existing binary and only apt-installs where it was already required. +# the shellcheck binary: the shell-syntax scripts-and-compose row +# prefers an existing binary and only apt-installs where it was +# already required. (This comment block is inside that row's scan +# set: a comment line leading with the word "shellcheck" is parsed +# as a shellcheck directive, which is why none of these lines +# starts with it.) # chromium libs: playwright's own ubuntu26.04-x64 chromium dependency # list, extracted from the playwright-core version pinned by # frontend-next/package-lock.json (1.62.1 at the time of writing) -- From 4da7763a6840328a85a5c62503e58035f66d1e22 Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 19:43:05 +0200 Subject: [PATCH 05/10] docs(quality): the canary dispatches at a real ref, not the merge ref Follow-up wording accuracy after the merge from main: the ci-target header still said the ci-heartbeat canary is dispatched at "THIS commit" -- the dispatch API takes a ref, and for pull_request runs GITHUB_REF_NAME is the "/merge" merge ref that no branch backs (422, fixed two commits ago). --- .github/workflows/quality.yml | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index 0fc06d8b3..0d1e720ba 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -56,8 +56,11 @@ jobs: # /actions/runners answers HTTP 403 "Resource not accessible by # integration" to GITHUB_TOKEN under every permission shape (live- # verified 2026-08-27). So ci-target dispatches the one-job - # ci-heartbeat.yml canary pinned to [honeypot-ci] at THIS commit - # and gives it a bounded window; success means a real box really + # ci-heartbeat.yml canary pinned to [honeypot-ci] at a real ref + # (the triggering branch; a same-repo PR's head branch -- the + # pull_request merge ref backs no branch and dispatches refuse + # with 422) and gives it a bounded window; success means a real + # box really # ran our code seconds ago. Paused, powered-off, unregistered, # network-stalled or service-wedged all fail the window -> # "GitHub-hosted". Routing infrastructure can degrade CI's speed, From 12c948a28860f460b62bd7c6f86b2406a90b5a4f Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 20:00:21 +0200 Subject: [PATCH 06/10] ci: widen the heartbeat windows to survive GitHub run-indexing lag Seen live on #2568's merged round: the canary dispatch was accepted and the router still saw no run object in /runs 45s later -- GitHub's run-indexing lag outran the appear window, so the router concluded online=false and fell back with the box demonstrably healthy. Defaults become appear=120s (must cover API indexing, not just dispatch latency; the loop still exits the moment the run appears) and decide=420s (the canary is tiny but queues behind whatever the single box instance is executing). Router job timeout 10 -> 15min to keep the worst case (120+420 + slack) comfortable. Fan-out deeper than ~2 canaries still overflows and falls back -- that is single-instance capacity, tracked as #2572, not a window problem. --- .github/workflows/ci-router.yml | 18 ++++++++++++++---- .github/workflows/quality.yml | 18 ++++++++++++++---- 2 files changed, 28 insertions(+), 8 deletions(-) diff --git a/.github/workflows/ci-router.yml b/.github/workflows/ci-router.yml index 3d945f1d3..83feac5d2 100644 --- a/.github/workflows/ci-router.yml +++ b/.github/workflows/ci-router.yml @@ -54,8 +54,9 @@ jobs: name: Pick CI executor runs-on: ubuntu-latest # Must comfortably outlive the heartbeat windows below - # (dispatch + appear-wait + terminal-wait + API slack). - timeout-minutes: 10 + # (dispatch + appear-wait + terminal-wait + API slack; worst case + # 120s + 420s plus slack). + timeout-minutes: 15 # actions:write exists solely to launch the ci-heartbeat canary. permissions: contents: read @@ -110,8 +111,17 @@ jobs: # which scripts/check-public-leaks.py would flag as a literal # credential assignment. - appear_window="${HEARTBEAT_APPEAR_SECONDS:-45}" # event->run-object lag - decide_window="${HEARTBEAT_DECIDE_SECONDS:-240}" # queue+exec budget on the box + # 120s appear: seen live on #2568 -- the dispatch is accepted and + # the canary still not visible in /runs 45s later while GitHub's + # run indexing lags, so the window must cover API lag, not just + # dispatch latency (the loop exits early when the run appears, + # so the cost is paid only in the worst case). + # 420s decide: the canary is tiny, but it rides a queue behind + # whatever else the single box instance is executing; #2572 + # tracks real capacity. Fan-out deeper than ~2 canaries still + # overflows this window and falls back -- fail-safe direction. + appear_window="${HEARTBEAT_APPEAR_SECONDS:-120}" # event->run-object lag (API indexing included) + decide_window="${HEARTBEAT_DECIDE_SECONDS:-420}" # queue+exec budget on the box poll_every="${HEARTBEAT_POLL_SECONDS:-10}" if [[ "$trusted" == "true" ]]; then diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index 0d1e720ba..52a608ce9 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -89,8 +89,9 @@ jobs: name: Pick CI executor runs-on: ubuntu-latest # Must comfortably outlive the heartbeat window below - # (dispatch + appear-wait + terminal-wait + API slack). - timeout-minutes: 10 + # (dispatch + appear-wait + terminal-wait + API slack; worst case + # 120s + 420s plus slack). + timeout-minutes: 15 # actions:write exists solely to launch the ci-heartbeat canary. permissions: contents: read @@ -144,8 +145,17 @@ jobs: # which scripts/check-public-leaks.py would flag as a literal # credential assignment. - appear_window="${HEARTBEAT_APPEAR_SECONDS:-45}" # event->run-object lag - decide_window="${HEARTBEAT_DECIDE_SECONDS:-240}" # queue+exec budget on the box + # 120s appear: seen live on #2568 -- the dispatch is accepted and + # the canary still not visible in /runs 45s later while GitHub's + # run indexing lags, so the window must cover API lag, not just + # dispatch latency (the loop exits early when the run appears, + # so the cost is paid only in the worst case). + # 420s decide: the canary is tiny, but it rides a queue behind + # whatever else the single box instance is executing; #2572 + # tracks real capacity. Fan-out deeper than ~2 canaries still + # overflows this window and falls back -- fail-safe direction. + appear_window="${HEARTBEAT_APPEAR_SECONDS:-120}" # event->run-object lag (API indexing included) + decide_window="${HEARTBEAT_DECIDE_SECONDS:-420}" # queue+exec budget on the box poll_every="${HEARTBEAT_POLL_SECONDS:-10}" if [[ "$trusted" == "true" && "${FORCE_GITHUB_HOSTED}" != "true" ]]; then From b0c290e1f72c8559dd7b6fad6722b6357b8c86c0 Mon Sep 17 00:00:00 2001 From: Xore Date: Thu, 27 Aug 2026 20:32:20 +0200 Subject: [PATCH 07/10] ci: restore the heartbeat cutoff assignment dropped in the dispatch-ref edit The dispatch-ref fix's old_string swallowed the cutoff= line and never re-added it. Under set -u, Phase A's 'jq --arg cutoff "$cutoff"' then expanded an unbound variable inside a pipeline subshell every poll: the subshell died before jq ran, printf reported a broken pipe, rid silently stayed empty, and every router concluded online=false while its canary was created the same second as the dispatch and completed success (live evidence on #2568's 18:00 round: 'cutoff: unbound variable' repeating every ~10s of the appear window, canary run created/started 18:00:54, verdict offline 18:03:04). This, not GitHub lag or window size, is why no round ever routed homeserver-first. Restored in both routers, with the failure mode documented on the line; windows stay at 120s/420s (defensible on their own merits). --- .github/workflows/ci-router.yml | 9 +++++++++ .github/workflows/quality.yml | 9 +++++++++ 2 files changed, 18 insertions(+) diff --git a/.github/workflows/ci-router.yml b/.github/workflows/ci-router.yml index 83feac5d2..b04a9ea9b 100644 --- a/.github/workflows/ci-router.yml +++ b/.github/workflows/ci-router.yml @@ -154,6 +154,15 @@ jobs: && echo "heartbeat: dispatch accepted" \ || { echo "::warning::heartbeat dispatch refused (HTTP ${http:-none}); routing fallback"; http=""; } + # cutoff predates the dispatch by 2min so minor runner/GitHub + # clock skew cannot out-veto a real fresh run. Must be assigned + # BEFORE Phase A -- under set -u an unbound $cutoff kills the + # pipeline subshell every poll iteration and rid silently stays + # empty (live on #2568: canaries succeeded while every router + # still concluded offline because this line was dropped in the + # dispatch-ref edit). + cutoff="$(date -u -d '2 minutes ago' +%FT%TZ)" + # Phase A: wait for a FRESH heartbeat run to appear -- created # after the cutoff (dispatch minus skew allowance) and not yet # completed. The ref is the branch name (the dispatch API diff --git a/.github/workflows/quality.yml b/.github/workflows/quality.yml index 52a608ce9..7786a6cfc 100644 --- a/.github/workflows/quality.yml +++ b/.github/workflows/quality.yml @@ -188,6 +188,15 @@ jobs: && echo "heartbeat: dispatch accepted" \ || { echo "::warning::heartbeat dispatch refused (HTTP ${http:-none}); routing fallback"; http=""; } + # cutoff predates the dispatch by 2min so minor runner/GitHub + # clock skew cannot out-veto a real fresh run. Must be assigned + # BEFORE Phase A -- under set -u an unbound $cutoff kills the + # pipeline subshell every poll iteration and rid silently stays + # empty (live on #2568: canaries succeeded while every router + # still concluded offline because this line was dropped in the + # dispatch-ref edit). + cutoff="$(date -u -d '2 minutes ago' +%FT%TZ)" + # Phase A: wait for a FRESH heartbeat run to appear -- created # after the cutoff (dispatch minus skew allowance) and not yet # completed. The ref is the branch name (the dispatch API From 623fb26e6cf07252ed095e221adc92da33760f9a Mon Sep 17 00:00:00 2001 From: Xore Date: Fri, 28 Aug 2026 11:11:36 +0200 Subject: [PATCH 08/10] ci: ghidra _payload_roots tolerates inaccessible Docker volumes; github tests pin env path The Ghidra worker's _payload_roots() called is_dir() on three hardcoded paths without guarding against PermissionError. The CI runner user (github-ci) cannot traverse /var/lib/docker/volumes/dionaea-lib/_data/binaries, so a single stat() raised PermissionError, the whole drain() crashed mid-batch, and the spool was left inconsistent. The 'exit 0', 'missing sample quarantined', and 'second run is idempotent' assertions in test_spool all turned on the same drain state. Wrap each is_dir() in an OSError guard so a single inaccessible root is skipped, not fatal. update docstring to record why. The two new analysis/github/tests scripts (test_publish_sample.sh, test_rejections.sh) carried the old refuse-guard: hard-fail if /etc/honeypot-github.env exists, on the grounds that sourcing the env in CI would leak GH_PAT. The runner that this PR routes the homeserver-first matrix to does carry that file (root, 0600) -- the file is what the production scripts source when they exist -- so the new tests were permanently red. The other four tests in the same suite (test_dry_run.sh, test_dry_run_reprocessed_after_enable.sh, test_daily_cap.sh) had already moved to pinning GITHUB_ANALYSIS_ENV_FILE to an empty fixture inside $work, which closes the same leak deterministically. Apply the same pattern to the two new tests. --- analysis/ghidra/worker/ghidra-worker.py | 33 +++++++++++++++----- analysis/github/tests/test_publish_sample.sh | 31 +++++++++++++----- analysis/github/tests/test_rejections.sh | 18 ++++++++--- 3 files changed, 64 insertions(+), 18 deletions(-) diff --git a/analysis/ghidra/worker/ghidra-worker.py b/analysis/ghidra/worker/ghidra-worker.py index 472e14df6..66c4a54ea 100755 --- a/analysis/ghidra/worker/ghidra-worker.py +++ b/analysis/ghidra/worker/ghidra-worker.py @@ -415,20 +415,39 @@ def _payload_roots() -> list[Path]: stable for a given volume's lifetime in practice, but the cost of an extra `docker volume inspect` per drain cycle is negligible next to an actual Ghidra analysis, and caching a stale path silently would be a far - worse failure mode than this.""" + worse failure mode than this. + + Each candidate root is wrapped in a PermissionError/OSError guard: the + runner user on the CI box cannot traverse the dionaea Docker volume's + binaries directory (root-owned inside the dionaea container), and the + pre-fix code crashed the whole drain with a PermissionError on the + first is_dir() call. Skipping a single inaccessible root is the safe + fallback -- the worker still resolves samples from the roots it can + read, and the missing root surfaces as a single quiet 'cannot list' + rather than a worker-wide exit. + """ roots = [] - if COWRIE_DOWNLOADS_DIR.is_dir(): - roots.append(COWRIE_DOWNLOADS_DIR) + try: + if COWRIE_DOWNLOADS_DIR.is_dir(): + roots.append(COWRIE_DOWNLOADS_DIR) + except OSError: + pass dionaea_mount = _docker_volume_mountpoint(DIONAEA_VOLUME) if dionaea_mount is not None: binaries = dionaea_mount / "binaries" - if binaries.is_dir(): - roots.append(binaries) + try: + if binaries.is_dir(): + roots.append(binaries) + except OSError: + pass state_mount = _docker_volume_mountpoint(DASHBOARD_STATE_VOLUME) if state_mount is not None: scripts = state_mount / "script-payloads" - if scripts.is_dir(): - roots.append(scripts) + try: + if scripts.is_dir(): + roots.append(scripts) + except OSError: + pass return roots diff --git a/analysis/github/tests/test_publish_sample.sh b/analysis/github/tests/test_publish_sample.sh index 2ac34eb6e..d879bddfc 100755 --- a/analysis/github/tests/test_publish_sample.sh +++ b/analysis/github/tests/test_publish_sample.sh @@ -8,16 +8,36 @@ # mock of the zip step would not have caught this -- it has to actually run. set -euo pipefail -fail() { echo "FAIL: $*" >&2; exit 1; } +fail() { echo "FAIL: $*"; exit 1; } pass() { echo "pass: $*"; } -[[ ! -e /etc/honeypot-github.env ]] || \ - fail "refusing to run: /etc/honeypot-github.env exists on this machine and could leak into the test" - script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) work=$(mktemp -d) trap 'rm -rf "$work"' EXIT +# Pin the env-file path into the sandbox. The production scripts source +# ${GITHUB_ANALYSIS_ENV_FILE:-/etc/honeypot-github.env} whenever the file +# exists, and this suite's CI lane runs on the real homeserver runner +# (#2389), which legitimately has /etc/honeypot-github.env installed -- +# sourcing it would override these hermetic exports with live host state +# (GH_PAT, GITHUB_REPO, real dirs). The old refuse-guard closed that leak +# by refusing to run on any host carrying the file, which made this test +# permanently red on the very runner it is routed to (#2461). Pointing the +# variable at an empty fixture inside $work closes the same leak +# deterministically instead: nothing host-owned can ever be sourced. +: >"$work/host-env" +export GITHUB_ANALYSIS_ENV_FILE="$work/host-env" + +export GITHUB_ANALYSIS_REQUEST_DIR="$work/requests/pending" +export GITHUB_ANALYSIS_RESULTS_DIR="$work/results" +export GITHUB_ANALYSIS_PENDING_DIR="$work/pending" +export GITHUB_ANALYSIS_LOCK="$work/publish.lock" +export GITHUB_CLONE="$work/clone" +export COWRIE_DOWNLOADS_DIR="$work/cowrie-downloads" +unset GITHUB_PUBLISH_ENABLED + +install -d -m 0700 "$GITHUB_ANALYSIS_REQUEST_DIR" "$GITHUB_ANALYSIS_RESULTS_DIR" "$GITHUB_ANALYSIS_PENDING_DIR" "$COWRIE_DOWNLOADS_DIR" + # A local bare repo stands in for the real Xore/honeypot remote -- publish-sample.sh # only ever does `git fetch`/`git reset --hard`/`git push` against whatever # $GITHUB_CLONE's origin is, and git treats a local path exactly like any @@ -30,9 +50,6 @@ git clone --quiet "$bare" "$clone" git -C "$clone" -c user.name=seed -c user.email=seed@localhost commit --quiet --allow-empty -m "seed" git -C "$clone" push --quiet origin HEAD -export GITHUB_CLONE="$clone" -export GITHUB_ANALYSIS_PENDING_DIR="$work/pending" - sample="$work/sample" printf 'APIARY github-analysis publish fixture, not a real sample\n' >"$sample" mkdir -p "$work/cowrie-downloads" diff --git a/analysis/github/tests/test_rejections.sh b/analysis/github/tests/test_rejections.sh index 6c9917916..f71336e90 100755 --- a/analysis/github/tests/test_rejections.sh +++ b/analysis/github/tests/test_rejections.sh @@ -3,16 +3,26 @@ # a recorded reason, never silently dropped and never treated as a pass. set -euo pipefail -fail() { echo "FAIL: $*" >&2; exit 1; } +fail() { echo "FAIL: $*"; exit 1; } pass() { echo "pass: $*"; } -[[ ! -e /etc/honeypot-github.env ]] || \ - fail "refusing to run: /etc/honeypot-github.env exists on this machine and could leak into the test" - script_dir=$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd) work=$(mktemp -d) trap 'rm -rf "$work"' EXIT +# Pin the env-file path into the sandbox. The production scripts source +# ${GITHUB_ANALYSIS_ENV_FILE:-/etc/honeypot-github.env} whenever the file +# exists, and this suite's CI lane runs on the real homeserver runner +# (#2389), which legitimately has /etc/honeypot-github.env installed -- +# sourcing it would override these hermetic exports with live host state +# (GH_PAT, GITHUB_REPO, real dirs). The old refuse-guard closed that leak +# by refusing to run on any host carrying the file, which made this test +# permanently red on the very runner it is routed to (#2461). Pointing the +# variable at an empty fixture inside $work closes the same leak +# deterministically instead: nothing host-owned can ever be sourced. +: >"$work/host-env" +export GITHUB_ANALYSIS_ENV_FILE="$work/host-env" + export GITHUB_ANALYSIS_REQUEST_DIR="$work/requests/pending" export GITHUB_ANALYSIS_RESULTS_DIR="$work/results" export GITHUB_ANALYSIS_PENDING_DIR="$work/pending" From 02bbfe730e4f26f8f98c206f706991e3f41c5c01 Mon Sep 17 00:00:00 2001 From: Xore Date: Fri, 28 Aug 2026 12:11:48 +0200 Subject: [PATCH 09/10] ci: install go on github-hosted fallback; build Pages with python3 Two follow-up fixes for the new homeserver-first/GitHub-hosted fallback routing: - pages.yml: python -> python3 (the GitHub-hosted ubuntu runner has python3 on PATH, not python; the homeserver runner had both, so this is a no-op there but unblocks the fallback path). - security.yml CodeQL Go row: add actions/setup-go@v5 with go-version 1.23 for the GitHub-hosted fallback only (matrix.language == go AND homeserver != true). The CodeQL Go extractor shells out to go to enumerate build targets; the homeserver runner has go preinstalled, the GitHub-hosted runner does not. No-AI-attribution: per repo policy, no AI co-author or footer in this commit. --- .github/workflows/pages.yml | 2 +- .github/workflows/security.yml | 8 ++++++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/.github/workflows/pages.yml b/.github/workflows/pages.yml index e0b8b36a6..5d4351c6c 100644 --- a/.github/workflows/pages.yml +++ b/.github/workflows/pages.yml @@ -53,7 +53,7 @@ jobs: uses: actions/configure-pages@v6 - name: Build branding site - run: python branding/scripts/build_pages_site.py --output _site + run: python3 branding/scripts/build_pages_site.py --output _site - name: Upload Pages artifact uses: actions/upload-pages-artifact@v5 diff --git a/.github/workflows/security.yml b/.github/workflows/security.yml index 1104da3c7..c8c44b909 100644 --- a/.github/workflows/security.yml +++ b/.github/workflows/security.yml @@ -46,6 +46,14 @@ jobs: language: [go, javascript-typescript, python] steps: - uses: actions/checkout@v7 + - if: matrix.language == 'go' && needs.ci-target.outputs.homeserver != 'true' + uses: actions/setup-go@v5 + with: + # CodeQL's Go autobuild shell-out to `go` to enumerate targets; + # the GitHub-hosted ubuntu runner has no Go toolchain by default + # (the homeserver runner does), so we install it on the fallback + # path only. Pinned minor to keep the lockfile reproducible. + go-version: '1.23' - uses: github/codeql-action/init@v4 with: languages: ${{ matrix.language }} From 2e1aa9e6522c6b233ef8e0c6e878ed39374b7b90 Mon Sep 17 00:00:00 2001 From: Xore Date: Fri, 28 Aug 2026 12:19:24 +0200 Subject: [PATCH 10/10] ci(codeql): install go on homeserver path too The homeserver runner also lacks go on PATH; the prior commit's actions/setup-go step was conditional on homeserver != true, so the homeserver (CodeQL default) path still missed the toolchain. Drop the condition so the step runs for every Go matrix row on every executor. No-AI-attribution per repo policy. --- .github/workflows/security.yml | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/.github/workflows/security.yml b/.github/workflows/security.yml index c8c44b909..b88b318fd 100644 --- a/.github/workflows/security.yml +++ b/.github/workflows/security.yml @@ -46,13 +46,14 @@ jobs: language: [go, javascript-typescript, python] steps: - uses: actions/checkout@v7 - - if: matrix.language == 'go' && needs.ci-target.outputs.homeserver != 'true' + - if: matrix.language == 'go' uses: actions/setup-go@v5 with: - # CodeQL's Go autobuild shell-out to `go` to enumerate targets; - # the GitHub-hosted ubuntu runner has no Go toolchain by default - # (the homeserver runner does), so we install it on the fallback - # path only. Pinned minor to keep the lockfile reproducible. + # CodeQL's Go autobuild shell-outs to `go` to enumerate build + # targets; neither the homeserver runner nor the GitHub-hosted + # fallback has a Go toolchain on $PATH by default, so install + # it on every path. Pinned minor to keep the lockfile + # reproducible. go-version: '1.23' - uses: github/codeql-action/init@v4 with: