diff --git a/.github/workflows/diagnostics.yml b/.github/workflows/diagnostics.yml index 3415d2dd..c6674fab 100644 --- a/.github/workflows/diagnostics.yml +++ b/.github/workflows/diagnostics.yml @@ -37,6 +37,42 @@ name: Diagnostics # adding coverage. A manual (workflow_dispatch) run never turns red for any # of the above; it keeps the original browsable report-only style, since a # human is already looking at it. +# +# #3312: this workflow ran 160 consecutive scheduled runs in failure, and the +# red X meant different things in each of them -- a real host fault, a lane +# that had never been able to see anything, a check asking from a vantage +# point the endpoint answers differently to, and a declared, deliberate +# stand-down -- all filed identically, with an ::error:: line that named none +# of them. Nobody triaged it, and that is the whole reason this file now +# has a vocabulary. Every finding is categorised on the way out, into one of +# exactly four categories, and each job ends with a ledger of what it found +# and what kind of thing each finding was: +# +# fault a real regression. Fatal, as before. +# runner-config the check could not run because of how the runner or this +# repository's environment is configured -- a helper not +# installed, a sudoers grant not applied, a secret unset. +# Still fatal: folding "could not tell" into a pass is the +# outcome that makes a check worse than not having it, and +# #3283 is what that costs when the lane is dark for six +# days. But it is its own category with its own annotation +# title, so it reads as an operator command to run rather +# than as a broken pipeline. +# expected a deliberate, declared absence. Never fatal. +# unmeasured the check did not run, and that is not a pass. Fatal. +# +# The vocabulary, and the four-category argument, live in +# scripts/diagnostics-lib.sh, which every step here sources. A check is never +# deleted and no finding is ever folded into a pass to make this run green -- +# the only thing this change does is make the run say which of the four +# situations it is in. +# +# Both jobs check the repo out now (#2908). Every other step still reads the +# deployed stack under /opt/stacks/apiary -- the checkout is for the scripts, +# so that a check asks its question with the code that was just fixed rather +# than with a copy refreshed only by a workflow_dispatch-only deploy. For the +# five weeks of #3312 the red X kept naming things #3338 had already fixed +# precisely because the deployed isolation-audit.sh was weeks behind. on: workflow_dispatch: @@ -78,30 +114,28 @@ jobs: runs-on: [self-hosted, linux, x64, honeypot-home] environment: production-home steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Inspect the home Compose stack shell: bash run: | set -uo pipefail stack=/opt/stacks/apiary - # #2222: on a scheduled run, alert() additionally turns the - # condition into the run's own red X (no human reads + # #2222: on a scheduled run, a categorised finding additionally turns + # the condition into the run's own red X (no human reads # GITHUB_STEP_SUMMARY on a cron trigger); on workflow_dispatch it # behaves exactly like report(). is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} schedule_failed=0 - report() { printf '%s\n' "$*" >> "$GITHUB_STEP_SUMMARY"; printf '%s\n' "$*"; } - alert() { - report "$*" - if [ "$is_schedule" = "true" ]; then - printf '::error::%s\n' "$*" - schedule_failed=1 - fi - } - section() { report ''; report "## $*"; report ''; } + # report/alert/note/section and the ledger come from the repo (#3312): + # five steps in two jobs have to agree on the vocabulary, and a + # second copy of it in YAML is a second thing to keep in step. + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" report "# Home stack diagnostics" if [ ! -d "$stack" ]; then - report "Stack directory is missing. The home stack has never been deployed to this host." + note expected "stack directory" "Stack directory is missing. The home stack has never been deployed to this host, so there is nothing here to inspect -- expected, not a fault." exit 0 fi @@ -142,7 +176,7 @@ jobs: if curl -fsS --max-time 10 "http://$bind:19090/healthz" >/dev/null 2>&1; then report "healthz: responding" else - alert "healthz: **unreachable** — the dashboard container is down or not bound" + alert fault "dashboard healthz" "healthz: **unreachable** — the dashboard container is down or not bound" fi # #3315: "green, healthy, running the old code" is indistinguishable @@ -228,7 +262,14 @@ jobs: # this workflow cannot make for them. helper=/opt/github-ci-runner-helpers/dashboard-source-health.sh if ! sudo -n -l "$helper" >/dev/null 2>&1; then - alert "metrics unavailable: $helper is not installed or not granted -- apply it with: sudo scripts/github-ci-runner/install-deploy-runner.sh --helpers-only (no runner restart; a full re-run also stops and starts the runner service)" + # #3312: the category is the point. This lane has never been able + # to see anything on a scheduled run -- the token is root-only and + # the job is not root -- and it was reported under the same red X + # as a dead sensor pipeline, which is why the ingest outage in + # #3283 went unread for six days. It stays fatal (an unmeasurable + # lane is not a pass), but as a runner-config gap whose fix is the + # one command below, not as a fault in the pipeline. + alert runner-config "sensor -> Elasticsearch -> dashboard" "metrics unavailable: $helper is not installed or not granted -- this lane has not been able to measure anything. Apply it with: sudo scripts/github-ci-runner/install-deploy-runner.sh --helpers-only (no runner restart; a full re-run also stops and starts the runner service)" elif source_health=$(sudo -n "$helper" 2>/dev/null) \ && printf '%s' "$source_health" | jq -e . >/dev/null 2>&1; then printf '%s' "$source_health" | jq -r ' @@ -239,7 +280,10 @@ jobs: "filebeat: \(.pipeline.state)", "unattributed_24h: \(.unattributed_24h)"' | tee -a "$GITHUB_STEP_SUMMARY" else - alert "metrics unavailable: source-health unreachable or unauthorized (401 = wrong/missing service token)" + # The grant is in place and the helper ran, so this is the + # pipeline's own answer and not a provisioning gap: the token is + # wrong (401) or the backend is not answering. + alert fault "sensor -> Elasticsearch -> dashboard" "metrics unavailable: source-health unreachable or unauthorized (401 = wrong/missing service token)" fi report '```' report 'cluster_status: 1 green, 2 yellow, 3 red. sensor state ACTIVE means the feed' @@ -261,7 +305,7 @@ jobs: # window (suricata-log-maintenance.sh prunes past it). suricata_dir="$stack/logs/suricata" if ! mountpoint -q "$suricata_dir"; then - alert "\`logs/suricata\` is **not a mount point** — the sshfs mount from the VPS is gone." + alert fault "VPS Suricata mount" "\`logs/suricata\` is **not a mount point** — the sshfs mount from the VPS is gone." report "Filebeat and EveBox are reading a stale local directory, if anything." else report "\`logs/suricata\` is mounted." @@ -272,9 +316,12 @@ jobs: total=$(du -cb "$suricata_dir"/eve-*.json 2>/dev/null | tail -1 | cut -f1) report "\`$(basename "$latest")\`: last written ${age}s ago. Retained eve-*.json total: ${total} bytes." if [ "$age" -gt 900 ]; then - alert "That is over fifteen minutes old — treat the Suricata feed as **stale**." + alert fault "VPS Suricata mount" "\`$(basename "$latest")\` was last written ${age}s ago — over fifteen minutes old. Treat the Suricata feed as **stale**." fi else + # Report-only, and deliberately not a second alert: when the mount + # is already gone, "no files" is the same finding seen from the + # other side, and one cause gets one row in the ledger (#3312). report "No \`eve-*.json\` files found." fi report "$(df -h "$suricata_dir" | tail -1 | awk '{print "VPS filesystem: "$2" total, "$4" available ("$5" used)"}')" @@ -300,19 +347,21 @@ jobs: shell: bash run: | set -uo pipefail - # #2908: measure how stale the deployed copy is BEFORE trusting its - # output. /opt/stacks/apiary is refreshed only by deploy.yml's rsync - # step, which is workflow_dispatch-only, so it drifts silently -- it - # was found 18 days behind main, running a pre-#2839 isolation-audit - # whose capability-posture section simply did not exist yet. Nothing - # measured that channel: #2858's drift reporting reads the Arcane - # gitops-syncs API, which has no record covering this directory at - # all, so every project could report lastSyncCommit == main while - # every rsynced script was weeks old. + is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} + schedule_failed=0 + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" + # #2908: measure how stale the deployed copy is, and report it, + # because a drifted copy is a real finding even now that the audit + # no longer depends on it. /opt/stacks/apiary is refreshed only by + # deploy.yml's rsync step, which is workflow_dispatch-only, so it + # drifts silently -- it was found 18 days behind main, running a + # pre-#2839 isolation-audit whose capability-posture section simply + # did not exist yet. Nothing measured that channel: #2858's drift + # reporting reads the Arcane gitops-syncs API, which has no record + # covering this directory at all, so every project could report + # lastSyncCommit == main while every rsynced script was weeks old. # - # This job deliberately has no actions/checkout (see the job header: - # it inspects the stack as deployed, not the triggering ref), but the - # deployed tree is itself a git clone, so its WORKING TREE can be + # The deployed tree is itself a git clone, so its WORKING TREE can be # diffed against origin/main in place -- which catches rsync-applied # changes too, not just an un-advanced HEAD. # @@ -328,8 +377,8 @@ jobs: else drift=$(git -C /opt/stacks/apiary -c safe.directory=/opt/stacks/apiary \ diff --stat origin/main -- scripts/ 2>/dev/null | tail -20) - echo "::warning title=Deployed scripts differ from main::isolation-audit.sh below ran from a copy that does not match origin/main (see #2908)" - printf 'deployed `scripts/` DIFFERS from `origin/main` -- the audit below ran from that copy:\n\n```\n%s\n```\n' \ + echo "::warning title=Deployed scripts differ from main::the deployed copy of isolation-audit.sh is not what main has; this step audits the host with the triggering ref's copy instead (see #2908)" + printf 'deployed `scripts/` DIFFERS from `origin/main` -- anything else on this host that runs a deployed script is still using that copy:\n\n```\n%s\n```\n' \ "$drift" >> "$GITHUB_STEP_SUMMARY" fi else @@ -337,20 +386,64 @@ jobs: fi printf '\n## Isolation audit\n\n```\n' >> "$GITHUB_STEP_SUMMARY" - # Runs the deployed stack's own copy, same as every other step in - # this job -- this workflow has no actions/checkout, by design - # (see this job's header comment): it inspects the stack as - # actually deployed, not whatever ref triggered the run. - /opt/stacks/apiary/scripts/isolation-audit.sh 2>&1 | tee -a "$GITHUB_STEP_SUMMARY" - status=${PIPESTATUS[0]} + # #2908/#3312: the audit runs from this job's checkout, not from + # /opt/stacks/apiary. A check that audits the host with a stale copy + # of its own question cannot report a fix as fixed, and that is + # precisely why the red X kept naming things #3338 had already + # fixed: the audit that named them had not been redeployed. The + # audit reads only the host (docker, libvirt, iptables, the + # stand-down declaration in /etc/apiary), so running it from the + # checkout changes which questions it asks, not what it asks them + # about. + audit_output=$("$GITHUB_WORKSPACE/scripts/isolation-audit.sh" 2>&1) || status=$? + printf '%s\n' "$audit_output" | tee -a "$GITHUB_STEP_SUMMARY" printf '```\n' >> "$GITHUB_STEP_SUMMARY" - exit "$status" + # The audit categorises its own findings and prints its own counts; + # the ledger takes that line verbatim rather than re-deriving it + # here, so there is one source of truth for them (#3312). + audit_categories=$(grep -m1 '^isolation-audit: categories --' <<<"$audit_output" || true) + audit_unmeasured=$(sed -n 's/^isolation-audit: categories -- \([0-9][0-9]*\) unmeasured.*/\1/p' <<<"$audit_categories") + audit_unmeasured=${audit_unmeasured:-0} + if [ "${status:-0}" -ne 0 ] && [ "$audit_unmeasured" -gt 0 ]; then + # An unread barrier is its own category: "this barrier's status + # is unknown" is a different claim from "this invariant was + # violated", and it must never read as a pass. + alert unmeasured "isolation audit (scripts/isolation-audit.sh)" "$audit_categories -- at least one barrier could not be read at all, which is not a pass" + elif [ "${status:-0}" -ne 0 ]; then + alert fault "isolation audit (scripts/isolation-audit.sh)" "$audit_categories" + elif [ "$audit_unmeasured" -gt 0 ]; then + # The audit's own verdict is exit 0: not every UNMEAS it prints is + # one it treats as fatal (AppArmor's is advisory). It is still not + # an agreement, so it takes a row of its own instead of being filed + # as clean -- and the exit code stays the audit's to decide, since + # second-guessing it here would change what the check asks. + diag_row unmeasured "isolation audit (scripts/isolation-audit.sh)" "$audit_categories -- not fatal to the audit, but an unanswered question rather than an agreement" + else + diag_row ok "isolation audit (scripts/isolation-audit.sh)" "$audit_categories" + fi + exit "${status:-0}" + + - name: Category ledger (#3312) + # The triage view, last, and if: always() so it is printed even when + # the steps above have already reddened the run. Report-only: the + # fatal signal is each step's own exit status, so this step cannot + # turn a run red on its own account. + if: always() + shell: bash + run: | + set -uo pipefail + is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" + diag_ledger_report vps: if: github.event_name == 'schedule' || inputs.target == 'vps' || inputs.target == 'both' runs-on: ubuntu-latest environment: production-vps steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false - name: Configure deployment key shell: bash env: @@ -367,19 +460,12 @@ jobs: VPS_PORT: ${{ secrets.VPS_PORT }} run: | set -uo pipefail - # #2222: on a scheduled run, alert() additionally turns the - # condition into the run's own red X; on workflow_dispatch it + # #2222: on a scheduled run, a categorised finding additionally turns + # the condition into the run's own red X; on workflow_dispatch it # behaves exactly like report(). is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} schedule_failed=0 - report() { printf '%s\n' "$*" >> "$GITHUB_STEP_SUMMARY"; printf '%s\n' "$*"; } - alert() { - report "$*" - if [ "$is_schedule" = "true" ]; then - printf '::error::%s\n' "$*" - schedule_failed=1 - fi - } + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" ssh_options=(-i ~/.ssh/honeypot_vps -p "${VPS_PORT:-2222}" -o StrictHostKeyChecking=accept-new) report "# VPS edge diagnostics" report '```' @@ -440,12 +526,15 @@ jobs: printf '%s\n' "$vps_output" | tee -a "$GITHUB_STEP_SUMMARY" if printf '%s\n' "$vps_output" | grep -q -- '--- end of vps report ---'; then if printf '%s\n' "$vps_output" | grep -q 'No compiled suricata.rules found'; then - alert "VPS: compiled suricata.rules is missing." + alert fault "VPS Suricata ruleset" "VPS: compiled suricata.rules is missing." elif printf '%s\n' "$vps_output" | grep -q 'over twelve hours old'; then - alert "VPS: compiled suricata.rules is over twelve hours old." + alert fault "VPS Suricata ruleset" "VPS: compiled suricata.rules is over twelve hours old." fi else - alert "SSH to the VPS failed" + # The sentinel never printed, so the report was not delivered. Every + # rule above it is then unknown, not fine -- the whole edge is one + # unmeasurable question, and this row says so once. + alert unmeasured "VPS edge" "SSH to the VPS failed, or the report did not run to its end sentinel: the ruleset age, the eve files, the disk and the WireGuard peers are all unmeasured, not healthy" fi report '```' if [ "$schedule_failed" -eq 1 ]; then @@ -477,22 +566,35 @@ jobs: VPS_PORT: ${{ secrets.VPS_PORT }} run: | set -uo pipefail - report() { printf '%s\n' "$*" >> "$GITHUB_STEP_SUMMARY"; printf '%s\n' "$*"; } + is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} + schedule_failed=0 + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" ssh_options=(-i ~/.ssh/honeypot_vps -p "${VPS_PORT:-2222}" -o StrictHostKeyChecking=accept-new) report "# Traefik origin certificate" cert=/root/vps/traefik/certs/origin.pem # shellcheck disable=SC2029 # $cert is meant to expand on this side if ! not_after=$(ssh "${ssh_options[@]}" "${VPS_USER:-root}@${VPS_HOST}" \ "openssl x509 -in $cert -noout -enddate" 2>/dev/null); then - report "::error::cannot read $cert on the VPS -- Traefik has no origin certificate to serve" + # Not fatal-by-schedule-gate: there is no expiry to be early for, + # so this is a plain fault and it fails a manual run too -- which is + # what the exit 1 below already did. + alert fault "Traefik origin certificate" "cannot read $cert on the VPS -- Traefik has no origin certificate to serve" exit 1 fi not_after=${not_after#notAfter=} days_left=$(( ( $(date -d "$not_after" +%s) - $(date +%s) ) / 86400 )) report "Expires ${not_after} (${days_left} days left)." if [ "$days_left" -lt 30 ]; then - report "::error::the Traefik origin certificate expires in ${days_left} days -- renew it (docs/CGNAT-DEPLOYMENT.md, Origin certificate)" - [ "${{ github.event_name }}" = "schedule" ] && exit 1 + # #3312: this used to be `report "::error::..."` followed by + # `[ "${{ github.event_name }}" = "schedule" ] && exit 1`, whose + # non-schedule case left the step exiting 1 anyway -- the compound + # was the last command in the block, so a manual run went red for a + # warning it was explicitly meant to report only. alert() has the + # schedule gate inside it, so the intent is now the behaviour. + alert fault "Traefik origin certificate" "the Traefik origin certificate expires in ${days_left} days -- renew it (docs/CGNAT-DEPLOYMENT.md, Origin certificate). A 2-year Origin CA cert does not renew itself: nothing else in the fleet notices until every proxied hostname starts failing at Cloudflare's edge with a 526." + if [ "$schedule_failed" -eq 1 ]; then + exit 1 + fi fi - name: OIDC discovery is reachable (#1225) @@ -505,16 +607,21 @@ jobs: DOMAIN: ${{ secrets.DOMAIN }} run: | set -uo pipefail - report() { printf '%s\n' "$*" >> "$GITHUB_STEP_SUMMARY"; printf '%s\n' "$*"; } + is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} + schedule_failed=0 + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" ssh_options=(-i ~/.ssh/honeypot_vps -p "${VPS_PORT:-2222}" -o StrictHostKeyChecking=accept-new) report "# OIDC discovery check" if ssh "${ssh_options[@]}" "${VPS_USER:-root}@${VPS_HOST}" \ 'grep -q "honeypot.example" /root/vps/traefik/dynamic.yml 2>/dev/null'; then - report "::error::traefik/dynamic.yml on the live VPS contains placeholder domains (auth.honeypot.example) -- every Traefik router keyed on the real domain is unreachable. Was this file touched outside deploy.yml's own domain-substitution step? See #1225." + alert fault "VPS traefik/dynamic.yml" "traefik/dynamic.yml on the live VPS contains placeholder domains (auth.honeypot.example) -- every Traefik router keyed on the real domain is unreachable. Was this file touched outside deploy.yml's own domain-substitution step? See #1225." exit 1 fi if [ -z "$DOMAIN" ]; then - report "DOMAIN secret is not set -- cannot check OIDC discovery reachability, only checked for the placeholder string above." + # A missing secret is a provisioning gap, not a broken endpoint, and + # it stays report-only as it always was -- but it now says which of + # the two it is, so nobody reads it as "OIDC is fine". + note runner-config "OIDC discovery" "DOMAIN secret is not set -- cannot check OIDC discovery reachability, only checked for the placeholder string above. This lane is unmeasured, not passing." exit 0 fi url="https://auth.${DOMAIN}/realms/apiary/.well-known/openid-configuration" @@ -531,6 +638,24 @@ jobs: runner_status=$(curl -s -o /dev/null -w '%{http_code}' --max-time 10 "$url" || echo "000") report "OIDC discovery HTTP status from the VPS: $status (from this GitHub-hosted runner: $runner_status, informational only)" if [ "$status" != "200" ]; then - report "::error::OIDC discovery at https://auth.\${DOMAIN}/realms/apiary/.well-known/openid-configuration returned $status from the VPS, not 200 -- a fresh dashboard container start will fail outright (newOIDCAuth requires this unconditionally at startup). See #1225." + if [ "$status" = "000" ]; then + # No HTTP status at all is a different finding from a status that + # is not 200: the first is "we could not ask", the second is "the + # endpoint answered, and not with a discovery document". + alert unmeasured "OIDC discovery" "OIDC discovery at https://auth.\${DOMAIN}/realms/apiary/.well-known/openid-configuration could not be reached from the VPS (no HTTP response at all), so its status is unknown rather than broken. See #1225." + else + alert fault "OIDC discovery" "OIDC discovery at https://auth.\${DOMAIN}/realms/apiary/.well-known/openid-configuration returned $status from the VPS, not 200 -- a fresh dashboard container start will fail outright (newOIDCAuth requires this unconditionally at startup). See #1225." + fi exit 1 fi + + - name: Category ledger (#3312) + # Report-only, and last: the triage view of everything the steps above + # categorised, printed even when they have already reddened the run. + if: always() + shell: bash + run: | + set -uo pipefail + is_schedule=${{ github.event_name == 'schedule' && 'true' || 'false' }} + . "$GITHUB_WORKSPACE/scripts/diagnostics-lib.sh" + diag_ledger_report diff --git a/docs/CI-CD.md b/docs/CI-CD.md index 66c06654..5d741985 100644 --- a/docs/CI-CD.md +++ b/docs/CI-CD.md @@ -1736,10 +1736,10 @@ scratch twice, reaching the same blocker both times. **No workflow edit is needed.** Every `secrets.VPS_*` / `secrets.DOMAIN` reference already sits inside a job that declares -`environment: production-vps` — `deploy.yml`'s `vps` job (`:207`, environment -at `:210`), `diagnostics.yml`'s `vps` job (`:252`/`:255`), and +`environment: production-vps` — `deploy.yml`'s `vps` job (`:233`, environment +at `:236`), `diagnostics.yml`'s `vps` job (`:439`/`:442`), and `vps-start-blackhole.yml`'s `start-blackhole-profile` job (`:22`/`:24`). The -`home` jobs (`deploy.yml:21`, `diagnostics.yml:76`) read none of the five. +`home` jobs (`deploy.yml:21`, `diagnostics.yml:112`) read none of the five. Environment secrets also shadow repository secrets of the same name, so *writing* the environment copies is non-breaking on its own. @@ -1812,12 +1812,65 @@ Two checks depend on host provisioning rather than on the workflow (#3312): `LIBVIRT_DEFAULT_URI=qemu:///system`, because a non-root `virsh` otherwise talks to the empty per-user session and reports every network missing. Group membership only takes effect across a runner restart, so unlike the - helper grant it does need a full installer run. + helper grant it does need a full installer run. The audit's own host-side + expectation — that the sandbox stack is *up* — is suspended by a dated + declaration while it is deliberately down; see + [honeypot-network-isolation.md](honeypot-network-isolation.md#5-declared-stand-down). The OIDC discovery probe runs **from the VPS** over the job's SSH key. Cloudflare answers 403 to GitHub-hosted runner address ranges, so the runner's own result is printed for information only and never fails the job. +### Every finding is categorised (#3312) + +This workflow failed 160 consecutive scheduled runs without anybody triaging +it, which means it carried no signal at all. The reason was not that the +findings were wrong — several of them were correct — but that a real host +fault, a lane that had never been able to see anything, a check asking from a +vantage point the endpoint answers differently to, and a deliberate stand-down +all produced the same thing: a red X, an `::error::` line with nothing on its +subject, and a body nobody had time to read. A run that lists five unrelated +things under one heading gets triaged by ignoring it. + +So every finding now carries one of four categories. The category is the +annotation's own title, and each job ends with a ledger — one table, one row +per finding, plus the counts — so triage is reading six lines rather than +reconstructing a run from its log: + +| category | Meaning | Fatal on a scheduled run | +|---|---|---| +| `fault` | A real regression: the pipeline or the host is broken | yes | +| `runner-config` | The check could not run because of how the runner or this repository's environment is configured — a helper not installed, a grant not applied, a secret unset | yes | +| `unmeasured` | The check did not run, and that is not a pass | yes | +| `expected` | A deliberate, declared absence | no | + +`runner-config` being fatal is deliberate, and it is the same position +`scripts/verify-deploy.sh` already takes with its exit 2: folding "could not +tell" into a pass is the outcome that makes a check worse than not having it. +#3283 is what that costs when it is wrong — Elasticsearch at 1000/1000 shards +with every sensor's events dead-lettered for six days, in a lane whose only +question is whether the pipeline is flowing. What changes is that it is its own +category with its own title, so "this lane has never been able to measure +anything" is tellable apart from "the pipeline is broken" without opening the +log, and the fix is the operator command the finding names rather than the +symptom. + +The vocabulary lives in `scripts/diagnostics-lib.sh`, which every step sources; +`alert` takes a category and refuses a non-fatal one, `note` records a finding +that must not redden the run, and `diag_ledger_report` prints the table. The +isolation audit has its own finer-grained labels and its own footer, and the +step carries that line into the ledger verbatim rather than re-deriving it — one +source of truth for the counts. + +Both jobs now check the repository out (#2908). Every other step still reads +the deployed stack under `/opt/stacks/apiary`; the checkout is for the scripts, +so a check asks its question with the code that was just fixed. Running +`isolation-audit.sh` from a copy refreshed only by a `workflow_dispatch`-only +deploy is why the red X kept naming things that had already been fixed. The +deployed copy is still diffed against `origin/main` and reported when it drifts, +because drift is a real finding for everything else on the host that runs a +deployed script. + ### Diagnostics vs. mutating deploy ```mermaid diff --git a/docs/honeypot-network-isolation.md b/docs/honeypot-network-isolation.md index 9ebe13c6..9d240910 100644 --- a/docs/honeypot-network-isolation.md +++ b/docs/honeypot-network-isolation.md @@ -109,7 +109,10 @@ here to implement and no issue to open. Verified by `.github/workflows/diagnostics.yml` `home` job. - **sshd** bound to the management address only, never the honeypot-facing one. - `ss -tlnp | grep :22` is the check. + The check asks `ss -tlnp` for a port-22 listener, and it distinguishes the two + answers that used to be the same one: "nothing is listening on :22" (`OK`) and + "the listen address could not be read at all" (`UNMEAS`, fatal). `ss … | + grep :22` exits 1 for both, which is why the audit does not use it. - **libvirt socket** — `/var/run/libvirt/libvirt-sock` as `srwxrwx--- root:libvirt`, and `listen_tcp = 0`. The TCP socket is off by default; the failure mode is someone enabling it while debugging remotely and leaving it. @@ -119,7 +122,78 @@ here to implement and no issue to open. Verified by - **Docker socket** never mounted into a honeypot-facing container. It is root equivalent — a container that has it has the host. -## 5. Pitfalls +## 5. Declared stand-down + +Freeing host RAM and CPU for a heavy leg is standing practice here: #3135 +paused honeypot containers for a training run, and the sandbox VMs are a much +bigger claim on the same memory. The problem is not the stand-down. The problem +is that a stand-down and a broken host are *absence*, so the audit reported +both identically — `FAIL`, exit 1, no difference a reader could act on. That is +the "cries wolf on a healthy host" case, and it is the same defect as running +no check at all, because a run that is always red is a run nobody reads. + +So absence gets a name, and the name is a dated declaration on the host: + +```bash +sudo scripts/sandbox-standdown.sh declare \ + --issue '#NNNN' --until YYYY-MM-DD --reason 'why, in one line' +sudo scripts/sandbox-standdown.sh show # ACTIVE or INACTIVE, with the reason +sudo scripts/sandbox-standdown.sh clear # when the stack comes back +``` + +The file is `/etc/apiary/sandbox-standdown` (root-owned, `0644`, because the +audit runs unprivileged as `github-deploy-runner`; overridable with +`APIARY_STANDDOWN_FILE` for testing). While a live declaration is in force the +audit reports the sandbox objects it covers as `EXPECT` and exits 0, and it says +so out loud — owner issue, expiry, and the reason — in the body, in its footer, +and in the diagnostics ledger. + +The declaration is deliberately hard to leave lying around, and every one of +these failures is itself a `FAIL` rather than a shrug: + +| Declaration | Why it does not count | +|---|---| +| no `issue:`, `until:` or `reason:` | Nothing to hold the exception to a window or an owner | +| `issue: 1` rather than `issue: #1` | Not a reference. An exception nobody can be held to is the thing that rots into a permanent excuse | +| `until:` in the past | A window that has run out excuses nothing | +| `until:` unparseable | The window is unknown, and an unknown window is not a long one | +| `until:` more than `APIARY_STANDDOWN_MAX_DAYS` (14) out | That is a permanent posture change, not a stand-down, and it wants an issue and a decision rather than a file | + +What a declaration does **not** cover, deliberately: anything that is present +and wrong. A libvirt network that exists and *forwards*, or an `ACCEPT` rule in +the `FORWARD` chain for `virbr-sandbox`, is a measured fault with a live +declaration in force — the exception is for absence, not for amnesty. + +## 6. What the audit's labels mean + +Every line the audit prints carries one, so a reader never has to count columns +or infer intent from a wording change: + +| Label | Meaning | Exit contribution | +|---|---|---| +| `OK` | Measured, and it agrees with the invariant | none | +| `FAIL` | A measured violation, an exception that no longer applies, or a barrier that could not be read | **exit 1** | +| `EXPECT` | Absent on purpose, behind a live dated declaration | none | +| `UNMEAS` | The check did not run. Counted and named, never folded into a pass | **exit 1** for the isolation barriers | +| `WARN` | A triaged gap with a named owner issue, tracked and visible | none | +| `--` | Not applicable, or a note | none | + +The footer prints the counts and one verdict line, so the answer to "is this +host safe" is the last line of the output rather than a reading of the whole +body: + +``` +isolation-audit: categories -- 0 unmeasured, 0 expected-by-declaration, 0 triaged gap(s), 0 fault(s) +isolation-audit: VERDICT PASS -- every check that could be measured agrees with the invariants (0 object(s) excused by a live declaration) +``` + +`UNMEAS` being fatal for the barriers is the load-bearing part. "The `FORWARD` +chain could not be read" and "the `FORWARD` chain is not DROP" are different +findings, and only the second one says the host is unprotected. Reporting +"could not tell" as agreement is the outcome that makes a check worse than not +having it. + +## 7. Pitfalls | Symptom | Cause | |---|---| diff --git a/scripts/diagnostics-lib.sh b/scripts/diagnostics-lib.sh new file mode 100755 index 00000000..6799eb80 --- /dev/null +++ b/scripts/diagnostics-lib.sh @@ -0,0 +1,209 @@ +#!/usr/bin/env bash +# diagnostics-lib.sh -- the reporting vocabulary every step in +# .github/workflows/diagnostics.yml shares. Sourced, never executed. +# +# Why this file exists (#3312). +# +# diagnostics.yml ran 160 consecutive scheduled runs in failure, and every one +# of them was filed the same way: a red X, an ::error:: line with nothing on +# its subject, and a body that had to be read to find out whether anything was +# actually broken. Of the conditions it kept naming, most were not faults in +# the pipeline: +# +# - a service token in a root-owned 0600 .env the runner user cannot read, so +# the metrics lane had never once been able to see anything (#3338's +# root-owned helper fixes the read; the lane still stayed red until an +# operator applied the grant, which a workflow cannot do for them), +# - a Cloudflare 403 seen only from GitHub-hosted runner address ranges on an +# endpoint that answers 200 from the VPS, +# - a declared stand-down of the sandbox isolation stack, which is a standing +# practice on this host (#3135). +# +# None of those is a reason to stop looking. All of them are a reason to stop +# filing them under the same red X as a dead sensor pipeline, because a run +# that lists five unrelated things under one heading is triaged by ignoring it. +# +# So every finding is categorised, and the category is carried in two places +# that cannot drift apart: the annotation's own title, and one row per finding +# in the ledger the job prints at the end. There are four categories: +# +# fault something is wrong with the pipeline or the host. A real +# regression. Fatal. +# runner-config the check could not run because of how the runner or this +# repository's environment is configured: a helper not +# installed, a sudoers grant not applied, a secret unset. +# Fatal, deliberately -- scripts/verify-deploy.sh already +# exits 2 for exactly this case, and folding "could not tell" +# into a pass is the one outcome that makes a check worse +# than not having it. (#3283 is what that position costs when +# it is wrong: Elasticsearch at 1000/1000 shards with every +# sensor's events dead-lettered for six days, in a lane whose +# only question is whether the pipeline is flowing.) +# What changes here is that it is its own category with its +# own title, so "this lane has never been able to see +# anything" is tellable apart from "the pipeline is broken" +# without re-reading the log -- and so the fix is the +# operator command, not the symptom. +# expected a deliberate, declared absence. Never fatal. `alert` +# refuses this category and `note` is the way to record one, +# so the ledger can say "this is on purpose" out loud rather +# than leaving the reader to infer it from silence. +# unmeasured the check did not run, and that is not a pass. Fatal, and +# it is the same case scripts/verify-deploy.sh's exit 2 is. +# Isolated here because the honest sentence for it is "this +# barrier's status is unknown", which is different from every +# other claim in the table. +# +# One fatal path per underlying cause. The same condition must not be alerted +# twice under two names: a run that lists one problem five times is as +# untriageable as one that lists it zero times, and the doubled lane is how +# "source-health is not measured" ended up reading as a second broken thing +# rather than as the same gap seen from a different angle. + +# Sourced, not run. Running it would exit 0 having done nothing, which is the +# exact shape of a check that passes without having run. +if [ "${BASH_SOURCE[0]}" = "$0" ]; then + echo "diagnostics-lib.sh is a library: source it, do not run it." >&2 + exit 64 +fi + +# The three fatal categories, and the one that is deliberately not fatal. +# The three fatal categories, the one that is deliberately not fatal, and the +# ledger-only one that records a check which ran and found nothing. `ok` is not +# a finding and is never alerted or noted -- it exists so a run that measured +# something and agreed says so in the same table, and so "nothing was +# unmeasured" is as visible as "this is a fault". +# shellcheck disable=SC2034 # read by the step that sources this, not by this file +DIAG_CATEGORIES="fault runner-config unmeasured expected ok" + +# #2222: a scheduled run's own red X is the alert, because nobody is reading +# GITHUB_STEP_SUMMARY on a cron trigger. A manual run keeps the browsable +# report-only style -- a human is already looking at it. Each step sets +# is_schedule explicitly; the GITHUB_EVENT_NAME fallback is what keeps a step +# that forgot to still behave. +if [ -z "${is_schedule:-}" ]; then + if [ "${GITHUB_EVENT_NAME:-}" = "schedule" ]; then + is_schedule=true + else + is_schedule=false + fi +fi + +# Steps run in separate shells, so the ledger is a file rather than a variable. +# RUNNER_TEMP is per job, so the two jobs of this workflow get one ledger each +# -- which is the right granularity: a row names a check, and a check lives in +# one job. +DIAG_LEDGER_ROWS="${RUNNER_TEMP:-/tmp}/diagnostics-ledger.rows" + +report() { printf '%s\n' "$*" >> "$GITHUB_STEP_SUMMARY"; printf '%s\n' "$*"; } +section() { report ''; report "## $*"; report ''; } + +# A table cell is one line and no wider than a screen. GitHub renders a +# newline inside a cell as a broken row, and a pipe inside a cell as a column +# boundary, so both are flattened here rather than at each call site. +diag_cell() { + printf '%s' "$*" | tr '\n|' ' ' | tr -s ' ' | cut -c1-200 +} + +# diag_row +# The ledger row, with no annotation and no exit code. `note` and `alert` are +# built on it; it is also the entry point for a finding some other step's +# output already categorised (the isolation audit carries its own vocabulary +# and is never re-categorised here -- its own footer is the authority). +diag_row() { + printf '| %s | %s | %s |\n' \ + "$(diag_cell "$1")" "$(diag_cell "$2")" "$(diag_cell "$3")" >> "$DIAG_LEDGER_ROWS" +} + +# note +# A categorised finding that does not redden the run. Two legitimate uses: a +# declared, deliberate absence (`expected`), and a real finding this step is +# deliberately not making fatal -- naming that is the point, because "we chose +# not to make this fatal" is a decision somebody has to be able to see. Any +# other category is a call-site mistake, so it says so out loud instead of +# quietly filing it as something it is not. +note() { + local category=$1 check=$2 + shift 2 + case "$category" in + expected) + report "$*" + diag_row expected "$check" "$*" + ;; + fault | runner-config | unmeasured) + report "$*" + diag_row "$category" "$check" "$*" + printf '::warning title=%s (noted, not fatal): %s::%s\n' "$category" "$check" "$*" + ;; + *) + printf '::error title=Unknown diagnostic category::%s: "%s" is not one of the categories this workflow defines (%s)\n' "$check" "$category" "$DIAG_CATEGORIES" + diag_row fault "$check" "$* [filed as a fault: the call site named an unknown category, \"$category\"]" + ;; + esac +} + +# alert +# A categorised finding that is fatal on a scheduled run (#2222) and still +# reported on a manual one. The category is the annotation's title, so the +# checks list on the run page says which of the four things this is without +# anyone opening the log. +alert() { + local category=$1 check=$2 + shift 2 + report "$*" + diag_row "$category" "$check" "$*" + case "$category" in + fault | runner-config | unmeasured) ;; + *) + # `expected` is the case that matters: an expected absence has no + # business reddening a run, and if one ever did it would be an accident + # dressed as a policy. Say so, and stay fatal anyway -- a miscategorised + # finding must never be able to turn a run green. + printf '::error title=Wrong category for a fatal finding::%s: "%s" is not a fatal category. An expected state is a note, not an alert -- fix the call site. Staying fatal so a miscategorised finding cannot turn this run green.\n' "$check" "$category" + ;; + esac + if [ "$is_schedule" = "true" ]; then + printf '::error title=%s: %s::%s\n' "$category" "$check" "$*" + # shellcheck disable=SC2034 # the step reads this after every alert() call + schedule_failed=1 + fi +} + +# diag_ledger_report +# The whole run on one screen: what was measured, what each finding was, and +# the counts. The counts are the triage. A run with 0 faults and 1 +# runner-config row is a healthy host with one misconfigured lane, and the fix +# is an operator command rather than an incident. Counts are derived from the +# rows rather than tallied as they are added, so a row cannot be recorded +# without being counted or counted without being recorded. +diag_ledger_report() { + local faults=0 runner_config=0 unmeasured=0 expected=0 ok=0 rows=0 + if [ -f "$DIAG_LEDGER_ROWS" ]; then + rows=$(wc -l < "$DIAG_LEDGER_ROWS") + faults=$(grep -c '^| fault |' "$DIAG_LEDGER_ROWS" || true) + runner_config=$(grep -c '^| runner-config |' "$DIAG_LEDGER_ROWS" || true) + unmeasured=$(grep -c '^| unmeasured |' "$DIAG_LEDGER_ROWS" || true) + expected=$(grep -c '^| expected |' "$DIAG_LEDGER_ROWS" || true) + ok=$(grep -c '^| ok |' "$DIAG_LEDGER_ROWS" || true) + fi + report '' + report '## Category ledger (#3312)' + report '' + report 'One row per finding, filed under what kind of thing it was. A green run' + report 'with a nonzero runner-config count is not green -- it is unmeasured.' + report '' + report '| category | check | finding |' + report '| --- | --- | --- |' + if [ "$rows" -gt 0 ]; then + while IFS= read -r line; do report "$line"; done < "$DIAG_LEDGER_ROWS" + else + report '| -- | -- | nothing to categorise: this run found no fault, no runner-config gap and nothing it could not measure |' + fi + report '' + report "counts: ${ok} check(s) measured and in agreement, ${faults} fault(s), ${runner_config} runner-config gap(s), ${unmeasured} unmeasured, ${expected} expected absence(s)." + if [ "$faults" -eq 0 ] && [ "$runner_config" -eq 0 ] && [ "$unmeasured" -eq 0 ]; then + report 'Verdict: every check that could run agreed, and nothing was left unmeasured.' + else + report 'Verdict: see the rows above. A run-config gap is an operator command; a fault is an incident.' + fi +} diff --git a/scripts/isolation-audit.sh b/scripts/isolation-audit.sh index 44e92df3..f1e857b7 100755 --- a/scripts/isolation-audit.sh +++ b/scripts/isolation-audit.sh @@ -14,6 +14,16 @@ # # Must never print HP_BIND or any WireGuard address (same rule # diagnostics.yml's own steps already follow). +# +# #3312: every line is prefixed with the category it hit -- OK, FAIL, EXPECT +# (deliberately absent, per a live dated declaration; see standdown_state), +# UNMEAS (could not be measured), WARN (triaged gap) or -- (informational) -- +# and the verdict prints the counts. This is not decoration. For the five weeks +# to 2026-09-27 this audit ended 160 consecutive scheduled runs in failure, +# and of the things it named, four were the audit's own blind spots and one +# was a real host fault; a reader of a red run could not tell them apart, so +# the red X meant nothing and #3312 was the result. An expected state that +# cannot be distinguished from a broken one is the same defect as no check. set -uo pipefail # #3312: every virsh call below is bare (no -c). As a non-root caller -- @@ -24,10 +34,40 @@ set -uo pipefail # Pin the system instance unless the caller explicitly chose otherwise. export LIBVIRT_DEFAULT_URI="${LIBVIRT_DEFAULT_URI:-qemu:///system}" -fail=0 +# #3312: every finding below is printed with the CATEGORY it hit, because for +# four days this audit's whole job was indistinguishable from noise: a run +# concluded failure on 2026-08-12 through 2026-09-27, and of the four things it +# named every one was either a check that could not see the object it was +# looking at (bare `virsh` on a non-root caller's qemu:///session, #3312/#3338) +# or a host fault with no category attached to it at all. Nobody could tell +# which was which from the red X, so nobody acted, and a diagnostic that +# cries wolf on a healthy host is worse than no diagnostic. +# +# Four categories, and the label is on every line: +# +# OK measured; the invariant holds. +# FAIL measured; the invariant is broken. Real, actionable, fatal. +# EXPECT the invariant is not measurable as stated because the operator +# has DECLARED the object deliberately absent -- a dated, +# issue-referencing declaration (standdown_state below, +# scripts/sandbox-standdown.sh). Not fatal while the declaration +# is live, and it never excuses anything except the absence of +# the object it names. +# UNMEAS the check could not run: no privilege, no tool, no daemon. Not +# fatal for the report-only host-posture observations (the section +# header already says so), fatal for the isolation barriers, where +# an unread FORWARD chain is not evidence of a DROP policy -- "could +# not tell" is never folded into a pass. +# +# The footer prints the counts per category, so a reader can tell at a glance +# whether a red run is a broken host or a broken check. +fails=0 +unmeas_fatal=0 warns=0 -ok() { printf ' OK %s\n' "$*"; } -bad() { printf ' FAIL %s\n' "$*"; fail=1; } +expected=0 +unmeas=0 +ok() { printf ' OK %s\n' "$*"; } +bad() { printf ' FAIL %s\n' "$*"; fails=$((fails + 1)); } # A known, triaged gap with a named owner issue: visible on every run, but it # does NOT fail the job. diagnostics.yml exits on this script's status, so a # finding nobody can act on today would make the whole isolation audit @@ -35,10 +75,110 @@ bad() { printf ' FAIL %s\n' "$*"; fail=1; } # stop reading, which is the exact failure mode #2366 exists to end. WARN is # for "triaged, tracked, not yet done"; FAIL stays for "nobody has looked at # this", which is always actionable. -warn() { printf ' WARN %s\n' "$*"; warns=$((warns + 1)); } -info() { printf ' -- %s\n' "$*"; } +warn() { printf ' WARN %s\n' "$*"; warns=$((warns + 1)); } +# Declared-absent, and therefore not a fault. The message always carries the +# owner issue and the expiry, so a reader never has to go looking for why a +# line is not red. +expect() { printf ' EXPECT %s\n' "$*"; expected=$((expected + 1)); } +# Could not be measured. $1 = '1' when that must fail the run anyway (an +# isolation barrier whose silence would be read as safety), anything else for +# the report-only observations. +unmeasured() { + if [ "${1:-0}" = "1" ]; then + printf ' UNMEAS %s (fatal: this barrier could not be read)\n' "$2" + unmeas_fatal=$((unmeas_fatal + 1)) + else + printf ' UNMEAS %s\n' "$2" + fi + unmeas=$((unmeas + 1)) +} +info() { printf ' -- %s\n' "$*"; } section() { printf '\n== %s ==\n' "$*"; } +# --------------------------------------------------------------------------- +# Declared stand-down of the libvirt-backed sandbox stack (#3312). +# +# The sandbox networks, the nwfilter and libvirt's socket are all *absence* +# invariants: nothing complains when they go missing, which is why this script +# exists at all. But "missing" has two very different causes, and until #3312 +# the audit reported both identically: +# +# broken -- the 2026-09-23 reboot let the modular per-driver libvirt +# units win the Conflicts= race against libvirtd and took the +# socket with them (#3338), or a rebuilt host never re-ran +# sandbox/install-host.sh, so the Linux lane's network and +# nwfilter were never restored (#3027). +# expected -- an operator stood the stack down on purpose, to hand its RAM +# and CPU to a training/benchmark leg. Freeing host resources +# for a heavy leg is a standing practice here (#3135). +# +# A dated declaration (scripts/sandbox-standdown.sh writes it; the audit never +# writes it) is what tells them apart. It is deliberately hard to leave lying +# around -- see rule 3 below -- because an exception that outlives its window +# is the failure mode this whole mechanism exists to prevent. +# +# Sets STANDDOWN_ACTIVE=1 and STANDDOWN_WHY to a human sentence when a live +# declaration is present; STANDDOWN_STALE to a reason when one is present but +# does not count. Paths and the max-window are env-overridable so the test +# suite can point them at a tmpfile. +STANDDOWN_FILE="${APIARY_STANDDOWN_FILE:-/etc/apiary/sandbox-standdown}" +STANDDOWN_MAX_DAYS="${APIARY_STANDDOWN_MAX_DAYS:-14}" +standdown_state() { + STANDDOWN_ACTIVE=0 + STANDDOWN_STALE='' + STANDDOWN_WHY='' + [ -f "$STANDDOWN_FILE" ] || return 0 + local sd_issue sd_until sd_reason + sd_issue=$(sed -n 's/^issue: //p' "$STANDDOWN_FILE" | head -1) + sd_until=$(sed -n 's/^until: //p' "$STANDDOWN_FILE" | head -1) + sd_reason=$(sed -n 's/^reason: //p' "$STANDDOWN_FILE" | head -1) + if [ -z "$sd_issue" ] || [ -z "$sd_until" ] || [ -z "$sd_reason" ]; then + STANDDOWN_STALE="malformed: issue, until and reason are all mandatory" + return 0 + fi + # The writer (scripts/sandbox-standdown.sh declare) refuses anything that + # does not name an issue, but the audit is what decides pass/fail and root can + # write anything, so it re-derives the rule rather than trusting the writer. + # An exception nobody can be held to is the thing that rots into a permanent + # excuse, and a bare number is the shape that rot arrives in. + case "$sd_issue" in + '#'[0-9]*) ;; + *) + STANDDOWN_STALE="malformed: issue must be an issue reference like '#1234', got '$sd_issue' -- a stand-down with no issue behind it has no owner to hold it to the window" + return 0 + ;; + esac + local now until_epoch + now=$(date +%s) + if ! until_epoch=$(date -d "$sd_until" +%s 2>/dev/null); then + STANDDOWN_STALE="until is not a date this host can parse: $sd_issue until '$sd_until'" + return 0 + fi + if [ "$until_epoch" -le "$now" ]; then + STANDDOWN_STALE="expired $sd_until ($sd_issue) -- a window that has run out excuses nothing" + return 0 + fi + if [ "$until_epoch" -gt $(( now + STANDDOWN_MAX_DAYS * 86400 )) ]; then + STANDDOWN_STALE="$sd_issue declares a stand-down to $sd_until, more than $STANDDOWN_MAX_DAYS days out -- that is a permanent posture change, not a stand-down" + return 0 + fi + STANDDOWN_ACTIVE=1 + STANDDOWN_WHY="declared stand-down, $sd_issue, until $sd_until: $sd_reason" +} +standdown_state + +# Reports a sandbox-libvirt object as declared-absent when a live stand-down +# covers it, and as a real fault otherwise. $1 = what is missing, $2 = the +# remediation hint. Kept as one function so no caller can accidentally skip +# the "but is this declared?" question. +standdown_absent() { + if [ "$STANDDOWN_ACTIVE" = "1" ]; then + expect "$1 -- EXPECTED, not a fault ($STANDDOWN_WHY)" + else + bad "$1 -- $2" + fi +} + # --------------------------------------------------------------------------- # Every isolated sandbox bridge/network this script audits, in one place the # iptables and route sections below both read (#2295: each used to hardcode @@ -58,14 +198,34 @@ GUARDED_BRIDGES=( "honeypot-sandbox virbr-hpsbx 198.18.0.0/24 198.18.0.1 forensic-egress" ) +# --------------------------------------------------------------------------- +# Read first, and loudly: it decides how every sandbox object below is +# categorised, so a reader must never have to infer it from three identical +# absence lines further down. +section "Declared stand-down of the sandbox isolation stack" +if [ -n "$STANDDOWN_STALE" ]; then + bad "a sandbox stand-down declaration exists at $STANDDOWN_FILE but does not count: $STANDDOWN_STALE. An exception that does not apply is not an exception: clear it (scripts/sandbox-standdown.sh clear) or re-declare it, and read every absent sandbox object below as the regression it is" +elif [ "$STANDDOWN_ACTIVE" = "1" ]; then + expect "sandbox isolation stack is deliberately absent ($STANDDOWN_WHY). Absent sandbox objects below are EXPECTED, not faults; anything else that breaks below is still a FAIL, and the declaration does not cover it" +else + info "no stand-down declaration at $STANDDOWN_FILE -- the full sandbox stack is expected here (scripts/sandbox-standdown.sh declare --issue '#NNNN' --until YYYY-MM-DD --reason '...' if it is deliberately down)" +fi + # --------------------------------------------------------------------------- section "Sandbox libvirt networks: no " # 'ghosts' is the one deliberate exception (#331: WAN-permitted by design, # NAT forward is intentional) -- every OTHER sandbox network must have none. +# +# #3312: a network that is not defined at all is a different question from one +# that is defined and forwards, and a host with the sandbox stack deliberately +# stood down answers the first way. The distinction is the stand-down +# declaration (standdown_absent), not a quieter FAIL. for entry in "${GUARDED_BRIDGES[@]}"; do read -r net _ <<<"$entry" if ! virsh net-info "$net" >/dev/null 2>&1; then - bad "libvirt network '$net' does not exist (expected active, isolated)" + standdown_absent \ + "libvirt network '$net' does not exist (expected active, isolated)" \ + "if the sandbox stack is meant to be up, sandbox/install-host.sh defines it and a stray is what this check is really for" continue fi forwards=$(virsh net-dumpxml "$net" 2>/dev/null | grep -c '/dev/null 2>&1; then - bad "iptables not found" + unmeasured 1 "iptables is not installed, so the FORWARD barrier cannot be read at all" elif iptables_rules=$(sudo -n iptables -S FORWARD 2>&1); then policy=$(grep '^-P FORWARD' <<<"$iptables_rules" | awk '{print $3}') if [ "$policy" != "DROP" ]; then @@ -107,7 +267,7 @@ elif iptables_rules=$(sudo -n iptables -S FORWARD 2>&1); then done fi else - bad "could not read iptables FORWARD chain (sudo -n iptables failed: needs the isolation-audit sudoers grant)" + unmeasured 1 "could not read the iptables FORWARD chain (sudo -n iptables failed -- it needs the isolation-audit sudoers grant). The default policy is unknown, and unknown is not DROP" fi # --------------------------------------------------------------------------- @@ -131,7 +291,9 @@ section "honeypot-sandbox-strict nwfilter" if virsh nwfilter-dumpxml honeypot-sandbox-strict >/dev/null 2>&1; then ok "'honeypot-sandbox-strict' nwfilter is defined" else - bad "'honeypot-sandbox-strict' nwfilter is missing" + standdown_absent \ + "'honeypot-sandbox-strict' nwfilter is missing" \ + "sandbox/install-host.sh restores it (it was lost the same way in #3027, when a rebuild never re-ran it)" fi # --------------------------------------------------------------------------- @@ -167,7 +329,21 @@ done # --------------------------------------------------------------------------- section "Stack containers (hp-*/sbx-* only -- this host also runs unrelated stacks: dockge, pihole, ghidra/ollama, ghosts-*, etc.)" -if containers=$(docker ps -a --format '{{.Names}}\t{{.Image}}' 2>&1 | grep -E '^(hp-|sbx-)'); then +# #3312: `if containers=$(docker ps -a | grep -E '^(hp-|sbx-)')` fused two +# different answers into one FAIL. grep exits 1 when nothing matches, so a host +# with no stack containers at all was reported as "could not enumerate +# containers (docker ps failed)" -- a claim about the tool, printed when the +# truth is a claim about the deployment, and the reader had no way to tell +# which. Enumeration and emptiness are now decided separately. +all_containers=$(docker ps -a --format '{{.Names}}\t{{.Image}}' 2>&1) +if [ $? -ne 0 ]; then + bad "could not enumerate containers (docker ps failed: ${all_containers//$'\n'/ })" + containers='' +elif ! containers=$(grep -E '^(hp-|sbx-)' <<<"$all_containers"); then + bad "no hp-* or sbx-* container exists on this host at all -- the honeypot stack is not deployed here (or every one of its containers was removed). This is not a docker failure" + containers='' +fi +if [ -n "$containers" ]; then privileged_others="" while IFS=$'\t' read -r name _; do [ -z "$name" ] && continue @@ -222,8 +398,6 @@ if containers=$(docker ps -a --format '{{.Names}}\t{{.Image}}' 2>&1 | grep -E '^ else ok "NET_ADMIN/NET_RAW confined to sbx-zeek/sbx-suricata/sbx-tcpdump" fi -else - bad "could not enumerate containers (docker ps failed)" fi # --------------------------------------------------------------------------- @@ -399,19 +573,34 @@ if [ -n "${containers:-}" ]; then printf ' (capability posture: %d hardened, %d tracked gaps, see WARN lines)\n' \ "$(wc -w <<<"$cap_hardened_names")" "$warns" else - bad "could not check capability posture (container enumeration failed above)" + # No second fault for one cause: the Stack containers section above has + # already reported, with its own category, why there is no inventory (the + # docker call failed, or the host has no stack containers at all). Two + # faults for one underlying condition is how a red run stops meaning + # anything. + info "no container inventory to audit capability posture -- the Stack containers section above reports why" fi # --------------------------------------------------------------------------- section "Host posture (reports only, does not fix)" -if ss_out=$(sudo -n ss -tlnp 2>/dev/null | grep ':22 '); then - if grep -qE '10\.8\.0\.|10\.10\.10\.' <<<"$ss_out"; then +# #3312: this check used to be `if ss_out=$(sudo -n ss -tlnp | grep ':22 ')`. +# That fuses two different outcomes into one branch: "ss could not run" and +# "nothing is listening on :22" both make the pipeline's last stage exit 1, so +# both printed 'could not confirm'. On the live homeserver that line has been +# printing on every run -- an sshd-listen-address check that has never once +# actually answered the question it exists to answer, and an UNMEAS line is +# exactly as easy to overlook as a wrong one. The stages are separated now: +# run ss, and only judge its output if it ran. +if ss_out=$(sudo -n ss -tlnp 2>/dev/null); then + if ! grep -q ':22 ' <<<"$ss_out"; then + ok "nothing is listening on :22, so sshd cannot be on a honeypot-facing address" + elif grep -qE '10\.8\.0\.|10\.10\.10\.' <<<"$ss_out"; then bad "sshd appears to be listening on a honeypot-facing address" else ok "sshd is not listening on a honeypot-facing address" fi else - info "could not confirm sshd listen address (needs the isolation-audit sudoers grant for 'ss -tlnp')" + unmeasured 0 "could not read the sshd listen address at all (sudo -n ss -tlnp failed -- it needs the isolation-audit sudoers grant, or ss is not on this host's PATH). This check has not run; it is not a pass" fi sock_path=/var/run/libvirt/libvirt-sock @@ -447,7 +636,9 @@ if [ -S "$sock_path" ]; then bad "libvirt socket is $mode $owner -- expected root:libvirt with no world access, or an active polkit rule gating org.libvirt.unix.manage" fi else - bad "libvirt socket not found at $sock_path" + standdown_absent \ + "libvirt socket not found at $sock_path" \ + "the monolithic libvirtd.socket is dead: the per-driver modular units (virtqemud, virtnetworkd, virtnwfilterd, ...) each Conflicts= libvirtd and win the race after a reboot unless install-homeserver.sh's monolithic branch disabled all of them (#3338). Enable it with: systemctl enable --now libvirtd.socket" fi sock_ro_path=/var/run/libvirt/libvirt-sock-ro @@ -470,7 +661,9 @@ if [ -S "$sock_ro_path" ]; then warn "libvirt read-only socket is $mode $owner -- unauthenticated read-only VM enumeration is possible (no polkit rule gates the RO monitor actions); accepted as read-only exposure on this honeypot host, tracked in #3039" fi else - bad "libvirt read-only socket not found at $sock_ro_path" + standdown_absent \ + "libvirt read-only socket not found at $sock_ro_path" \ + "it ships with the monolithic libvirtd stack and comes back with it (#3338); if libvirtd is up, the -ro socket should be too" fi if grep -qE '^\s*listen_tcp\s*=\s*1' /etc/libvirt/libvirtd.conf 2>/dev/null; then bad "libvirtd.conf has listen_tcp = 1 -- the TCP socket is enabled" @@ -486,18 +679,41 @@ if command -v aa-status >/dev/null 2>&1; then bad "a libvirt/QEMU AppArmor profile is in complain mode, or aa-status output didn't match expectations -- check manually" fi else - info "could not run aa-status (needs the isolation-audit sudoers grant)" + unmeasured 0 "could not run aa-status (needs the isolation-audit sudoers grant) -- the libvirt/QEMU AppArmor posture is unmeasured, not fine" fi else info "AppArmor not installed on this host" fi # --------------------------------------------------------------------------- +# The verdict, with its category counts (#3312). The point of printing them is +# that a reader must be able to answer "is this host broken, or is this check +# broken?" from the last five lines, without re-deriving it from the body -- +# the failure this file exists to end is a red run whose meaning has to be +# reconstructed by hand. printf '\n' -if [ "$fail" -eq 0 ]; then - printf 'isolation-audit: all checks passed (or skipped as expected)\n' +# `fail` counts measured violations; `unmeas_fatal` counts barriers that could +# not be read at all. Both are faults, and the second is the more dangerous of +# the two -- an unread FORWARD chain tells you nothing about whether the +# default policy is still DROP. +faults=$(( fails + unmeas_fatal )) +printf 'isolation-audit: categories -- %d unmeasured, %d expected-by-declaration, %d triaged gap(s), %d fault(s)\n' \ + "$unmeas" "$expected" "$warns" "$faults" +if [ "$STANDDOWN_ACTIVE" = "1" ]; then + printf 'isolation-audit: a declared stand-down is in force (%s) -- the sandbox objects it covers were not checked for presence, and everything else was\n' \ + "$STANDDOWN_WHY" +fi +if [ "$unmeas" -gt 0 ]; then + # Also non-fatal on its own: the fatal ones are already counted in + # $faults above, so this line is the count a reader needs, not a verdict. + printf 'isolation-audit: %d check(s) could not be measured at all -- every UNMEAS line above is an unanswered question, not a pass\n' "$unmeas" +fi +if [ "$faults" -eq 0 ]; then + printf 'isolation-audit: VERDICT PASS -- every check that could be measured agrees with the invariants (%s)\n' \ + "$([ "$expected" -gt 0 ] && printf '%d object(s) excused by a live declaration' "$expected" || printf 'nothing excused')" else - printf 'isolation-audit: ONE OR MORE CHECKS FAILED -- see FAIL lines above\n' + printf 'isolation-audit: VERDICT FAIL -- %d fault(s) above: each is a measured violation, an exception that no longer applies, or a barrier that could not be read (%d of the latter). None of them is a configuration problem with this script\n' \ + "$faults" "$unmeas_fatal" fi if [ "$warns" -gt 0 ]; then # Deliberately does not affect the exit status: these are triaged gaps with @@ -505,4 +721,4 @@ if [ "$warns" -gt 0 ]; then # they stay visible without turning the job red forever (#2366 review). printf 'isolation-audit: %d triaged gap(s) reported as WARN -- tracked, not failing this run\n' "$warns" fi -exit "$fail" +exit "$([ "$faults" -gt 0 ] && echo 1 || echo 0)" diff --git a/scripts/sandbox-standdown.sh b/scripts/sandbox-standdown.sh new file mode 100755 index 00000000..eff58385 --- /dev/null +++ b/scripts/sandbox-standdown.sh @@ -0,0 +1,210 @@ +#!/usr/bin/env bash +# sandbox-standdown.sh -- declare, inspect or clear a DELIBERATE stand-down of +# the libvirt-backed sandbox isolation stack (#3312). +# +# Why this exists. scripts/isolation-audit.sh asserts that the guarded sandbox +# networks (sandbox, honeypot-sandbox), the honeypot-sandbox-strict nwfilter +# and libvirt's own socket are present. Those assertions were all "FAIL" and +# nothing else, so the audit could not tell two very different hosts apart: +# +# * a broken one -- e.g. the 2026-09-23 reboot, where the per-driver modular +# libvirt units (virtqemud/virtnetworkd/... each Conflicts=libvirtd) won +# the race and left libvirtd.socket dead, taking every sandbox object with +# it (#3338), or a post-rebuild host where install-host.sh was never re-run +# and the network and nwfilter were never restored (#3027). Both are real +# regressions and must stay red. +# * an expected one -- a host whose operator has deliberately stood the +# sandbox stack down to free the RAM and CPU it holds for a heavy +# training or benchmark leg, which is a standing practice (#3135), or any +# other deliberate, temporary choice. +# +# Before this, both hosts read as the same red line, so a reader could not +# triage it, and a diagnostic that cries wolf identically on a healthy and a +# broken host is a diagnostic whose red X nobody acts on -- which is exactly +# what #3312 was filed about. +# +# The declaration is a small dated file, world-readable and owned by root, at +# /etc/apiary/sandbox-standdown. The audit reads it (see the standdown_state() +# function in isolation-audit.sh) and applies three rules, all of which fail +# CLOSED, i.e. towards "still a fault": +# +# 1. A live declaration excuses ONLY the absence of the objects it names +# (the guarded networks, the nwfilter, the libvirt socket). It never +# excuses a element, a host route, a privileged container or +# any other invariant -- a stand-down is not an amnesty. +# 2. A declaration is bounded. `until` must be in the future, and at most +# STANDDOWN_MAX_DAYS ahead, so an exception cannot be declared once and +# then quietly live forever -- the same anti-rot rule the audit already +# applies to its capability lists (see the list-hygiene pass). +# 3. A declaration that is present but NOT active -- expired, malformed, or +# missing a mandatory field -- is itself a FAIL. An expired exception is +# not an exception, and a typo must never be able to neuter the audit. +# +# This script only writes that file. It changes no libvirt, network or +# container state: standing the stack down and bringing it back is the +# operator's own systemctl/virsh work, and the audit's job is to report what +# the host actually looks like, not to fix it. +set -euo pipefail + +STANDDOWN_FILE="${APIARY_STANDDOWN_FILE:-/etc/apiary/sandbox-standdown}" +# Kept in sync with isolation-audit.sh's own copy. Both are overridable by +# env so the audit's test suite can point them at a tmpfile; the constant +# itself is duplicated rather than shared because isolation-audit.sh is +# deployed and run on its own (diagnostics.yml, and the root systemd timer +# its header mentions) and must not depend on a second file being present. +STANDDOWN_MAX_DAYS="${APIARY_STANDDOWN_MAX_DAYS:-14}" + +usage() { + cat >&2 <&2; usage ;; + esac + done + [ -n "$issue" ] || { echo "declare needs --issue '#NNNN'" >&2; usage; } + [ -n "$until" ] || { echo "declare needs --until YYYY-MM-DD" >&2; usage; } + [ -n "$reason" ] || { echo "declare needs --reason 'why'" >&2; usage; } + # An owner issue is mandatory and must look like one: the audit's own + # capability-gap lists work the same way, and an untraceable exception is + # the thing that rots into a permanent excuse. + case "$issue" in + '#'[0-9]*) ;; + *) echo "--issue must be an issue reference like '#1234', got: $issue" >&2; exit 2 ;; + esac + # Validate the date the same way the audit does, before writing anything: + # a declaration that cannot be parsed is one the audit must treat as a + # FAIL, and the operator should find that out here instead. + if ! until_epoch=$(date -d "$until" +%s 2>/dev/null); then + echo "--until is not a date the host can parse: $until" >&2 + exit 2 + fi + now_epoch=$(date +%s) + max_epoch=$(( now_epoch + STANDDOWN_MAX_DAYS * 86400 )) + if [ "$until_epoch" -le "$now_epoch" ]; then + echo "--until is in the past ($until): the audit would treat this as an expired exception and fail" >&2 + exit 2 + fi + if [ "$until_epoch" -gt "$max_epoch" ]; then + echo "--until is more than $STANDDOWN_MAX_DAYS days out ($until)." >&2 + echo "A stand-down this long is a permanent posture change, not a stand-down:" >&2 + echo "stand the stack down, then leave the audit honest, or open an issue for the change." >&2 + exit 2 + fi + [ "$(id -u)" = "0" ] || { + echo "declare must run as root: $STANDDOWN_FILE is root-owned" >&2 + exit 1 + } + # Only ever create the directory. An existing one (including a bind + # mount, or /tmp under a test override) is left exactly as it is: a + # recursive-looking chmod/chown of a parent the operator did not ask us + # to touch is not this script's business. + standdown_dir=$(dirname "$STANDDOWN_FILE") + if [ ! -d "$standdown_dir" ]; then + install -d -m 0755 -o root -g root "$standdown_dir" + fi + # Written to a temp file and moved into place, so a reader (the audit, + # running as the unprivileged runner user) never sees a half-written + # declaration and never has to tolerate one. + tmp=$(mktemp "${STANDDOWN_FILE}.XXXXXX") + trap 'rm -f "$tmp"' EXIT + cat > "$tmp" </dev/null); then + echo "status: INACTIVE (until is not a date this host can parse: $until)" + exit 1 + fi + now_epoch=$(date +%s) + if [ "$until_epoch" -le "$now_epoch" ]; then + echo "status: INACTIVE (expired $until -- the audit is failing the sandbox checks again)" + exit 1 + fi + if [ "$until_epoch" -gt $(( now_epoch + STANDDOWN_MAX_DAYS * 86400 )) ]; then + echo "status: INACTIVE (until $until is more than $STANDDOWN_MAX_DAYS days out)" + exit 1 + fi + echo "status: ACTIVE until $until ($issue): $reason" + echo "The audit will report the sandbox stack as EXPECTED, not FAIL, until then." + ;; + clear) + if [ ! -f "$STANDDOWN_FILE" ]; then + echo "no declaration at $STANDDOWN_FILE -- nothing to clear" + exit 0 + fi + [ "$(id -u)" = "0" ] || { + echo "clear must run as root: $STANDDOWN_FILE is root-owned" >&2 + exit 1 + } + rm -f "$STANDDOWN_FILE" + echo "removed $STANDDOWN_FILE -- the audit expects the full sandbox stack again" + ;; + *) + usage + ;; +esac diff --git a/tests/docs/test_3312_signal_categories.py b/tests/docs/test_3312_signal_categories.py new file mode 100644 index 00000000..59fdc800 --- /dev/null +++ b/tests/docs/test_3312_signal_categories.py @@ -0,0 +1,913 @@ +#!/usr/bin/env python3 +"""#3312: a diagnostic that cries wolf on a healthy host is worse than no +diagnostic -- and this is the machinery that makes the three outcomes +distinguishable. + +`diagnostics.yml` ended 160 consecutive scheduled runs in failure (2026-08-12 +through 2026-09-27) and of the four things it named every one was either a +check that could not see the object it was looking at or a real host fault +with no category attached. The red X was the same red X for all of them, so +nobody triaged it and #3312 is the result. These tests pin the three +outcomes separately, because that separation is the whole fix: + + measured fault a real regression. Exit 1. + expected state an object deliberately absent, with a dated, issue- + referencing declaration on the host. Exit 0, labelled + EXPECT, naming the owner issue and the expiry. + unmeasurable the check could not run. Counted and named, never folded + into a pass; fatal for the isolation barriers, whose + silence would otherwise read as safety. + +Each test drives the real scripts/isolation-audit.sh against a synthesised +host (every external command is a PATH stub), so what is asserted here is the +behaviour on a host, not the presence of a string in the source. +""" +from __future__ import annotations + +import os +import pathlib +import shutil +import stat +import subprocess +import sys + +import pytest + +REPO_ROOT = pathlib.Path(__file__).resolve().parents[2] +AUDIT = REPO_ROOT / "scripts" / "isolation-audit.sh" +STANDDOWN = REPO_ROOT / "scripts" / "sandbox-standdown.sh" +LIB = REPO_ROOT / "scripts" / "diagnostics-lib.sh" +WORKFLOW = REPO_ROOT / ".github" / "workflows" / "diagnostics.yml" + +sys.path.insert(0, str(REPO_ROOT / "tests" / "docs")) +from test_2295_isolation_audit_full_coverage import ( # noqa: E402 + CLEAN_RULES, + FAKE_IP, + FAKE_IPTABLES, + FAKE_SUDO, + FAKE_SYSTEMCTL, +) + +# --------------------------------------------------------------------------- +# stubs +# --------------------------------------------------------------------------- + +# STATE is baked in per-invocation, not exported, so each test gets a +# self-contained host. +FAKE_VIRSH = """#!/usr/bin/env bash +case "$1 $2" in + "net-info ghosts") exit 0 ;; + "net-info "*) + [ "$STATE" = "healthy" ] && exit 0 || exit 1 ;; + "net-dumpxml "*) + [ "$STATE" = "healthy" ] || exit 1 + if [ "$FAKE_FORWARD" = "1" ]; then + echo "" + else + echo "sandbox" + fi + exit 0 ;; + "nwfilter-dumpxml "*) + [ "$STATE" = "healthy" ] && { echo ""; exit 0; } || exit 1 ;; +esac +exit 1 +""" + +# #3312: a real `ss` that answers, so the sshd check has something to judge. +# FAKE_SS_22 is the port-22 line; unset means nothing is listening on :22. +FAKE_SS = """#!/usr/bin/env bash +[ -n "${FAKE_SS_BROKEN:-}" ] && exit 1 +[ -n "${FAKE_SS_22:-}" ] && echo "LISTEN 0 128 $FAKE_SS_22 0.0.0.0:*" +exit 0 +""" + +FAKE_DOCKER = """#!/usr/bin/env bash +if [ "$1" = "ps" ]; then + [ -n "${FAKE_DOCKER_PS_FAIL:-}" ] && { echo "Cannot connect to the Docker daemon" >&2; exit 1; } + [ -n "${FAKE_NO_STACK_CONTAINERS:-}" ] && exit 0 + printf 'hp-cowrie\\thoneypot/cowrie\\nhp-arcane\\thoneypot/arcane\\n' + exit 0 +fi +if [ "$1" = "inspect" ]; then + case "$2" in + hp-cowrie) echo '[ALL]'; exit 0 ;; # hardened + hp-arcane) echo ''; exit 0 ;; # the documented CAP_EXCEPTIONS entry + esac + exit 1 +fi +if [ "$1" = "network" ]; then exit 1; fi # sandbox compose not up +exit 0 +""" + +# The libvirt sockets are `[ -S ]`-tested at hardcoded absolute paths, so a +# faithful stand-in has to create real sockets. stat is stubbed to report the +# RHEL/Debian posture a healthy sandbox host actually has (root:libvirt, no +# world access), because a test namespace cannot create a real `libvirt` group +# without writing /etc/group on the host. +FAKE_STAT = """#!/usr/bin/env bash +fmt=""; target="" +while [ $# -gt 0 ]; do + case "$1" in + -c) fmt=$2; shift 2 ;; + *) target=$1; shift ;; + esac +done +case "$target" in + /var/run/libvirt/*) + case "$fmt" in + '%a') echo 750 ;; + '%U:%G') echo root:libvirt ;; + *) echo 0 ;; + esac ;; + *) exec /usr/bin/stat "$fmt" "$target" ;; +esac +""" + + +def _stub(path: pathlib.Path, body: str) -> None: + path.write_text(body, encoding="utf-8") + path.chmod(path.stat().st_mode | stat.S_IEXEC | stat.S_IXGRP | stat.S_IXOTH) + + +@pytest.fixture() +def fake_bin(tmp_path): + bindir = tmp_path / "bin" + bindir.mkdir() + for name, body in ( + ("sudo", FAKE_SUDO), + ("iptables", FAKE_IPTABLES), + ("ip", FAKE_IP), + ("systemctl", FAKE_SYSTEMCTL), + ("virsh", FAKE_VIRSH), + ("ss", FAKE_SS), + ("docker", FAKE_DOCKER), + ("stat", FAKE_STAT), + ): + _stub(bindir / name, body) + rules = tmp_path / "forward-rules.txt" + rules.write_text(CLEAN_RULES, encoding="utf-8") + return bindir, rules + + +# --------------------------------------------------------------------------- +# running the audit against a synthesised host +# --------------------------------------------------------------------------- + + +def _can_sandbox() -> bool: + """True when this host can give the audit real /var/run/libvirt sockets.""" + if shutil.which("unshare") is None: + return False + if not pathlib.Path("/var/run/libvirt").is_dir(): + return False + probe = subprocess.run( + ["unshare", "-rm", "bash", "-c", "mount --bind /var/run/libvirt /var/run/libvirt"], + capture_output=True, + ) + return probe.returncode == 0 + + +SANDBOX_OK = _can_sandbox() + +# The stand-down declaration is written as root by scripts/sandbox-standdown.sh +# and read as the unprivileged runner user, so 0644 root:root is a contract the +# test asserts rather than an accident. +def _standdown_body(issue: str, until: str, reason: str) -> str: + return f"# written by the test\nissue: {issue}\nuntil: {until}\nreason: {reason}\n" + + +def _run(fake_bin, tmp_path, state="healthy", standdown=None, audit=None, extra_env=None): + bindir, rules_file = fake_bin + socket_dir = tmp_path / "libvirt-run" + socket_dir.mkdir(exist_ok=True) + declaration = tmp_path / "sandbox-standdown" + if standdown is not None: + declaration.write_text(standdown, encoding="utf-8") + audit = audit or AUDIT + inner = f""" +set -e +mount --bind '{socket_dir}' /var/run/libvirt +chown 0:0 /var/run/libvirt/* 2>/dev/null || true +chmod 0750 /var/run/libvirt/* 2>/dev/null || true +STATE='{state}' +export STATE +if [ '{state}' = healthy ]; then + python3 -c "import socket,sys; s=socket.socket(socket.AF_UNIX); s.bind('/var/run/libvirt/libvirt-sock'); s.close()" + python3 -c "import socket,sys; s=socket.socket(socket.AF_UNIX); s.bind('/var/run/libvirt/libvirt-sock-ro'); s.close()" + chown 0:0 /var/run/libvirt/* 2>/dev/null || true +fi +exec bash '{audit}' +""" + env = dict(os.environ) + env["PATH"] = f"{bindir}:{env['PATH']}" + # Clear any FAKE_* inherited from the ambient environment *before* setting + # this run's own values, so a developer's shell cannot decide a verdict. + for key in list(env): + if key.startswith("FAKE_"): + del env[key] + env["FAKE_RULES_FILE"] = str(rules_file) + env["APIARY_STANDDOWN_FILE"] = str(declaration) + if extra_env: + env.update(extra_env) + return subprocess.run( + ["unshare", "-rm", "bash", "-c", inner], + env=env, + capture_output=True, + text=True, + cwd=REPO_ROOT, + timeout=120, + ) + + +def _future(days: int = 7) -> str: + import datetime + + return (datetime.date.today() + datetime.timedelta(days=days)).isoformat() + + +def _fault_lines(stdout: str) -> list: + """The findings the audit actually categorised as faults. + + Matched on the category column, not as a substring: the audit's own prose + says the word ('anything else that breaks below is still a FAIL') when a + stand-down is in force, and a substring test would then be asserting on its + explanation rather than on its findings. + """ + import re + + return re.findall(r"^\s*FAIL\s+\S.*$", stdout, re.MULTILINE) + + +def _expect_lines(stdout: str) -> list: + """The findings the audit actually categorised as expected-by-declaration.""" + import re + + return re.findall(r"^\s*EXPECT\s+\S.*$", stdout, re.MULTILINE) + + +# --------------------------------------------------------------------------- +# the three outcomes, distinguished +# --------------------------------------------------------------------------- + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_healthy_host_exits_zero(fake_bin, tmp_path): + """The baseline every other test is measured against: nothing wrong, so + nothing red. If this ever fails, the audit is crying wolf again.""" + result = _run(fake_bin, tmp_path, state="healthy") + assert "VERDICT PASS" in result.stdout, result.stdout + assert result.returncode == 0, result.stdout + assert not _fault_lines(result.stdout), result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_broken_host_exits_one_and_names_the_fault(fake_bin, tmp_path): + """The real-regression case: libvirt and its sandbox objects are gone with + no declaration behind them (the 2026-09-23 state, #3338). Must stay red, + and must name the cause rather than the symptom.""" + result = _run(fake_bin, tmp_path, state="broken") + assert result.returncode == 1, result.stdout + assert "libvirt network 'sandbox' does not exist" in result.stdout + assert "'honeypot-sandbox-strict' nwfilter is missing" in result.stdout + assert "libvirt socket not found" in result.stdout + assert "VERDICT FAIL" in result.stdout + assert not _expect_lines(result.stdout), result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_declared_standdown_exits_zero_where_broken_exits_one(fake_bin, tmp_path): + """The expected-state case, and the reason this file exists. Same host as + the test above -- libvirtd down, both networks gone, nwfilter gone -- with + a live dated declaration standing behind it. + + Before #3312 this state was indistinguishable from the one above: both + printed the same FAIL lines and both exited 1, which is why a legitimate + stand-down and a real regression looked the same to whoever read the run. + """ + result = _run( + fake_bin, + tmp_path, + state="broken", + standdown=_standdown_body("#3135", _future(7), "training leg holds the sandbox RAM/CPU"), + ) + assert "EXPECT" in result.stdout, result.stdout + assert "declared stand-down, #3135, until" in result.stdout + assert "training leg holds the sandbox RAM/CPU" in result.stdout + assert result.returncode == 0, ( + "a live, unexpired declaration must make the absence EXPECTED and the run " + f"green; got exit {result.returncode}:\n{result.stdout}" + ) + assert "VERDICT PASS" in result.stdout + assert "excused by a live declaration" in result.stdout + # The identical host, same everything but the declaration, is still red. + (tmp_path / "sandbox-standdown").unlink() + without = _run(fake_bin, tmp_path, state="broken") + assert without.returncode == 1, without.stdout + # ...and the specific lines that were red are now EXPECT, not deleted. + assert "libvirt network 'sandbox' does not exist" in result.stdout + assert not _fault_lines(result.stdout), result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_expired_declaration_stops_excusing_anything(fake_bin, tmp_path): + """Anti-rot. An exception that outlived its window is not an exception: it + is a stale file pretending to be an expectation, and it must be named and + fatal, not quietly honoured.""" + result = _run( + fake_bin, + tmp_path, + state="broken", + standdown=_standdown_body("#3135", "2020-01-01", "a leg that ended in january"), + ) + assert result.returncode == 1, result.stdout + assert "does not count" in result.stdout + assert "expired 2020-01-01" in result.stdout + assert "clear it" in result.stdout + assert "libvirt network 'sandbox' does not exist" in result.stdout + assert not _expect_lines(result.stdout), result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +@pytest.mark.parametrize( + "body,why", + [ + ("until: %s\nreason: r\n" % _future(3), "no issue"), + ("issue: #1\nreason: r\n", "no until"), + ("issue: #1\nuntil: %s\n" % _future(3), "no reason"), + ("issue: 1\nuntil: %s\nreason: r\n" % _future(3), "issue is not a reference"), + ("issue: #1\nuntil: not-a-date\nreason: r\n", "unparseable date"), + ("issue: #1\nuntil: %s\nreason: r\n" % _future(400), "further out than the max window"), + ], +) +def test_malformed_declaration_fails_closed(fake_bin, tmp_path, body, why): + """A typo must never be able to neuter the audit. Every shape of a bad + declaration -- missing field, wrong issue syntax, unparseable date, or a + window so long it is really a permanent posture change -- has to land on + the fault side of the line.""" + result = _run(fake_bin, tmp_path, state="broken", standdown=body) + assert "does not count" in result.stdout, f"{why}:\n{result.stdout}" + assert result.returncode == 1, f"{why}:\n{result.stdout}" + assert not _expect_lines(result.stdout), f"{why}:\n{result.stdout}" + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_declaration_never_excuses_a_forwarding_network(fake_bin, tmp_path): + """A stand-down is an excuse for absence, not an amnesty. A network that IS + defined and DOES forward is the exact thing the isolation audit exists to + catch, and no declaration may cover it.""" + result = _run( + fake_bin, + tmp_path, + state="healthy", + standdown=_standdown_body("#3135", _future(7), "standing down anyway"), + extra_env={"FAKE_FORWARD": "1"}, + ) + assert "can route to the internet" in result.stdout, result.stdout + assert result.returncode == 1, result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_declaration_never_excuses_a_planted_forward_accept(fake_bin, tmp_path): + """Same rule on the iptables side, which is the barrier that actually + keeps a NAT-mode sandbox off the internet.""" + bindir, rules_file = fake_bin + rules_file.write_text(CLEAN_RULES + "-A FORWARD -i virbr-sandbox -o eth0 -j ACCEPT\n", encoding="utf-8") + result = _run( + fake_bin, + tmp_path, + state="healthy", + standdown=_standdown_body("#3135", _future(7), "standing down anyway"), + ) + assert "an explicit ACCEPT rule references virbr-sandbox" in result.stdout + assert result.returncode == 1, result.stdout + + +# --------------------------------------------------------------------------- +# unmeasurable is its own category, and is not a pass +# --------------------------------------------------------------------------- + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_sshd_check_answers_when_sshd_is_simply_not_on_22(fake_bin, tmp_path): + """The check used to be `ss -tlnp | grep ':22 '`, which exits 1 both when + nothing listens on :22 and when ss could not run at all -- so on the live + homeserver it had never once answered, and it reported 'could not confirm' + about a host where the answer was simply 'nothing is listening'. That is a + check that cannot tell a healthy host from a blind one, which is the same + defect as no check at all.""" + result = _run(fake_bin, tmp_path, state="healthy") + assert "nothing is listening on :22" in result.stdout, result.stdout + assert "could not confirm" not in result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_sshd_check_says_so_when_ss_cannot_run(fake_bin, tmp_path): + """...and the other half: when ss really is unreadable the line has to say + the check did not run, in its own category, counted in the footer.""" + result = _run(fake_bin, tmp_path, state="healthy", extra_env={"FAKE_SS_BROKEN": "1"}) + assert "UNMEAS" in result.stdout, result.stdout + assert "could not read the sshd listen address" in result.stdout + assert "not a pass" in result.stdout + assert "0 unmeasured" not in result.stdout + assert "check(s) could not be measured at all" in result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_sshd_on_a_honeypot_facing_address_is_still_a_fault(fake_bin, tmp_path): + """Splitting the stages must not weaken the finding that matters.""" + result = _run( + fake_bin, tmp_path, state="healthy", extra_env={"FAKE_SS_22": "10.8.0.2:22"} + ) + assert "sshd appears to be listening on a honeypot-facing address" in result.stdout + assert result.returncode == 1, result.stdout + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_no_stack_containers_is_not_reported_as_a_docker_failure(fake_bin, tmp_path): + """`docker ps | grep -E '^(hp-|sbx-)'` exits 1 when the grep finds nothing, + so a host with no stack containers at all was reported as 'could not + enumerate containers (docker ps failed)' -- a claim about the tool printed + when the truth is a claim about the deployment. Both are faults; they are + not the same fault, and only one of them is fixable by reinstalling + docker.""" + result = _run(fake_bin, tmp_path, state="healthy", extra_env={"FAKE_NO_STACK_CONTAINERS": "1"}) + assert "no hp-* or sbx-* container exists on this host at all" in result.stdout, result.stdout + assert "This is not a docker failure" in result.stdout + assert "could not enumerate containers (docker ps failed" not in result.stdout + # ...and it is reported once, not doubled up by the capability section. + assert result.stdout.count("no hp-* or sbx-* container exists") == 1 + + +@pytest.mark.skipif(not SANDBOX_OK, reason="needs a user+mount namespace to stage /var/run/libvirt") +def test_docker_itself_failing_is_still_named_as_such(fake_bin, tmp_path): + result = _run(fake_bin, tmp_path, state="healthy", extra_env={"FAKE_DOCKER_PS_FAIL": "1"}) + assert "could not enumerate containers (docker ps failed" in result.stdout, result.stdout + assert result.returncode == 1, result.stdout + + +# --------------------------------------------------------------------------- +# the verdict itself is legible +# --------------------------------------------------------------------------- + + +def test_every_line_carries_a_category_prefix(): + """A reader has to be able to tell what kind of line they are looking at + without counting columns, and the four labels have to stay stable, because + the workflow's summary quotes them.""" + audit = AUDIT.read_text(encoding="utf-8") + for label in ("' OK %s\\n'", "' FAIL %s\\n'", "' EXPECT %s\\n'", "' UNMEAS %s\\n'"): + assert label in audit, f"the {label} label is gone -- the category vocabulary moved" + + +def test_verdict_prints_category_counts(): + audit = AUDIT.read_text(encoding="utf-8") + assert "isolation-audit: categories --" in audit + assert "VERDICT PASS" in audit and "VERDICT FAIL" in audit + assert "could not be measured at all" in audit + # The exit code has to follow the categories, including the fatal + # unmeasurable ones: an unread barrier must not exit 0. + assert 'unmeasured 1 "could not read the iptables FORWARD chain' in audit + assert 'unmeasured 1 "iptables is not installed' in audit + + +def test_standdown_helper_and_audit_agree_on_the_path_and_the_window(): + audit = AUDIT.read_text(encoding="utf-8") + helper = STANDDOWN.read_text(encoding="utf-8") + for text in (audit, helper): + assert "APIARY_STANDDOWN_FILE:-/etc/apiary/sandbox-standdown" in text + assert "APIARY_STANDDOWN_MAX_DAYS:-14" in text + + +# --------------------------------------------------------------------------- +# the declaration tool: round trip, and the refusals +# --------------------------------------------------------------------------- + + +@pytest.mark.skipif( + shutil.which("unshare") is None, reason="needs a user+mount namespace to act as root" +) +def test_declare_show_clear_round_trip(tmp_path): + """Runs the real tool as (namespaced) root, so the file it writes is a real + root-owned 0644 file -- the exact thing the unprivileged audit has to be + able to read.""" + declaration = tmp_path / "sandbox-standdown" + env = dict(os.environ, APIARY_STANDDOWN_FILE=str(declaration)) + inner = f""" +set -e +'{STANDDOWN}' declare --issue '#3135' --until '{_future(7)}' --reason 'a training leg' +stat -c '%a %U' '{declaration}' +'{STANDDOWN}' show +'{STANDDOWN}' clear +'{STANDDOWN}' show +""" + proc = subprocess.run( + ["unshare", "-rm", "bash", "-c", inner], env=env, capture_output=True, text=True + ) + assert proc.returncode == 0, proc.stdout + proc.stderr + # World-readable: the audit runs as github-deploy-runner, not root. + assert "644 root" in proc.stdout, proc.stdout + assert "status: ACTIVE" in proc.stdout + assert "no declaration at" in proc.stdout + + +def test_declare_refuses_what_the_audit_would_refuse_too_honour(tmp_path): + """The tool and the audit must not disagree about validity -- a window the + writer accepts but the reader ignores would be an exception that exists on + disk and nowhere else.""" + declaration = tmp_path / "sandbox-standdown" + env = dict(os.environ, APIARY_STANDDOWN_FILE=str(declaration)) + for args, why in ( + ("declare --until 2099-01-01 --reason r", "no issue"), + ("declare --issue '#1' --reason r", "no until"), + ("declare --issue '#1' --until 2099-01-01", "no reason"), + ("declare --issue 'one' --until 2099-01-01 --reason r", "issue is not a reference"), + ("declare --issue '#1' --until 2020-01-01 --reason r", "until is in the past"), + ("declare --issue '#1' --until 2099-01-01 --reason r", "too far out"), + ): + proc = subprocess.run( + ["unshare", "-rm", str(STANDDOWN)] + args.split(), + env=env, + capture_output=True, + text=True, + ) + assert proc.returncode == 2, f"{why}: expected a refusal, got {proc.returncode}\n{proc.stderr}" + assert not declaration.exists(), f"{why}: a refused declaration wrote a file anyway" + + +# --------------------------------------------------------------------------- +# the workflow: same vocabulary, and one ledger +# --------------------------------------------------------------------------- + + +def test_workflow_alerts_are_categorised(): + """Every fatal condition in the workflow has to say which of the four + things it is. #3312's whole content is that 'a red run' and 'a red run + because the runner cannot read a root-only token' are different events, + and only one of them is a fault in the pipeline.""" + workflow = WORKFLOW.read_text(encoding="utf-8") + lib = LIB.read_text(encoding="utf-8") + # No bare alert() left: every call site names a category. + import re + + calls = re.findall(r"^\s*alert [^\n]*$", workflow, re.MULTILINE) + assert calls, "no alert call sites found -- the workflow's shape changed" + for call in calls: + assert re.match(r"^\s*alert (fault|runner-config|unmeasured) ", call), ( + f"alert call site has no fatal category: {call.strip()!r}" + ) + # ...and only categories the library defines, so a finding cannot be filed + # under a label nothing knows how to treat. + declared = re.search(r'DIAG_CATEGORIES="([^"]+)"', lib) + assert declared, "scripts/diagnostics-lib.sh no longer declares DIAG_CATEGORIES" + categories = set(declared.group(1).split()) + assert {c.split()[1] for c in calls} <= categories, lib + # Same for note(), which carries the non-fatal ones. + notes = re.findall(r"^\s*note [^\n]*$", workflow, re.MULTILINE) + for call in notes: + assert re.match(r"^\s*note (fault|runner-config|unmeasured|expected) ", call), ( + f"note call site has no category: {call.strip()!r}" + ) + assert {c.split()[1] for c in notes} <= categories, lib + # The categories are documented where they are defined, not only used. + for category in ("fault", "runner-config", "expected", "unmeasured"): + assert f"# {category} " in lib, ( + f"{category} is defined without a line saying what it means" + ) + + +def test_workflow_prints_a_category_ledger(): + """One table, at the end, listing what was measured and what each finding + was. This is what makes a red run triageable in seconds rather than by + re-reading the body -- and it is the artefact that says 'nothing was + unmeasured' just as loudly as it says 'this is a fault'.""" + import re + + workflow = WORKFLOW.read_text(encoding="utf-8") + lib = LIB.read_text(encoding="utf-8") + assert "| category | check | finding |" in lib + assert "diag_ledger" in lib + # The verdict line has to name the counts, not just the exit code. + assert "unmeasured" in lib.split("diag_ledger_report() {")[1][:2000] + # Every job that can produce a finding ends with the ledger: steps are + # separate shells, so one job's findings cannot reach the other's table. + assert workflow.count("- name: Category ledger (#3312)") == 2 + assert workflow.count("diag_ledger_report") == 2 + for step in re.findall( + r"- name: Category ledger \(#3312\).*?(?=\n - name:|\n \w+:)", workflow, re.S + ): + assert "if: always()" in step, ( + "a ledger step gated on success would print nothing on exactly the " + "runs that need it" + ) + assert "exit 1" not in step, ( + "the ledger step is the triage view, not the verdict -- the fatal " + "signal is each step's own exit status" + ) + + +def test_workflow_still_fails_on_a_real_fault_and_still_fails_when_unmeasured(): + """The two fatal categories stay fatal. This is the anti-wolf guard: the + temptation with a job that has been red for four days is to make the thing + that cannot be measured stop mattering, and that would leave the sensor -> + Elasticsearch -> dashboard pipeline unmonitored while the run went green.""" + workflow = WORKFLOW.read_text(encoding="utf-8") + for job in ("home", "vps"): + section = workflow.split(f" {job}:", 1)[1].split("\n # ---", 1)[0] + assert "schedule_failed=0" in section + assert 'if [ "$schedule_failed" -eq 1 ]; then\n exit 1' in section, ( + f"the {job} job no longer turns a categorised finding into a red run (#3312)" + ) + + +def test_workflow_runs_the_audit_from_the_triggering_ref_not_the_deployed_copy(): + """#2908: /opt/stacks/apiary is refreshed only by deploy.yml, which is + workflow_dispatch-only, so a fix merged to isolation-audit.sh sat inert + until a human remembered to deploy -- and for the five weeks of #3312 that + is exactly why the red X kept naming things #3338 had already fixed. A + check auditing the host with a stale copy of its own question cannot report + a fix as fixed.""" + import re + + workflow = WORKFLOW.read_text(encoding="utf-8") + for job in ("home", "vps"): + section = workflow.split(f" {job}:", 1)[1].split("\n vps:", 1)[0] + assert "- uses: actions/checkout@" in section, ( + f"the {job} job has no checkout, so the scripts it runs are whatever " + "the last deploy left behind (#2908)" + ) + isolation_step = workflow.split("- name: Isolation invariants", 1)[1] + assert "$GITHUB_WORKSPACE/scripts/isolation-audit.sh" in isolation_step + assert "/opt/stacks/apiary/scripts/isolation-audit.sh" not in isolation_step + # ...and the deployed copy is still measured, because drift is a real + # finding (#2908) even when the audit no longer depends on it. + assert "diff --quiet origin/main -- scripts/" in workflow + # Pinning is the repo's rule for every checkout (quality.yml's zizmor gate + # blocks an unpinned one), and a workflow that reads the service token + # through a root-owned helper must not keep a credential in the tree it + # checked out. + for line in re.findall(r"^\s*- uses: actions/checkout@.*$", workflow, re.MULTILINE): + assert re.search(r"@[0-9a-f]{40} # v", line), ( + f"unpinned checkout, or a pin with no version beside it to bump: {line.strip()}" + ) + assert workflow.count("persist-credentials: false") >= 2 + + +def _lib_run(body: str, is_schedule: str = "true"): + """Sources the real library in a real bash, the way a step does, and + returns the log, the annotations and the step's exit status.""" + import re + import tempfile + + with tempfile.TemporaryDirectory() as tmp: + env = dict( + os.environ, + RUNNER_TEMP=tmp, + GITHUB_STEP_SUMMARY=f"{tmp}/summary.md", + ) + env.pop("GITHUB_EVENT_NAME", None) + script = f""" +set -uo pipefail +is_schedule={is_schedule} +schedule_failed=0 +. '{LIB}' +{body} +diag_ledger_report +exit "$schedule_failed" +""" + proc = subprocess.run( + ["bash", "-c", script], env=env, capture_output=True, text=True, timeout=60 + ) + summary = pathlib.Path(f"{tmp}/summary.md").read_text(encoding="utf-8") + # A title ends at the first "::", not at the first ":" -- the titles here +# are ": ". + annotations = re.findall(r"^::(error|warning) title=(.*?)::", proc.stdout, re.MULTILINE) + return proc, summary, annotations + + +def test_the_library_keeps_a_runner_config_gap_distinct_and_still_fatal(): + """The load-bearing property, and the one most likely to be broken by a + well-meaning edit: a lane that has never been able to measure anything must + (a) not be reported as a fault in the pipeline, and (b) still fail the + scheduled run. Dropping (b) is how a green run comes to mean "we stopped + looking"; #3283 is what that costs.""" + proc, summary, annotations = _lib_run( + 'alert runner-config "sensor -> ES -> dashboard" "metrics unavailable: helper not installed"' + ) + assert proc.returncode == 1, "a runner-config gap stopped being fatal" + assert ("error", "runner-config: sensor -> ES -> dashboard") in annotations, annotations + assert not any(title.startswith("fault:") for _, title in annotations), ( + f"the gap was filed as a pipeline fault: {annotations}" + ) + assert "| runner-config | sensor -> ES -> dashboard | metrics unavailable: helper not installed |" in summary + assert "1 runner-config gap(s)" in summary + + +def test_a_declared_absence_is_reported_and_never_fatal(): + """The case #3312 was filed about: something deliberately down must be + legible in the summary without reddening a run that is doing its job.""" + proc, summary, annotations = _lib_run( + 'note expected "sandbox stand-down" "the sandbox stack is deliberately down until 2026-10-04"' + ) + assert proc.returncode == 0, "a declared, expected absence reddened the run" + assert not annotations, f"an expected absence raised an annotation: {annotations}" + assert "| expected | sandbox stand-down |" in summary + assert "1 expected absence(s)" in summary + assert "Verdict: every check that could run agreed" in summary + + +def test_unmeasurable_is_its_own_category_and_never_a_pass(): + proc, summary, annotations = _lib_run( + 'alert unmeasured "isolation audit" "FORWARD chain could not be read at all"' + ) + assert proc.returncode == 1, "an unread barrier stopped being fatal" + assert ("error", "unmeasured: isolation audit") in annotations, annotations + assert "1 unmeasured" in summary + assert "Verdict: see the rows above" in summary + + +def test_a_manual_run_reports_everything_and_fails_for_nothing(): + """#2222: a workflow_dispatch run is a human reading a report. Every + finding is still labelled and still filed, and nothing is fatal -- the two + behaviours have to be independent, because a category that only works on + one trigger is a category nobody can trust.""" + proc, summary, annotations = _lib_run( + 'alert fault "dashboard healthz" "unreachable"\nalert runner-config "metrics lane" "no grant"', + is_schedule="false", + ) + assert proc.returncode == 0, "a manual run went red on a reported finding" + assert not annotations, f"a manual run raised annotations: {annotations}" + assert "| fault | dashboard healthz | unreachable |" in summary + assert "| runner-config | metrics lane | no grant |" in summary + assert "1 fault(s), 1 runner-config gap(s)" in summary + + +def test_a_miscategorised_finding_cannot_turn_a_run_green(): + """`alert expected` is a call-site bug. The honest response is to say so + loudly -- and to stay fatal anyway, because the one thing a category + system must never do is let a mistake in it become a quiet pass.""" + proc, _summary, annotations = _lib_run( + 'alert expected "sandbox stand-down" "deliberately down"' + ) + assert proc.returncode == 1, "a fatal call site with a non-fatal category went green" + assert any("Wrong category" in title for _, title in annotations), annotations + + +def test_the_library_refuses_to_be_run(): + """Running it would exit 0 having measured nothing, which is the exact + shape of the problem this issue is about.""" + proc = subprocess.run(["bash", str(LIB)], capture_output=True, text=True, timeout=60) + assert proc.returncode != 0, "sourcing the library also runs it when executed" + assert "source it, do not run it" in proc.stderr + + +# --------------------------------------------------------------------------- +# the isolation step, run as the step itself runs it +# --------------------------------------------------------------------------- + + +def _step_run_body(step_name_prefix: str, is_schedule: str = "true") -> str: + """The literal `run:` body of a workflow step, de-indented, with the one + ${{ }} expression these steps use already substituted. + + Text slicing rather than a YAML parse: PyYAML is not a declared dependency + of this suite, and a test that needs one the docs job does not install is a + test that fails in CI and passes locally. + """ + workflow = WORKFLOW.read_text(encoding="utf-8") + start = workflow.index(f"- name: {step_name_prefix}") + run_at = workflow.index(" run: |\n", start) + body = [] + for line in workflow[run_at + len(" run: |\n") :].splitlines(): + if line.strip() and not line.startswith(" " * 10): + break + body.append(line[10:] if line.startswith(" " * 10) else line) + return "\n".join(body).replace( + "${{ github.event_name == 'schedule' && 'true' || 'false' }}", is_schedule + ) + + +@pytest.mark.parametrize( + "unmeasured,audit_exit,category", + [ + (0, 1, "fault"), + (2, 1, "unmeasured"), + ], +) +def test_the_isolation_step_files_the_audit_verdict_under_its_own_category( + fake_bin, tmp_path, unmeasured, audit_exit, category +): + """The step's own half of the job, executed as the runner executes it. + + It has to do three things at once, and it is easy to break any of them + silently: run the audit from the checkout rather than the deployed copy, + take the audit's own counts rather than re-deriving them (one source of + truth, so the two cannot disagree about how many faults there were), and + file the result under a category that says whether the run failed on a + measured violation or on a barrier nobody could read. + """ + bindir, _rules_file = fake_bin + # The step's #2908 freshness block shells out to git against the deployed + # tree. A stub that fails is the same shape as an offline runner, and it + # keeps this test from doing a real network fetch on a machine that has + # /opt/stacks/apiary (the homeserver CI runner does). + _stub(bindir / "git", "#!/usr/bin/env bash\nexit 1\n") + + workspace = tmp_path / "workspace" + (workspace / "scripts").mkdir(parents=True) + shutil.copy(LIB, workspace / "scripts" / "diagnostics-lib.sh") + audit = workspace / "scripts" / "isolation-audit.sh" + audit.write_text( + "#!/usr/bin/env bash\n" + f"echo 'isolation-audit: categories -- {unmeasured} unmeasured," + f" 0 expected-by-declaration, 0 triaged gap(s), {audit_exit} fault(s)'\n" + f"exit {audit_exit}\n", + encoding="utf-8", + ) + audit.chmod(0o755) + + run_tmp = tmp_path / "runtmp" + run_tmp.mkdir() + script = tmp_path / "step.sh" + script.write_text(_step_run_body("Isolation invariants"), encoding="utf-8") + env = dict(os.environ) + env.update( + { + "GITHUB_WORKSPACE": str(workspace), + "RUNNER_TEMP": str(run_tmp), + "GITHUB_STEP_SUMMARY": str(run_tmp / "summary.md"), + "PATH": f"{bindir}:{env['PATH']}", + } + ) + proc = subprocess.run( + ["bash", str(script)], env=env, capture_output=True, text=True, cwd=REPO_ROOT, timeout=120 + ) + assert proc.returncode == audit_exit, ( + f"the step exited {proc.returncode}, the audit exited {audit_exit}\n" + f"{proc.stdout}\n{proc.stderr}" + ) + assert f"::error title={category}: isolation audit" in proc.stdout, proc.stdout + # ...and the row reaches the reader. Steps are separate shells, so the row + # is a file until the ledger step renders it -- running only the first + # step and looking for the table would pass while the ledger was broken. + ledger = tmp_path / "ledger.sh" + ledger.write_text(_step_run_body("Category ledger"), encoding="utf-8") + subprocess.run( + ["bash", str(ledger)], env=env, capture_output=True, text=True, cwd=REPO_ROOT, timeout=120 + ) + summary = (run_tmp / "summary.md").read_text(encoding="utf-8") + assert "| category | check | finding |" in summary, summary + assert f"| {category} | isolation audit (scripts/isolation-audit.sh) |" in summary, summary + # The audit's own line is carried rather than reworded here: the counts in + # the ledger are the ones the audit printed, so the two cannot disagree. + assert f"{unmeasured} unmeasured" in summary, summary + # ...and the deployed copy is still measured for drift, whatever its state. + assert "could not reach origin to compare" in summary, summary + + +def test_a_clean_isolation_run_is_filed_as_measured_and_in_agreement(fake_bin, tmp_path): + """The other end of the same step: a run with nothing wrong still says + what it measured. Without this row the ledger is silent on a green run, + and silence is the one reading a reader cannot tell apart from a lane + that never ran.""" + bindir, _rules_file = fake_bin + _stub(bindir / "git", "#!/usr/bin/env bash\nexit 1\n") + workspace = tmp_path / "workspace" + (workspace / "scripts").mkdir(parents=True) + shutil.copy(LIB, workspace / "scripts" / "diagnostics-lib.sh") + audit = workspace / "scripts" / "isolation-audit.sh" + audit.write_text( + "#!/usr/bin/env bash\n" + "echo 'isolation-audit: categories -- 0 unmeasured, 0 expected-by-declaration," + " 0 triaged gap(s), 0 fault(s)'\n" + "exit 0\n", + encoding="utf-8", + ) + audit.chmod(0o755) + run_tmp = tmp_path / "runtmp" + run_tmp.mkdir() + script = tmp_path / "step.sh" + script.write_text(_step_run_body("Isolation invariants"), encoding="utf-8") + ledger = tmp_path / "ledger.sh" + ledger.write_text(_step_run_body("Category ledger"), encoding="utf-8") + env = dict( + os.environ, + GITHUB_WORKSPACE=str(workspace), + RUNNER_TEMP=str(run_tmp), + GITHUB_STEP_SUMMARY=str(run_tmp / "summary.md"), + PATH=f"{bindir}:{os.environ['PATH']}", + ) + proc = subprocess.run( + ["bash", str(script)], env=env, capture_output=True, text=True, cwd=REPO_ROOT, timeout=120 + ) + assert proc.returncode == 0, proc.stdout + proc.stderr + assert "::error" not in proc.stdout, proc.stdout + subprocess.run( + ["bash", str(ledger)], env=env, capture_output=True, text=True, cwd=REPO_ROOT, timeout=120 + ) + summary = (run_tmp / "summary.md").read_text(encoding="utf-8") + assert "| ok | isolation audit (scripts/isolation-audit.sh) |" in summary, summary + assert "0 unmeasured, 0 expected absence(s)" in summary, summary + + +if __name__ == "__main__": + sys.exit(pytest.main([__file__, "-v"]))