From cc60a47454fc513d03fdec6a74e4b7915c745d69 Mon Sep 17 00:00:00 2001 From: G <41178744+catomean@users.noreply.github.com> Date: Sat, 29 Aug 2026 09:11:22 +0200 Subject: [PATCH] feat(ops): audit what is DEPLOYED, not what is committed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit bitbaum/fleet already has hosted-supabase-audit.sh, whose stated job is "does any repo still point at a Supabase we retired?". It ran on 2026-08-28 and reported SUCCESS while printcraft returned 500 in production, pointing at a hosted project whose DNS no longer resolved. It missed because it runs `git grep` over repo checkouts, and the reference lived in /opt/printcraft/shared/.env — not in git, deliberately, because the box is the environment SSOT. The gate was scoped to the wrong substrate: it checked the artifact while the configuration lived in the running system. A check that looks in the wrong place is worse than no check, because it answers. The rule this implements: SOURCE audits run in CI against git, RUNTIME audits run on the box against what is actually deployed. The fleet had a good collection of the first kind and none of the second. Five checks, each a shape that has already bitten: - an env still naming a retired managed host (supabase.co, neon, planetscale) - a configured host that no longer resolves - a process running with keys its .env can no longer provide — config broken on disk for hours while the process holds the old values, which is exactly how 25 vitareba cron jobs died at the next restart - a declared node engine the box cannot satisfy - an app unit that is not active Found on its first run, before it was committed: revamp-info declares node >=22, box runs 20.20.2 — and it IS deployed revampit points at a dead trycloudflare.com quick-tunnel host aoz-demo unit inactive Alerts per finding, never one aggregate: host-check spent six weeks unable to fire because one permanently-failed unit pinned a single boolean at `bad`. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01UvjGNAS9CMfEGNW26tUR4P --- scripts/ci/runtime-conformance-audit.sh | 107 ++++++++++++++++++ .../hetzner/install-runtime-conformance.sh | 87 ++++++++++++++ 2 files changed, 194 insertions(+) create mode 100755 scripts/ci/runtime-conformance-audit.sh create mode 100755 scripts/hetzner/install-runtime-conformance.sh diff --git a/scripts/ci/runtime-conformance-audit.sh b/scripts/ci/runtime-conformance-audit.sh new file mode 100755 index 00000000..79044a36 --- /dev/null +++ b/scripts/ci/runtime-conformance-audit.sh @@ -0,0 +1,107 @@ +#!/usr/bin/env bash +# runtime-conformance-audit.sh — check the RUNNING system, not the source. +# +# WHY THIS EXISTS. bitbaum/fleet already has hosted-supabase-audit.sh, whose +# stated job is "does any repo still point at a Supabase we retired?". It ran on +# 2026-08-28 and reported SUCCESS while printcraft's production API returned 500, +# because it pointed at a hosted project whose DNS no longer resolved. +# +# It missed because it runs `git grep` over repo checkouts, and the reference +# lived in /opt/printcraft/shared/.env — not in git, deliberately, because the +# box is the environment SSOT. The gate was scoped to the wrong substrate: it +# checked the artifact while the configuration lived in the running system. +# +# So the rule this file implements: SOURCE audits run in CI against git; +# RUNTIME audits run here against what is actually deployed. The fleet had a +# good collection of the first kind and none of the second. +# +# Read-only. Prints one line per finding, exits 1 if any. Never prints a secret. +set -u +export LC_ALL=C + +BOX_NODE="$(node --version 2>/dev/null | tr -d 'v')" +BOX_NODE_MAJOR="${BOX_NODE%%.*}" +findings=0 +unreadable=0 + +note() { printf '%s\n' "$*"; } +finding() { findings=$((findings+1)); printf 'FINDING|%s|%s\n' "$1" "$2"; } +cannot() { unreadable=$((unreadable+1)); printf 'UNREADABLE|%s|%s\n' "$1" "$2"; } + +# Hosts we have deliberately left. A managed service we no longer pay for is +# not a dependency we still want discovered by an outage. +RETIRED_RE='supabase\.co|neon\.tech|planetscale\.com|railway\.app|render\.com|\.vercel-storage\.com' + +for dir in /opt/*/; do + app="$(basename "$dir")" + case "$app" in _appcron|backups|monitoring|supabase) continue ;; esac + + env="" + for cand in "$dir/shared/.env" "$dir/app/.env"; do + [ -f "$cand" ] && { env="$cand"; break; } + done + [ -n "$env" ] || continue + [ -r "$env" ] || { cannot "$app" "env not readable: $env"; continue; } + + # ---- 1. a retired managed host still referenced ------------------------- + hits="$(grep -oE "https?://[A-Za-z0-9._-]*($RETIRED_RE)" "$env" 2>/dev/null | sort -u | tr '\n' ' ')" + if [ -n "${hits// /}" ]; then + finding "$app" "env still points at a retired managed host: $hits" + fi + + # ---- 2. every configured host actually resolves ------------------------- + # A URL that no longer resolves is the shape printcraft was in for weeks + # while its pages still returned 200. + while read -r host; do + [ -z "$host" ] && continue + case "$host" in localhost|127.0.0.1|0.0.0.0) continue ;; esac + if ! getent hosts "$host" >/dev/null 2>&1; then + finding "$app" "configured host does not resolve: $host" + fi + done < <(grep -oE '^[A-Z_]+=[^#]*https?://[A-Za-z0-9._-]+' "$env" 2>/dev/null \ + | grep -oE 'https?://[A-Za-z0-9._-]+' | sed 's#https\?://##' | sort -u) + + # ---- 3. the running process vs the file that is supposed to produce it --- + # Config can be broken on disk for hours while the process still holds the + # old values in memory. That is a PENDING outage, and a restart detonates it. + unit="${app}-app.service" + if systemctl cat "$unit" >/dev/null 2>&1; then + pid="$(systemctl show "$unit" -p MainPID --value 2>/dev/null)" + if [ -n "$pid" ] && [ "$pid" != "0" ] && [ -r "/proc/$pid/environ" ]; then + disk="$(grep -oE '^[A-Z][A-Z0-9_]*=' "$env" | sed 's/=$//' | sort -u)" + proc="$(tr '\0' '\n' < "/proc/$pid/environ" | grep -oE '^[A-Z][A-Z0-9_]*=' | sed 's/=$//' \ + | grep -vE '^(PATH|HOME|USER|LOGNAME|SHELL|PWD|OLDPWD|LANG|LC_ALL|TERM|SHLVL|_|HOSTNAME|PORT|NODE_ENV|NODE_VERSION|INVOCATION_ID|JOURNAL_STREAM|SYSTEMD_EXEC_PID|MEMORY_PRESSURE_WATCH|MEMORY_PRESSURE_WRITE|NOTIFY_SOCKET|MANAGERPID)$' \ + | sort -u)" + lost="$(comm -13 <(printf '%s\n' "$disk") <(printf '%s\n' "$proc") | tr '\n' ' ')" + if [ -n "${lost// /}" ]; then + finding "$app" "running with keys its .env can no longer provide (lost on next restart): $lost" + fi + fi + fi + + # ---- 4. declared node engine vs the node that will run it --------------- + for pj in "$dir/app/package.json" "$dir/package.json"; do + [ -f "$pj" ] || continue + want="$(grep -oE '"node"[[:space:]]*:[[:space:]]*"[^"]+"' "$pj" 2>/dev/null | head -1 | grep -oE '[0-9]+' | head -1)" + if [ -n "$want" ] && [ -n "$BOX_NODE_MAJOR" ] && [ "$want" -gt "$BOX_NODE_MAJOR" ]; then + finding "$app" "declares node >=$want but this box runs $BOX_NODE" + fi + break + done +done + +# ---- 5. every app unit that exists is actually serving --------------------- +for u in /etc/systemd/system/*-app.service; do + [ -e "$u" ] || continue + unit="$(basename "$u")"; app="${unit%-app.service}" + state="$(systemctl is-active "$unit" 2>/dev/null)" + [ "$state" = "active" ] || finding "$app" "unit is $state" +done + +note "checked $(ls -d /opt/*/ 2>/dev/null | wc -l) app dirs on node $BOX_NODE" +note "findings=$findings unreadable=$unreadable" +# Absence of a finding across zero readable apps is not a pass. +if [ "$unreadable" -gt 0 ] && [ "$findings" -eq 0 ]; then + note "NOTE: $unreadable app(s) could not be read — clean is not proven" +fi +[ "$findings" -eq 0 ] diff --git a/scripts/hetzner/install-runtime-conformance.sh b/scripts/hetzner/install-runtime-conformance.sh new file mode 100755 index 00000000..bc50ba4a --- /dev/null +++ b/scripts/hetzner/install-runtime-conformance.sh @@ -0,0 +1,87 @@ +#!/usr/bin/env bash +# install-runtime-conformance.sh — run the runtime conformance audit on the box +# and alert to Telegram, per finding, on transition. +# +# The audit itself ships in the repo (scripts/ci/runtime-conformance-audit.sh) +# and the deploy rsyncs it to /opt/fleetcrown/app, so there is ONE copy. This +# installs only the wrapper and the timer. +# +# Per-FINDING alert keys, never one aggregate: host-check spent six weeks unable +# to fire because a single permanently-failed unit pinned one boolean at `bad`. +# A new finding must always transition, whatever else is already broken. +# +# Idempotent. Usage: bash scripts/hetzner/install-runtime-conformance.sh +set -euo pipefail +source "$(dirname "${BASH_SOURCE[0]}")/lib.sh" + +ssh -o BatchMode=yes "$BOX" 'sudo bash -s' <<'REMOTE' +set -euo pipefail +MON=/opt/monitoring +mkdir -p "$MON/state" + +cat > "$MON/runtime-conformance.sh" <<'CHK' +#!/usr/bin/env bash +# Wrapper: run the audit, alert per finding. Invoked by the timer. +# +# NOT `set -e`: the audit exits non-zero precisely when it has something to say, +# and -e would kill this before it could say it. +set -u +MON="${MON:-/opt/monitoring}" +. "$MON/lib-alert.sh" +AUDIT=/opt/fleetcrown/app/scripts/ci/runtime-conformance-audit.sh + +# "Could not look" gets its OWN key so it can never overwrite a real finding +# or be mistaken for a clean run. +if [ ! -r "$AUDIT" ]; then + alert_transition rtc_probe bad "🔎" "Runtime conformance CANNOT RUN: $AUDIT missing (deploy changed?)" + exit 0 +fi +alert_transition rtc_probe ok "🔎" "" + +out=$(bash "$AUDIT" 2>&1) + +# One key per app+check, so a new finding always transitions. +declare -A seen=() +while IFS='|' read -r kind app msg; do + [ "$kind" = "FINDING" ] || continue + key="rtc_$(printf '%s' "$app $msg" | tr -c 'a-zA-Z0-9' '_' | cut -c1-60)" + seen[$key]=1 + alert_transition "$key" bad "🧩" "$app: $msg" +done <<< "$out" + +# A finding that has gone away clears its own key, or its next occurrence +# would be silent. +for sf in "$MON"/state/host_rtc_*; do + [ -e "$sf" ] || continue + k=$(basename "$sf"); k=${k#host_} + case "$k" in rtc_probe) continue ;; esac + [ -n "${seen[$k]:-}" ] || alert_transition "$k" ok "" "" +done +exit 0 +CHK +chmod +x "$MON/runtime-conformance.sh" + +cat > /etc/systemd/system/fleetcrown-runtime-conformance.service <<'SVC' +[Unit] +Description=Runtime conformance audit (what is deployed, not what is committed) +OnFailure=notify-failure@%n.service +[Service] +Type=oneshot +ExecStart=/opt/monitoring/runtime-conformance.sh +SVC + +cat > /etc/systemd/system/fleetcrown-runtime-conformance.timer <<'TMR' +[Unit] +Description=Run the runtime conformance audit every 6h +[Timer] +OnCalendar=*-*-* 02,08,14,20:00:00 +RandomizedDelaySec=300 +Persistent=true +[Install] +WantedBy=timers.target +TMR + +systemctl daemon-reload +systemctl enable --now fleetcrown-runtime-conformance.timer >/dev/null +systemctl list-timers --all --no-pager | grep runtime-conformance || true +REMOTE