diff --git a/.github/workflows/main-health-watch.yml b/.github/workflows/main-health-watch.yml new file mode 100644 index 00000000..d862a7b4 --- /dev/null +++ b/.github/workflows/main-health-watch.yml @@ -0,0 +1,41 @@ +name: Main health watch + +# #3324: keep one main-red-alarm issue open while main's own Quality/Containers +# runs are red, close it when they are green, and start CI on a main head that +# never got any (merges made by github-actions start no push runs). See +# scripts/main-health-watch.py for the lifecycle. +on: + workflow_run: + workflows: [Quality, Containers] + types: [completed] + branches: [main] + # Hourly as well: an untested head produces no completed run to react to, + # which is exactly the case the schedule exists for. + schedule: + - cron: "41 * * * *" + workflow_dispatch: + +permissions: {} + +concurrency: + group: main-health-watch + cancel-in-progress: false + +jobs: + sweep: + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + issues: write + actions: write + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + sparse-checkout: scripts/main-health-watch.py + sparse-checkout-cone-mode: false + - name: Sweep main + env: + GH_TOKEN: ${{ github.token }} + run: python3 scripts/main-health-watch.py diff --git a/docs/CI-CD.md b/docs/CI-CD.md index 798f02c3..6ed84264 100644 --- a/docs/CI-CD.md +++ b/docs/CI-CD.md @@ -146,6 +146,26 @@ ruleset's enforcement to *Disabled* (Settings → Rules) and must re-enable it afterwards; there is deliberately no standing bypass actor, since automation merges with the owner's token. +## Main health watch (#3324) + +`main-health-watch.yml` runs after every Quality/Containers run on `main` +and hourly (`scripts/main-health-watch.py`): + +- **Red:** when the newest completed run of either workflow on `main` failed, + it opens one `main-red-alarm` issue naming the failing jobs, the first red + and last green commit, and the commits in between. It comments again only + when the failing head changes, and closes the issue itself once both are + green. It never reverts or re-runs a failed run. +- **Untested:** merges made by `github-actions` (Dependabot auto-merge) start + no push runs, because `GITHUB_TOKEN` events don't trigger workflows. That's + how the 2026-09-25 breakage sat on `main` with no red run at all (#3311). + When `main`'s head is over an hour old and a watched workflow never ran on + it, the watch dispatches that workflow on `main` (`workflow_dispatch` is + the exception `GITHUB_TOKEN` may start), and the next sweep judges it. + +Replay any past moment without side effects: +`GITHUB_REPOSITORY=Xore/APIARY python3 scripts/main-health-watch.py --before 2026-09-25T10:00:00Z`. + ## Pull request workflow ### No AI attribution (#3329) diff --git a/scripts/main-health-watch.py b/scripts/main-health-watch.py new file mode 100644 index 00000000..577ea78b --- /dev/null +++ b/scripts/main-health-watch.py @@ -0,0 +1,217 @@ +#!/usr/bin/env python3 +"""Keep exactly one issue open while main's own gates are red (#3324). + +On 2026-09-25 five dependabot merges left main's Rust backend red for about +seven hours (#3311) and nothing said so: Quality/Containers ran on the push, +failed, and a red push run is read by nobody. The other *-watch workflows +already follow the one-labeled-issue lifecycle for disk, backups and compose +drift; this is the same thing for main itself: + +- A single open `main-red-alarm` issue at a time. The newest completed push + run of each watched workflow on main decides the state; cancelled runs are + skipped (a newer push superseded them, they say nothing about the code). +- Red: open the issue, or comment on it -- but only when the failing head + commit changed since the last comment, so a sweep does not repeat itself. + The report names the failing jobs, the first red and last green commit per + workflow, and the commits in between (the suspects). +- Untested heads get tested: when main's head is older than an hour and a + watched workflow never ran on it (merges made by github-actions start no + push runs -- 2026-09-25's actual case), the watch dispatches that workflow + on main; the next sweep judges the result like any other run. +- Green again: close the issue with the green run as evidence. +- Never reverts, never re-runs anything. + +Usage: main-health-watch.py [--dry-run] [--before ISO8601] (needs gh, GITHUB_REPOSITORY) +""" +from __future__ import annotations + +import argparse +import json +import os +import re +import subprocess +import sys +from datetime import datetime, timedelta, timezone + +LABEL = "main-red-alarm" +WORKFLOWS = ("quality.yml", "containers.yml") +MARKER = "" +MARKER_RE = re.compile(r"") +HISTORY = 30 # push runs to look back through for the last green one +UNTESTED_AFTER = timedelta(hours=1) # CI for a fresh head may simply still be queued + + +def gh(*args: str) -> str: + out = subprocess.run(["gh", *args], capture_output=True, text=True) + if out.returncode != 0: + print(f"FAIL: gh {' '.join(args[:2])}: {out.stderr.strip()}", file=sys.stderr) + sys.exit(1) + return out.stdout + + +def gh_json(*args: str): + return json.loads(gh(*args) or "null") + + +def push_runs(repo: str, workflow: str, before: str | None = None) -> list[dict]: + runs = gh_json( + "run", "list", "-R", repo, "--workflow", workflow, "--branch", "main", + "--status", "completed", "--limit", str(HISTORY * (4 if before else 2)), + "--json", "databaseId,headSha,conclusion,url,createdAt,event", + ) + # push runs, plus the workflow_dispatch runs this watch starts for heads + # no push run ever tested (see untested_head). + runs = [ + r for r in runs + if r["event"] in ("push", "workflow_dispatch") and r["conclusion"] not in ("cancelled", "skipped") + ] + if before: # replay: judge main as it stood at that moment + runs = [r for r in runs if r["createdAt"] < before] + return runs[:HISTORY] + + +def failed_jobs(repo: str, run_id: int) -> list[str]: + jobs = gh_json("run", "view", str(run_id), "-R", repo, "--json", "jobs")["jobs"] + return [j["name"] for j in jobs if j["conclusion"] in ("failure", "timed_out")] + + +def assess(repo: str, before: str | None = None) -> dict: + """Per workflow: latest conclusion, and for a red one, the red streak.""" + state = {} + for wf in WORKFLOWS: + runs = push_runs(repo, wf, before) + if not runs: + continue + latest = runs[0] + entry = {"latest": latest, "red": latest["conclusion"] != "success"} + if entry["red"]: + green = next((r for r in runs if r["conclusion"] == "success"), None) + streak = runs[: runs.index(green)] if green else runs + entry.update(first_red=streak[-1], last_green=green, jobs=failed_jobs(repo, latest["databaseId"])) + state[wf] = entry + return state + + +def untested_head(repo: str, before: str | None = None) -> dict | None: + """main's head commit if it is older than UNTESTED_AFTER and a watched + workflow has no run for it at all -- not red, not queued: never started. + + That is what 2026-09-25 actually looked like (#3311): merges made by + github-actions with GITHUB_TOKEN start no workflow runs, so five broken + dependabot merges produced no red run to notice, only silence. + """ + query = f"repos/{repo}/commits?sha=main&per_page=1" + (f"&until={before}" if before else "") + [head] = gh_json("api", query) + committed = datetime.fromisoformat(head["commit"]["committer"]["date"].replace("Z", "+00:00")) + now = datetime.fromisoformat(before.replace("Z", "+00:00")) if before else datetime.now(timezone.utc) + if now - committed < UNTESTED_AFTER: + return None + missing = [ + wf for wf in WORKFLOWS + if not gh_json("run", "list", "-R", repo, "--workflow", wf, "--commit", head["sha"], "--json", "databaseId") + ] + if not missing: + return None + return {"headSha": head["sha"], "committed": head["commit"]["committer"]["date"], + "subject": head["commit"]["message"].splitlines()[0][:100], "missing": missing} + + +def suspects(repo: str, last_green: dict | None, first_red: dict) -> list[str]: + if not last_green: + return [] + cmp = gh_json("api", f"repos/{repo}/compare/{last_green['headSha']}...{first_red['headSha']}") + return [f"{c['sha'][:10]} {c['commit']['message'].splitlines()[0][:100]}" for c in cmp["commits"]][-15:] + + +def report(repo: str, state: dict) -> str: + lines = ["`main`'s own gates are red. Found by `scripts/main-health-watch.py` (#3324); this issue closes itself when they are green again.", ""] + for wf, e in state.items(): + if not e["red"]: + lines.append(f"- **{wf}**: green ({e['latest']['url']})") + continue + lg = e["last_green"] + lines += [ + f"- **{wf}**: {e['latest']['conclusion']} on `{e['latest']['headSha'][:10]}` ({e['latest']['url']})", + f" - failing jobs: {', '.join(e['jobs']) or '(none reported)'}", + f" - first red: `{e['first_red']['headSha'][:10]}` ({e['first_red']['createdAt']})", + f" - last green: `{lg['headSha'][:10]}` ({lg['createdAt']})" if lg else f" - no green run in the last {HISTORY}", + ] + if sus := suspects(repo, lg, e["first_red"]): + lines += [" - commits since last green (suspects):", *[f" - {s}" for s in sus]] + head = max((e["latest"] for e in state.values() if e["red"]), key=lambda r: r["createdAt"]) + lines += ["", MARKER.format(sha=head["headSha"])] + return "\n".join(lines) + + +def open_issue(repo: str) -> dict | None: + found = gh_json("issue", "list", "-R", repo, "--label", LABEL, "--state", "open", "--json", "number,comments,body") + return found[0] if found else None + + +def last_marker(issue: dict) -> str | None: + texts = [issue["body"], *(c["body"] for c in issue["comments"])] + for text in reversed(texts): + if m := MARKER_RE.search(text or ""): + return m.group(1) + return None + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument("--dry-run", action="store_true") + parser.add_argument("--before", metavar="ISO8601", help="replay: judge main as of this time (implies --dry-run)") + args = parser.parse_args(argv) + repo = os.environ.get("GITHUB_REPOSITORY", "") + if not repo: + print("FAIL: GITHUB_REPOSITORY must be set", file=sys.stderr) + return 1 + + if args.before: + args.dry_run = True + state = assess(repo, args.before) + untested = untested_head(repo, args.before) + if untested: + # workflow_dispatch is the one event GITHUB_TOKEN may start runs with, + # so test the head instead of alarming; the next sweep judges the runs. + print(f"untested head {untested['headSha'][:10]} ({untested['subject']}): no run of {', '.join(untested['missing'])}") + for wf in untested["missing"]: + if args.dry_run: + print(f" dry run: would dispatch {wf} on main") + else: + gh("workflow", "run", wf, "-R", repo, "--ref", "main") + print(f" dispatched {wf} on main") + red = any(e["red"] for e in state.values()) + issue = None if args.dry_run else open_issue(repo) + + if not red: + print("main is green: " + ", ".join(f"{wf} {e['latest']['headSha'][:10]}" for wf, e in state.items())) + if issue: + evidence = "\n".join(f"- {wf}: green on `{e['latest']['headSha'][:10]}` ({e['latest']['url']})" for wf, e in state.items()) + gh("issue", "close", str(issue["number"]), "-R", repo, "--comment", f"`main` is green again.\n\n{evidence}") + print(f"closed #{issue['number']}") + return 0 + + body = report(repo, state) + if args.dry_run: + print(body) + return 0 + head = MARKER_RE.search(body).group(1) + subprocess.run( + ["gh", "label", "create", LABEL, "-R", repo, "-d", "main's own Quality/Containers gates are red (scripts/main-health-watch.py)", "--color", "B60205"], + capture_output=True, text=True, + ) + if issue: + if last_marker(issue) == head: + print(f"#{issue['number']} already reports head {head[:10]}; nothing new") + return 0 + gh("issue", "comment", str(issue["number"]), "-R", repo, "--body", body) + print(f"updated #{issue['number']}") + else: + gh("issue", "create", "-R", repo, "--label", LABEL, "--label", "ops", "--label", "bug", + "--title", "ci: main is red -- Quality/Containers failing on main (#3324 watch)", "--body", body) + print("opened main-red-alarm issue") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/tests/test_main_health_watch.py b/scripts/tests/test_main_health_watch.py new file mode 100644 index 00000000..3ba67c85 --- /dev/null +++ b/scripts/tests/test_main_health_watch.py @@ -0,0 +1,114 @@ +"""#3324 main-health-watch: issue lifecycle and untested-head dispatch against a fake gh.""" +from __future__ import annotations + +import importlib.util +import json +import os +import unittest +from pathlib import Path +from unittest import mock + +SCRIPT = Path(__file__).resolve().parents[1] / "main-health-watch.py" +_spec = importlib.util.spec_from_file_location("main_health_watch", SCRIPT) +mhw = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(mhw) + +OLD = "2026-09-25T06:00:00Z" +GREEN_SHA, RED_SHA, HEAD_SHA = "a" * 40, "b" * 40, "c" * 40 + + +def run(sha, conclusion, created, event="push", rid=1): + return {"databaseId": rid, "headSha": sha, "conclusion": conclusion, "url": f"u/{rid}", "createdAt": created, "event": event} + + +class FakeGh: + """Answers the gh calls the script makes; records the mutating ones.""" + + def __init__(self, runs, head_sha=GREEN_SHA, head_date=OLD, runs_for_head=True, issue=None): + self.runs, self.head_sha, self.head_date = runs, head_sha, head_date + self.runs_for_head, self.issue, self.calls = runs_for_head, issue, [] + + def __call__(self, *args): + a = list(args) + if a[:2] == ["run", "list"] and "--commit" in a: + return json.dumps([{"databaseId": 9}] if self.runs_for_head else []) + if a[:2] == ["run", "list"]: + return json.dumps(self.runs) + if a[:2] == ["run", "view"]: + return json.dumps({"jobs": [{"name": "Dashboard backend-service (Rust)", "conclusion": "failure"}, {"name": "ok", "conclusion": "success"}]}) + if a[0] == "api" and "/compare/" in a[1]: + return json.dumps({"commits": [{"sha": RED_SHA, "commit": {"message": "chore(deps): bump rand\n\nbody"}}]}) + if a[0] == "api" and "/commits?" in a[1]: + return json.dumps([{"sha": self.head_sha, "commit": {"committer": {"date": self.head_date}, "message": "x"}}]) + if a[:2] == ["issue", "list"]: + return json.dumps([self.issue] if self.issue else []) + self.calls.append(a) + return "" + + +class WatchTests(unittest.TestCase): + def go(self, fake, *argv): + with mock.patch.object(mhw, "gh", fake), mock.patch.object(mhw.subprocess, "run"), \ + mock.patch.dict(os.environ, {"GITHUB_REPOSITORY": "o/r"}): + return mhw.main(list(argv)) + + def red_runs(self): + return [run(RED_SHA, "failure", "2026-09-25T07:00:00Z", rid=2), run(GREEN_SHA, "success", OLD, rid=1)] + + def test_red_main_opens_one_issue_naming_jobs_and_suspects(self): + fake = FakeGh(self.red_runs()) + self.assertEqual(self.go(fake), 0) + [create] = [c for c in fake.calls if c[:2] == ["issue", "create"]] + body = create[create.index("--body") + 1] + self.assertIn("Dashboard backend-service (Rust)", body) + self.assertIn("bump rand", body) + self.assertIn(f"`{GREEN_SHA[:10]}`", body) + self.assertIn(mhw.MARKER.format(sha=RED_SHA), body) + + def test_same_red_head_is_not_reported_twice(self): + issue = {"number": 5, "body": mhw.MARKER.format(sha=RED_SHA), "comments": []} + fake = FakeGh(self.red_runs(), issue=issue) + self.go(fake) + self.assertEqual([c for c in fake.calls if c[0] == "issue"], []) + + def test_new_red_head_appends(self): + issue = {"number": 5, "body": mhw.MARKER.format(sha=GREEN_SHA), "comments": []} + fake = FakeGh(self.red_runs(), issue=issue) + self.go(fake) + self.assertTrue(any(c[:3] == ["issue", "comment", "5"] for c in fake.calls)) + + def test_green_again_closes_the_issue(self): + issue = {"number": 5, "body": mhw.MARKER.format(sha=RED_SHA), "comments": []} + fake = FakeGh([run(GREEN_SHA, "success", "2026-09-25T08:00:00Z")], issue=issue) + self.go(fake) + self.assertTrue(any(c[:3] == ["issue", "close", "5"] for c in fake.calls)) + + def test_cancelled_runs_do_not_decide(self): + runs = [run(RED_SHA, "cancelled", "2026-09-25T07:00:00Z", rid=2), run(GREEN_SHA, "success", OLD)] + fake = FakeGh(runs) + self.go(fake) + self.assertEqual([c for c in fake.calls if c[0] == "issue"], []) + + def test_untested_head_is_dispatched_not_alarmed(self): + fake = FakeGh([run(GREEN_SHA, "success", OLD)], head_sha=HEAD_SHA, runs_for_head=False) + self.go(fake) + dispatched = [c for c in fake.calls if c[:2] == ["workflow", "run"]] + self.assertEqual(sorted(c[2] for c in dispatched), sorted(mhw.WORKFLOWS)) + self.assertTrue(all(c[c.index("--ref") + 1] == "main" for c in dispatched)) + self.assertEqual([c for c in fake.calls if c[0] == "issue"], []) + + def test_fresh_head_gets_a_grace_period(self): + now = mhw.datetime.now(mhw.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + fake = FakeGh([run(GREEN_SHA, "success", OLD)], head_sha=HEAD_SHA, head_date=now, runs_for_head=False) + self.go(fake) + self.assertEqual([c for c in fake.calls if c[:2] == ["workflow", "run"]], []) + + def test_dispatched_runs_count_as_main_runs(self): + runs = [run(HEAD_SHA, "failure", "2026-09-25T09:00:00Z", event="workflow_dispatch", rid=3), run(GREEN_SHA, "success", OLD)] + fake = FakeGh(runs) + self.go(fake) + self.assertTrue(any(c[:2] == ["issue", "create"] for c in fake.calls)) + + +if __name__ == "__main__": + unittest.main()