Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 41 additions & 19 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -391,11 +391,17 @@ jobs:
name: Playwright E2E (production)
runs-on: ubuntu-latest
# A backstop above the test step's own cap, not the primary bound: a job
# timeout is a cancellation and would skip the rollback below, whereas the
# step cap marks the step failed and keeps it reachable.
# timeout cancels the test step, and a cancelled step is not a verdict the
# rollback job acts on, whereas the step cap marks it failed.
timeout-minutes: 30
needs: [production-smoke-tests]
if: github.ref == 'refs/heads/main' && github.event_name == 'push'
# The test step's own outcome, for rollback-production-on-e2e-failure. Not
# the job's result: a failed checkout, npm ci or Chrome report fails the job
# too, and none of them says anything about whether production is healthy.
# Only tests that ran and failed roll production back.
outputs:
tests: ${{ steps.e2e.outcome }}
# The browser is the Google Chrome the runner image ships
# (e2e/playwright.config.ts) rather than one installed here: `npx
# playwright install --with-deps` puts an apt update and Playwright's CDN in
Expand All @@ -418,9 +424,9 @@ jobs:
- name: Report the runner's Chrome
run: node e2e/report-chrome.js
# Bounded at the step rather than relying on the job cap: a step timeout
# marks the step failed, which still trips the rollback below, whereas a
# job timeout is a cancellation and would skip it. A production wedged
# badly enough to hang Playwright is exactly when the rollback is wanted.
# marks the step failed, which still trips the rollback job, whereas a
# job timeout cancels it and would not. A production wedged badly enough
# to hang Playwright is exactly when the rollback is wanted.
- name: Run E2E tests (production)
id: e2e
# See the staging job: the step cap has to clear nodes.spec.ts's retried
Expand All @@ -430,13 +436,29 @@ jobs:
env:
CI: "true"
RADAR_API_KEY: ${{ secrets.RADAR_API_KEY }}
# Scoped to the test step's own outcome, as the production smoke job is.
# A bare failure() also fires when checkout, npm ci or the Chrome report
# fails, and anything that breaks before the tests run is a statement
# about the runner, not about production. Those leave the job red for a
# human to read; only tests that ran and failed roll production back.
- name: Rollback production on E2E failure
if: failure() && steps.e2e.outcome == 'failure'

# A job of its own, so that production's root key is never handed to a job
# that has just run npm's install scripts. test_deploy_workflows.py holds
# every workflow to that.
rollback-production-on-e2e-failure:
name: Roll back production if its E2E tests failed
runs-on: ubuntu-latest
timeout-minutes: 20
needs: [e2e-prod]
# See rollback-production-on-deploy-failure: a following push's deploy must
# not be inside the deploy directory while rollback.sh is.
concurrency:
group: production-deploy
cancel-in-progress: false
# !cancelled() rather than always(): e2e-prod failing is when this has to
# run, but an operator cancelling the run, say over a known flake, must
# still be able to stop it. The main-push gate is repeated so no other event
# can open production SSH.
if: >-
!cancelled() && github.ref == 'refs/heads/main' && github.event_name == 'push' &&
needs.e2e-prod.outputs.tests == 'failure'
steps:
- name: Roll back production
uses: appleboy/ssh-action@0ff4204d59e8e51228ff73bce53f80d53301dee2 # v1.2.5
with:
host: ${{ secrets.SERVER_HOST }}
Expand Down Expand Up @@ -934,8 +956,8 @@ jobs:

# Reached only once the app answered /api/health. The box is no
# longer mid-deploy, so drop the marker; any failure after this
# point is a smoke/E2E failure, which the rollback steps on those
# jobs already cover.
# point is a smoke/E2E failure, which the smoke job's rollback step
# and rollback-production-on-e2e-failure already cover.
rm -f ${{ env.APP_DIR }}/.deploy-in-progress

rollback-production-on-deploy-failure:
Expand All @@ -959,9 +981,9 @@ jobs:
# would then SSH into production to consider rolling back a deploy that
# never ran. Keyed to the deploy job's own result, it cannot.
#
# This is the gap the two existing rollbacks do not cover. Both hang off
# jobs DOWNSTREAM of the deploy (production-smoke-tests, e2e-prod), and a
# failed deploy skips those jobs and their rollback steps with them.
# This is the gap the two other rollbacks do not cover. Both hang off jobs
# DOWNSTREAM of the deploy (production-smoke-tests, e2e-prod), and a failed
# deploy skips those jobs, and so the rollbacks keyed to them.
# 'cancelled' as well as 'failure'. Cancelling mid-deploy leaves production
# exactly as half-deployed as a crash does, and deploy-production sets
# cancel-in-progress: false, so this can only be a human pressing cancel —
Expand Down Expand Up @@ -1368,7 +1390,7 @@ jobs:
echo ""
smoke_summary
[ "$FAIL" -eq 0 ] || exit 1
# Auto-rollback on smoke failure, mirroring the prod E2E job. A failed
# Auto-rollback on smoke failure, as on an E2E failure. A failed
# smoke test means a bad build is live serving prod; revert to
# last-known-good rather than leaving it up for a human to catch. The
# condition is scoped to the smoke step's own outcome, so a failure in the
Expand All @@ -1383,7 +1405,7 @@ jobs:
key: ${{ secrets.SERVER_SSH_KEY }}
envs: APP_DIR
script: |
# See the identical assertion in e2e-prod's rollback step.
# See the identical assertion in rollback-production-on-e2e-failure.
[ "$(hostname)" = "retina-prod" ] || { echo "::error::Refusing to roll back on '$(hostname)' — expected retina-prod."; exit 1; }
echo "Production smoke tests failed — rolling back production..."
cd ${{ env.APP_DIR }} && bash deploy/rollback.sh
176 changes: 163 additions & 13 deletions backend/tests/test_deploy_workflows.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
and pin what must hold in every copy.
"""

import functools
import re
from pathlib import Path

Expand All @@ -12,6 +13,12 @@

WORKFLOWS = Path(__file__).resolve().parents[2] / ".github" / "workflows"


@functools.cache
def _workflow(workflow: str) -> dict:
return yaml.safe_load((WORKFLOWS / workflow).read_text())


DEPLOYS = [
pytest.param("ci.yml", "deploy-production", id="production"),
pytest.param("staging-deploy-verify.yml", "deploy", id="staging"),
Expand All @@ -26,7 +33,7 @@


def _job(workflow: str, job: str) -> dict:
jobs = yaml.safe_load((WORKFLOWS / workflow).read_text())["jobs"]
jobs = _workflow(workflow)["jobs"]
assert job in jobs, f"{workflow} has no job {job!r}"
return jobs[job]

Expand Down Expand Up @@ -205,22 +212,165 @@ def test_rollback_job_reads_back_what_the_deploy_wrote(workflow, deploy, rollbac
assert "always()" in condition


def _allowed(workflow: str, job: str, values: dict[str, str], *, cancelled: bool = False) -> bool:
"""A job's `if:`, evaluated with each context field replaced by a value."""
condition = " ".join(_job(workflow, job)["if"].split())
for field, value in values.items():
# Whole fields only: github.ref must not also rewrite github.ref_name.
condition = re.sub(rf"(?<![\w.-]){re.escape(field)}(?![\w-])", repr(value), condition)
# The rest is equality and boolean operators over concrete strings, whose
# semantics are the same in Actions and Python. Any other function is left
# undefined, and so fails the test rather than guessing.
condition = condition.replace("!cancelled()", repr(not cancelled)).replace("always()", "True")
assert not re.search(r"!(?!=)", condition), f"no Python equivalent for a negation in {condition!r}"
condition = condition.replace("&&", " and ").replace("||", " or ")
return eval(condition, {"__builtins__": {}})


@pytest.mark.parametrize("ref", ["refs/heads/main", "refs/heads/topic", "refs/pull/417/merge"])
@pytest.mark.parametrize("event", ["push", "pull_request", "workflow_dispatch"])
@pytest.mark.parametrize("result", ["failure", "cancelled", "success", "skipped"])
def test_production_rollback_requires_a_failed_or_cancelled_main_push(ref, event, result):
@pytest.mark.parametrize("cancelled", [False, True])
def test_production_rollback_requires_a_failed_or_cancelled_main_push(ref, event, result, cancelled):
# A cancelled PR run can report a skipped deploy as cancelled. Evaluate
# the real job condition: the remote marker check is too late to prevent
# an unauthorized workflow event from opening a production SSH session.
condition = " ".join(_job("ci.yml", "rollback-production-on-deploy-failure")["if"].split())
for field, value in (
("github.ref", ref),
("github.event_name", event),
("needs.deploy-production.result", result),
):
condition = condition.replace(field, repr(value))
# This expression uses only equality and boolean operators, whose
# semantics are the same for these concrete strings in Actions/Python.
condition = condition.replace("always()", "True").replace("&&", "and").replace("||", "or")
allowed = eval(condition, {"__builtins__": {}})
# Cancelling the run does not stop it: a cancel mid-deploy leaves the box
# as half-deployed as a crash does.
allowed = _allowed(
"ci.yml",
"rollback-production-on-deploy-failure",
{"github.ref": ref, "github.event_name": event, "needs.deploy-production.result": result},
cancelled=cancelled,
)
assert allowed is (ref == "refs/heads/main" and event == "push" and result in {"failure", "cancelled"})


# ── Third-party code never shares a job with a droplet key ───────────────────
# A package manager runs install scripts from its registry, an action runs its
# author's code, and a container image is someone else's filesystem. Any of them
# can tamper with the environment a later step hands a root key to, so a job
# that runs one must not be able to read a key. Rollbacks that follow such a job
# run in a job of their own.

# A tripwire for the honest mistake, not a sandbox: flags may sit between a
# tool and its verb, and a script a step calls is not followed.
THIRD_PARTY_CODE = re.compile(
r"\b(?:(?:npm|pip3?|apt(?:-get)?|gem|cargo|go|bun|poetry)\b[^\n;&|]*\binstall"
r"|npm\b[^\n;&|]*\b(?:ci|i|exec)|npx|yarn|pnpm|pipx|uvx|apk\s+add|curl\b[^\n;&]*\|\s*(?:ba)?sh"
r"|uv\b[^\n;&|]*\b(?:sync|add|run|pip|tool)|pre-commit"
r"|docker(?:-compose|\b[^\n;&|]*\b(?:run|build|pull|compose)))\b",
re.IGNORECASE,
)
# The actions a key-holding job may use: the checkout, and the ssh transport the
# key is for. Any other action counts as third-party code.
KEY_JOB_ACTIONS = ("actions/checkout@", "appleboy/ssh-action@")
# The droplet keys are all named *SSH*KEY. A dynamic lookup, the whole secrets
# object, or `secrets: inherit` could be any of them. Case-blind, as Actions
# expressions are.
SSH_KEY = re.compile(
r"\bsecrets(?:\.\w*SSH\w*KEY\w*\b|\[\s*['\"]\w*SSH\w*KEY\w*['\"]\s*\]|\[\s*(?!['\"])|:\s*inherit\b)"
r"|\btoJSON\(\s*secrets\s*\)",
re.IGNORECASE,
)


def _workflow_files() -> list[Path]:
return sorted([*WORKFLOWS.glob("*.yml"), *WORKFLOWS.glob("*.yaml")])


def _workflow_jobs() -> list:
return [
pytest.param(path.name, name, job, id=f"{path.name}:{name}")
for path in _workflow_files()
for name, job in _workflow(path.name)["jobs"].items()
]


def _runs_third_party_code(job: dict) -> bool:
steps = job.get("steps", [])
return (
"container" in job
or "services" in job
# Another repository's reusable workflow.
or ("uses" in job and not str(job["uses"]).startswith("./"))
or any("uses" in step and not str(step["uses"]).startswith(KEY_JOB_ACTIONS) for step in steps)
or any(THIRD_PARTY_CODE.search(step.get("run", "")) for step in steps)
)


def test_the_guard_recognises_what_the_jobs_run():
# Guards the guard: patterns that matched nothing would pass every job.
assert _runs_third_party_code(_job("ci.yml", "web-build"))
assert _runs_third_party_code(_job("ci.yml", "backend-tests"))
assert _runs_third_party_code(_job("ci.yml", "lint"))
assert _runs_third_party_code(_job("ci.yml", "e2e-prod"))
assert _runs_third_party_code({"services": {"db": {"image": "postgres"}}, "steps": []})
assert _runs_third_party_code({"steps": [{"uses": "docker://alpine:3"}]})
assert _runs_third_party_code({"steps": [{"uses": "./.github/actions/setup-uv"}]})
assert _runs_third_party_code({"steps": [{"run": "uv run python scripts/check.py"}]})
assert _runs_third_party_code({"steps": [{"run": "docker run --rm some/image"}]})
assert _runs_third_party_code({"steps": [{"run": "npm --prefix frontend ci"}]})
assert _runs_third_party_code({"steps": [{"run": "sudo apt-get -y install jq"}]})
assert _runs_third_party_code({"uses": "someorg/repo/.github/workflows/x.yml@main"})
assert not _runs_third_party_code(_job("ci.yml", "staging"))
assert not _runs_third_party_code(_job("ci.yml", "deploy-production"))
assert not _runs_third_party_code(_job("ci.yml", "production-smoke-tests"))
assert SSH_KEY.search(yaml.safe_dump(_job("ci.yml", "deploy-production")))
assert SSH_KEY.search("key: ${{ secrets['STAGING_SSH_KEY'] }}")
assert SSH_KEY.search("key: ${{ secrets[format('{0}_SSH_KEY', inputs.env)] }}")
assert SSH_KEY.search("ALL: ${{ toJson(secrets) }}")
assert SSH_KEY.search("key: ${{ secrets.server_ssh_key }}")
assert SSH_KEY.search("key: ${{ secrets.DEPLOY_SSH_PRIVATE_KEY }}")
assert SSH_KEY.search(yaml.safe_dump({"uses": "someorg/repo/.github/workflows/x.yml@main", "secrets": "inherit"}))
assert not SSH_KEY.search("RADAR_API_KEY: ${{ secrets.RADAR_API_KEY }}")


@pytest.mark.parametrize(("workflow", "name", "job"), _workflow_jobs())
def test_no_job_that_runs_third_party_code_can_read_a_droplet_key(workflow, name, job):
if _runs_third_party_code(job):
assert not SSH_KEY.search(yaml.safe_dump(job)), f"{workflow}:{name} runs third-party code"


@pytest.mark.parametrize("path", _workflow_files(), ids=lambda path: path.name)
def test_no_droplet_key_is_handed_to_every_job(path):
# Workflow-level env reaches every job, the npm ones included.
assert not SSH_KEY.search(yaml.safe_dump(_workflow(path.name).get("env", {})))


# ── Production's E2E rollback ────────────────────────────────────────────────


def test_the_e2e_rollback_acts_on_the_test_step_alone():
# A failed pull, checkout or npm ci fails e2e-prod as well, and says
# nothing about production. Only the test step's own verdict may count.
e2e = _job("ci.yml", "e2e-prod")
assert [step for step in e2e["steps"] if step.get("id") == "e2e" and "test:e2e:prod" in step["run"]]
assert e2e["outputs"]["tests"] == "${{ steps.e2e.outcome }}"


def test_the_e2e_rollback_holds_the_production_lock_on_the_runner():
rollback = _job("ci.yml", "rollback-production-on-e2e-failure")
assert "container" not in rollback
assert rollback["needs"] == ["e2e-prod"]
assert rollback["concurrency"] == {"group": "production-deploy", "cancel-in-progress": False}
script = _script("ci.yml", "rollback-production-on-e2e-failure")
identity = _index(script, re.escape('[ "$(hostname)" = "retina-prod" ]'))
assert identity < _index(script, r"\bbash deploy/rollback\.sh$")


@pytest.mark.parametrize("ref", ["refs/heads/main", "refs/heads/topic", "refs/pull/417/merge"])
@pytest.mark.parametrize("event", ["push", "pull_request", "workflow_dispatch"])
@pytest.mark.parametrize("tests", ["failure", "cancelled", "success", "skipped", ""])
@pytest.mark.parametrize("cancelled", [False, True])
def test_the_e2e_rollback_requires_failed_tests_on_a_main_push(ref, event, tests, cancelled):
# "" is what the output reads when e2e-prod never reached the test step,
# or was skipped outright. Cancelling the run stops it: an operator cancels
# to keep a known flake from reverting a healthy production.
allowed = _allowed(
"ci.yml",
"rollback-production-on-e2e-failure",
{"github.ref": ref, "github.event_name": event, "needs.e2e-prod.outputs.tests": tests},
cancelled=cancelled,
)
assert allowed is (ref == "refs/heads/main" and event == "push" and tests == "failure" and not cancelled)
Loading