diff --git a/.github/workflows/asvs-prove-absences.yml b/.github/workflows/asvs-prove-absences.yml new file mode 100644 index 00000000..9defecb9 --- /dev/null +++ b/.github/workflows/asvs-prove-absences.yml @@ -0,0 +1,288 @@ +name: ASVS prove-absences + +# WHAT THIS WIRES, AND WHY IT WAS WORTH WIRING. +# +# `scripts/asvs/scorecard.py --prove-absences` shipped on 2026-08-07. It copies the tree to a scratch +# dir, asserts a named observable is green, applies a stated reintroduction, and requires the +# observable to go RED -- real mutation testing of a control, and the only check in the ASVS toolchain +# that proves a claim by EXECUTION rather than by grep. Measured on 2026-08-09 against vault +# `origin/main` (1a59e4a1) and engine `main`: +# +# absence claims on the record : 276 +# carrying `observable` : 0 +# carrying `mutation_path` : 0 +# `--prove-absences` invoked in CI : 0 references under .github/, in EITHER repo +# +# A mode nothing invokes cannot go red whatever is inside it, so those 276 green absence claims were +# exactly as strong the day after that merge as the day before. This file is the invocation. It is +# deliberately the FIRST half of a two-step: wire it while adoption is zero, harden it after. Making +# it fail on zero adoption would be correct and entirely inert, because nothing was running it. +# +# WHY HERE AND NOT IN THE VAULT'S asvs-scorecard.yml. That job is `timeout-minutes: 5`, has no install +# step, and its own comment states the stdlib-only constraint is deliberate "so this job cannot rot on +# a lockfile it does not own". Proving needs the full engine install plus pytest, which is what this +# repo already has. Adding it there would couple a seconds-long stdlib gate to this repo's lockfile. +# +# --------------------------------------------------------------------------------------------------- +# THE INPUT PROBLEM. DECIDED 2026-08-09: option (b) -- the scheduled pass runs in THE VAULT. +# +# The three options and their real costs are kept below because the decision is only readable if the +# rejected alternatives are, and because option (a) stays IMPLEMENTED-AND-OFF in this file: a +# dispatch-only run here is how the engine-side path is exercised on demand without a standing +# credential. What moved is the SCHEDULE, not the capability. +# +# The scorecard lives in THE VAULT, which is private; this repository is public. The vault's own +# workflow reads the engine freely -- `repository: MEFORORG/MessageFoundry` with the comment "public: +# no token needed" -- but the reverse direction has no free version. Every honest option costs +# something: +# +# (a) THIS REPO HOLDS A READ CREDENTIAL FOR THE VAULT. What the two knobs below implement, and it +# is OFF: neither `vars.ASVS_VAULT_REPO` nor `secrets.ASVS_VAULT_READ_TOKEN` exists today, and +# this workflow does not create them. It is off because the vault exists precisely so that a +# compromise of the public repo does not yield the security corpus, and a vault-read token in +# the public repo's secret store collapses that boundary to one credential. If it is ever +# switched on it MUST be a fine-grained, read-only, contents-scoped token for that one repo, and +# the sparse-checkout below keeps the materialised blast radius to the single scorecard file +# rather than the whole `docs/security` tree. That mitigates the checkout; it does not mitigate +# the token. +# +# (b) THE PROVER RUNS IN THE VAULT INSTEAD, as a NEW workflow beside `asvs-scorecard.yml` rather +# than inside it -- own job, own install, own timeout, so the stdlib-only constraint above is +# untouched. The vault already checks the engine out with no token at all, so this needs NO new +# credential in EITHER direction. Its cost is that the vault pays an install against a lockfile +# it does not own. That is a maintenance cost; (a) is a security-boundary cost. +# +# (c) A SELF-HOSTED RUNNER that already holds both checkouts. Cheapest operationally, and a +# self-hosted runner attached to a PUBLIC repo is its own well-known hazard. Owner's call. +# +# There is a SECOND half to this and it points the same way. The prover's problem lines name the cell +# and the control that would not prove -- a ranked list of the weakest controls on the record. This +# repo's run logs are world-readable, so `prove_report.py` suppresses those lines by default and +# prints only counts (`--detail` opts them back in, for a private log). So the public repo is the +# wrong host for the OUTPUT as well as the wrong holder of the INPUT, and neither of those is fixable +# by moving the environment, whereas (b)'s only cost IS the environment. +# +# ==> DECIDED: (b). The same two scripts run unchanged in the vault, pointed at a local scorecard +# and an `engine/` checkout. `prove_report.py` therefore ships HERE and is mirrored THERE, on +# the same ADR 0156 §7 footing as `scorecard.py` -- one tool, developed in the repo whose code +# it constrains, run in the repo that holds the data. The vault-side scheduled workflow is +# sequenced separately and is deliberately NOT built from this branch. +# +# --------------------------------------------------------------------------------------------------- +# WHY THE `prove` JOB IS DISPATCH-ONLY, AND WHY IT STILL EXITS 2 ON NO INPUT. +# +# Those are two different questions and conflating them is what a `schedule:` here would have done. +# +# ADVISORY APPLIES TO FINDINGS, NEVER TO THE INSTRUMENT. A claim that will not prove is reported and +# does not fail this job (see `vars.ASVS_PROVE_STRICT`). A run that could not obtain a scorecard +# scanned ZERO claims and is not evidence about any of them, so it exits non-zero -- the rule +# `scorecard.py` already states for its own loader ("Fail closed, never skip ... refusing to report a +# pass on a missing file"). That is unchanged and must stay unchanged: never make the no-input path +# green, which would restore exactly the "green check that never ran" state this whole exercise +# exists to end. +# +# But a job that fails closed on no input must not be SCHEDULED to obtain no input. With option (a) +# off, a nightly run here would be RED EVERY DAY BY CONSTRUCTION -- not reporting a finding, just +# re-announcing a decision already recorded in this file. A gate whose first act is to fail is a gate +# somebody switches off, and a disabled workflow is indistinguishable from a passing one at a glance. +# So the failing-closed behaviour stays and the cron goes to the repo that can actually feed it. +# +# The `selftest` job below keeps no cron either, and that is not an oversight: its harness also runs +# as `tests/test_asvs_prove_absences_wiring.py::test_selftest_all_limbs_pass` in the unfiltered +# `ci.yml` suite on every code PR and push, so a nightly re-run here would re-measure something +# already measured. The job exists for the paths-filtered case -- a change to the wiring itself. +# +# --------------------------------------------------------------------------------------------------- +# NOT A REQUIRED CHECK. Neither job context is in `.github/required-contexts.txt`, so nothing here can +# gate a merge or wedge auto-merge. `tests/test_asvs_prove_absences_wiring.py` pins that, the absent +# write scopes, and the advisory default. + +on: + # NO `schedule:`. The scheduled pass lives in the vault (see the decision above). Adding a cron back + # here re-creates a job that is red every day for a reason nobody can act on from this repository. + workflow_dispatch: + # The gate must be able to observe changes to ITSELF (the lesson written up at length above + # asvs-scorecard.yml's own path filter). On these events only the `selftest` job runs -- see its + # `if:` -- because the PR-time question is "does the wiring still work", which needs no credential, + # while "what does the record say" needs one and would paint every such PR red for a reason that has + # nothing to do with the PR. + pull_request: + paths: &wiring_paths + - ".github/workflows/asvs-prove-absences.yml" + - "scripts/asvs/**" + - "tests/test_asvs_prove_absences_wiring.py" + push: + branches: [main] + paths: *wiring_paths + +# Deny by default; each job grants only `contents: read`. Nothing here writes. +permissions: {} + +concurrency: + group: asvs-prove-absences-${{ github.ref }} + cancel-in-progress: false + +jobs: + selftest: + # THE FAIL-ON-PURPOSE GATE, and the reason anything below is believable. + # + # `prove_report.py selftest` builds fixture trees in a temp dir and drives the wiring through nine + # limbs that must each come out a specific way: a biting claim proves (L1), a non-biting claim is + # reported but not fatal in advisory mode (L2), the SAME claim fails under --strict (L3, so + # "advisory" is a choice rather than the only behaviour), a missing scorecard is an instrument + # failure and never a pass (L4), the census actually sees the claim (L5), the public-log detail + # suppression is attacked from both sides (L6), an unparseable prover summary is an instrument + # failure rather than a report of zeros (L7), and a prover that stopped iterating is caught by + # reconciliation (L8 -- its own summary line looks identical whether it walked 276 claims or 2). + # + # Every limb was confirmed to go red by injecting the matching defect and checking the defect + # landed on disk first; a mutation that never applied reads exactly like a pass. + name: prove-absences wiring selftest + runs-on: ubuntu-latest + timeout-minutes: 20 + permissions: + contents: read + steps: + - name: Check out the source + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.14" + + - name: Set up uv + uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 + with: + cache-dependency-glob: | + pyproject.toml + requirements.lock + + - name: Install the project + # `--constraint constraints.lock` for the same reason every other install here carries it: a + # bare `-e ".[dev]"` re-resolves from pyproject's `>=` floors and adopts whatever upstream + # published since. The prover spawns pytest, so `[dev]` is the minimum that makes it real. + run: uv pip install --system --constraint constraints.lock -e ".[dev]" + + - name: Prove the wiring can go red + run: python scripts/asvs/prove_report.py selftest + + prove: + # The real pass over the record. Advisory: findings are reported, never fatal (see + # `vars.ASVS_PROVE_STRICT`). An instrument failure IS fatal -- that distinction is the point. + name: prove absence claims (advisory) + needs: selftest + # ON DEMAND ONLY. On a PR there is no credential and nothing to prove; the wiring question is + # `selftest`'s and it already ran. There is no `schedule` arm to match either -- kept as an + # explicit event test rather than deleted, so that re-adding a cron above does NOT silently start + # running this job: someone would have to change this line too, and this line is next to the + # reason not to. + if: github.event_name == 'workflow_dispatch' + runs-on: ubuntu-latest + # Generous because the prover spawns a pytest run per provable claim, and that is what the budget + # is for. The tree is copied ONCE for the whole pass (save/apply/run/restore against one pristine + # copy, plus a rebuild if a run writes into it), so the copy cost no longer scales with adoption + # -- it used to be one copytree per claim at roughly 1.2s each. At today's adoption -- zero -- the + # whole pass is under a second, so this budget is for the future, not the present. + timeout-minutes: 60 + permissions: + contents: read + steps: + - name: Check out the engine + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # A subdirectory, not the workspace root, so `--root engine` bounds the prover's scratch + # copy to the engine tree and can never sweep the vault checkout beside it into a + # world-default temp dir. `scorecard.py`'s `_scratch_ignore` defends the same property from + # the other side; this is the layout that means it never has to. + path: engine + persist-credentials: false + + - name: Report where the scorecard is coming from + id: input + env: + # Hoisted into `env` rather than interpolated into the shell body (zizmor + # template-injection). Same shape the vault's ASVS job uses for its anchor SHA. + VAULT_REPO: ${{ vars.ASVS_VAULT_REPO }} + run: | + set -euo pipefail + if [ -n "${VAULT_REPO}" ]; then + echo "scorecard input : private vault repository, sparse checkout of the scorecard alone" + echo "mode=vault" >> "$GITHUB_OUTPUT" + else + echo "scorecard input : NONE" + echo "mode=none" >> "$GITHUB_OUTPUT" + echo "::error::ASVS prove-absences has NO SCORECARD INPUT, so this run scanned zero absence claims and is not evidence about any of them. The scorecard lives in the vault, which is private, while this repository is public; see the block at the top of .github/workflows/asvs-prove-absences.yml for the three options and the recommendation. Do not make this path green -- either configure an input or disable the workflow." + fi + + - name: Check out ONLY the scorecard from the vault + if: steps.input.outputs.mode == 'vault' + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: ${{ vars.ASVS_VAULT_REPO }} + token: ${{ secrets.ASVS_VAULT_READ_TOKEN }} + path: vault + persist-credentials: false + fetch-depth: 1 + # ONE FILE. `docs/security` is the maintainer-internal corpus -- remediation plans, the + # fails register, the risk-acceptance register. A cone-mode-off sparse checkout of the single + # path means a credential that can read all of it materialises none of the rest on a runner + # whose logs are public. + sparse-checkout: docs/security/asvs-scorecard.toml + sparse-checkout-cone-mode: false + + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.14" + + - name: Set up uv + uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0 + with: + cache-dependency-glob: | + engine/pyproject.toml + engine/requirements.lock + + - name: Install the project + working-directory: engine + run: uv pip install --system --constraint constraints.lock -e ".[dev]" + + - name: Prove the absence claims, and report what was scanned + working-directory: engine + env: + # Both default OFF. STRICT is the whole ratchet -- one variable, proved to bite by selftest + # limb L3 -- and it should stay off until adoption is non-zero, because a strict run over + # 276 unprovable claims fails on the absence of work rather than on a defect. DETAIL prints + # the per-claim problem lines and is ONLY appropriate where the run log is private. + STRICT: ${{ vars.ASVS_PROVE_STRICT }} + DETAIL: ${{ vars.ASVS_PROVE_DETAIL }} + SCORECARD: ${{ github.workspace }}/vault/docs/security/asvs-scorecard.toml + run: | + set -euo pipefail + # An ARRAY, not a string: an unquoted "${flags}" would rely on word splitting (shellcheck + # SC2086, and actionlint runs shellcheck over every run body). Explicit `if` blocks rather + # than `[ ... ] && flags=...` because under `bash -e` a false test at the end of an `&&` + # chain aborts the step -- which would turn "STRICT is off" into a failed job. + flags=() + if [ "${STRICT:-}" = "true" ]; then + flags+=(--strict) + fi + if [ "${DETAIL:-}" = "true" ]; then + flags+=(--detail) + fi + echo "flags: ${flags[*]:-none (advisory, counts only)}" + # --timeout is set BELOW the job's timeout-minutes on purpose. If the prover overruns, the + # script's own "did not finish within Ns" instrument failure is what a reader sees; a bare + # job kill truncates the log and looks like infrastructure rather than a measurement that + # did not complete. A gate that stops must say so in its own words. + # + # No pipe, so the exit code reaching the runner is this command's own (SDS-3.8). The three + # outcomes it can return are distinct and all three matter: 0 clean-or-advisory, + # 1 findings-under-strict, 2 the instrument could not measure. + python scripts/asvs/prove_report.py run \ + --scorecard "${SCORECARD}" \ + --root . \ + --timeout 2700 \ + "${flags[@]}" diff --git a/docs/ASVS-L2-PHASE0-CHANGES.md b/docs/ASVS-L2-PHASE0-CHANGES.md index 502a536a..a999a1c3 100644 --- a/docs/ASVS-L2-PHASE0-CHANGES.md +++ b/docs/ASVS-L2-PHASE0-CHANGES.md @@ -103,6 +103,7 @@ Update it whenever a crypto dependency, algorithm, or key source changes. | Config fingerprint ([ADR 0041](adr/0041-load-path-attestation-and-change-attribution.md)) | SHA-256 content digest of a loaded config bundle — path-relative Merkle fold over every loaded file (`*.py` incl `_*.py`, `connections.toml`, `codesets/*`, `environments/*.toml`); `hashlib` in `config/fingerprint.py` | Recorded in the `config_reload` audit detail (not stored as a secret) | Recomputed per reload/startup; binds reviewed-commit → loaded-bytes (integrity/attribution, not confidentiality) | | Engine wheel attestation ([ADR 0041](adr/0041-load-path-attestation-and-change-attribution.md) D3) | SHA-256 over each **loaded** first-party `messagefoundry` module file, compared to the installed wheel's `*.dist-info/RECORD` baseline (a base64 `sha256=` manifest already in the wheel); `hashlib` in `integrity.py` | Drift recorded in the hash-chained `startup_integrity` audit row (not a secret); RECORD baseline read from site-packages metadata | Recomputed at startup + on demand; in-place-tamper tripwire (integrity, not confidentiality). Alert-only by default; `[integrity].fail_closed_on_drift` refuses to start on drift; no-op on an editable install | | ASVS corpus pin ([ADR 0156](adr/0156-asvs-scorecard-as-data-a-derived-count-verified-evidence-anchors-and-a-fail-closed-drift-gate.md)) | SHA-256 over the **OWASP ASVS 5.0.0 corpus file**, recorded in `[scorecard].corpus_sha256` and recomputed on every verifier run; `hashlib` in `scripts/asvs/scorecard.py`. **Integrity of a build input, not a security control** — no secret, no key, no message authentication, and nothing user- or PHI-derived is hashed. It exists because the corpus was originally fetched from `master` (the bleeding-edge branch, where a rolling "latest" release republishes identical filenames) and matched the tagged `v5.0.0_release` asset only by luck; the digest is now recorded and checked rather than assumed, because ASVS requirement ids are **not stable across versions** (bare `1.2.5` is *Architecture* in 4.0.3 and *Encoding and Sanitization* in 5.0.0), so a corpus that moves silently re-points every id in the scorecard | Not a secret: the digest is committed alongside the corpus it pins | Recomputed on every scorecard verification; a mismatch fails the gate and forces re-verification before any verdict is trusted | +| ASVS scorecard revision identifier ([ADR 0156](adr/0156-asvs-scorecard-as-data-a-derived-count-verified-evidence-anchors-and-a-fail-closed-drift-gate.md)) | SHA-256 over the **ASVS scorecard file**, printed truncated to 16 hex characters by a `--prove-absences` run; `hashlib` in `scripts/asvs/prove_report.py`. Same class as the corpus pin above and **not a security control** for the same reasons — no secret, no key, no message authentication, nothing user- or PHI-derived. It differs only in what it covers: the record itself rather than a build input, and it is never compared against a declared value. It exists so a run states *which* revision of the record it read — two runs reporting different counts are otherwise indistinguishable from one run whose input moved underneath it | Not a secret: it is an identifier in a run log, and the scorecard it covers is private for unrelated reasons | Recomputed on every run; nothing is gated on it, so a change is information for a reader rather than a failure | | Outbound message signing (opt-in) | Detached JWS (RFC 7515) — RS256/PS256 (RSA) or ES256 (ECDSA P-256), SHA-256; `cryptography` in `transports/signing.py` (ASVS 4.1.5, [ADR 0018](adr/0018-per-message-signatures-accepted-risk.md)) | Operator-supplied PEM **private** signing key per connection (inline via `env()` or a PEM file path; encrypted-key passphrase via `env()`); the **public** key is shared with the partner out-of-band. **Usage scope:** this private key **only** signs this connection's outbound per-message JWS — a message-**authenticity/integrity** key in transit; it is never used for at-rest encryption or session/token material, and the partner holds only the matching **public** verification half | **OFF by default**; per-connection opt-in. `kid` carried in the JWS header so key rotation / a managed provider ([ADR 0019](adr/0019-pluggable-keyprovider-hsm-kms-vault.md)) slots in without a wire change | | DIRECT S/MIME (opt-in, [ADR 0085](adr/0085-direct-hisp-smime-connector.md)) | CMS **sign-then-encrypt** in `transports/direct.py` (core `cryptography` `serialization.pkcs7`): PKCS#7 signature over the body with a **SHA-256** digest, the public-key signature algorithm (RSA / ECDSA) following the loaded signing key type (not pinned to RSA), then a PKCS#7 **envelope** to the partner's recipient cert. The envelope content-encryption cipher is the **`cryptography` pkcs7 library default** — no algorithm is pinned in code | Sender **signing cert** + PEM **private key** (optional `signing_key_password`) and the per-partner **`recipient_cert`**, all operator-supplied files; the recipient cert is trust-verified at construction against an operator `trust_anchor` (one-level direct-issuance check); key/cert mismatch refused. **Usage scope:** the sender signing key signs the CMS body and the partner's `recipient_cert` encrypts the CMS envelope — this material protects the **confidentiality + authenticity of a DIRECT message to one partner in transit**; it is not an at-rest store key and encrypts nothing in the store | **OFF by default** — only when a DIRECT Connection is configured, and its HISP relay host is gated by the **opt-in** `[egress].allowed_direct` allow-list (empty by default = unrestricted; an unlisted host is refused only once the list is populated, or outright when `[security].block_unlisted_outbound` is set). Signing key + recipient certs rotate on the schedule below | | OIDC IdP JWKS verification keys (opt-in, [ADR 0142](adr/0142-federated-sso-oidc-authorization-code-pkce-relying-party-hybrid-ad-backed.md)) | **Public** verifying keys fetched from the IdP JWKS: **RS256/PS256** (RSA, ≥ 2048-bit floor) and **ES256/ES384** (EC P-256/P-384) — rebuilt from each JWK by `cryptography` in [`auth/oidc/jwks.py`](../messagefoundry/auth/oidc/jwks.py); the closed `SignatureAlgorithm` enum forecloses `alg:none` and RS256→HS256 confusion. Bounded, TTL-cached (`DEFAULT_JWKS_TTL_SECONDS`), a 512 KiB body cap, a global min-refetch floor (fetch-amplification bound), and a hard refusal of a duplicate `kid`; a key below the floor is skipped/refused, never merely warned. **Usage scope:** these are **public**, non-secret keys used **only** to verify the IdP's id-token signature at console login — they encrypt nothing and can protect no data; the engine holds no private half. | Fetched from the IdP JWKS URI over the CA-pinned no-redirect opener (row below); held process-local in `JwksCache`, never persisted, never logged | Refetched per TTL / on an unknown `kid` within the amplification bound; rolls when the IdP rotates its signing keys | diff --git a/docs/CI.md b/docs/CI.md index 5938d8fa..8729a4a5 100644 --- a/docs/CI.md +++ b/docs/CI.md @@ -24,6 +24,7 @@ claims move with it. | `zizmor.yml` | Lints the workflow files themselves for insecure patterns (template injection, over-broad tokens), and runs `actionlint` on the workflow syntax. Hard-fails, but **not a required check** — it is paths-filtered, so it does not report on a PR that touches no workflow, and requiring it would wedge every such PR. The `actionlint` pre-commit hook is the local half. | | `dast.yml` | Authenticated authorization sweep against a live loopback listener in front of a real engine. **Not a required check** — nightly / release-tag / manual dispatch only, with no `pull_request` trigger, so it never reports on a PR and cannot wedge one. It is NOT `continue-on-error`: it goes red on a finding. See [ADR 0155](adr/0155-dast-dynamic-security-testing-of-the-running-engine.md). | | `quality-advisory.yml` | Advisory quality measurement — complexity (ruff `C901`), duplication (`jscpd`), diff-coverage (`diff-cover`) and mutation testing (`mutmut`). **Every job is advisory and none is in branch protection.** See below for how each signal reaches a reviewer. | +| `asvs-prove-absences.yml` | Runs `scripts/asvs/scorecard.py --prove-absences`: applies each absence claim's stated reintroduction to a scratch tree and requires its named observable to go red. **Advisory and not in branch protection.** Two jobs. `selftest` runs on any PR touching the wiring, needs no credential, and is what stops the tool rotting in the repo that develops it. `prove` is **`workflow_dispatch` only** — the scheduled pass runs in the vault, the only repo holding the scorecard, per the 2026-08-09 location decision recorded in that workflow's own header block. A dispatch here still fails closed with exit 2 when no input is configured, because a run that scanned nothing must not report success; it is simply not *scheduled* to obtain nothing. `scripts/asvs/prove_report.py` ships here and `MIRRORED_TOOLS` in `tests/test_asvs_verifier_vault_contract.py` holds it to that list's **contract** — stdlib-only, so a vault copy would run on the bare interpreter there. That contract is in force *before* any mirror exists, deliberately, because the cheap moment to hold a tool to it is before it acquires a dependency. **Do not read that entry as evidence a vault copy exists: it does not.** The vault's mirror automation is scoped to `scorecard.py` alone, and `MIRRORED_TOOLS` asserts the stdlib property, never that a vault copy exists — so widening the vault's automation is the open half, tracked with the vault-side scheduled pass. | Several heavier legs (server-DB store tests, load/throughput, service-smoke, DICOM/FHIR breadth) run **nightly on a schedule** and/or only when a PR touches their paths, so an ordinary PR does not pay for diff --git a/scripts/asvs/prove_report.py b/scripts/asvs/prove_report.py new file mode 100644 index 00000000..2510d65d --- /dev/null +++ b/scripts/asvs/prove_report.py @@ -0,0 +1,629 @@ +# SPDX-License-Identifier: AGPL-3.0-or-later +# Copyright (C) 2026 MessageFoundry Organization and contributors +"""Run ``scorecard.py --prove-absences`` from CI and report WHAT IT SCANNED. + +``--prove-absences`` shipped on 2026-08-07 and, measured on 2026-08-09, ran in **no workflow in +either repo** (0 references under ``.github/`` in the engine and in the vault). A mode nothing +invokes cannot go red whatever is inside it, so the 276 absence claims on the record were exactly as +strong after that merge as before it. This module is the wiring, and it exists as a script rather +than as shell inside the workflow for three reasons, each of which was a defect somewhere first: + +1. **The prover's total is derived from the prover's own loop.** ``_run_prove_absences`` does lead + with ``saw N absence claim(s)``, but that ``N`` is ``findings.checked_absences`` -- a counter + incremented INSIDE the proving loop. It is the right denominator for the counters printed beside + it, which were measured on the same pass, and it is the wrong instrument for the question *did + this pass walk the whole record*: a prover that stopped after two claims prints ``saw 2``, which + is internally consistent and indistinguishable from a two-claim scorecard. A total that shrinks + along with the pass cannot detect the pass shrinking, so the check has to come from outside the + loop -- which is the next point. (The counters themselves still do not close to that total, and + deliberately: a static-screened claim may *also* raise a SUSPECT problem, and eight outcomes raise + a problem while incrementing no counter. See ``Findings.proved_absences`` for the identity.) +2. **The census must not share the prover's arithmetic.** :func:`census` reads the TOML with plain + ``tomllib`` and deliberately does **not** call ``scorecard.load_scorecard``. A total derived from + the loader would agree with the loader by construction -- the same "value generated from the thing + it validates" defect the ``mutation`` field exists to close, arriving through the fix. Two + independent counts are what make :func:`reconcile` able to say anything at all. +3. **Exit codes do not survive a shell pipeline.** Advisory-versus-blocking, and instrument-failure + versus finding, are three outcomes that must stay distinguishable; deciding that in ``run:`` YAML + is how ``$?``-after-a-pipe bugs are born (SDS-3.8). + +**Advisory applies to FINDINGS, never to the INSTRUMENT.** A claim that will not prove is reported +and, until adoption is non-zero, does not fail the job (:data:`EXIT_FINDINGS` is returned only under +``--strict``). A run that could not obtain a scorecard, could not run the prover, or could not +reconcile its own numbers returns :data:`EXIT_INSTRUMENT` unconditionally -- the rule +``scorecard.py`` already states for its own loader failure ("could not measure -- never 0, never +confused with clean"). + +**Per-claim detail is suppressed by default because this repo is public.** The prover's problem +lines name the cell and the control that would not prove, which is a ranked list of the weakest +controls on the record -- the same disclosure that got the "verdict-attributed anchor manifest" +rejected in the 2026-08-08 tracking-rework diagnosis. Counts are safe and are always printed; +``--detail`` opts the lines in, and is only appropriate where the run log is private. +""" + +from __future__ import annotations + +import argparse +import contextlib +import hashlib +import io +import os +import re +import subprocess # nosec B404 - fixed argv, no shell; see _invoke_prover +import sys +import tempfile +import tomllib +from dataclasses import dataclass +from pathlib import Path + +HERE = Path(__file__).resolve().parent +SCORECARD_PY = HERE / "scorecard.py" + +#: Clean: the instrument ran, and either found nothing or is in advisory mode. +EXIT_OK = 0 +#: Findings, under ``--strict`` only. The ratchet Stream C / the owner flips once adoption is real. +EXIT_FINDINGS = 1 +#: The instrument could not measure. NEVER suppressed by advisory mode, and never 0. +EXIT_INSTRUMENT = 2 + +#: Parses ``_run_prove_absences``'s one summary line. Coupled to that format ON PURPOSE and with a +#: hard failure on a miss: if that line is reformatted, this must go red rather than quietly report +#: zeros. An unparseable instrument is an instrument failure, not a clean run. +#: +#: `.*?` after the prefix spans the `saw N absence claim(s);` total that summary carries, which this +#: module deliberately does NOT read: the whole reason :func:`census` exists is that a loop-derived +#: total cannot check the loop (module docstring, point 1). Deliberately narrow tolerance -- it +#: absorbs a leading insertion without becoming loose enough to match arbitrary text, which selftest +#: limb L7 asserts by feeding it exactly that. +_SUMMARY_RE = re.compile( + r"prove-absences:.*?proved\s+(?P\d+)\s+by mutation;\s*" + r"(?P\d+)\s+static-screened;\s*" + r"(?P\d+)\s+skipped;\s*" + r"(?P\d+)\s+problem" +) + + +@dataclass(frozen=True) +class Census: + """The claim population, counted independently of the prover (see the module docstring).""" + + cells: int + cells_with_claims: int + claims: int + #: ``observable`` AND ``mutation_path``: the prover attempts a live execution proof. + provable: int + #: ``mutation_path`` only: the prover applies the coarse static backstop. A screen, not a proof. + static_only: int + #: No ``mutation_path``: the prover SKIPS it. This is the population the whole exercise is about. + unprovable: int + #: ``observable`` authored but ``mutation_path`` left empty. The prover's first branch keys on + #: ``mutation_path``, so such a claim is skipped SILENTLY and its observable never runs -- an + #: authoring slip that looks identical to a claim nobody has touched. Reported separately. + orphan_observable: int + #: Whitespace-only ``mutation_path``. Truthy to the loader (which does not strip) and empty to a + #: reader. Named so the two cannot disagree without someone being told. + blank_mutation_path: int + + +def census(scorecard: Path) -> Census: + """Count the absence-claim population straight from the TOML. + + Deliberately not ``scorecard.load_scorecard``: this number's whole job is to be checkable against + the prover's, which means it must not come from the prover's own loader. + """ + data = tomllib.loads(scorecard.read_text(encoding="utf-8")) + cells = data.get("cell", []) + claims = provable = static_only = unprovable = orphan = blank = 0 + cells_with_claims = 0 + for cell in cells: + absences = cell.get("absence", []) + if absences: + cells_with_claims += 1 + for a in absences: + claims += 1 + # `str(...)` with no strip, matching `load_scorecard` in scorecard.py exactly -- the + # classification has to agree with what the prover will actually branch on. Named, not + # cited by line: this comment previously pinned `scorecard.py:339-340`, which resolved in + # PR #304 and resolves to unrelated docstring prose here, because the port moved the + # loader. An anchor that has to be re-checked by hand decays the first time either file + # moves, and nothing in the repo checks a line citation inside a Python comment. + raw_path = str(a.get("mutation_path", "")) + raw_obs = str(a.get("observable", "")) + if raw_path and not raw_path.strip(): + blank += 1 + if not raw_path: + unprovable += 1 + if raw_obs: + orphan += 1 + elif raw_obs: + provable += 1 + else: + static_only += 1 + return Census( + cells=len(cells), + cells_with_claims=cells_with_claims, + claims=claims, + provable=provable, + static_only=static_only, + unprovable=unprovable, + orphan_observable=orphan, + blank_mutation_path=blank, + ) + + +@dataclass(frozen=True) +class ProverResult: + """What the prover subprocess reported.""" + + returncode: int + proved: int + screened: int + skipped: int + problems: int + problem_lines: tuple[str, ...] + stdout: str + stderr: str + + +def _invoke_prover( + scorecard: Path, root: Path, python: str, timeout: float +) -> tuple[int, str, str]: + """Run ``scorecard.py --prove-absences`` as a subprocess and hand back the raw streams.""" + proc = subprocess.run( # nosec B603 - fixed argv, no shell; paths come from argparse + [ + python, + str(SCORECARD_PY), + "--scorecard", + str(scorecard), + "--root", + str(root), + "--prove-absences", + ], + capture_output=True, + text=True, + check=False, + timeout=timeout, + ) + return proc.returncode, proc.stdout, proc.stderr + + +def parse_prover(returncode: int, stdout: str, stderr: str) -> ProverResult | None: + """Parse the prover's summary line. ``None`` means the output did not contain one at all, which + is an instrument failure and never a clean run.""" + m = _SUMMARY_RE.search(stdout) + if m is None: + return None + lines = tuple( + ln.strip()[5:].strip() for ln in stderr.splitlines() if ln.strip().startswith("FAIL") + ) + return ProverResult( + returncode=returncode, + proved=int(m.group("proved")), + screened=int(m.group("screened")), + skipped=int(m.group("skipped")), + problems=int(m.group("problems")), + problem_lines=lines, + stdout=stdout, + stderr=stderr, + ) + + +def reconcile(c: Census, r: ProverResult) -> list[str]: + """Check the prover's four counters against the independent census. + + This is the prover's own accounting identity (stated once, on ``scorecard.Findings``) evaluated + against a census the prover did not produce: the residual ``claims - (proved + screened + + skipped)`` is the number of claims that raised a problem and incremented no counter. Anything + that makes it negative, or larger than the reported problem count, means the two sides are not + describing the same population -- which is the only way to notice a prover that stopped + iterating, since its own summary looks identical either way. + + The bound is ``residual <= problems`` and NOT ``residual == problems``, deliberately: a + static-screened claim can also raise a SUSPECT problem, so a problem and a problem-only claim are + different things. Tightening this to equality would make an honest SUSPECT finding look like a + stalled prover. + """ + bad: list[str] = [] + if r.skipped != c.unprovable: + bad.append( + f"prover skipped {r.skipped} claims but {c.unprovable} carry no `mutation_path` -- the " + "prover skips exactly those, so these must be equal" + ) + if r.proved > c.provable: + bad.append( + f"prover proved {r.proved} claims but only {c.provable} carry both `observable` and " + "`mutation_path`" + ) + if r.screened > c.static_only: + bad.append( + f"prover static-screened {r.screened} claims but only {c.static_only} carry a " + "`mutation_path` without an `observable`" + ) + residual = c.claims - (r.proved + r.screened + r.skipped) + if residual < 0: + bad.append( + f"prover accounted for {r.proved + r.screened + r.skipped} outcomes across only " + f"{c.claims} claims -- more outcomes than claims" + ) + elif residual > r.problems: + bad.append( + f"{residual} claims produced no outcome counter but only {r.problems} problem(s) were " + "reported -- claims went missing between the scorecard and the prover" + ) + return bad + + +def _sha256(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _git_head(root: Path) -> str: + try: + proc = subprocess.run( # nosec B603 B607 - fixed argv, no shell + ["git", "-C", str(root), "rev-parse", "HEAD"], + capture_output=True, + text=True, + check=False, + timeout=30, + ) + except (OSError, subprocess.SubprocessError): + return "unknown" + return proc.stdout.strip() or "unknown" + + +def _emit_summary(lines: list[str]) -> None: + """Append to the GitHub job summary when running under Actions; a no-op elsewhere.""" + dest = os.environ.get("GITHUB_STEP_SUMMARY") + if not dest: + return + with open(dest, "a", encoding="utf-8") as fh: + fh.write("\n".join(lines) + "\n") + + +def _report( + scorecard: Path, root: Path, c: Census, r: ProverResult, *, detail: bool, strict: bool +) -> None: + """Print the census, the prover's own counters, and the adoption rate. + + This block is the whole point of the module: a run that scanned nothing and a run that scanned + everything and found nothing must not print the same thing. + """ + pct = (100.0 * (c.provable + c.static_only) / c.claims) if c.claims else 0.0 + mode = "STRICT (findings fail the job)" if strict else "ADVISORY (findings report only)" + out = [ + "ASVS prove-absences", + f" mode : {mode}", + f" scorecard : {scorecard} (sha256 {_sha256(scorecard)[:16]})", + f" root : {root} (HEAD {_git_head(root)})", + f" cells : {c.cells}", + f" cells with claims : {c.cells_with_claims}", + f" absence claims SEEN : {c.claims}", + f" live-provable : {c.provable} (observable + mutation_path)", + f" static-screen only : {c.static_only} (mutation_path, no observable)", + f" NOT provable : {c.unprovable} (no mutation_path -- the prover skips these)", + f" orphan observable : {c.orphan_observable} (observable authored, mutation_path empty " + "-- SILENTLY skipped)", + f" blank mutation_path: {c.blank_mutation_path} (whitespace-only)", + f" prover reported : proved {r.proved} / static-screened {r.screened} / " + f"skipped {r.skipped} / problems {r.problems} (exit {r.returncode})", + f" reconciliation : {'OK' if not reconcile(c, r) else 'MISMATCH'}", + f" ADOPTION : {c.provable + c.static_only} of {c.claims} claims " + f"({pct:.1f}%) can be proved or screened at all", + ] + print("\n".join(out)) + + if r.problems: + if detail: + for line in r.problem_lines: + print(f" FAIL {line}", file=sys.stderr) + else: + print( + f" {r.problems} problem line(s) SUPPRESSED -- they name the cell and the control " + "that would not prove, and this repo's run logs are public. Re-run with --detail " + "where the log is private.", + file=sys.stderr, + ) + + _emit_summary( + [ + "### ASVS `--prove-absences`", + "", + f"Mode: **{mode.split(' (')[0]}**. Scorecard `{_sha256(scorecard)[:16]}`, " + f"root HEAD `{_git_head(root)}`.", + "", + "| what | n |", + "|---|---:|", + f"| absence claims seen | {c.claims} |", + f"| live-provable (observable + mutation_path) | {c.provable} |", + f"| static-screen only (mutation_path only) | {c.static_only} |", + f"| not provable (no mutation_path -- skipped) | {c.unprovable} |", + f"| orphan observable (silently skipped) | {c.orphan_observable} |", + f"| proved by mutation | {r.proved} |", + f"| static-screened | {r.screened} |", + f"| problems | {r.problems} |", + "", + f"**Adoption: {c.provable + c.static_only} of {c.claims} ({pct:.1f}%).** A claim with no " + "`observable` is *not yet proven by execution* -- never *proven vacuous*.", + ] + ) + + +def run(args: argparse.Namespace) -> int: + scorecard = Path(args.scorecard) + root = Path(args.root) + if not scorecard.is_file(): + # Fail closed. This is the limb that fires when the vault scorecard could not be obtained, + # and it must never be mistaken for "nothing to report". + print( + f"error: scorecard not found at {scorecard} -- refusing to report a pass on a run that " + "scanned nothing", + file=sys.stderr, + ) + _emit_summary( + [ + "### ASVS `--prove-absences` DID NOT RUN", + "", + "No scorecard input was available, so **zero** absence claims were scanned. This " + "run is not evidence about any claim.", + ] + ) + return EXIT_INSTRUMENT + if not root.is_dir(): + print(f"error: --root {root} is not a directory", file=sys.stderr) + return EXIT_INSTRUMENT + + try: + c = census(scorecard) + except (tomllib.TOMLDecodeError, OSError) as exc: + print(f"error: could not read the scorecard: {exc}", file=sys.stderr) + return EXIT_INSTRUMENT + + try: + rc, stdout, stderr = _invoke_prover(scorecard, root, args.python, args.timeout) + except subprocess.TimeoutExpired: + print(f"error: the prover did not finish within {args.timeout}s", file=sys.stderr) + return EXIT_INSTRUMENT + + if rc == EXIT_INSTRUMENT: + print(f"error: the prover could not measure (exit 2):\n{stderr}", file=sys.stderr) + return EXIT_INSTRUMENT + + r = parse_prover(rc, stdout, stderr) + if r is None: + print( + "error: the prover printed no summary line -- its output could not be parsed, so this " + f"run measured nothing checkable. stdout was:\n{stdout}", + file=sys.stderr, + ) + return EXIT_INSTRUMENT + + _report(scorecard, root, c, r, detail=args.detail, strict=args.strict) + + mismatches = reconcile(c, r) + if mismatches: + for m in mismatches: + print(f" RECONCILE {m}", file=sys.stderr) + return EXIT_INSTRUMENT + + if r.problems and args.strict: + return EXIT_FINDINGS + return EXIT_OK + + +# --- the fail-on-purpose harness ------------------------------------------------------------------ +# +# Everything above is a report, and a report is the easiest thing in the world to have quietly stop +# working. These limbs make the wiring go RED on purpose and assert that it did -- including the +# advisory/strict split and the detail suppression, both of which are controls in their own right and +# neither of which is exercised by tests/test_asvs_scorecard.py (those test `prove_absences` and the +# `main` CLI, not this module's policy). Run it locally with `python scripts/asvs/prove_report.py +# selftest`; the workflow runs it before it believes anything the report says. + +_SCANNER = "def scan(p):\n return 'clean'\n" +_OBS_TEST = "from scanner import scan\n\n\ndef test_clean():\n assert scan('x') == 'clean'\n" +_MUTATION = 'def scan(p): return "infected"' + + +def _fixture(tree: Path, mutation_path: str, observable: str) -> Path: + """A two-file tree plus a one-claim scorecard. Modelled on the fixtures in + tests/test_asvs_scorecard.py so the shape is one the prover is already known to handle.""" + tree.mkdir(parents=True, exist_ok=True) + (tree / "scanner.py").write_text(_SCANNER, encoding="utf-8") + (tree / "unrelated.py").write_text("VALUE = 1\n", encoding="utf-8") + (tree / "test_scanner.py").write_text(_OBS_TEST, encoding="utf-8") + sc = tree.parent / f"{tree.name}-scorecard.toml" + # TOML LITERAL strings (single quotes) throughout: `_MUTATION` contains double quotes, and a + # basic-string fixture silently produced an unparseable scorecard whose only symptom was the + # instrument limb firing on every case -- L1-L3 red, L4 green, which reads like a broken prover + # rather than a broken fixture. Caught by running it; kept as a comment so it stays caught. + sc.write_text( + "[[cell]]\n" + "id = '1.1.1'\n" + "level = 1\n" + "verdict = 'fail'\n" + "residual = 'selftest fixture'\n" + "[[cell.absence]]\n" + "pattern = 'irrelevant-to-proving'\n" + "positive_control = 'irrelevant-to-proving'\n" + f"mutation = '{_MUTATION}'\n" + f"mutation_path = '{mutation_path}'\n" + f"observable = '{observable}'\n", + encoding="utf-8", + ) + return sc + + +def _limb(name: str, got: int, want: int, detail: str = "") -> bool: + ok = got == want + print(f" [{'PASS' if ok else 'FAIL'}] {name}: exit {got} (wanted {want}) {detail}") + return ok + + +def selftest(args: argparse.Namespace) -> int: + """Prove the wiring can go red, and that each outcome is DISTINCT from the others.""" + del args + ok = True + with tempfile.TemporaryDirectory(prefix="asvs_prove_selftest_") as td: + base = Path(td) + + # L1 -- the mutation shadows `scan`, so the observable goes red and the claim BITES. + biting = _fixture(base / "biting", "scanner.py", "test_scanner.py::test_clean") + ns = argparse.Namespace( + scorecard=str(biting), + root=str(base / "biting"), + python=sys.executable, + timeout=300.0, + detail=False, + strict=False, + ) + ok &= _limb("L1 biting claim, advisory", run(ns), EXIT_OK, "-- proved by mutation") + + # L2 -- the mutation lands in a file the observable never imports, so it reddens nothing. + # ADVISORY: reported, exit 0. This is the limb that proves advisory is a real state. + nonbiting = _fixture(base / "inert", "unrelated.py", "test_scanner.py::test_clean") + ns = argparse.Namespace( + scorecard=str(nonbiting), + root=str(base / "inert"), + python=sys.executable, + timeout=300.0, + detail=False, + strict=False, + ) + ok &= _limb("L2 non-biting claim, advisory", run(ns), EXIT_OK, "-- reported, not fatal") + + # L3 -- the SAME fixture under --strict must fail. One flag is the whole ratchet, and if it + # does not bite then "advisory" was never a choice, it was the only behaviour. + ns = argparse.Namespace( + scorecard=str(nonbiting), + root=str(base / "inert"), + python=sys.executable, + timeout=300.0, + detail=False, + strict=True, + ) + ok &= _limb("L3 non-biting claim, strict", run(ns), EXIT_FINDINGS, "-- the ratchet bites") + + # L4 -- no scorecard at all: the no-input limb. Must be 2 (instrument), never 0 and never 1; + # advisory mode must NOT be able to turn a run that scanned nothing into a pass. + ns = argparse.Namespace( + scorecard=str(base / "does-not-exist.toml"), + root=str(base / "biting"), + python=sys.executable, + timeout=300.0, + detail=False, + strict=False, + ) + ok &= _limb("L4 missing scorecard, advisory", run(ns), EXIT_INSTRUMENT, "-- fails closed") + + # L5 -- the census must SEE the claim. A report that says 0 claims when there is 1 is the + # exact failure this module exists to make impossible, and it would pass L1-L4 unnoticed. + c = census(biting) + seen_ok = (c.claims, c.provable, c.unprovable) == (1, 1, 0) + print(f" [{'PASS' if seen_ok else 'FAIL'}] L5 census sees the claim: {c}") + ok &= seen_ok + + # L6 -- the disclosure control, attacked rather than assumed. Without --detail the problem + # text must NOT reach the log; with --detail it must. A suppression nobody tried to defeat is + # a claim, not a control. + for detail, want in ((False, False), (True, True)): + r = _invoke_prover(nonbiting, base / "inert", sys.executable, 300.0) + parsed = parse_prover(*r) + assert parsed is not None, "L6 could not parse the prover output" + buf_out, buf_err = io.StringIO(), io.StringIO() + with contextlib.redirect_stdout(buf_out), contextlib.redirect_stderr(buf_err): + _report( + nonbiting, + base / "inert", + census(nonbiting), + parsed, + detail=detail, + strict=False, + ) + leaked = "UNPROVEN" in (buf_out.getvalue() + buf_err.getvalue()) + limb_ok = leaked == want + print( + f" [{'PASS' if limb_ok else 'FAIL'}] L6 detail={detail}: " + f"claim text {'present' if leaked else 'suppressed'} (wanted " + f"{'present' if want else 'suppressed'})" + ) + ok &= limb_ok + + # L7 -- a prover whose summary line stopped matching must be an INSTRUMENT failure, not a + # report of zeros. This is the coupling to `_run_prove_absences`'s print format, and it is + # the one that decays silently: reformat that line and, without this, every future run would + # report "0 claims, 0 problems" and read exactly like a healthy record. + unparseable = parse_prover(0, "prove-absences ran, trust me", "") is None + print( + f" [{'PASS' if unparseable else 'FAIL'}] L7 unparseable summary -> instrument failure" + ) + ok &= unparseable + + # L8 -- reconciliation must catch a prover that stopped iterating. Its own summary line looks + # identical whether it walked 276 claims or 2, so this is the ONLY thing standing between a + # half-run prover and a green report. Attacked directly: 276 claims, 2 skipped. + stalled = reconcile( + Census( + cells=345, + cells_with_claims=209, + claims=276, + provable=0, + static_only=0, + unprovable=276, + orphan_observable=0, + blank_mutation_path=0, + ), + ProverResult( + returncode=0, + proved=0, + screened=0, + skipped=2, + problems=0, + problem_lines=(), + stdout="", + stderr="", + ), + ) + caught = len(stalled) > 0 + print( + f" [{'PASS' if caught else 'FAIL'}] L8 stalled prover caught: {len(stalled)} mismatch(es)" + ) + ok &= caught + + print(f"selftest: {'ALL LIMBS PASS' if ok else 'FAILED'}") + return EXIT_OK if ok else EXIT_FINDINGS + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + sub = ap.add_subparsers(dest="cmd", required=True) + + r = sub.add_parser("run", help="census the claims, run the prover, reconcile, report") + r.add_argument("--scorecard", required=True) + r.add_argument("--root", default=".", help="tree the mutations are applied to (the engine)") + r.add_argument("--python", default=sys.executable) + r.add_argument("--timeout", type=float, default=3600.0) + r.add_argument( + "--detail", + action="store_true", + help="print the per-claim problem lines. ONLY where the run log is private -- they rank the " + "weakest controls on the record.", + ) + r.add_argument( + "--strict", + action="store_true", + help="findings fail the job. Off while adoption is zero; this is the whole ratchet.", + ) + r.set_defaults(func=run) + + s = sub.add_parser("selftest", help="make the wiring go red on purpose and assert that it did") + s.set_defaults(func=selftest) + + args = ap.parse_args(argv) + result: int = args.func(args) + return result + + +if __name__ == "__main__": # pragma: no cover + raise SystemExit(main()) diff --git a/scripts/asvs/scorecard.py b/scripts/asvs/scorecard.py index c73edafe..46c2eedb 100644 --- a/scripts/asvs/scorecard.py +++ b/scripts/asvs/scorecard.py @@ -321,16 +321,27 @@ class Findings: #: ``mutation_path`` (nothing to apply). #: #: **These three do NOT sum to the population, and that is why ``checked_absences`` above must be - #: printed beside them.** FIVE outcomes raise a problem and increment no counter at all — a - #: ``mutation_path`` that escapes the tree, one that is not a file, a baseline that is not green, - #: an UNPROVEN mutated-green, and a mutated run that errored rather than failed. The arithmetic - #: that closes is:: + #: printed beside them.** The arithmetic that closes is:: #: #: checked_absences - proved_absences - static_screened - skipped_absences #: == claims that ended in a problem-only branch #: - #: and note that ``len(problems)`` is NOT that number: a SUSPECT finding rides along with a claim - #: already counted in ``static_screened``, so problems and claims are different populations. + #: A PROBLEM-ONLY BRANCH raises a problem and increments no counter at all. There are EIGHT of + #: them today, in the order :func:`_prove_one` can reach them — a ``mutation_path`` that escapes + #: the tree, one that is not a file, a mutation that is not valid Python, a mutation that + #: redefines a symbol with a signature the target does not declare (those two are + #: :func:`_screen_mutation`, and the reason a refusal is problem-only rather than counted is + #: argued there), a baseline that is not green, a scratch target that did not restore after the + #: mutated run, an UNPROVEN mutated-green, and a mutated run that errored rather than failed. + #: + #: **The identity does not depend on that eight.** It holds for any number of problem-only + #: branches, which is what makes it safe to add one; the count is an aid to the reader and the + #: only thing a new branch falsifies. It is asserted by + #: ``test_prove_absences_counters_close_against_the_population``, which computes BOTH sides from a + #: real run rather than checking the left side against a typed-in constant. + #: + #: Note that ``len(problems)`` is NOT the right-hand side: a SUSPECT finding rides along with a + #: claim already counted in ``static_screened``, so problems and claims are different populations. proved_absences: int = 0 static_screened: int = 0 skipped_absences: int = 0 @@ -1259,12 +1270,163 @@ def _landing_swallows(source: str) -> bool: ) +# --- pre-flight screens on the MUTATION itself ---------------------------------------------------- +# +# The proving loop counts ``mutated == 1`` as "the control bit". That is only sound if the observable +# went red because of the SEMANTIC change the claim describes. Application is append-based, so a +# reintroduction breaks the target by REDEFINITION SHADOWING -- and two mutations that shadow nothing +# semantic still redden the observable, at exit 1, indistinguishably from a surgical proof: +# +# * WRONG ARITY. ``def scan(p)`` reintroduced as ``def scan()`` raises ``TypeError`` at every call +# site. Every test touching it fails, exit 1, counted as PROVED. The claim proved that calling a +# function with the wrong number of arguments breaks it -- which is true of every function in the +# repository and evidence about no control at all. This is the "wrecking ball that reads as a +# surgical ablation" case, and it is not hypothetical here: the mutation is authored in a TOML +# file in a DIFFERENT REPOSITORY from the signature it copies, with nothing keeping the two in +# step (8 signature edits across the anchored surface in 149 commits). +# * A MUTATION THAT DOES NOT PARSE. Appending invalid Python breaks import of the target. When the +# observable imports it at module scope pytest reports a COLLECTION error (exit 2) and the +# existing fail-closed branch catches it -- but an import inside the test body surfaces as an +# ordinary test failure at exit 1, and is counted as a proof. +# +# NOT IMPLEMENTED, DELIBERATELY: a blanket refusal of ``raise`` in a mutation. That rule belongs to a +# schema of typed mutation kinds where an "ablate" limb weakens a control and must never throw. This +# module has no kinds -- every mutation is a REINTRODUCTION -- and a reintroduction that raises is an +# anticipated, legitimate shape: the static backstop above exists precisely to flag one landing in a +# swallowing handler, with two tests pinning that behaviour. Banning `raise` would delete the case the +# backstop was written for. + + +def _toplevel(source: str) -> tuple[dict[str, ast.arguments], set[str]] | None: + """Top-level function signatures by name, plus every top-level name bound. ``None`` if `source` + does not parse. + + TOP LEVEL ONLY, and that is the point rather than a simplification: the mutation is APPENDED to + the module, so it can only shadow a module-level binding. A method inside a class body is not + reachable by this mechanism and must not be compared against. + """ + try: + tree = ast.parse(source) + except SyntaxError: + return None + sigs: dict[str, ast.arguments] = {} + bound: set[str] = set() + for node in tree.body: + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)): + sigs[node.name] = node.args + bound.add(node.name) + elif isinstance(node, ast.ClassDef): + bound.add(node.name) + elif isinstance(node, ast.Assign): + bound.update(t.id for t in node.targets if isinstance(t, ast.Name)) + elif isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name): + bound.add(node.target.id) + return sigs, bound + + +def _signature(args: ast.arguments) -> tuple[object, ...]: + """The call-compatible shape of a signature: what a CALL SITE can observe. + + Parameter NAMES are included, not just counts -- a rename breaks every keyword call, which is the + same wrecking-ball failure as a wrong count and is invisible to an arity-only comparison. Default + values are compared by COUNT and never by value: a mutation legitimately changes what a default + IS (that can be the whole reintroduction), while changing how many there are moves the arity. + """ + return ( + tuple(p.arg for p in args.posonlyargs), + tuple(p.arg for p in args.args), + tuple(p.arg for p in args.kwonlyargs), + args.vararg.arg if args.vararg else None, + args.kwarg.arg if args.kwarg else None, + len(args.defaults), + sum(1 for d in args.kw_defaults if d is not None), + ) + + +def _screen_mutation(a: Absence, cell_id: str, target: Path, findings: Findings) -> bool: + """Refuse a mutation that would redden the observable for a reason other than the change it + claims. ``False`` means refused, and a PROVE-ERROR has been recorded. + + Runs on the live-proof AND the static-screen path: a mutation whose signature does not match the + symbol it shadows is a defective claim whether or not anyone executes it. + + A REFUSAL IS PROBLEM-ONLY: it appends a PROVE-ERROR and increments NO counter, which is why the + accounting identity on :class:`Findings` still closes. That was a choice between three options and + the other two are worse. Folding a refusal into ``static_screened`` would be a lie about kind: + that counter says "this claim's evidence is a screen rather than a proof", and a refused claim + carries no evidence at all -- inflating it would make the instrument report more coverage than it + has, which is the exact overstatement this module keeps correcting. Giving refusals their own + counter would be arbitrary: every one of the problem-only branches has an equal claim to one, and + the reason none of them has one is that the PROBLEM is the record. So a refusal joins them. + """ + if target.suffix != ".py": + return True # nothing to parse; the append is opaque text by design + mutation = _toplevel(a.mutation) + if mutation is None: + findings.problems.append( + f"{cell_id}: absence claim PROVE-ERROR — the mutation is not valid Python, so applying it " + f"breaks import of {a.mutation_path} rather than reintroducing anything. An observable " + "that imports inside a test body reddens at exit 1 and would be counted as a proof" + ) + return False + source = _toplevel(target.read_text(encoding="utf-8", errors="replace")) + if source is None: + # The TARGET does not parse. Not this claim's defect, and not something to refuse it over -- + # leave it to execution, where the baseline will already be red and fail closed. + return True + mut_sigs, _ = mutation + src_sigs, _ = source + for name, margs in mut_sigs.items(): + real = src_sigs.get(name) + if real is None: + continue # shadows no module-level function of that name; nothing to compare + if _signature(margs) != _signature(real): + findings.problems.append( + f"{cell_id}: absence claim PROVE-ERROR — the mutation redefines {name}() with a " + f"different signature than {a.mutation_path} declares " + f"({_signature(margs)} vs {_signature(real)}). Applied, that raises TypeError at " + "every call site, so the observable would go red for the arity and not for the " + "reintroduction — a wrecking ball wearing the shape of a surgical proof. Copy the " + "live signature; the mutation lives in a different repository from the symbol it " + "shadows, so nothing else keeps the two in step" + ) + return False + return True + + +def _inventory(tree: Path) -> dict[str, int]: + """Relative path -> size for every file in the scratch tree, skipping derived bytecode. + + The pristine copy is reused across claims, so something has to notice if a pytest run WROTE into + it -- residue from claim N would otherwise be attributed to claim N+1's mutation. Names and sizes + are a stat-only sweep, roughly two orders of magnitude cheaper than the copy it replaces. + Honest residual: it cannot see a same-length in-place rewrite. + """ + out: dict[str, int] = {} + for p in tree.rglob("*"): + if p.is_file() and "__pycache__" not in p.parts: + out[str(p.relative_to(tree))] = p.stat().st_size + return out + + +def _drop_pycache(target: Path) -> None: + """Remove the bytecode cache beside a restored file. + + Belt and braces. CPython invalidates a ``.pyc`` on either source mtime or size, and a restore + changes both relative to the mutated run, so this should never be load-bearing -- but the cost of + being wrong is a stale module silently serving a mutation that was already reverted, which is a + false proof, and the fix is one directory removal. + """ + shutil.rmtree(target.parent / "__pycache__", ignore_errors=True) + + def _prove_one( a: Absence, cell_id: str, root: Path, - scratch_dir: Path, + scratch: Path, findings: Findings, + baselines: dict[str, int], *, python: str, timeout: float, @@ -1291,9 +1453,18 @@ def _prove_one( ) return + if not _screen_mutation(a, cell_id, target_in_root, findings): + return + if a.observable: - scratch = _copy_scratch(root, scratch_dir) - baseline = _run_node(scratch, a.observable, python, timeout) + # Baselines are CACHED BY NODE ID. The baseline is a property of the pristine tree and the + # node, and the tree is pristine by construction at this point -- re-running it per claim + # re-measured a constant, at one pytest subprocess each. Two claims naming the same observable + # now pay for one baseline. + baseline = baselines.get(a.observable) + if baseline is None: + baseline = _run_node(scratch, a.observable, python, timeout) + baselines[a.observable] = baseline if baseline != 0: # An already-red or uncollectable observable cannot attribute its red to the mutation. findings.problems.append( @@ -1302,8 +1473,26 @@ def _prove_one( "be attributed to the mutation" ) return - _apply_mutation(scratch, a.mutation_path, a.mutation) - mutated = _run_node(scratch, a.observable, python, timeout) + # SAVE / APPLY / RUN / RESTORE against ONE pristine copy, rather than a fresh copytree per + # claim (measured at roughly 1.2s a copy). `finally` so a crash in the run cannot leave the + # shared tree mutated -- that would silently poison every later claim, which is the hazard the + # per-claim copy was buying protection from and the reason the restore is verified below. + target_in_scratch = scratch / a.mutation_path + original = target_in_scratch.read_bytes() + try: + _apply_mutation(scratch, a.mutation_path, a.mutation) + mutated = _run_node(scratch, a.observable, python, timeout) + finally: + target_in_scratch.write_bytes(original) + _drop_pycache(target_in_scratch) + if target_in_scratch.read_bytes() != original: + # Asserted, not assumed. A restore that silently did not happen turns every subsequent + # claim's result into a fact about the previous claim's mutation. + findings.problems.append( + f"{cell_id}: absence claim PROVE-ERROR — the scratch copy of {a.mutation_path} did " + "not restore after the mutated run, so no later claim in this pass is attributable" + ) + return if mutated == 1: findings.proved_absences += 1 # live proof: the control bit elif mutated == 0: @@ -1350,12 +1539,18 @@ def prove_absences( """ findings = Findings() resolved_root = root.resolve() + #: Baseline exit code per observable node id. See the caching note in :func:`_prove_one`. + baselines: dict[str, int] = {} with tempfile.TemporaryDirectory(prefix="asvs_prove_") as td_base: base = Path(td_base) - i = 0 + # ONE pristine copy for the whole pass. It was one per claim, which re-copied the entire tree + # to apply a few lines and then threw it away -- at 1.2s a copy that is pure overhead + # proportional to adoption, and adoption is the thing this mode exists to grow. + generation = 0 + scratch = _copy_scratch(resolved_root, base / f"tree_{generation}") + inventory = _inventory(scratch) for c in cells: for a in c.absence: - i += 1 # The POPULATION, recorded before any outcome branch, so it is right whichever branch # this claim takes. Without it the outcome counters float free of what was scanned. findings.checked_absences += 1 @@ -1363,11 +1558,28 @@ def prove_absences( a, c.id, resolved_root, - base / f"scratch_{i}", + scratch, findings, + baselines, python=python, timeout=timeout, ) + # Reusing one tree is only sound while the tree stays pristine. A test that writes + # into it leaves residue that the NEXT claim's mutated run would be blamed for, so the + # reuse is CHECKED rather than assumed: on any change beyond the file just restored, + # rebuild and say so. Cached baselines are dropped with it -- they were measured + # against a tree that no longer exists. + now = _inventory(scratch) + if now != inventory: + generation += 1 + scratch = _copy_scratch(resolved_root, base / f"tree_{generation}") + inventory = _inventory(scratch) + baselines.clear() + findings.advisories.append( + f"{c.id}: the scratch tree was written to during this claim's run, so it was " + "rebuilt and cached baselines were dropped. The claim's own result stands; " + "an observable that writes into the tree it is measuring is worth a look" + ) return findings @@ -1919,7 +2131,7 @@ def _run_prove_absences(scorecard: Path, root: Path) -> int: findings = prove_absences(cells, root) # `saw N` is the denominator, and it comes FIRST because the parts are unreadable without it: a # run over 276 claims and a run over zero otherwise print counter sets that look equally - # plausible. The three outcome counters deliberately do not sum to it -- five problem-only + # plausible. The three outcome counters deliberately do not sum to it -- eight problem-only # outcomes increment nothing -- so the remainder is derivable and the gap is the point rather # than a rounding error. See `Findings.proved_absences` for the closing arithmetic. print( @@ -1928,6 +2140,12 @@ def _run_prove_absences(scorecard: Path, root: Path) -> int: f"{findings.static_screened} static-screened; {findings.skipped_absences} skipped; " f"{len(findings.problems)} problem(s)" ) + # Advisories are reported, never fatal -- but they have to be REPORTED. A scratch tree rebuilt + # mid-pass is the only signal that an observable wrote into the tree it was measuring, and it + # would otherwise be visible nowhere at all: `ok` ignores advisories and the summary counts them + # in nothing. + for a in findings.advisories: + print(f" NOTE {a}", file=sys.stderr) for p in findings.problems: print(f" FAIL {p}", file=sys.stderr) return 0 if findings.ok else 1 diff --git a/scripts/security/crypto_inventory_check.py b/scripts/security/crypto_inventory_check.py index 9dcb94fe..e93df250 100644 --- a/scripts/security/crypto_inventory_check.py +++ b/scripts/security/crypto_inventory_check.py @@ -364,6 +364,13 @@ # recorded and recomputed rather than assumed. Non-cryptographic alternatives were rejected only # because SHA-256 is already the tree's convention for file pinning. "scripts/asvs/scorecard.py": frozenset({"hashlib"}), + # Same class as the line above, registered for the same reason: SHA-256 over the SCORECARD FILE, + # printed truncated so a run states WHICH revision of the record it read. Two runs reporting + # different counts are otherwise indistinguishable from one run whose input changed underneath + # it. No secret, no key, no message authentication, nothing user- or PHI-derived — the digest is + # an identifier in a log line. It covers the record itself rather than a build input, which is + # the only way it differs from the entry above. + "scripts/asvs/prove_report.py": frozenset({"hashlib"}), "scripts/security/dast_target.py": frozenset({"secrets"}), # BACKLOG #1220: SHA-256 over the DISCOVERED engine/console seam surface, truncated to 16 hex # characters, to give ENGINE_UI_SEAM an identity nobody chooses by hand. A CHANGE DETECTOR, not a diff --git a/tests/test_asvs_prove_absences_wiring.py b/tests/test_asvs_prove_absences_wiring.py new file mode 100644 index 00000000..46538dde --- /dev/null +++ b/tests/test_asvs_prove_absences_wiring.py @@ -0,0 +1,217 @@ +# SPDX-License-Identifier: AGPL-3.0-or-later +# Copyright (C) 2026 MessageFoundry Organization and contributors +"""Pin the guarantees of .github/workflows/asvs-prove-absences.yml and scripts/asvs/prove_report.py. + +`--prove-absences` shipped on 2026-08-07 and ran in no workflow in either repo until this wiring +landed. The wiring's own safety properties are strings in a YAML file plus policy in one script: +one edit can grant a write scope, turn the no-input path green, or make advisory the only behaviour, +and nothing else in the repo would notice. Same reason +`tests/test_quality_advisory_invariants.py` exists for its workflow. + +Two claims are pinned here that are easy to state and easy to lose: + +* **Advisory applies to FINDINGS, never to the INSTRUMENT.** A claim that will not prove is reported; + a run that could not obtain a scorecard exits non-zero. If those two ever collapse into one + outcome, a run that scanned nothing reads exactly like a clean record -- the failure mode the whole + ASVS proving exercise exists to end. +* **Per-claim problem lines stay out of a public run log by default.** They name the cell and the + control that would not prove, which is a ranked list of the weakest controls on the record. + +The wiring's *behaviour* is proved by `prove_report.py selftest`, which drives nine limbs through +real prover subprocesses; `test_selftest_all_limbs_pass` runs it so the CI suite covers it too. +""" + +from __future__ import annotations + +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + +_REPO = Path(__file__).resolve().parents[1] +_WORKFLOW = _REPO / ".github" / "workflows" / "asvs-prove-absences.yml" +_SCRIPT = _REPO / "scripts" / "asvs" / "prove_report.py" +_REQUIRED_CONTEXTS = _REPO / ".github" / "required-contexts.txt" + + +@pytest.fixture(scope="module") +def raw() -> str: + return _WORKFLOW.read_text(encoding="utf-8") + + +@pytest.fixture(scope="module") +def workflow(raw: str) -> dict: + # `on:` is the YAML 1.1 boolean `True` once parsed -- a well-known Actions/PyYAML collision. + return yaml.safe_load(raw) + + +def test_the_workflow_exists_and_invokes_the_prover(raw: str) -> None: + """The point of the whole exercise: something must actually run the mode. + + Falsified by deleting the invocation -- the mode returns to being wired to nothing, which is the + state this file was written to end. + """ + assert "prove_report.py run" in raw + assert "prove_report.py selftest" in raw + + +def test_the_workflow_grants_no_permissions_by_default(workflow: dict) -> None: + assert workflow["permissions"] == {} + + +def test_no_job_holds_any_write_scope(workflow: dict) -> None: + for name, job in workflow["jobs"].items(): + perms = job.get("permissions", {}) + assert perms == {"contents": "read"}, f"job {name} holds {perms!r}, wanted contents: read" + + +def test_neither_job_is_a_required_context() -> None: + """Advisory is only real if these contexts cannot gate a merge. Branch protection is server-side, + so the checked-in claim is what is assertable from here.""" + declared = _REQUIRED_CONTEXTS.read_text(encoding="utf-8") + for context in ("prove absence claims", "prove-absences wiring selftest"): + assert context not in declared + + +def test_every_action_is_sha_pinned_with_a_version_comment(raw: str) -> None: + uses = [ln.strip() for ln in raw.splitlines() if ln.strip().startswith("- uses:")] + uses += [ln.strip() for ln in raw.splitlines() if ln.strip().startswith("uses:")] + assert uses, "no actions found -- the assertion below would pass vacuously" + for line in uses: + ref = line.split("uses:")[1].strip() + assert "@" in ref, line + sha = ref.split("@")[1].split()[0] + assert len(sha) == 40 and all(c in "0123456789abcdef" for c in sha), line + assert "#" in ref, f"{line} -- a bare SHA with no version comment is unreviewable" + + +def test_checkouts_do_not_persist_credentials(workflow: dict) -> None: + """The vault checkout carries a read token for the vault, which is private. Persisting it into + .git/config would leave it available to every later step in the job for no reason -- only the + files are wanted.""" + found = 0 + for job in workflow["jobs"].values(): + for step in job["steps"]: + if "checkout" in str(step.get("uses", "")): + found += 1 + assert step.get("with", {}).get("persist-credentials") is False, step + assert found >= 3, f"expected at least 3 checkouts, saw {found}" + + +def test_the_vault_checkout_is_sparse_to_the_scorecard_alone(workflow: dict) -> None: + """A vault-read credential in a public repo is a boundary cost taken deliberately and minimised. + A cone-mode checkout, or a widened path, would materialise the maintainer-internal + `docs/security` corpus on a runner whose logs are world-readable.""" + steps = [s for j in workflow["jobs"].values() for s in j["steps"]] + vault = [s for s in steps if "vault" in str(s.get("with", {}).get("path", ""))] + assert len(vault) == 1, "expected exactly one vault checkout" + with_ = vault[0]["with"] + assert with_["sparse-checkout"].strip() == "docs/security/asvs-scorecard.toml" + assert with_["sparse-checkout-cone-mode"] is False + + +def test_strict_and_detail_are_variables_and_default_off(raw: str) -> None: + """The ratchet and the disclosure control are both one repository variable, both unset, so the + committed default is advisory-with-counts-only. Anything that hardcodes `--strict` or `--detail` + into the run body takes the choice away from the owner.""" + assert "vars.ASVS_PROVE_STRICT" in raw + assert "vars.ASVS_PROVE_DETAIL" in raw + run_bodies = "\n".join( + str(s.get("run", "")) for j in yaml.safe_load(raw)["jobs"].values() for s in j["steps"] + ) + assert "flags+=(--strict)" in run_bodies + assert "--strict --root" not in run_bodies and "--root . --strict" not in run_bodies + + +def test_no_expression_interpolation_inside_run_bodies(workflow: dict) -> None: + """Workflow expressions are hoisted into `env:`, never expanded into a shell body -- the + template-injection shape zizmor exists to catch. Same rule the vault's ASVS job states for its + anchor SHA.""" + for job in workflow["jobs"].values(): + for step in job["steps"]: + assert "${{" not in str(step.get("run", "")), step.get("name") + + +def test_the_prove_job_is_dispatch_only(workflow: dict) -> None: + """On a PR there is no credential, so the prove job would fail for a reason unrelated to the PR. + The PR-time question is the wiring's, and `selftest` answers it without a secret. + + DISPATCH-ONLY as of the 2026-08-09 location decision: the scheduled pass runs in the vault, which + is the only repo that can feed it. Both halves are asserted, and the second is the load-bearing + one -- a `schedule` arm here would be red every day for a reason nobody can act on from this + repository, which is how a gate gets switched off. + """ + gate = workflow["jobs"]["prove"]["if"] + assert "workflow_dispatch" in gate + assert "schedule" not in gate + assert "pull_request" not in gate + + +def test_the_workflow_has_no_schedule_trigger(raw: str) -> None: + """The decision is a property of the TRIGGER, not only of the job gate, and the two can drift + apart: a cron could be re-added above while the job's `if:` still excludes it (a workflow that + runs nightly to do nothing), or the `if:` widened while no cron exists. Assert the trigger + directly so neither half can move alone. + """ + parsed = yaml.safe_load(raw) + on = parsed[True] if True in parsed else parsed["on"] + assert "schedule" not in on, ( + "the scheduled prove pass lives in the vault -- a cron here is red every day by construction" + ) + assert "workflow_dispatch" in on + + +def test_the_prove_job_depends_on_the_selftest(workflow: dict) -> None: + """A report from wiring that was never proved able to go red is a decoration.""" + assert workflow["jobs"]["prove"]["needs"] == "selftest" + + +def test_the_workflow_triggers_on_changes_to_itself(raw: str) -> None: + """A gate excluded from its own trigger cannot observe changes to itself -- written up at length + on the vault's asvs-scorecard.yml after a broken gate merged green.""" + parsed = yaml.safe_load(raw) + on = parsed[True] if True in parsed else parsed["on"] + for event in ("pull_request", "push"): + paths = on[event]["paths"] + assert ".github/workflows/asvs-prove-absences.yml" in paths + assert "scripts/asvs/**" in paths + + +# --- the policy claims, asserted against the script rather than the YAML -------------------------- + + +def _run(*args: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, str(_SCRIPT), *args], + capture_output=True, + text=True, + check=False, + timeout=900, + ) + + +def test_a_missing_scorecard_is_an_instrument_failure_not_a_pass(tmp_path: Path) -> None: + """The no-input limb, which is the state this repo is in TODAY and will stay in until the + credential question is answered. Exit 2, never 0. + + Falsified by returning EXIT_OK from that branch: the assertion below goes red, and a daily run + that obtained nothing would report success. Confirmed by injection. + """ + proc = _run("run", "--scorecard", str(tmp_path / "absent.toml"), "--root", str(tmp_path)) + assert proc.returncode == 2, proc.stderr + assert "scanned nothing" in proc.stderr + + +def test_selftest_all_limbs_pass() -> None: + """Run the nine-limb harness in the suite so a regression in the wiring surfaces on any PR, not + only on the workflow's own paths filter.""" + proc = _run("selftest") + assert proc.returncode == 0, proc.stdout + proc.stderr + assert "ALL LIMBS PASS" in proc.stdout + # Naming the limbs here, not just the verdict: a harness that silently stopped running six of + # them would still print ALL LIMBS PASS. + for limb in ("L1", "L2", "L3", "L4", "L5", "L6", "L7", "L8"): + assert f"[PASS] {limb}" in proc.stdout, f"{limb} did not run" + assert proc.stdout.count("[PASS]") == 9, proc.stdout diff --git a/tests/test_asvs_scorecard.py b/tests/test_asvs_scorecard.py index 3810aa41..da6210f6 100644 --- a/tests/test_asvs_scorecard.py +++ b/tests/test_asvs_scorecard.py @@ -12,6 +12,7 @@ from __future__ import annotations +import ast import json import os import re @@ -22,6 +23,7 @@ import pytest +import scripts.asvs.scorecard as scorecard_module from scripts.asvs.scorecard import ( _DESCEND_ONLY, _TRANSPARENT, @@ -35,6 +37,7 @@ Verdict, _copy_scratch, _humanise_age, + _signature, anchor_form, check_absences, check_anchors, @@ -1560,6 +1563,252 @@ def test_prove_absences_refuses_a_mutation_path_that_escapes_the_scratch_tree( assert findings.proved_absences == 0 +# --- pre-flight screens: a red observable is not a proof unless the MUTATION is what reddened it --- +# +# Application is append-based, so a reintroduction bites by redefinition shadowing. Two mutations that +# shadow nothing semantic still redden the observable at exit 1 -- indistinguishable from a surgical +# proof to every check that existed before these screens. Both holes were MEASURED with the screen +# neutered, and both reported `proved=1, problems=0`: a clean green false proof. + + +def test_prove_absences_refuses_a_mutation_whose_signature_does_not_match_the_symbol( + tmp_path: Path, +) -> None: + """A wrong-arity reintroduction raises TypeError at every call site. Every test touching the + symbol fails, exit 1, and the claim would be counted as PROVED -- having demonstrated that calling + a function with the wrong number of arguments breaks it, which is true of every function in the + repository and is evidence about no control at all. + + This is not a hypothetical shape: the mutation is authored in a TOML file in a DIFFERENT + REPOSITORY from the signature it copies, and nothing keeps the two in step. + + Falsified by making `_screen_mutation` return True unconditionally (the pre-change behaviour): + `proved_absences` becomes 1 and `problems` empty, so `not findings.ok` and the + `proved_absences == 0` assertion both go RED. Measured, not reasoned. Restored. + """ + _module(tmp_path, "scanner.py", _SCANNER) + _obs_test(tmp_path, "test_scanner.py", _OBS_TEST) + claim = _live_claim( + 'def scan():\n return "infected"', # real symbol is scan(p) + "scanner.py", + "test_scanner.py::test_clean", + ) + findings = prove_absences([claim], tmp_path) + assert not findings.ok + assert any("PROVE-ERROR" in p and "different signature" in p for p in findings.problems), ( + findings.problems + ) + assert findings.proved_absences == 0 + + +def test_prove_absences_refuses_a_mutation_that_is_not_valid_python(tmp_path: Path) -> None: + """Appending invalid Python breaks IMPORT of the target rather than reintroducing anything. + + When the observable imports at module scope pytest reports a collection error (exit 2) and the + shipped fail-closed branch already catches it. This fixture imports INSIDE the test body, where + the same breakage surfaces as an ordinary test failure at exit 1 -- and was counted as a proof. + That difference is the whole reason the screen is static rather than left to exit codes. + + Falsified by making `_screen_mutation` return True unconditionally: `proved_absences` becomes 1 + with no problems recorded. Restored. + """ + _module(tmp_path, "scanner.py", _SCANNER) + _obs_test( + tmp_path, + "test_scanner.py", + "def test_clean():\n from scanner import scan\n\n assert scan('x') == 'clean'\n", + ) + claim = _live_claim( + 'def scan(p) return "infected"', # missing colon + "scanner.py", + "test_scanner.py::test_clean", + ) + findings = prove_absences([claim], tmp_path) + assert not findings.ok + assert any("PROVE-ERROR" in p and "not valid Python" in p for p in findings.problems), ( + findings.problems + ) + assert findings.proved_absences == 0 + + +def test_the_screens_do_not_refuse_an_honest_claim(tmp_path: Path) -> None: + """The negative control the two tests above need. A screen that refused everything would satisfy + both of them while destroying the mode, and neither would notice. + + A matching signature and valid Python must still prove. Falsified by making `_screen_mutation` + return False unconditionally: this goes RED while both refusal tests stay green. + """ + _module(tmp_path, "scanner.py", _SCANNER) + _obs_test(tmp_path, "test_scanner.py", _OBS_TEST) + claim = _live_claim( + 'def scan(p):\n return "infected"', "scanner.py", "test_scanner.py::test_clean" + ) + findings = prove_absences([claim], tmp_path) + assert findings.ok, findings.problems + assert findings.proved_absences == 1 + + +def test_signature_compares_names_not_only_counts() -> None: + """A parameter RENAME breaks every keyword call, the same wrecking-ball failure as a wrong count + and invisible to an arity-only comparison. Defaults are compared by COUNT, never by value: a + mutation legitimately changes what a default IS, which can be the entire reintroduction. + + Falsified by reducing `_signature` to argument counts: the rename pair compares EQUAL and the + first assertion goes RED, while the default-value pair stays equal either way. + """ + + def sig(src: str) -> tuple[object, ...]: + fn = ast.parse(src).body[0] + assert isinstance(fn, ast.FunctionDef) + return _signature(fn.args) + + assert sig("def f(path): ...") != sig("def f(p): ...") + assert sig("def f(a, *, b): ...") != sig("def f(a, b): ...") + assert sig("def f(a, timeout=1): ...") == sig("def f(a, timeout=999): ...") + + +def test_prove_absences_runs_one_baseline_per_observable_and_one_tree_copy( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Two claims naming the same observable pay for ONE baseline and ONE tree copy. + + The baseline is a property of the pristine tree and the node, so re-measuring it per claim spent a + pytest subprocess re-deriving a constant; the tree was re-copied per claim to apply a few lines and + then thrown away. Both are counted here rather than asserted in prose. + + Falsified by reverting to a `_copy_scratch` per claim: `copies` becomes 2. Falsified separately by + dropping the `baselines` cache: the run count becomes 4. Each moves ONE counter, so the two + changes are independently pinned. + """ + _module(tmp_path, "scanner.py", _SCANNER) + _module(tmp_path, "other.py", "def other(q):\n return 'clean'\n") + _obs_test(tmp_path, "test_scanner.py", _OBS_TEST) + + copies = 0 + real_copy = scorecard_module._copy_scratch + + def counting_copy(root: Path, dest: Path) -> Path: + nonlocal copies + copies += 1 + return real_copy(root, dest) + + runs: list[str] = [] + real_run = scorecard_module._run_node + + def counting_run(scratch: Path, node: str, python: str, timeout: float) -> int: + runs.append(node) + return real_run(scratch, node, python, timeout) + + monkeypatch.setattr(scorecard_module, "_copy_scratch", counting_copy) + monkeypatch.setattr(scorecard_module, "_run_node", counting_run) + + claims = [ + _live_claim( + 'def scan(p):\n return "infected"', "scanner.py", "test_scanner.py::test_clean" + ), + _live_claim( + 'def other(q):\n return "infected"', "other.py", "test_scanner.py::test_clean" + ), + ] + findings = prove_absences(claims, tmp_path) + + assert copies == 1, f"expected one pristine copy, saw {copies}" + # 1 baseline + 2 mutated runs. The second claim reuses the cached baseline. + assert len(runs) == 3, runs + assert findings.proved_absences == 1, findings.problems + # The SECOND claim mutates a module the observable never imports, so it is honestly UNPROVEN -- + # and that verdict is only trustworthy if claim one's mutation was restored before it ran. This + # assertion is therefore the restore check: a leaked mutation would keep the observable red and + # claim two would come back PROVED. + assert any("UNPROVEN" in p for p in findings.problems), findings.problems + + +def test_prove_absences_keys_the_baseline_cache_on_the_OBSERVABLE_not_the_claim( + tmp_path: Path, +) -> None: + """The baseline cache must be keyed on the observable NODE ID. Coarsening that key does not slow + the prover down -- it manufactures a FALSE PROOF, which is the worst outcome this tool has. + + The mechanism: a claim whose observable is already red on the pristine tree must be refused, + because a red that was already there cannot be attributed to the mutation. That refusal is + reached only via the claim's OWN baseline. Under a coarser key the second claim below inherits + the first's green baseline, sails past the refusal, mutates, watches an already-failing test fail, + and books it as proof of a control that never fired. + + The fixture is one cell carrying two claims with the SAME `mutation_path` and DIFFERENT + observables, so it collides under every coarsening that was actually plausible here -- a constant + slot, `cell_id` (`_live_claim` hardcodes 1.1.1, so cell id does not separate claims at all), and + `mutation_path`. Only the node id separates them. + + Falsified by re-keying the cache to a constant (`baselines.get("k")` / `baselines["k"] = ...`): + MEASURED RED here, `proved_absences` 2 instead of 1 and no "not green on the pristine tree" + problem -- i.e. the false proof, observed. The whole rest of the prover suite stayed GREEN under + that same injection, which is what makes this test worth its lines. Restored, `git status` clean. + """ + _module(tmp_path, "scanner.py", _SCANNER) + _obs_test(tmp_path, "test_scanner.py", _OBS_TEST) + # Red on the PRISTINE tree, before any mutation is applied. + _obs_test(tmp_path, "test_broken.py", "def test_already_red():\n assert False\n") + + two_observables = Cell( + id="1.1.1", + level=1, + verdict="fail", + absence=( + Absence( + pattern="x", + positive_control="y", + mutation='def scan(p):\n return "infected"', + mutation_path="scanner.py", + observable="test_scanner.py::test_clean", # green: baseline 0, caches under ITS id + ), + Absence( + pattern="x2", + positive_control="y2", + mutation="UNUSED_CONSTANT = 1", + mutation_path="scanner.py", # same path on purpose: a path key would collide + observable="test_broken.py::test_already_red", # red: must be refused, not proved + ), + ), + ) + + findings = prove_absences([two_observables], tmp_path) + + assert findings.proved_absences == 1, ( + "the already-red observable was scored as a proof, so the baseline cache handed it a " + f"baseline belonging to a different node: {findings.problems}" + ) + assert any( + "not green on the pristine tree" in p and "test_broken.py::test_already_red" in p + for p in findings.problems + ), findings.problems + + +def test_prove_absences_restores_the_scratch_target_between_claims(tmp_path: Path) -> None: + """One shared tree is only sound while it stays pristine, so the restore is asserted rather than + assumed. Two claims against the SAME file: the second must see the original bytes. + + Falsified by deleting the `finally:` restore: claim two's baseline is taken on an already-mutated + tree, comes back red, and the run reports PROVE-ERROR "not green on the pristine tree" instead of + the expected UNPROVEN -- so this test's assertion goes RED and names exactly what leaked. + """ + _module(tmp_path, "scanner.py", _SCANNER) + _obs_test(tmp_path, "test_scanner.py", _OBS_TEST) + # First mutates scanner.py and bites; second appends something inert to the same file. + claims = [ + _live_claim( + 'def scan(p):\n return "infected"', "scanner.py", "test_scanner.py::test_clean" + ), + _live_claim("UNUSED_CONSTANT = 1", "scanner.py", "test_scanner.py::test_clean"), + ] + findings = prove_absences(claims, tmp_path) + assert findings.proved_absences == 1, findings.problems + assert any("UNPROVEN" in p for p in findings.problems), findings.problems + assert not any("not green on the pristine tree" in p for p in findings.problems), ( + findings.problems + ) + + def test_copy_scratch_excludes_secrets_store_and_vault_posture(tmp_path: Path) -> None: """The scratch copy the vault mutation-run reads must never carry secrets, the local store, or the vault posture tree -- CLAUDE.md §9 forbids this module reading them at all. `_copy_scratch` skips @@ -2096,59 +2345,119 @@ def test_prove_absences_records_the_population_it_saw(tmp_path: Path) -> None: def test_prove_absences_counters_close_against_the_population(tmp_path: Path) -> None: - """The arithmetic that makes the summary readable, asserted rather than asserted-in-prose. - - FIVE outcomes raise a problem and increment no counter (an escaping `mutation_path`, one that is - not a file, a baseline that is not green, an UNPROVEN mutated-green, and a mutated run that - errored). So the closing identity is population minus the three counters, and this fixture drives - one claim into each of four different branches to check it holds across them. - - Note what is NOT asserted: that `len(problems)` equals the problem-only count. It does not, and - cannot -- a SUSPECT finding rides along with a claim already counted in `static_screened`, so - problems and claims are different populations. Asserting that equality would pin a false identity. + """The accounting identity on `Findings`, with BOTH sides computed from a real run:: + + checked_absences - proved_absences - static_screened - skipped_absences + == claims that ended in a problem-only branch + + The left side is the counters. The right side is DERIVED FROM `problems` -- the set of distinct + cell ids that raised one -- rather than typed in as a constant, and that difference is the whole + point of this test. The version it replaces asserted `remainder == 1` against a hand-chosen + fixture: true of that fixture, silent about the identity. It is exactly why three more + problem-only branches could be added (the two mutation screens and the restore verification) + while the prose enumerating them went stale, and nothing moved. + + The derivation is valid only while no claim is BOTH counted and problem-bearing -- the static + backstop's SUSPECT finding is precisely that shape -- so the screened claim here deliberately + carries no `raise`, and the assertion below pins that it stayed out of `problems`. Cell 1.1.2 + carries TWO claims and every other cell carries one; because both of 1.1.2's end in a COUNTED + branch, no cell id can reach `problems` twice, so a cell id still names at most one problem-only + claim and the set-dedupe below stays exact. A cell with two PROBLEM-only claims would break that + derivation silently, which is why the multi-claim cell is a counted one on purpose. + + What is NOT asserted, and must not be: that `len(problems)` is the right-hand side. It is not, + for the SUSPECT reason above; asserting that equality would pin a false identity. + + Falsified by giving a `_screen_mutation` refusal a counter (`static_screened += 1` beside the + PROVE-ERROR -- the fold this design rejected): MEASURED RED at the identity assertion, reporting + "counters leave 2 unaccounted for, but 3 claim(s) ended in a problem-only branch". Falsified + again by guarding the `checked_absences` increment on `a.mutation_path`, which undercounts the + population: MEASURED RED at the fixture-shape assertion below (5 != 6), and the identity would + have caught it a line later (remainder 2 against 3 problem-only claims) had the shape check not + named it first. Falsified a third time, independently, by giving the not-a-file PROVE-ERROR a + counter (`skipped_absences += 1`): MEASURED RED at the identity assertion, "counters leave 2 + unaccounted for, but 3 claim(s) ended in a problem-only branch: ['1.1.4', '1.1.5', '1.1.6']". + Falsified a fourth time by degenerating the prover's inner loop to `c.absence[:1]`, which is what + the two-claim cell exists to catch: MEASURED RED at the fixture-shape assertion (6 != 7). All + injections applied to the code and removed after measuring, each verified by a clean `git status`. """ _module(tmp_path, "scanner.py", _SCANNER) _module(tmp_path, "quiet.py", "VALUE = 1\n") _obs_test(tmp_path, "test_scanner.py", _OBS_TEST) - proved = _live_claim( - 'def scan(p): return "infected"', "scanner.py", "test_scanner.py::test_clean" - ) + def _cell(cell_id: str, **absence: str) -> Cell: + return Cell( + id=cell_id, + level=1, + verdict="fail", + absence=(Absence(pattern="x", positive_control="y", **absence),), + ) + + # Three counted outcomes, one per counter. + proved = _cell( + "1.1.1", + mutation='def scan(p): return "infected"', + mutation_path="scanner.py", + observable="test_scanner.py::test_clean", + ) + # 1.1.2 carries TWO claims, and it is the only fixture in this file that carries more than one: + # every other `absence=` tuple in the suite has length 1, so the prover's inner `for a in + # c.absence` loop is otherwise never driven past its first element and `checked_absences` + # counting CLAIMS is indistinguishable from it counting CELLS. That is the one property of + # #304's `test_the_prove_summary_reports_the_total_it_saw` this test did not supersede when it + # replaced it. Both claims are `skipped` (no `mutation_path`) so the cell stays out of + # `problems` -- see the docstring on why the multi-claim cell must be a counted one. skipped = Cell( id="1.1.2", level=1, verdict="fail", - absence=(Absence(pattern="x", positive_control="y", mutation="import x"),), - ) - screened = Cell( - id="1.1.3", - level=1, - verdict="fail", absence=( - Absence( - pattern="x", positive_control="y", mutation="VALUE = 2", mutation_path="quiet.py" - ), + Absence(pattern="x", positive_control="y", mutation="import x"), + Absence(pattern="x2", positive_control="y2", mutation="import y"), ), ) - problem_only = Cell( - id="1.1.4", - level=1, - verdict="fail", - absence=( - Absence( - pattern="x", - positive_control="y", - mutation="import x", - mutation_path="not_a_file.py", - ), - ), + screened = _cell("1.1.3", mutation="VALUE = 2", mutation_path="quiet.py") # no `raise` + + # Three of the eight problem-only branches, chosen to span the pre-existing and the new: a + # mutation_path that is not a file, one that escapes the tree, and a screen refusal. + not_a_file = _cell("1.1.4", mutation="import x", mutation_path="not_a_file.py") + escapes = _cell( + "1.1.5", + mutation='def scan(p): return "infected"', + mutation_path="../escape.py", + observable="test_scanner.py::test_clean", ) + refused = _cell( + "1.1.6", + mutation='def scan():\n return "infected"', # real symbol is scan(p) + mutation_path="scanner.py", + observable="test_scanner.py::test_clean", + ) + + f = prove_absences([proved, skipped, screened, not_a_file, escapes, refused], tmp_path) - f = prove_absences([proved, skipped, screened, problem_only], tmp_path) - assert f.checked_absences == 4 + # Non-vacuity: the identity holds trivially over an empty population, so pin the fixture shape + # before asserting anything derived from it. SEVEN claims across SIX cells -- the inequality is + # the point, and it is what makes a claim/cell confusion in the prover reddable. + assert f.checked_absences == 7 + + # THE IDENTITY, asserted before the per-counter checks below so that it -- and not a narrower + # assertion that happens to sit earlier in the file -- is what a miscount reddens. + problem_claims = {p.split(":", 1)[0] for p in f.problems} + assert "1.1.3" not in problem_claims, ( + "the screened claim raised a problem too, so it is counted AND problem-bearing and the " + f"right-hand side below is no longer derivable this way: {f.problems}" + ) remainder = f.checked_absences - f.proved_absences - f.static_screened - f.skipped_absences - assert remainder == 1 # exactly the PROVE-ERROR claim, derived rather than counted - assert f.proved_absences == 1 and f.skipped_absences == 1 and f.static_screened == 1 + assert remainder == len(problem_claims), ( + f"identity broken: counters leave {remainder} unaccounted for, but " + f"{len(problem_claims)} claim(s) ended in a problem-only branch: {sorted(problem_claims)}" + ) + + # Attribution, not the identity: which counter moved, so a break names itself. `skipped` is 2, + # not 1, because cell 1.1.2 carries two claims -- a prover that iterated cells instead of claims + # would read 1 here even with the identity closing, since it would undercount both sides. + assert f.proved_absences == 1 and f.skipped_absences == 2 and f.static_screened == 1 def test_prove_absences_summary_prints_the_population_before_the_parts( diff --git a/tests/test_asvs_verifier_vault_contract.py b/tests/test_asvs_verifier_vault_contract.py index 7b4de230..e0e42bb9 100644 --- a/tests/test_asvs_verifier_vault_contract.py +++ b/tests/test_asvs_verifier_vault_contract.py @@ -129,6 +129,11 @@ def scan(rel: str = VERIFIER_REL) -> tuple[str, dict[str, list[int]], dict[str, # This one has NO engine-side subject at all -- the record it reads exists only in the vault -- # so this list is the only thing standing between it and a dependency nothing here would notice. "scripts/docs/asvs_residual_lint.py", + # The CI wrapper around `--prove-absences`. docs/CI.md states it "ships here and is mirrored into + # the vault, on the same footing as scorecard.py"; this entry is what makes that a checked claim + # rather than a sentence. It passes today because the module is stdlib-only, which is exactly the + # moment the comment above says to add it -- BEFORE it acquires a dependency. + "scripts/asvs/prove_report.py", ) diff --git a/tests/test_key_usage_scope_inventory.py b/tests/test_key_usage_scope_inventory.py index fbcaef3c..e4cca75f 100644 --- a/tests/test_key_usage_scope_inventory.py +++ b/tests/test_key_usage_scope_inventory.py @@ -69,6 +69,9 @@ "is hashed, and the digest is committed in source on both sides of the seam", "ASVS corpus pin": "a keyless content hash over a build input (the OWASP ASVS corpus file), " "not a key, a secret, or a message authenticator", + "ASVS scorecard revision identifier": "a keyless content hash over the scorecard file, printed " + "truncated so a --prove-absences run states which revision of the record it read; not a key, a " + "secret, or a message authenticator, and unlike the corpus pin above nothing is gated on it", "Engine wheel attestation": "a keyless digest over the installed distribution, verified against " "a recorded value; no key is involved on either side", "AD transport": "a TLS hop whose key material is the OS/directory trust store, not engine-held", diff --git a/tests/test_security_static.py b/tests/test_security_static.py index 021e9bbb..a1bb3178 100644 --- a/tests/test_security_static.py +++ b/tests/test_security_static.py @@ -1013,6 +1013,9 @@ def test_xml_import_scanner_sees_indented_imports() -> None: _CRYPTO_SITES_OUTSIDE_THE_PACKAGE = { # ADR 0156: SHA-256 over the ASVS corpus FILE to pin it to the tagged release. No key. "scripts/asvs/scorecard.py": frozenset({"hashlib"}), + # ADR 0156: SHA-256 over the ASVS SCORECARD file, printed truncated so a --prove-absences run + # states which revision of the record it read. No key. + "scripts/asvs/prove_report.py": frozenset({"hashlib"}), "messagefoundry_webconsole/_security.py": frozenset({"secrets"}), "tee/__main__.py": frozenset({"ssl"}), "tee/anon/keying.py": frozenset({"hashlib"}),