diff --git a/.asdd.example.yml b/.asdd.example.yml index 28ee6df..0a0f275 100644 --- a/.asdd.example.yml +++ b/.asdd.example.yml @@ -101,6 +101,9 @@ protected_paths: # style: # spelling: british # banned_chars: ["–", "—"] # en dash, em dash - checked on ADDED lines only +# # Your project's own test/lint/type command. The asdd-preflight.yml workflow RUNS this on every PR as +# # a deterministic, blocking gate (distinct from the model test agent), so a regression the review missed +# # is caught by the real suite. Make it a required check. Unset => the gate is an opt-in no-op. # preflight: "ruff check . && ruff format --check . && mypy pkg && pytest -n 4" # exemplars: # merged PRs that exemplify house style, cheaper than prose # - "https://github.com/OWNER/REPO/pull/123" diff --git a/.github/asdd/audit-export.sh b/.github/asdd/audit-export.sh index 540818a..50354e7 100755 --- a/.github/asdd/audit-export.sh +++ b/.github/asdd/audit-export.sh @@ -38,6 +38,15 @@ yaml_in() { } SINK="$(yaml_in sink)"; SINK="${SINK:-none}" +# The same-repo refusal below (and the commit message) need the governed repo's name. CI sets +# GITHUB_REPOSITORY; a host produce or council run does NOT, which would SILENTLY SKIP the same-repo +# refusal. Derive it from the git remote when unset, so the refusal is enforced everywhere - CI and host - +# with no per-caller wiring (the same anti-miss philosophy as exporting from the wrapper). +if [ -z "${GITHUB_REPOSITORY:-}" ]; then + GITHUB_REPOSITORY="$(git -C "$ROOT" config --get remote.origin.url 2>/dev/null \ + | sed -E 's#(git@|https?://)[^/:]+[/:]##; s#\.git$##')" +fi + SINK_REPO="$(yaml_in sink_repo)" SINK_CMD="$(yaml_in sink_command)" diff --git a/.github/asdd/audit-export.test.sh b/.github/asdd/audit-export.test.sh index 435820a..628ba87 100755 --- a/.github/asdd/audit-export.test.sh +++ b/.github/asdd/audit-export.test.sh @@ -34,6 +34,16 @@ out="$(run_with_cfg 'audit: echo "$out" | grep -q "repository being governed" && echo "ok: refuses the governed repo as sink" \ || { echo "FAIL: should refuse the governed repo (got: $out)"; fail=1; } +# 2b. On a HOST run (GITHUB_REPOSITORY unset), the governed repo is derived from the git remote, so the +# same-repo refusal holds off-CI too instead of being silently skipped. +d="$TMP/hostrepo"; rm -rf "$d"; mkdir -p "$d/.github/asdd" "$d/cli" +cp "$EX" "$d/.github/asdd/audit-export.sh"; cp "$ROOT/cli/audit.py" "$d/cli/audit.py" +( cd "$d" && git init -q && git remote add origin https://github.com/acme/project.git ) >/dev/null 2>&1 +printf 'audit:\n sink: repo\n sink_repo: acme/project\n' > "$d/.asdd.yml" +out="$(env -u GITHUB_REPOSITORY bash "$d/.github/asdd/audit-export.sh" "$L" 2>&1)" +echo "$out" | grep -q "repository being governed" && echo "ok: derives the governed repo from the git remote on a host run" \ + || { echo "FAIL: host-run same-repo refusal not enforced (got: $out)"; fail=1; } + # 3. Unverifiable visibility fails closed (no token -> cannot prove it is private). out="$(run_with_cfg 'audit: sink: repo diff --git a/.github/asdd/intake-check.sh b/.github/asdd/intake-check.sh index 828eb7e..794e3e9 100755 --- a/.github/asdd/intake-check.sh +++ b/.github/asdd/intake-check.sh @@ -202,20 +202,37 @@ if [ "$overridden" = "true" ] && [ "${#problems[@]}" -gt 0 ]; then problems+=("Owner override in effect: the above are advisory only for this PR (owner-override label).") fi +# Lane hygiene (advisory, never a failure): a `chore` PR (spec-exempt) that ADDS or edits a spec is +# self-contradictory - a change that authors a spec is usually a feature or fix, not a chore, so the lane +# is probably mislabelled. WARN so it surfaces; chore still passes. +warnings=() +if [ "$is_chore" = "true" ] && [ -f "$WORKDIR/changed.txt" ] && \ + SPEC_RE="^(${spec_re})$" awk -F'\t' 'BEGIN{re=ENVIRON["SPEC_RE"]} $1 ~ /^[AMR]/ && $NF ~ re {f=1} END{exit f?0:1}' \ + "$WORKDIR/changed.txt"; then + warnings+=("Lane is 'chore' (spec-exempt) but this change adds or edits a spec. A change that authors a spec is usually a feature or fix, not a chore; check the lane. Chore still passes intake.") +fi + # Emit intake JSON. if [ "${#problems[@]}" -gt 0 ]; then probs_json="$(printf '%s\n' "${problems[@]}" | jq -R . | jq -s 'map(select(length>0))')" else probs_json='[]' fi +if [ "${#warnings[@]}" -gt 0 ]; then + warns_json="$(printf '%s\n' "${warnings[@]}" | jq -R . | jq -s 'map(select(length>0))')" +else + warns_json='[]' +fi jq -n \ --argjson pr "${pr_number}" --arg head "${head_sha}" \ --argjson disc "$disclosed" --argjson sign "$signed_off" --argjson lane "$laned" \ - --argjson flood "$flood_ok" --argjson spec "$spec_ok" --argjson conv "$conv_ok" --argjson ovr "$overridden" --argjson probs "$probs_json" \ + --argjson flood "$flood_ok" --argjson spec "$spec_ok" --argjson conv "$conv_ok" --argjson ovr "$overridden" \ + --argjson probs "$probs_json" --argjson warns "$warns_json" \ '{schema:"asdd/intake/v0.1", pr_number:$pr, head_sha:$head, disclosed:$disc, signed_off:$sign, laned:$lane, flood_ok:$flood, spec_ok:$spec, conventions_ok:$conv, override:$ovr, passed:($ovr or ($disc and $sign and $lane and $flood and $spec and $conv)), - problems:$probs}' > "$OUT" + problems:$probs, warnings:$warns}' > "$OUT" +[ "${#warnings[@]}" -gt 0 ] && printf 'intake WARNING: %s\n' "${warnings[@]}" >&2 echo "intake: disclosed=$disclosed signed_off=$signed_off ($signed/$total signed) laned=$laned flood_ok=$flood_ok spec_ok=$spec_ok conventions_ok=$conv_ok override=$overridden" [ -s "$OUT" ] || { echo "intake-check: no output" >&2; exit 1; } diff --git a/.github/asdd/intake-check.test.sh b/.github/asdd/intake-check.test.sh index 9677c72..2adc5ea 100755 --- a/.github/asdd/intake-check.test.sh +++ b/.github/asdd/intake-check.test.sh @@ -105,6 +105,30 @@ case_run "a renamed-to spec counts (R status, new path is \$NF)" \ case_run "a chore PR skips the spec requirement" \ asdd-default.yml 'chore' "$DISC" "$(printf 'M\tsrc/a.py\n')" true true +# Lane hygiene: a `chore` PR is spec-exempt, so authoring a spec inside it is self-contradictory (the +# change is really a feature or fix). The gate still PASSES chore (spec is not required), but it must +# surface a non-failing warning so the mislabelled lane is visible. +fw="$TMP/w"; rm -rf "$fw"; mkdir -p "$fw" +printf '%s' "$DISC" > "$fw/body.md"; printf 'chore' > "$fw/labels.txt" +printf 'feat: a change\n\nSigned-off-by: A Dev \0' > "$fw/commits.txt" +printf 'A\tdocs/specs/new.md\n' > "$fw/changed.txt" +printf 'pr_number=1\nhead_sha=abc123\nrequire_spec=true\n' > "$fw/meta.env" +( cd "$TREE" && ASDD_CONFIG="asdd-default.yml" bash "$GATE" "$fw" "$fw/out.json" >/dev/null 2>&1 ) +if [ "$(jq -r '.passed' "$fw/out.json")" = "true" ] \ + && jq -e '.warnings | map(select(startswith("Lane is "))) | length > 0' "$fw/out.json" >/dev/null; then + echo " ok a chore PR that adds a spec passes but warns the lane is mislabelled" +else + echo " FAIL chore+spec-delta: expected pass with a lane-hygiene warning"; fail=1 +fi +# A chore with NO spec delta must carry no such warning (no false positive). +printf 'M\tsrc/a.py\n' > "$fw/changed.txt" +( cd "$TREE" && ASDD_CONFIG="asdd-default.yml" bash "$GATE" "$fw" "$fw/out.json" >/dev/null 2>&1 ) +if jq -e '.warnings | map(select(startswith("Lane is "))) | length == 0' "$fw/out.json" >/dev/null; then + echo " ok a chore with no spec delta carries no lane-hygiene warning" +else + echo " FAIL chore-without-spec should not warn"; fail=1 +fi + # --- a foreign layout: OpenSpec ----------------------------------------------------------------------- case_run "OpenSpec: adding a proposal satisfies the gate" \ asdd-openspec.yml "$OK_LABELS" "$DISC" "$(printf 'A\topenspec/changes/add-auth/proposal.md\n')" true true diff --git a/.github/asdd/operate/runner-trail.test.sh b/.github/asdd/operate/runner-trail.test.sh new file mode 100755 index 0000000..2e4e841 --- /dev/null +++ b/.github/asdd/operate/runner-trail.test.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# The operate runners (test.sh, docsync.sh) leave an audit trail on EVERY exit path, and their export half +# is INERT without a sink credential: record locally, do NOT attempt a push, and do NOT fail the run. A stub +# audit-export.sh here proves invocation vs non-invocation directly, so a regression that drops the +# AUDIT_SINK_TOKEN guard (turning "record on every exit" into "try to export on every exit") is caught. +set -uo pipefail +HERE="$(cd "$(dirname "$0")" && pwd)" # .github/asdd/operate +ROOT="$(cd "$HERE/../../.." && pwd)" +fail=0 +T="$(mktemp -d)"; trap 'rm -rf "$T"' EXIT + +# A throwaway repo so the runner resolves a STUB audit-export.sh and the real deps it needs. +D="$T/repo"; mkdir -p "$D/cli" "$D/.github/asdd/operate" "$D/recipes" +cp "$ROOT/cli/audit.py" "$ROOT/cli/operate-guard.py" "$D/cli/" +cp "$ROOT/recipes/test-runner.yaml" "$D/recipes/" 2>/dev/null || true +printf '#!/usr/bin/env bash\necho called >> "%s/exp.log"\n' "$T" > "$D/.github/asdd/audit-export.sh" +chmod +x "$D/.github/asdd/audit-export.sh" + +for pair in "test.sh:test-runner" "docsync.sh:documentation"; do + runner="${pair%%:*}"; role="${pair##*:}" + cp "$ROOT/cli/templates/operate/$runner" "$D/.github/asdd/operate/$runner" + [ "$runner" = "docsync.sh" ] && cp "$ROOT/recipes/documentation.yaml" "$D/recipes/" 2>/dev/null || true + rm -f "$T/exp.log" + + # 1. No sink credential: must record locally, exit 0, and NOT invoke the exporter. + env -u ASDD_MODEL_URL -u ASDD_RUNTIME_TOKEN -u AUDIT_SINK_TOKEN ASDD_ACTIVITY_LOG="$T/a.jsonl" \ + bash "$D/.github/asdd/operate/$runner" cx "$T/o.md" >/dev/null 2>&1; rc=$? + if [ "$rc" = 0 ] && [ -s "$T/a.jsonl" ] && grep -q "\"role\":\"$role\"" "$T/a.jsonl" && [ ! -f "$T/exp.log" ]; then + echo " ok $runner: no-token run records locally, no export, no failure" + else + echo " FAIL $runner: no-token path (rc=$rc record=$([ -s "$T/a.jsonl" ] && echo y) export=$([ -f "$T/exp.log" ] && echo attempted))"; fail=1 + fi + + # 2. With a sink credential (trusted post-merge context): the export half fires. + rm -f "$T/exp.log" "$T/b.jsonl" + env -u ASDD_MODEL_URL -u ASDD_RUNTIME_TOKEN AUDIT_SINK_TOKEN=x ASDD_ACTIVITY_LOG="$T/b.jsonl" \ + bash "$D/.github/asdd/operate/$runner" cx "$T/o2.md" >/dev/null 2>&1 + [ -f "$T/exp.log" ] && echo " ok $runner: a sink credential triggers the export" \ + || { echo " FAIL $runner: export not attempted with a token"; fail=1; } +done + +echo +[ "$fail" = 0 ] && echo "runner-trail self-test: PASS" || echo "runner-trail self-test: FAIL" +exit "$fail" diff --git a/.github/asdd/preflight.sh b/.github/asdd/preflight.sh new file mode 100755 index 0000000..60b74f5 --- /dev/null +++ b/.github/asdd/preflight.sh @@ -0,0 +1,42 @@ +#!/usr/bin/env bash +# ASDD - the deterministic preflight gate. Runs the adopter's OWN test/lint/type command +# (conventions.preflight in .asdd.yml) as a real, blocking check, so a regression the model review missed +# is still caught by the actual suite. This is DISTINCT from the model test-runner agent (asdd-test.yml): +# that agent judges with a model post-merge; this runs the deterministic suite and its exit status IS the +# gate. A spec-and-test framework that never runs the tests deterministically is the gap this closes. +# +# The command is the adopter's own trusted config (not untrusted PR text), so running it in a shell is the +# intended interface (they write `ruff ... && pytest ...`). The workflow that calls this holds no secrets, +# so executing the change's tests on a PR cannot exfiltrate; a fork PR still needs the maintainer's +# fork-workflow approval before it runs at all. +# +# Usage: preflight.sh (reads .asdd.yml, runs conventions.preflight, exits with its status) +# Env: ASDD_CONFIG overrides the config path. +set -uo pipefail +ROOT="$(cd "$(dirname "$0")/../.." && pwd)" +CFG="${ASDD_CONFIG:-$ROOT/.asdd.yml}" + +# Read conventions.preflight (nested one level under conventions:), no YAML dependency. The value is a +# quoted scalar; a `#` inside it is part of the command, so inline-comment stripping is deliberately NOT +# applied here (unlike the lane/path list readers, whose tokens never contain #). +cmd="$(awk ' + /^conventions:/ { inc=1; next } + inc && /^[A-Za-z]/ { inc=0 } + inc && /^[[:space:]]+preflight:[[:space:]]*/ { + line=$0; sub(/^[[:space:]]+preflight:[[:space:]]*/, "", line); print line; exit + }' "$CFG" 2>/dev/null)" +# Strip ONE matching pair of outer quotes (a double-quoted value may legitimately end in a single quote, +# e.g. "python3 -c 'x'", so a blanket strip of both would eat the command's own inner quote). +case "$cmd" in + \"*\") cmd="${cmd#\"}"; cmd="${cmd%\"}" ;; + \'*\') cmd="${cmd#\'}"; cmd="${cmd%\'}" ;; +esac + +if [ -z "$cmd" ]; then + echo "asdd preflight: no conventions.preflight configured; nothing to run. Declare your project's own" + echo "test/lint/type command (e.g. \"ruff check . && pytest -n 4\") under conventions: to enable the gate." + exit 0 +fi + +echo "asdd preflight: running the project's own suite -> $cmd" +bash -c "$cmd" # the command's exit status IS the gate: a non-zero here fails the check and blocks merge. diff --git a/.github/asdd/preflight.test.sh b/.github/asdd/preflight.test.sh new file mode 100755 index 0000000..0a52f24 --- /dev/null +++ b/.github/asdd/preflight.test.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Self-test for the deterministic preflight gate (.github/asdd/preflight.sh): it runs the adopter's own +# conventions.preflight command and its exit status IS the gate - a passing suite passes, a failing suite +# (a deliberately broken test) fails the check, and an unconfigured preflight is a no-op that passes. +set -uo pipefail +HERE="$(cd "$(dirname "$0")" && pwd)" +PF="$HERE/preflight.sh" +T="$(mktemp -d)"; trap 'rm -rf "$T"' EXIT +fail=0 +ok() { echo " ok $1"; } +bad() { echo " FAIL $1"; fail=1; } + +run() { ASDD_CONFIG="$1" bash "$PF" >/dev/null 2>&1; echo "$?"; } + +# 1. A passing suite passes the gate. +printf 'conventions:\n preflight: "exit 0"\n' > "$T/pass.yml" +[ "$(run "$T/pass.yml")" = 0 ] && ok "a passing preflight passes the gate" || bad "passing preflight" + +# 2. A failing suite fails the gate (a regression is caught deterministically). +printf 'conventions:\n preflight: "exit 7"\n' > "$T/fail.yml" +[ "$(run "$T/fail.yml")" != 0 ] && ok "a failing preflight fails the gate" || bad "failing preflight did not fail" + +# 3. A real command: a passing then a deliberately broken python check. +printf "conventions:\n preflight: \"python3 -c 'raise SystemExit(0)'\"\n" > "$T/realok.yml" +[ "$(run "$T/realok.yml")" = 0 ] && ok "a real passing check passes" || bad "real passing check" +printf "conventions:\n preflight: \"python3 -c 'raise SystemExit(1)'\"\n" > "$T/realbad.yml" +[ "$(run "$T/realbad.yml")" != 0 ] && ok "a real broken check fails the gate" || bad "real broken check did not fail" + +# 4. No preflight configured: a no-op that passes and says so (opt-in). +printf 'lanes:\n - feature\n' > "$T/none.yml" +out="$(ASDD_CONFIG="$T/none.yml" bash "$PF" 2>&1)"; rc=$? +if [ "$rc" = 0 ] && printf '%s' "$out" | grep -q "nothing to run"; then + ok "unconfigured preflight is an opt-in no-op" +else + bad "unconfigured preflight (rc=$rc)" +fi + +echo +[ "$fail" = 0 ] && echo "preflight self-test: PASS" || echo "preflight self-test: FAIL" +exit "$fail" diff --git a/.github/workflows/asdd-preflight.yml b/.github/workflows/asdd-preflight.yml new file mode 100644 index 0000000..484102f --- /dev/null +++ b/.github/workflows/asdd-preflight.yml @@ -0,0 +1,32 @@ +# ASDD preflight - the deterministic test gate. +# +# Runs the adopter's OWN test/lint/type command (conventions.preflight in .asdd.yml) on every PR, so a +# regression the model review missed is still caught by the real suite. This is DISTINCT from the model +# test-runner agent (asdd-test.yml, which judges post-merge with a model): here the deterministic suite's +# exit status IS the gate. Make it a required check in branch protection to block a red suite from merging. +# +# It holds NO secrets (contents: read only), so executing the change's own tests on a PR cannot exfiltrate; +# a fork PR still needs the maintainer's fork-workflow approval before it runs at all. It is a no-op that +# passes until conventions.preflight is set, so installing it is safe before a project wires its suite. +name: ASDD preflight + +on: + pull_request: + types: [opened, synchronize, reopened] + +permissions: + contents: read + +concurrency: + group: asdd-preflight-${{ github.event.pull_request.number }} + cancel-in-progress: true + +jobs: + preflight: + if: github.event.pull_request.draft == false + runs-on: ubuntu-latest + steps: + - name: Check out the change (the PR merged into its base) + uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 + - name: Run the project's deterministic preflight + run: bash .github/asdd/preflight.sh diff --git a/CHANGELOG.md b/CHANGELOG.md index f2a92ee..e16b661 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,41 @@ draft, so pin a conformance claim to a commit or date. ## [Unreleased] ### Added +- **The audit trail is complete: every agent's records reach the sink, not only the reviewer's.** Before, + only the review exported, so a deployment that turned on `audit.sink` captured the reviewer's decisions + but lost the produce side (the developer/council "coding"), the test and documentation agents, and the + operator agents. Now every path records through one route and exports from a credential-safe point (the + produce wrapper, the runners, the council), the untrusted PR review keeps its record/publish split, and + `doctor` runs a **dynamic completeness check** that enumerates every record-writing path and warns on any + with no export route, so a new agent added later cannot silently drop its trail. The check also covers the + CI workflow surface with a stricter rule: because the sink credential lives in the job, a workflow that + runs a recorder which does not itself export (for example `dev-council.py` wired straight into a workflow + instead of through the exporting `dev-council.sh`) must export in that same job, and cannot lean on a + runner it bypassed. `audit-export.sh` also derives the governed repo from the git remote when + `GITHUB_REPOSITORY` is unset, so its "never export into the repo you govern" refusal holds on host runs, + not only in CI. +- **A deterministic preflight gate runs the project's real test suite.** `conventions.preflight` (the + adopter's own `ruff`/`pytest`/`mypy` command) was declared and surfaced to the agents but never actually + run. The new `asdd-preflight.yml` workflow runs it on every PR through `.github/asdd/preflight.sh`, so a + regression the model review missed is still caught by the real suite; its exit status is the gate. It is + distinct from the model test-runner agent, holds no secrets, and is an opt-in no-op until the command is + set. A spec-and-test framework should run the tests, not only reason about them. +- **`doctor` warns when the git identity is unset.** A contributor whose `user.name` / `user.email` are + not set produces commits that cannot be signed off or attributed, so intake rejects them after the work + is done. The preflight now flags this up front, as a warning, so it is fixed before the first commit. +- **`connect-check --json` for tooling and CI.** The per-role connected/dry-run status is now available as + machine-readable JSON with the same accounting and exit code as the human output, so a setup script or + pipeline can gate on it without scraping text. +- **Intake warns when a `chore` change authors a spec.** The `chore` lane is spec-exempt, so a change that + adds or edits a spec while labelled `chore` is almost certainly a mislabelled feature or fix. Intake now + surfaces this as a non-failing warning (the change still passes) so the lane can be corrected. +- **Editor pointers for a bring-your-own developer.** `init` now writes a thin pointer for the common + coding assistants (`CLAUDE.md` for Claude Code and the Claude app, `.cursor/rules/asdd.mdc` for Cursor; + Codex and other AGENTS.md-convention tools read `AGENTS.md` directly), so a contributor's assistant loads + the contribution constitution with no manual setup. Each pointer references `AGENTS.md` and is skipped if + the file already exists, so an existing rule file is never overwritten. A new guide, + [bring your own developer](docs/guides/bring-your-own-developer.md), covers the non-engineer spec path + and that the operate agents run on open-source Goose. - **`doctor` and `setup` warn when the reviewer is a heavy reasoning model.** A reasoning model reasons at length and can exceed a hosted inference window on a real code diff, so the review times out and posts no lenses while trivial or docs-only diffs still pass and look fine. The preflight and the setup wizard now diff --git a/README.md b/README.md index 7a96f20..441ef40 100644 --- a/README.md +++ b/README.md @@ -87,9 +87,9 @@ The intake gate runs on the next pull request immediately. The review runs in a Any OpenAI-compatible provider works. The analysis job holds `contents: read` only; the write scope stays in the publish job, which never reads untrusted PR content. That split is the security invariant, do not merge the two jobs. -### 3. From a coding assistant +### 3. Develop from your coding assistant -The same steps are slash commands: [`/asdd:setup`, `/asdd:spec`, `/asdd:review`, `/asdd:status`](docs/guides/slash-commands.md). They are thin prompts over the CLI, so they port to any assistant. The CLI also runs the deterministic gates locally and a read-only [dashboard](docs/guides/governance-dashboard.md); a spec that passes locally passes on the PR, because it is the same code. +`init` writes a pointer for the common assistants (`CLAUDE.md` for Claude Code and the Claude app, `.cursor/rules/asdd.mdc` for Cursor, and `AGENTS.md` for Codex and any tool that follows the AGENTS.md convention), so whatever you bring reads the contribution constitution with no extra setup. From there the workflow is a set of slash commands: [`/asdd:setup`, `/asdd:spec`, `/asdd:review`, `/asdd:status`](docs/guides/slash-commands.md), thin prompts over the CLI that port to any assistant. You do not have to be an engineer to start: describe the change to the spec agent in plain language and it drafts the spec the pipeline needs. The CLI also runs the deterministic gates and a read-only [dashboard](docs/guides/governance-dashboard.md) locally; a spec that passes locally passes on the PR, because it is the same code. See [bring your own developer](docs/guides/bring-your-own-developer.md). ## Bring your own assistant, and your own spec tool @@ -107,7 +107,7 @@ Two things make it work. **Your conventions are declared, and the agents are hel ## Running the operate layer with Goose -`init` above wires the **govern** layer (the CI gates). The **operate** layer, the agents that do the work, is optional and runtime-neutral: implement the contract in [`agents/runtime.md`](agents/runtime.md) on your own harness, or use the ready-to-run kit for unmodified [Goose](https://block.github.io/goose/). +`init` above wires the **govern** layer (the CI gates). The **operate** layer, the agents that do the work, is optional and runtime-neutral: implement the contract in [`agents/runtime.md`](agents/runtime.md) on your own harness, or use the ready-to-run kit for unmodified [Goose](https://block.github.io/goose/), which is open source and free to run. You bring an API key for a model; you do not buy the harness. > **Status: alpha.** The Goose operate kit is usable and dogfooded, but its recipes and interfaces may still change. diff --git a/cli/README.md b/cli/README.md index 6d1cdde..4fbc1b0 100644 --- a/cli/README.md +++ b/cli/README.md @@ -162,6 +162,7 @@ Use `--no-ping` for a network-free config-completeness check. ```bash asdd connect-check # ping every role (+ the council if configured) asdd connect-check --no-ping # "is the runtime configured?" without a model call +asdd connect-check --json # the same per-role result as JSON, for a setup script or CI to gate on ``` ## kit-check @@ -412,7 +413,10 @@ With no model wired it prints a labelled dry run. The Goose kit ships the runner `asdd doctor [CONFIG]` preflights the operate path before you rely on it. It checks Python, Goose, the selected spec CLI, the roster's heterogeneity rule, the runtime key, the declared conventions and the -recipes, reporting each as OK, a warning, or a blocking issue with the exact next step. +recipes, reporting each as OK, a warning, or a blocking issue with the exact next step. It also warns on +the setup traps that pass silently: a git identity that is not set (a contributor's commits could not be +signed off), a reviewer that is a heavy reasoning model (it can time out on a real diff), and, when an +audit sink is configured, any record-writing path or CI job with no export route. The part worth knowing: it tells **"not installed"** apart from **"installed but not on your PATH"**. A plain `which` reports the second as the first, which sends people reinstalling a tool they already have. diff --git a/cli/connect-check.py b/cli/connect-check.py index a381f86..765490a 100644 --- a/cli/connect-check.py +++ b/cli/connect-check.py @@ -129,6 +129,7 @@ def main(): ap.add_argument("config", nargs="?", default=".asdd.yml") ap.add_argument("--role", action="append", default=[], help="check only these roster roles") ap.add_argument("--no-ping", action="store_true", help="report config completeness only; no model call") + ap.add_argument("--json", action="store_true", help="emit the per-role result as JSON for tooling/CI") a = ap.parse_args() if not os.path.isfile(a.config): sys.stderr.write(f"connect-check: no config at {a.config} (run `asdd init --goose` first).\n") @@ -141,20 +142,29 @@ def main(): if not a.role: checks += [(name, m, u, t) for name, m, u, t in council_members(a.config)] - print("asdd connect-check - are the agents connected, or dry-running?\n") + rows = [] connected = configured = 0 for name, model, url, token in checks: if not model and name.startswith("council"): continue configured += 1 if model else 0 state, detail = classify(name, model, url, token, not a.no_ping) - mark = {"live": "LIVE ", "ready": "READY", "dry-run": "DRY ", "error": "ERR ", - "no-model": "- "}[state] if state in ("live", "ready"): connected += 1 - print(f" [{mark}] {name:<14} {detail}") - + rows.append({"role": name, "model": model, "state": state, "detail": detail, + "connected": state in ("live", "ready")}) live, total = connected, configured + + if a.json: + print(json.dumps({"roles": rows, "connected": live, "configured": total, + "all_connected": total > 0 and live == total, "pinged": not a.no_ping})) + return 0 if (total > 0 and live == total) else 1 + + print("asdd connect-check - are the agents connected, or dry-running?\n") + for r in rows: + mark = {"live": "LIVE ", "ready": "READY", "dry-run": "DRY ", "error": "ERR ", + "no-model": "- "}[r["state"]] + print(f" [{mark}] {r['role']:<14} {r['detail']}") print() if total == 0: print("No role has a model in the roster. Run `asdd setup` to assign models, then connect a runtime.") diff --git a/cli/connect-check.test.sh b/cli/connect-check.test.sh index 5f51def..3c8a372 100644 --- a/cli/connect-check.test.sh +++ b/cli/connect-check.test.sh @@ -73,6 +73,15 @@ assert ok and len(calls) == 1 and 'max_completion_tokens' not in calls[0], "norm PY then ok "reasoning-model max_completion_tokens fallback (mocked)"; else bad "reasoning-model max_completion_tokens fallback"; fi +# --json emits a machine-readable per-role result for CI/tooling, with the same connected/total accounting +# and exit code as the human output. No ping (config-completeness only), so it is hermetic. +out="$(python3 "$CC" "$T/.asdd.yml" --no-ping --json 2>&1)"; rc=$? +if printf '%s' "$out" | python3 -c "import json,sys; d=json.load(sys.stdin); assert isinstance(d['roles'],list) and 'all_connected' in d and 'connected' in d and 'configured' in d; assert all('state' in r and 'connected' in r for r in d['roles'])" 2>/dev/null; then + ok "--json emits valid machine-readable per-role status" +else + bad "--json output invalid or incomplete" +fi + rm -rf "$T" echo [ "$fail" = 0 ] && echo "connect-check self-test: PASS" || echo "connect-check self-test: FAIL" diff --git a/cli/doctor.py b/cli/doctor.py index c079a68..63e50f6 100644 --- a/cli/doctor.py +++ b/cli/doctor.py @@ -21,6 +21,7 @@ """ import argparse import os +import re import shutil import subprocess import sys @@ -88,6 +89,31 @@ def _read_scalar(config, key): return None +def _git_config(key): + try: + r = subprocess.run(["git", "config", "--get", key], capture_output=True, text=True) + return r.stdout.strip() if r.returncode == 0 else "" + except Exception: + return "" + + +def check_git_identity(rep): + """A bring-your-own developer must be able to produce a CONFORMANT commit: a git identity for the author + attribution and the DCO sign-off intake requires. doctor reported READY while git identity was unset + (locally and globally), so a contributor then could not sign off or attribute a commit and every PR + would fail intake. WARN, with the exact fix.""" + name = _git_config("user.name") + email = _git_config("user.email") + if name and email: + rep.add(OK, "git identity", f"{name} <{email}> - commits can be attributed and DCO-signed") + return + missing = " and ".join(k for k, v in (("user.name", name), ("user.email", email)) if not v) + rep.add(WARN, "git identity is not set", + f"{missing} unset, so a commit cannot carry the author trailer or a DCO sign-off, and intake " + "(which requires disclosure + DCO) would reject every PR", + "git config --global user.name 'Your Name' && git config --global user.email 'you@example.com'") + + def check_python(rep): v = sys.version_info s = f"{v.major}.{v.minor}.{v.micro}" @@ -242,6 +268,89 @@ def check_reviewer_reasoning(rep, config): "per-call timeout will otherwise fail fast and name this cause") +def _audit_sink(config): + """The audit.sink value (none|repo|command) from .asdd.yml, or None if no audit block.""" + inblk = False + try: + with open(config) as fh: + for line in fh: + s = line.split("#", 1)[0].rstrip() + if s.strip() == "audit:": + inblk = True + continue + if inblk: + if s and not s[0].isspace(): + break + t = s.strip() + if t.startswith("sink:"): + return t[len("sink:"):].strip().strip('"').strip("'") + except OSError: + return None + return None + + +def check_audit_export_completeness(rep, config): + """When a sink is configured, every path that RECORDS must also EXPORT, or its trail is silently + dropped. Enumerate the record-writers dynamically (a NEW recorder added later with no export is then + caught, instead of a hardcoded list going stale), and WARN on any with no export route. A shared library + recorder is covered when its own file calls audit-export or a runner that invokes it by name does. A CI + workflow is held to a stricter rule: because the sink credential lives in the job, a workflow that runs a + non-self-exporting recorder must export in that same job, not lean on a runner it bypassed.""" + sink = _audit_sink(config) + if not sink or sink == "none": + return # opting in is required; with no sink there is nothing to export. + root = os.path.dirname(os.path.abspath(config)) or "." + dirs = [os.path.join(root, "cli"), os.path.join(root, ".github", "asdd")] + rec_re = re.compile(r"audit\.py\s+(append|from-review)|from_review|audit\.py['\"]?\s*,\s*['\"]?append") + files = [] + for d in dirs: + for dp, _sub, fs in os.walk(d) if os.path.isdir(d) else []: + for f in fs: + if f.endswith((".sh", ".py", ".yml")) and not f.endswith((".test.sh", ".test.py")): + files.append(os.path.join(dp, f)) + texts = {f: _read(f) for f in files} + exporters = " ".join(os.path.basename(f) for f, t in texts.items() if "audit-export" in t) + missing = [] + for f, t in texts.items(): + if not rec_re.search(t): + continue + if "audit-export" in t or os.path.basename(f) in exporters: + continue # exports itself, or a runner/workflow that names it exports. + missing.append(os.path.relpath(f, root)) + + # The workflow surface has a stricter rule than a shared library file. A CI job is where the sink + # credential lives, so a workflow that runs a recorder which does not itself export MUST export in that + # same job; it cannot lean on a sibling runner it bypassed. A deployment that wires dev-council.py (or a + # raw audit.py append) straight into a workflow, instead of through the exporting dev-council.sh runner, + # would otherwise drop the produce trail with the sibling-runner escape hiding it. Match that case by + # its own file, not by a runner that names the tool. + wf_dir = os.path.join(root, ".github", "workflows") + wf_rec_re = re.compile(r"dev-council\.py|audit\.py\s+(append|from-review)") + for dp, _sub, fs in os.walk(wf_dir) if os.path.isdir(wf_dir) else []: + for f in fs: + if not f.endswith((".yml", ".yaml")): + continue + p = os.path.join(dp, f) + if wf_rec_re.search(_read(p)) and "audit-export" not in _read(p): + missing.append(os.path.relpath(p, root)) + if missing: + rep.add(WARN, "an agent records but has no export route", + "sink is '" + sink + "', but these record to the ledger with no export step, so their " + "trail is dropped: " + ", ".join(sorted(missing)), + "add an audit-export.sh call at the end of the runner (trusted context), or a publish job " + "that exports, so the produce and support trails reach the sink like the review's") + else: + rep.add(OK, "audit export completeness", "every record-writing path has an export route") + + +def _read(path): + try: + with open(path, encoding="utf-8", errors="replace") as fh: + return fh.read() + except OSError: + return "" + + def main(): ap = argparse.ArgumentParser(description="Preflight the ASDD Goose operate path.") ap.add_argument("config", nargs="?", default=".asdd.yml", @@ -257,10 +366,12 @@ def main(): return 1 check_python(rep) + check_git_identity(rep) check_goose(rep) check_spec_tool(rep, a.config) check_roster(rep, a.config) check_reviewer_reasoning(rep, a.config) + check_audit_export_completeness(rep, a.config) check_conventions(rep, a.config) check_runtime_key(rep) check_recipes(rep, a.config) diff --git a/cli/doctor.test.sh b/cli/doctor.test.sh index 06ca292..7812b32 100644 --- a/cli/doctor.test.sh +++ b/cli/doctor.test.sh @@ -65,4 +65,35 @@ out="$(python3 "$DOC" "$TMP/fast.yml" 2>&1)" grep -qi 'reasoning model' <<<"$out" && bad "fast reviewer wrongly warned" \ || ok "fast reviewer does not trigger the reasoning warn" +# Dynamic audit-export completeness: a sink configured + a recorder with no export route must WARN and name +# the file (the anti-miss guard); with no sink it is moot and silent. +RC="$TMP/reco"; mkdir -p "$RC/cli" "$RC/.github/asdd" +printf 'audit:\n sink: repo\n sink_repo: acme/ledger\n' > "$RC/.asdd.yml" +printf '#!/usr/bin/env bash\npython3 cli/audit.py append --role triage --action x\n' > "$RC/.github/asdd/rogue.sh" +out="$(python3 "$DOC" "$RC/.asdd.yml" 2>&1)" +grep -qi "records but has no export" <<<"$out" && grep -q "rogue.sh" <<<"$out" \ + && ok "audit-export completeness flags an unexported recorder" || bad "completeness check missed a rogue recorder" +printf 'audit:\n sink: none\n' > "$RC/none.yml" +out="$(python3 "$DOC" "$RC/none.yml" 2>&1)" +grep -qi "no export route" <<<"$out" && bad "completeness warned on sink none" || ok "completeness is moot on sink none" + +# The workflow surface is stricter: a CI job that runs a recorder which does not self-export must export in +# that same job. A workflow that wires dev-council.py directly, bypassing the exporting dev-council.sh +# runner, drops the produce trail, and the sibling-runner escape must NOT hide it (the gap a real deployment +# hit). With an in-job export step the same workflow is clean. +WF="$TMP/wf"; mkdir -p "$WF/.github/workflows" +printf 'audit:\n sink: repo\n sink_repo: acme/ledger\n' > "$WF/.asdd.yml" +printf 'name: council\njobs:\n run:\n steps:\n - run: python3 cli/dev-council.py --change x\n' > "$WF/.github/workflows/dev-council.yml" +out="$(python3 "$DOC" "$WF/.asdd.yml" 2>&1)" +grep -qi "records but has no export" <<<"$out" && grep -q "dev-council.yml" <<<"$out" \ + && ok "a workflow running dev-council.py with no in-job export is flagged" || bad "workflow-surface recorder not flagged" +printf 'name: council\njobs:\n run:\n steps:\n - run: python3 cli/dev-council.py --change x\n - run: bash .github/asdd/audit-export.sh .asdd-work/audit.jsonl\n' > "$WF/.github/workflows/dev-council.yml" +out="$(python3 "$DOC" "$WF/.asdd.yml" 2>&1)" +grep -q "dev-council.yml" <<<"$out" && bad "workflow with an in-job export wrongly flagged" \ + || ok "a workflow that exports in the same job is clean" +# git identity: unset -> WARN (a BYO developer could not sign a commit or attribute it) but still READY. +out="$(cd "$TMP" && env GIT_CONFIG_GLOBAL=/dev/null GIT_CONFIG_SYSTEM=/dev/null HOME="$TMP" python3 "$DOC" "$ROOT/.asdd.example.yml" 2>&1)" +grep -qi "git identity is not set" <<<"$out" && grep -q "RESULT: READY" <<<"$out" \ + && ok "git identity unset WARNs but stays READY" || bad "git-identity WARN missing" + [ "$fail" = "0" ] && { echo "doctor self-test: PASS"; exit 0; } || { echo "doctor self-test: FAIL"; exit 1; } diff --git a/cli/init.sh b/cli/init.sh index 6b783ce..8723b1f 100755 --- a/cli/init.sh +++ b/cli/init.sh @@ -78,6 +78,12 @@ echo "ASDD - scaffolding governance into: $TARGET" step "1. Constitution (AGENTS.md)" copy "$SELF/AGENTS.md" "$TARGET/AGENTS.md" say "edit the sections marked (adapt); keep the (fixed) ones." +# Editor pointers: the developer is bring-your-own, so hand the common assistants the constitution the way +# each one loads context. Thin pointers to AGENTS.md; skip-if-exists never clobbers an existing rule file. +# Codex and other AGENTS.md-convention tools read AGENTS.md directly, so they need no separate pointer. +copy "$SELF/cli/templates/editors/CLAUDE.md" "$TARGET/CLAUDE.md" +copy "$SELF/cli/templates/editors/cursor-asdd.mdc" "$TARGET/.cursor/rules/asdd.mdc" +say "a contributor's Claude Code / Cursor / Codex picks up AGENTS.md automatically (docs/guides/bring-your-own-developer.md)." step "2. Config (.asdd.yml)" # Rewrite the example's 3-line header so the generated config reads as this repo's @@ -128,10 +134,10 @@ step "5. The gates (intake + review pipeline) + the dashboard" # This IS the reference implementation. It lives in this repo's own .github/, so an adopter gets exactly # what ASDD runs on itself: the deterministic intake gate, and the model review pipeline (dry-run until a # model is wired). Read-only analysis is split from the write-scoped publish job (the security invariant). -for w in asdd-intake.yml asdd-intake-feedback.yml pr-review.yml pr-review-publish.yml; do +for w in asdd-intake.yml asdd-intake-feedback.yml pr-review.yml pr-review-publish.yml asdd-preflight.yml; do copy "$SELF/.github/workflows/$w" "$TARGET/.github/workflows/$w" done -for s in intake-check.sh owner-override.sh run-review.sh post-review.sh policy-check.sh set-status.sh security_scan.py audit-export.sh; do +for s in intake-check.sh owner-override.sh run-review.sh post-review.sh policy-check.sh set-status.sh security_scan.py audit-export.sh preflight.sh; do copy "$SELF/.github/asdd/$s" "$TARGET/.github/asdd/$s" done for r in generic.sh openai-compat.sh extract-json.py; do diff --git a/cli/init.test.sh b/cli/init.test.sh index 4f411d9..05fe940 100755 --- a/cli/init.test.sh +++ b/cli/init.test.sh @@ -99,6 +99,20 @@ if bash "$ROOT/cli/init.sh" "$P" >/dev/null 2>&1; then for a in review-code review-security review-spec review-impact review-quality; do [ -f "$P/.github/asdd/agents/$a.md" ] || { echo "FAIL: plain init did not copy the $a lens doc the runtime runs"; fail=1; } done + # The deterministic preflight gate is a workflow+script pair like the others: both must land, or the + # copied asdd-preflight.yml calls a preflight.sh that is not there. + [ -f "$P/.github/workflows/asdd-preflight.yml" ] || { echo "FAIL: plain init did not copy asdd-preflight.yml"; fail=1; } + [ -f "$P/.github/asdd/preflight.sh" ] || { echo "FAIL: plain init copied the preflight workflow but not preflight.sh it calls"; fail=1; } + # Editor pointers ship in base (the developer is bring-your-own in every profile): the common assistants + # must find AGENTS.md through the pointer for their tool, and each pointer must actually reference it. + [ -f "$P/CLAUDE.md" ] || { echo "FAIL: plain init did not write the CLAUDE.md pointer"; fail=1; } + [ -f "$P/.cursor/rules/asdd.mdc" ] || { echo "FAIL: plain init did not write the Cursor rule pointer"; fail=1; } + grep -q 'AGENTS.md' "$P/CLAUDE.md" || { echo "FAIL: CLAUDE.md pointer does not reference AGENTS.md"; fail=1; } + grep -q 'AGENTS.md' "$P/.cursor/rules/asdd.mdc" || { echo "FAIL: Cursor pointer does not reference AGENTS.md"; fail=1; } + # skip-if-exists must never clobber a contributor's own rule file. + echo "MINE" > "$P/CLAUDE.md" + bash "$ROOT/cli/init.sh" "$P" >/dev/null 2>&1 + grep -qx "MINE" "$P/CLAUDE.md" || { echo "FAIL: init overwrote an existing CLAUDE.md"; fail=1; } else echo "FAIL: plain init exited non-zero"; fail=1 fi diff --git a/cli/operate-run.py b/cli/operate-run.py index 57cdef4..92642f2 100644 --- a/cli/operate-run.py +++ b/cli/operate-run.py @@ -31,6 +31,7 @@ HERE = os.path.dirname(os.path.realpath(__file__)) AUDIT = os.path.join(HERE, "audit.py") +EXPORT = os.path.normpath(os.path.join(HERE, "..", ".github", "asdd", "audit-export.sh")) # The valid audit roles, from the ledger itself, so the wrapper refuses a bad role BEFORE running a whole # agent only to be unable to record it. A record that cannot be written is exactly the loss this exists to @@ -89,6 +90,21 @@ def emit(role, ledger, instructed_by, result, goose_exit): return rc == 0 +def export_if_configured(ledger): + """Ship the just-written trail to the adopter's sink when a sink credential is present. The wrapper + runs in a TRUSTED context (the operator's own produce loop, or a post-merge job), so holding the sink + credential here is safe, unlike the untrusted PR review which keeps record and export in separate jobs. + audit-export.sh is itself a no-op when audit.sink is none and refuses a public or same-repo sink, so + calling it whenever a token is present cannot leak. No token means no sink is wired for this run; skip + quietly. Never fails the run: an export problem must not cost the operator their agent's outcome.""" + if not os.environ.get("AUDIT_SINK_TOKEN") or not os.path.isfile(EXPORT): + return + try: + subprocess.run(["bash", EXPORT, ledger], check=False) + except Exception: + sys.stderr.write("operate-run: WARNING audit export failed; the record was still written.\n") + + def main(): ap = argparse.ArgumentParser(description="Run a Goose operate agent and record it deterministically.") ap.add_argument("--role", required=True, help="the operate role (test-runner, documentation, ...)") @@ -141,6 +157,7 @@ def main(): ok = emit(a.role, a.ledger, a.instructed_by, result, goose_exit) if not ok: sys.stderr.write("operate-run: WARNING the audit record could not be written.\n") + export_if_configured(a.ledger) kind = "rich" if result else "minimal (no structured result)" sys.stderr.write(f"operate-run: {a.role} recorded ({kind}); goose exit {goose_exit}.\n") # Surface the run's real outcome to the caller (CI, a human), not the emit's. diff --git a/cli/operate-run.test.sh b/cli/operate-run.test.sh index 031a25e..3531cb9 100644 --- a/cli/operate-run.test.sh +++ b/cli/operate-run.test.sh @@ -58,4 +58,17 @@ python3 "$OR" --role test-runner --recipe /dev/null --ledger "$TMP/d.jsonl" --sk python3 "$OR" --role not-a-role --recipe /dev/null --ledger "$TMP/e.jsonl" --skip-goose >/dev/null 2>&1 [ "$?" = "2" ] && ok "an unknown role is rejected up front" || bad "unknown role not rejected" +# 8. Export by trust class: with a sink credential present the wrapper ships the trail (trusted context); +# with none it does not. A throwaway repo layout so operate-run.py resolves a STUB audit-export.sh. +R="$TMP/repo"; mkdir -p "$R/cli" "$R/.github/asdd" +cp "$OR" "$R/cli/operate-run.py"; cp "$(dirname "$OR")/audit.py" "$R/cli/audit.py" +printf '#!/usr/bin/env bash\necho "called $1" >> "%s/exp.log"\n' "$TMP" > "$R/.github/asdd/audit-export.sh" +chmod +x "$R/.github/asdd/audit-export.sh" +echo '{"action":"test-runner.run","verdict":"ok","reasoning":"x"}' > "$R/res.json" +( cd "$R" && AUDIT_SINK_TOKEN=x python3 cli/operate-run.py --role test-runner --recipe /dev/null --skip-goose --result res.json --ledger "$R/l.jsonl" >/dev/null 2>&1 ) +[ -f "$TMP/exp.log" ] && ok "a sink credential triggers the export step" || bad "export not called with a token" +rm -f "$TMP/exp.log" +( cd "$R" && python3 cli/operate-run.py --role test-runner --recipe /dev/null --skip-goose --result res.json --ledger "$R/l2.jsonl" >/dev/null 2>&1 ) +[ -f "$TMP/exp.log" ] && bad "export called without a token" || ok "no export without a sink credential" + [ "$fail" = "0" ] && { echo "operate-run self-test: PASS"; exit 0; } || { echo "operate-run self-test: FAIL"; exit 1; } diff --git a/cli/run-agent.sh b/cli/run-agent.sh index 0bb227c..f5cd8e8 100755 --- a/cli/run-agent.sh +++ b/cli/run-agent.sh @@ -97,6 +97,11 @@ emit_record() { --role "$ROLE" --action "agent.$AGENT.run" \ --authorizing-decision "operator-run (advisory; a human acts on the output)" \ --verdict "$1" --reasoning "operator ran the $AGENT agent on $INPUT ($2)" >/dev/null 2>&1 || true + # Trusted operator-run context: ship the trail to the adopter's sink if one is wired. audit-export.sh + # no-ops on sink:none and refuses a public or same-repo sink, so this cannot leak; never fails the run. + local led="${ASDD_ACTIVITY_LOG:-.asdd-work/audit.jsonl}" + [ -n "${AUDIT_SINK_TOKEN:-}" ] && [ -x "$ROOT/.github/asdd/audit-export.sh" ] \ + && bash "$ROOT/.github/asdd/audit-export.sh" "$led" >/dev/null 2>&1 || true } # Default to the bundled OpenAI-compatible model command when an endpoint is configured. diff --git a/cli/templates/editors/CLAUDE.md b/cli/templates/editors/CLAUDE.md new file mode 100644 index 0000000..f2ac27a --- /dev/null +++ b/cli/templates/editors/CLAUDE.md @@ -0,0 +1,20 @@ +# Contributing to this repository with a coding agent + +This repository is governed by ASDD (Agentic Spec-Driven Development). Before you write code here, +read [AGENTS.md](AGENTS.md): it is the contribution constitution, and it binds every assistant, this +one included. This file is a pointer to it, not a second set of rules. + +The rules that shape a change: + +- **Disclose the agent.** Say in the pull request and the commit trailer that an AI agent helped, and + sign every commit off with `git commit -s`. +- **One lane per pull request.** Tag the change with exactly one lane label (the set is in `.asdd.yml`). +- **Spec first.** A feature or fix points to a spec that says what "done" is; open or reference one + before building. The `chore` lane is the only spec-exempt lane. +- **Tests run on a different model.** You are the developer. The test agents run on a model that must + differ from yours, so your own tests are not the check that decides the change. +- **A human merges.** Agents review and recommend; a named human approves and merges. You do not merge + your own work. + +Turn an idea into a spec by talking it through with the spec agent (`/asdd:spec`), and run `asdd status` +to see where a change stands. Full guide: [bring your own developer](docs/guides/bring-your-own-developer.md). diff --git a/cli/templates/editors/cursor-asdd.mdc b/cli/templates/editors/cursor-asdd.mdc new file mode 100644 index 0000000..796e90e --- /dev/null +++ b/cli/templates/editors/cursor-asdd.mdc @@ -0,0 +1,15 @@ +--- +description: ASDD contribution rules for this repository +alwaysApply: true +--- + +This repository is governed by ASDD (Agentic Spec-Driven Development). Read AGENTS.md, the contribution +constitution, before writing code here. These points are a pointer to it, not a second set of rules. + +- Disclose the agent in the pull request and the commit trailer, and sign every commit off (`git commit -s`). +- Exactly one lane label per pull request; the set is in `.asdd.yml`. +- A feature or fix points to a spec that defines "done"; only the `chore` lane is spec-exempt. +- You are the developer. The test agents run on a different model, so your own tests are not the deciding check. +- A human merges. Agents review and recommend; you do not merge your own work. + +Full guide: docs/guides/bring-your-own-developer.md diff --git a/cli/templates/operate/dev-council.sh b/cli/templates/operate/dev-council.sh index ff5a231..a1039c2 100644 --- a/cli/templates/operate/dev-council.sh +++ b/cli/templates/operate/dev-council.sh @@ -15,4 +15,13 @@ ROOT="$(cd "$(dirname "$0")/../../.." && pwd)" # .github/asdd/operate/ -> repo CHANGE="${1:-}" [ -n "$CHANGE" ] || { echo "usage: dev-council.sh [args...]" >&2; exit 2; } shift || true -exec python3 "$ROOT/cli/dev-council.py" --change "$CHANGE" --root "$ROOT" "$@" +python3 "$ROOT/cli/dev-council.py" --change "$CHANGE" --root "$ROOT" "$@" +rc=$? +# The council records its trail (dev-council.py: role developer, action dev-council.*). Ship it to the +# adopter's sink if one is wired, like every other agent - host-run, trusted context, so holding the sink +# credential here is safe (audit-export no-ops on sink:none and refuses a public or same-repo sink). Same +# ledger expression the council writes to, and the council's own exit code is preserved. +LEDGER="${ASDD_ACTIVITY_LOG:-.asdd-work/audit.jsonl}" +[ -n "${AUDIT_SINK_TOKEN:-}" ] && [ -x "$ROOT/.github/asdd/audit-export.sh" ] \ + && bash "$ROOT/.github/asdd/audit-export.sh" "$LEDGER" >/dev/null 2>&1 || true +exit "$rc" diff --git a/cli/templates/operate/docsync.sh b/cli/templates/operate/docsync.sh index 72101c6..270fe20 100644 --- a/cli/templates/operate/docsync.sh +++ b/cli/templates/operate/docsync.sh @@ -19,6 +19,21 @@ OUT="${2:?docsync: out file required}" REPO_ROOT="$(cd "$(dirname "$0")/../../.." && pwd)" RECIPE="$REPO_ROOT/recipes/documentation.yaml" +# Leave a trail on ANY exit path (a real run, a dry run with no model, or a guard refusal), so this agent +# is never invisible to the ledger and the corpus as it was before. Record the run, then in this trusted +# post-merge context ship it to the sink if one is wired (audit-export no-ops on sink:none and refuses a +# public or same-repo sink). Neither step fails the run. +_trail() { + local led="${ASDD_ACTIVITY_LOG:-.asdd-work/audit.jsonl}" v=completed + [ -f "$OUT" ] && grep -qi 'dry run' "$OUT" 2>/dev/null && v=dry-run + python3 "$REPO_ROOT/cli/audit.py" append --ledger "$led" --role documentation --action documentation.run \ + --authorizing-decision "post-merge documentation agent (trusted)" --verdict "$v" \ + --reasoning "documentation agent ran on ${CHANGE_REF}" >/dev/null 2>&1 || true + [ -n "${AUDIT_SINK_TOKEN:-}" ] && [ -x "$REPO_ROOT/.github/asdd/audit-export.sh" ] \ + && bash "$REPO_ROOT/.github/asdd/audit-export.sh" "$led" >/dev/null 2>&1 || true +} +trap _trail EXIT + # Enforce the operate-agent security classification. This runner is post-merge (trusted), but assert it # so the rule is mechanical: a tool-using recipe is refused on untrusted input (see cli/operate-guard.py). python3 "$REPO_ROOT/cli/operate-guard.py" "$RECIPE" --input trusted \ diff --git a/cli/templates/operate/test.sh b/cli/templates/operate/test.sh index e689b60..c85c976 100644 --- a/cli/templates/operate/test.sh +++ b/cli/templates/operate/test.sh @@ -19,6 +19,21 @@ OUT="${2:?test: out file required}" REPO_ROOT="$(cd "$(dirname "$0")/../../.." && pwd)" RECIPE="$REPO_ROOT/recipes/test-runner.yaml" +# Leave a trail on ANY exit path (a real run, a dry run with no model, or a guard refusal), so this agent +# is never invisible to the ledger and the corpus as it was before. Record the run, then in this trusted +# post-merge context ship it to the sink if one is wired (audit-export no-ops on sink:none and refuses a +# public or same-repo sink). Neither step fails the run. +_trail() { + local led="${ASDD_ACTIVITY_LOG:-.asdd-work/audit.jsonl}" v=completed + [ -f "$OUT" ] && grep -qi 'dry run' "$OUT" 2>/dev/null && v=dry-run + python3 "$REPO_ROOT/cli/audit.py" append --ledger "$led" --role test-runner --action test.run \ + --authorizing-decision "post-merge test agent (trusted)" --verdict "$v" \ + --reasoning "test agent ran on ${CHANGE_REF}" >/dev/null 2>&1 || true + [ -n "${AUDIT_SINK_TOKEN:-}" ] && [ -x "$REPO_ROOT/.github/asdd/audit-export.sh" ] \ + && bash "$REPO_ROOT/.github/asdd/audit-export.sh" "$led" >/dev/null 2>&1 || true +} +trap _trail EXIT + # Enforce the operate-agent security classification. This runner is post-merge (trusted), but assert it # so the rule is mechanical: a tool-using recipe is refused on untrusted input (see cli/operate-guard.py). # This is the guard that keeps the tester off untrusted pre-merge PR content. diff --git a/docs/README.md b/docs/README.md index 0411905..0d35943 100644 --- a/docs/README.md +++ b/docs/README.md @@ -25,6 +25,7 @@ governed, disclosed, secure, and quality-gated. These docs are organised by what - [Deploy ASDD with Goose (end to end)](guides/deploy.md): the full setup, gates + agents + knowledge + interfaces, with a "what you'll need" checklist. Start here for a full deployment. - [Adopt the govern layer](guides/adopt-govern.md): the CI gates that make a project conformant. - [Using ASDD as one person](guides/using-asdd-solo.md): run your agents under their own identity so you, a solo maintainer, can approve and merge their PRs. +- [Bring your own developer](guides/bring-your-own-developer.md): point a contributor's coding assistant at the constitution, and take an idea from plain language to a merged change (no engineering background needed). - [Adopt into a project that already exists](guides/adopt-existing-project.md): the brownfield path. Declare the spec layout, changelog format, impact log and house style your project already has, so the agents conform to them instead of guessing. Gates judge the change, never your existing tree. diff --git a/docs/guides/bring-your-own-developer.md b/docs/guides/bring-your-own-developer.md new file mode 100644 index 0000000..56307d3 --- /dev/null +++ b/docs/guides/bring-your-own-developer.md @@ -0,0 +1,57 @@ +# Bring your own developer + +ASDD does not ship a developer. The agent that writes the code is always the contributor's own: their +coding assistant, or their hands. A deployment never runs a standing developer against your repository, +because the point of the pipeline is to govern what any developer produces, not to be one. + +That leaves a practical question: how does a contributor's assistant learn the rules of this repository? +The answer is that the rules live in the repository as files, so the assistant reads them the same way it +reads the code. + +## Your assistant reads the constitution automatically + +`asdd init` writes [AGENTS.md](../../AGENTS.md), the contribution constitution, and a thin pointer for the +common assistants so each one loads it without any manual setup: + +| Assistant | Reads | Written by `init` | +|-----------|-------|-------------------| +| Claude Code, Claude in the app | `CLAUDE.md` | pointer to `AGENTS.md` | +| Cursor | `.cursor/rules/asdd.mdc` | pointer to `AGENTS.md` | +| Codex, and any assistant following the AGENTS.md convention | `AGENTS.md` | the constitution itself | + +Each pointer is a few lines that hand the assistant the five rules that shape a change (disclose the +agent, one lane per pull request, spec first, tests on a different model, a human merges) and send it to +`AGENTS.md` for the rest. `init` never overwrites one you already have, so an existing `CLAUDE.md` or +Cursor rule set is left alone; delete the pointer if you do not want it. + +You do not have to use one of these assistants. The rules are files and CI, so a contributor on a tool +nobody has heard of is governed exactly the same way. The pointers are a convenience for the common case, +not a requirement. + +## Starting from an idea, not a spec + +A change begins with a spec, but you do not have to write one first. Talk the idea through with the spec +agent (`/asdd:spec`, or the `spec` recipe on the operate kit) and it drafts the spec with you: the +outcome, the scope, the constraints, and how "done" is checked. It parks an idea that is not ready yet +instead of forcing a half-formed one through. This is the path that lets a non-engineer bring a real, +reviewable change: describe what you want in plain language, and the spec agent turns it into the artefact +the pipeline needs. + +## The agents that come with it are free to run + +The developer is yours, but the operate agents (test author, test runner, documentation, interaction) are +provided, and they run on [Goose](https://block.github.io/goose/), which is open source and free. You +bring an API key for a model; you do not buy the harness. The [Goose quickstart](operate-goose.md) includes +a no-keys "prove it runs" check so you can see the loop work before wiring a model. + +## What a contributor actually does + +1. Point your assistant at the repository. It reads `AGENTS.md` through the pointer for your tool. +2. Turn the idea into a spec with the spec agent, or reference an existing one. +3. Build the change and open a pull request: disclose the agent, sign the commits off, one lane label. +4. Intake, then review, run on the pull request. Fix what they flag. +5. A human merges. + +See also: [the constitution](../../AGENTS.md), [slash commands](slash-commands.md), and +[using ASDD solo](using-asdd-solo.md) for giving your agents their own GitHub identity so they open pull +requests you approve. diff --git a/docs/guides/deploy.md b/docs/guides/deploy.md index 1b381bf..fdd5fde 100644 --- a/docs/guides/deploy.md +++ b/docs/guides/deploy.md @@ -57,9 +57,11 @@ asdd init --goose /path/to/your-repo Or run `bash cli/init.sh --goose /path/to/your-repo` from a checkout. -This writes the constitution (`AGENTS.md`), `.asdd.yml`, the PR template and `CODEOWNERS`, the lane -labels, and the operate kit (recipes, the deterministic gates, the `asdd-gates` MCP, the operate-agent -guard, the docsync workflow). Details: [adopt the govern layer](adopt-govern.md) and +This writes the constitution (`AGENTS.md`) and a pointer to it for the common assistants (`CLAUDE.md`, +`.cursor/rules/asdd.mdc`; Codex reads `AGENTS.md` directly), `.asdd.yml`, the PR template and `CODEOWNERS`, +the lane labels, and the operate kit (recipes, the deterministic gates, the `asdd-gates` MCP, the +operate-agent guard, the docsync workflow). A pointer is skipped if the file already exists. Details: +[adopt the govern layer](adopt-govern.md), [bring your own developer](bring-your-own-developer.md), and [operate with Goose](operate-goose.md). ## 3. Turn the gates on (govern) diff --git a/docs/guides/operate-goose.md b/docs/guides/operate-goose.md index 54d65f9..12794de 100644 --- a/docs/guides/operate-goose.md +++ b/docs/guides/operate-goose.md @@ -56,6 +56,7 @@ reachable. `asdd setup` runs the check at the end, and you can re-run it any tim ```bash asdd connect-check # LIVE or NOT CONNECTED, per role. Non-zero until every configured agent is live. +asdd connect-check --json # same result as machine-readable JSON, for a setup script or CI to gate on. ``` Connect a runtime by setting `ASDD_MODEL_URL` (variable) + `ASDD_RUNTIME_TOKEN` (secret), or the per-role / diff --git a/docs/reference/README.md b/docs/reference/README.md index c9e4a15..a27b090 100644 --- a/docs/reference/README.md +++ b/docs/reference/README.md @@ -34,6 +34,7 @@ The role definitions the operate layer runs. See [agents/](https://github.com/On ## The CLI and gates - [cli/README.md](https://github.com/OneHillAI/ASDD/blob/main/cli/README.md): the `asdd` unified CLI and every deterministic gate - (`spec-check`, `openspec-gate`, `claim-check`, `merge-eligibility`, `check-models`, `audit` (append / - verify / tip / graft / trail / corpus / knowledge as OKGF pages), `audit-ship`, `operate-run`, `run-agent`, `dev-council`, `audit-check`, `conventions-check`, `doctor`, `workflow-lint`, + (`spec-check`, `openspec-gate`, `claim-check`, `merge-eligibility`, `check-models`, `connect-check`, + `audit` (append / verify / tip / graft / trail / corpus / knowledge as OKGF pages), `audit-ship`, + `operate-run`, `run-agent`, `dev-council`, `audit-check`, `conventions-check`, `doctor`, `workflow-lint`, `asdd-mcp`). diff --git a/docs/specs/agent-trajectory-capture.md b/docs/specs/agent-trajectory-capture.md new file mode 100644 index 0000000..7b0023f --- /dev/null +++ b/docs/specs/agent-trajectory-capture.md @@ -0,0 +1,39 @@ +# Spec: agent trajectory capture (the data-generation layer) + +Builtin-format mirror of the OpenSpec change +[`agent-trajectory-capture`](../../openspec/changes/agent-trajectory-capture/). The change's `proposal.md` +and `specs/trajectory-capture/spec.md` carry the full requirements and scenarios; this file satisfies the +repo's builtin spec gate and states the problem, requirements, and acceptance in brief. + +## Problem + +Every governed change runs agents that read an input and produce an output, but only a content-safe digest +reaches the ledger, so the full labelled trajectories, the highest-value artefact an agentic organisation +produces, are discarded. An organisation pursuing software-development sovereignty should own that data: a +knowledge base of how the project builds, and, when it chooses, the training data to tune its own open +models. Value compounds, so capturing from day one is strictly better than deciding later. + +## Requirements + +1. An opt-in capture store, separate from the ledger, holding the full input and output of each agent call; + deployment-owned, never public (reusing the ledger export's public/same-repo refusals); the ledger keeps + its digest-only content-safety; disabled means no behaviour change. +2. A per-record schema: role, model, `model_class` (frontier|open), input, output, change context (id, PR, + lane, spec), content hash, outcome label. +3. Outcome labelling from governance signals: start `pending`, `positive` on a clean merge, `negative` when + a defect names the change via `Escaped-from: #N`, never a false label from an absent signal. +4. Two derived views from the one store: knowledge (wiki/OKGF pages) and corpus (labelled input-output + pairs grouped by role and `model_class`, exportable as training records). +5. Opt-in via a `.asdd.yml` `capture:` block, off by default, within the deployment's privacy boundary, + capturing regardless of provider or host; the docs state the corpus accrues only from when enabled. + +Non-goals: no training or tuning in this change (capture, label, export only); no change to the ledger's +content-safety; no mandated storage backend (a path by default). + +## Acceptance criteria + +- Disabled is inert; an enabled capture writes the full record with `model_class`. +- The label transitions `pending` to `positive` to `negative` on the real signals and never fabricates one. +- Knowledge and corpus both derive from the same capture store. +- The capture store refuses a public or same-repo destination. +- An export can select only `model_class: open` rows without hand-filtering. diff --git a/openspec/changes/agent-trajectory-capture/.openspec.yaml b/openspec/changes/agent-trajectory-capture/.openspec.yaml new file mode 100644 index 0000000..5e6d53a --- /dev/null +++ b/openspec/changes/agent-trajectory-capture/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-24 diff --git a/openspec/changes/agent-trajectory-capture/proposal.md b/openspec/changes/agent-trajectory-capture/proposal.md new file mode 100644 index 0000000..2795891 --- /dev/null +++ b/openspec/changes/agent-trajectory-capture/proposal.md @@ -0,0 +1,57 @@ +## Why + +Every governed change already runs agents that each read an input and produce an output: a review, a test, +a doc, a synthesis. The audit ledger records that these happened, but only a content-safe digest reaches +it, so the full labelled trajectories, the input paired with the output and the outcome, are discarded at +run end. That trajectory set is the highest-value artefact an agentic software organisation produces. + +An organisation pursuing software-development sovereignty should own it. Kept, it is a knowledge base of +how the project actually builds, and, when the organisation chooses, the training data to tune its own +open models and reduce dependence on frontier vendors. The value compounds: every change from day one is a +labelled example, so capturing earlier is strictly better than deciding to capture later. Capture now, +decide training later. + +This is the natural companion to the completed audit-trail work: that made the content-safe DIGEST reach +the sink from every agent; this captures the FULL input and output, into a separate, deployment-owned, +never-published store, so the training corpus is possible without weakening the ledger's content-safety. + +## What Changes + +Add an opt-in **capture** layer, distinct from the governed ledger and off by default. + +1. **A capture sink separate from the ledger.** The ledger stays thin (digest plus rationale, shareable); + the capture sink holds the full input and output per agent call, on a deployment-owned path, never + published. The two are separate stores with separate rules. +2. **A per-record schema:** role, model, `model_class` (frontier or open), input, output, the change + context (change id, PR, lane, spec), a content hash, and an outcome label. +3. **Outcome labelling from signals governance already produces.** A record starts `pending`; it becomes + `positive` when its change merges clean, and `negative` when a later defect names its change through the + escaped-defect convention (`Escaped-from: #N`). No new human labelling. +4. **Two derived views from the one capture:** a **knowledge** view (wiki / OKGF pages, extending the + existing synthesis-to-exemplar and rejected-to-rejected derivation, now backed by the full record) and + a **corpus** view (labelled input-output pairs grouped by role and `model_class`, exportable as training + records). +5. **Config:** a `capture:` block in `.asdd.yml` (enable, path, roles, redaction, retention), off by + default. The docs state plainly that the corpus only accrues from when it is enabled, so an adopter + turns it on at adoption rather than later. +6. **Provider- and host-agnostic, and sovereign.** Capture happens on the ASDD side regardless of where + the model runs, and `model_class` separates the freely-trainable OPEN rows from the FRONTIER rows a + vendor's terms may restrict (distilling closed-frontier outputs into open models is a vendor-terms grey + area; open-to-open is clean). + +## Non-Goals + +- **No training or tuning in this change.** Capture, label, and export only. Whether and how to train is a + separate, later decision the captured corpus makes possible. +- **No change to the ledger's content-safety.** The ledger keeps its digest-only rule; capture is a + separate, private store with its own boundary. +- **No mandated storage backend.** A filesystem path by default; a deployment may point capture elsewhere. + +## Impact + +- Affected specs: new capability `trajectory-capture`. +- Affected code (later implementation, not this change): the record path (`cli/audit.py` / the runners) to + also write a full-fidelity capture record when `capture:` is enabled; `cli/audit.py corpus` and + `knowledge` to read the capture store for the two views; `.asdd.yml` schema for the `capture:` block. +- New config: the `capture:` block; reuses the existing `Escaped-from: #N` outcome signal. +- Privacy: the capture store is deployment-owned and never published; disabled means no behaviour change. diff --git a/openspec/changes/agent-trajectory-capture/specs/trajectory-capture/spec.md b/openspec/changes/agent-trajectory-capture/specs/trajectory-capture/spec.md new file mode 100644 index 0000000..6082e23 --- /dev/null +++ b/openspec/changes/agent-trajectory-capture/specs/trajectory-capture/spec.md @@ -0,0 +1,69 @@ +## ADDED Requirements + +### Requirement: A full-fidelity capture sink distinct from the ledger +The system MUST support an opt-in capture store, separate from the governed audit ledger, that holds the +full input and output of each agent call. It MUST be deployment-owned and never published: it MUST refuse +a public destination and MUST NOT be written to the governed repository, reusing the ledger export's +refusals. The governed ledger MUST keep its digest-only content-safety unchanged; capture is a separate +store with its own boundary. When capture is disabled there MUST be no behaviour change. + +#### Scenario: Capture is separate from and does not weaken the ledger +- **WHEN** capture is enabled +- **THEN** the full input and output are written to the capture store, and the ledger still receives only + its digest and rationale +- **AND** the capture store refuses a public or same-repo destination the way the ledger export does + +#### Scenario: Disabled is inert +- **WHEN** no `capture:` block is configured, or it is disabled +- **THEN** no capture record is written and behaviour is exactly as before + +### Requirement: Per-record capture schema with model class +Each capture record MUST carry: the agent role, the model, a `model_class` of `frontier` or `open`, the +input, the output, the change context (change id, PR number, lane, spec reference where available), a +content hash, and an outcome label. `model_class` MUST let an export separate the freely-trainable OPEN +rows from the FRONTIER rows whose provider terms may restrict reuse. + +#### Scenario: A record carries the fields a corpus export needs +- **WHEN** an agent call is captured +- **THEN** its record has role, model, `model_class`, input, output, change context, content hash, and an + outcome label +- **AND** an export can select only `model_class: open` rows for training without hand-filtering + +### Requirement: Outcome labelling from governance signals +The outcome label MUST be derived from signals governance already produces, not new human labelling. A +record MUST start `pending`; it MUST become `positive` when its change merges clean; and it MUST become +`negative` when a later defect names its change through the escaped-defect convention (`Escaped-from: #N`). +A record MUST NOT be given a false label from an absent signal: with no merge and no defect link it stays +`pending`. + +#### Scenario: The label follows the change's real outcome +- **WHEN** a captured change merges with no later escaped-defect link +- **THEN** its capture records are labelled `positive` +- **WHEN** a later defect names the change via `Escaped-from: #N` +- **THEN** its records are relabelled `negative` +- **WHEN** neither signal has occurred yet +- **THEN** the records stay `pending`, never a fabricated label + +### Requirement: Two derived views from one capture +From the one capture store the system MUST derive two views: a **knowledge** view (wiki / OKGF pages, +extending the existing synthesis-to-exemplar and rejected-to-rejected derivation, now backed by the full +record) and a **corpus** view (labelled input-output pairs grouped by role and `model_class`, exportable as +training records). Both MUST read the same capture store; neither view is a second source of truth. + +#### Scenario: Knowledge and corpus both derive from the capture +- **WHEN** the capture store holds labelled records +- **THEN** the knowledge view emits pages and the corpus view emits labelled input-output pairs grouped by + role and `model_class`, both from that one store + +### Requirement: Opt-in, off by default, within the privacy boundary +Capture MUST be opt-in through a `capture:` block in `.asdd.yml` (enable, path, roles, redaction, +retention), off by default. It MUST write only within the deployment's privacy boundary, and MUST capture +the input-output pair regardless of where the model runs (provider- and host-agnostic). The documentation +MUST state plainly that the corpus only accrues from when capture is enabled, so an adopter turns it on at +adoption rather than expecting retroactive history. + +#### Scenario: Off by default, and honest about accrual +- **WHEN** an adopter has not configured `capture:` +- **THEN** nothing is captured +- **WHEN** an adopter enables it +- **THEN** capture begins from that point, the docs having stated the corpus does not include prior runs diff --git a/openspec/changes/agent-trajectory-capture/tasks.md b/openspec/changes/agent-trajectory-capture/tasks.md new file mode 100644 index 0000000..86110c4 --- /dev/null +++ b/openspec/changes/agent-trajectory-capture/tasks.md @@ -0,0 +1,29 @@ +## 1. Config and boundary +- [ ] 1.1 Add the `capture:` block to `.asdd.yml` (enable, path, roles, redaction, retention), off by + default, read with the kit's no-YAML-dependency scan. +- [ ] 1.2 Reuse the ledger export's refusals for the capture store: never a public destination, never the + governed repo. Fail closed on an unverifiable destination. + +## 2. Capture +- [ ] 2.1 On each agent call, when capture is enabled, write a full-fidelity record (role, model, + model_class, input, output, change context, content hash, outcome label) to the capture store, + distinct from the ledger, which keeps its digest-only record. +- [ ] 2.2 Resolve `model_class` (frontier|open) for the record so an export can separate freely-trainable + rows from vendor-restricted ones. + +## 3. Outcome labelling +- [ ] 3.1 Start each record `pending`; relabel `positive` on a clean merge of its change and `negative` + when a defect names the change via `Escaped-from: #N`. Never label from an absent signal. + +## 4. Derived views +- [ ] 4.1 `asdd audit knowledge` reads the capture store (extending synthesis-to-exemplar, + rejected-to-rejected) for pages backed by the full record. +- [ ] 4.2 `asdd audit corpus` emits labelled input-output pairs grouped by role and model_class, as + exportable training records, from the same store. + +## 5. Docs and tests +- [ ] 5.1 Document the `capture:` block, the sink separation from the ledger, the outcome-label rules, and + state plainly that the corpus accrues only from when capture is enabled. +- [ ] 5.2 Tests: disabled is inert; an enabled capture writes the full record with model_class; the label + transitions pending -> positive -> negative on the real signals and never fabricates; the corpus and + knowledge views read the same store; the store refuses a public or same-repo destination. diff --git a/validation/run-base.py b/validation/run-base.py index 30fe939..874c9bf 100755 --- a/validation/run-base.py +++ b/validation/run-base.py @@ -32,6 +32,12 @@ "zero", "the advisory comment is a public artefact and must meet the writing standard"), ("audit-log properties (P1-P6, P9)", ["sh", "validation/audit-check.test.sh"], "zero", "for-all invariants over the audit trail"), + ("audit-export refusals + host derivation", ["bash", ".github/asdd/audit-export.test.sh"], + "zero", "the sink refuses the governed repo (deriving it from the git remote on a host run, not only " + "in CI), a public sink, and a broken chain, so the trail cannot leak"), + ("operate runners leave + export a trail", ["bash", ".github/asdd/operate/runner-trail.test.sh"], + "zero", "test/docsync record on every exit (dry-run, refusal, real run) and export only when a sink " + "credential is present, so no run is invisible and a tokenless run never tries to push"), ("operate-run deterministic emission", ["bash", "cli/operate-run.test.sh"], "zero", "the run wrapper emits exactly one record even when the agent run produced nothing, so a " "provider timeout mid-run cannot silently lose the action"), @@ -77,6 +83,9 @@ "zero", "PRs bucketed by governance stage + releases + contributors into a self-contained HTML page"), ("operate-agent security guard", ["sh", "cli/operate-guard.test.sh"], "zero", "a tool-using recipe is refused on untrusted input; execution-free is allowed"), + ("deterministic preflight gate (F2)", ["bash", ".github/asdd/preflight.test.sh"], + "zero", "the adopter's own test/lint/type command runs as a real blocking gate: a passing suite " + "passes, a broken test fails the check, and an unconfigured preflight is an opt-in no-op"), ("host conventions gate (brownfield)", ["bash", "cli/conventions-check.test.sh"], "zero", "agent output is held to the host project's DECLARED conventions, judging only the change " "and checking style on added lines only, so a mature repo can adopt and ratchet"),