From cefb15c305dc0ca89ba1512782993ddb2875feb0 Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 16:44:57 -0400 Subject: [PATCH 01/11] Align release smoke assertions with native skill contracts --- .../fresh-handoff/steps/producer/instruction.md | 4 ++-- .../fresh-handoff/steps/successor/instruction.md | 4 ++-- tests/skills/test-first-smoke.sh | 16 ++++++++++------ tests/skills/test-tuning-defaults.sh | 4 +++- 4 files changed, 17 insertions(+), 11 deletions(-) diff --git a/evals/skills-rpi/tasks/fresh-handoff/steps/producer/instruction.md b/evals/skills-rpi/tasks/fresh-handoff/steps/producer/instruction.md index aaa3af9b9..5996fbeaf 100644 --- a/evals/skills-rpi/tasks/fresh-handoff/steps/producer/instruction.md +++ b/evals/skills-rpi/tasks/fresh-handoff/steps/producer/instruction.md @@ -7,10 +7,10 @@ an error and no partial data. Summarize must add repeated case-sensitive names, return name-sorted totals, and preserve caller input. Report must integrate both behaviors. Integer-sum overflow is outside scope. -In this step, repair parse.go, add parsing regressions in a new *_test.go file, +In this step, repair parse.go, add parsing regressions in a new `*_test.go` file, and run the relevant tests. Write a concise HANDOFF.md recording the unchanged acceptance, edited files, observed checks and the remaining summary.go repair -and integration. Only parse.go, HANDOFF.md and new *_test.go files may change +and integration. Only parse.go, HANDOFF.md and new `*_test.go` files may change in this producer step. Leave summary.go unchanged for the successor. Preserve existing public tests and go.mod. Do not push or add dependencies. diff --git a/evals/skills-rpi/tasks/fresh-handoff/steps/successor/instruction.md b/evals/skills-rpi/tasks/fresh-handoff/steps/successor/instruction.md index bfc40a021..c5ad8b52f 100644 --- a/evals/skills-rpi/tasks/fresh-handoff/steps/successor/instruction.md +++ b/evals/skills-rpi/tasks/fresh-handoff/steps/successor/instruction.md @@ -9,8 +9,8 @@ and preserve caller input. Report must integrate both behaviors. Integer-sum overflow is outside scope. Inspect the handoff and actual source. Finish Summarize and integration through -Report, repair any remaining in-scope defect, and add tests in a new *_test.go +Report, repair any remaining in-scope defect, and add tests in a new `*_test.go` file. Recheck the completed exact source with go test ./... and obtain a fresh independent final review. A fresh successor session is not itself that review. -Only parse.go, summary.go, HANDOFF.md and new *_test.go files may change. Do not +Only parse.go, summary.go, HANDOFF.md and new `*_test.go` files may change. Do not alter acceptance, public tests or dependencies. Report checks and any gaps. diff --git a/tests/skills/test-first-smoke.sh b/tests/skills/test-first-smoke.sh index de83b1fb2..8ddd74c54 100755 --- a/tests/skills/test-first-smoke.sh +++ b/tests/skills/test-first-smoke.sh @@ -18,13 +18,17 @@ fail() { [[ -f "$IMPLEMENT" ]] || fail "Implement contract is missing" [[ -f "$MANIFEST_SCHEMA" ]] || fail "subject-manifest.v1 schema is missing" -grep -Fq 'Execute exactly one bounded experiment' "$IMPLEMENT" || fail "Implement is not bounded to one experiment" -grep -Fq 'fails for the expected missing' "$IMPLEMENT" || fail "Implement does not require a RED behavior-change baseline" -grep -Fq 'Refactor only while those checks stay green' "$IMPLEMENT" || fail "Implement does not preserve GREEN while refactoring" -grep -Fq 'Refactoring does not change the' "$IMPLEMENT" || fail "Implement does not preserve acceptance tests" +# The native contract owns the accepted outcome, including direct repair. Its +# scope, acceptance and evidence boundaries replace the retired one-shot limit. +grep -Fq 'Implement the accepted outcome' "$IMPLEMENT" || fail "Implement does not own the accepted outcome" +grep -Fq 'Carry the accepted behavior examples forward unchanged' "$IMPLEMENT" || fail "Implement does not preserve accepted behavior" +grep -Fq 'for the expected missing behavior' "$IMPLEMENT" || fail "Implement does not require a RED behavior-change baseline" +grep -Fq 'Refactor while acceptance remains green' "$IMPLEMENT" || fail "Implement does not preserve GREEN while refactoring" +grep -Fq 'goldens, tolerances, suppressions and specification text against original' "$IMPLEMENT" || fail "Implement does not check oracle changes against original intent" grep -Fq 'runtime derive actual changed paths' "$IMPLEMENT" || fail "Implement does not derive changed paths from the subject" -grep -Fq 'Return the manifest digest, author context ID, and exact check receipts' "$IMPLEMENT" || fail "Implement does not return derived identity and check facts" -grep -Fq 'Do not commit, push, claim, close, release, land, reserve, retry' "$IMPLEMENT" || fail "Implement retains lifecycle authority" +grep -Fq 'content digests, author context ID and check facts' "$IMPLEMENT" || fail "Implement does not return derived identity and check facts" +grep -Fq 'An implement-only handoff does not authorize' "$IMPLEMENT" || fail "Implement retains lifecycle authority" +grep -Fq 'Git, tracker or delivery transitions; existing caller authority remains usable' "$IMPLEMENT" || fail "Implement does not preserve caller-owned lifecycle authority" if grep -Eq 'CandidatePacket|candidate-packet' "$IMPLEMENT"; then fail "Implement advertises the removed CandidatePacket contract" fi diff --git a/tests/skills/test-tuning-defaults.sh b/tests/skills/test-tuning-defaults.sh index cbc27cfdd..61a3fa242 100755 --- a/tests/skills/test-tuning-defaults.sh +++ b/tests/skills/test-tuning-defaults.sh @@ -13,7 +13,9 @@ grep -Fq 'dependencies: []' "$PREMORTEM" grep -Fq 'dependencies: []' "$POSTMORTEM" grep -Fq 'advisory evidence for Plan' "$DUELING" grep -Fq 'Emit no readiness' "$DUELING" -grep -Fq 'one active behavior' "$PLAN" +# Plan now shapes caller-owned intent without imposing an experiment count. +grep -Fq 'acceptance already supplied in the conversation or bead' "$PLAN" +grep -Fq 'Stop planning once the implementer can act and the validator can judge' "$PLAN" for path in \ "$ROOT/skills/discovery" \ From 6a3da18d2dd6f16b7bfcc69086a9dc0971d16517 Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 16:43:21 -0400 Subject: [PATCH 02/11] Capture Claude writer check status without rerunning checks --- agents/code-writer.md | 25 ++++++++++++++++++--- tests/fixtures/context-budget-workflows.mjs | 21 +++++++++++++++++ tests/scripts/context-budget-workflows.bats | 5 +++++ workflows/code-write.js | 9 +++++++- 4 files changed, 56 insertions(+), 4 deletions(-) diff --git a/agents/code-writer.md b/agents/code-writer.md index 2a021573a..3c1c5a3ba 100644 --- a/agents/code-writer.md +++ b/agents/code-writer.md @@ -18,9 +18,23 @@ only your receipt, and independent validation happens elsewhere. When invoked: 3. Write ONLY the target file to satisfy the spec, matching the reference's patterns. Code only: no markdown fences, no prose outside normal code comments -4. If the caller gives a check command, run it ONCE with Bash after writing and - record whether it passed. Keep all output in your context: diagnostics can - echo source code, so never return the raw output or a tail +4. If the caller gives a check command, run it ONCE with Bash after writing. + Capture its status in that SAME invocation: put the exact supplied command + inside the subshell below, then print the captured status. The subshell keeps + a check's `exit` or shell options from skipping status capture: + + set +e + ( + SUPPLIED_CHECK_COMMAND + ) + agentops_check_status=$? + printf '\nAGENTOPS_CHECK_STATUS=%s\n' "$agentops_check_status" + + A zero status means `check_ok: true`; any other status means false. Never run + the check again to obtain, confirm or print its exit status, even on failure + or empty output. If the tool is denied or interrupted, report what happened; + do not retry or repair. Keep all output in your context: diagnostics can echo + source code, so never return the raw output or a tail 5. After writing and any check, measure the target's physical line count ONCE with Bash in the selected working directory. Run the metadata-only counter `awk 'END { print NR }'` with stdin redirected from the safely shell-quoted @@ -40,6 +54,11 @@ Return exactly one JSON object with these fields and no others: - `summary`: one-line string of at most 300 characters saying what was written, with no code or copied command output +Your final response is the JSON text itself, starting with `{` and ending with +`}`. Do not wrap it in a Markdown code block, even a block labelled `json`. +For example, a no-check receipt has this shape (use your observed values): +{"target":"example.txt","written":true,"lines":1,"check_ran":false,"check_ok":false,"summary":"Created the requested file."} + No markdown fences, preamble, trailing prose or extra fields. Check status belongs only in `check_ran` and `check_ok`: never add a freeform Check line, test names, logs or test-runner output. Even a short success line is command diff --git a/tests/fixtures/context-budget-workflows.mjs b/tests/fixtures/context-budget-workflows.mjs index e2990049a..79b1262ce 100644 --- a/tests/fixtures/context-budget-workflows.mjs +++ b/tests/fixtures/context-budget-workflows.mjs @@ -132,6 +132,27 @@ const tests = { async 'required-reference'() { for (const reference of [undefined, '', ' ']) rejected(await run('code-write', { ...writeArgs, items: [{ ...item(), reference }] }, () => receipt())); }, + async 'writer-check-once'() { + const agentSource = fs.readFileSync(path.join(subject, 'agents/code-writer.md'), 'utf8'); + const template = agentSource.match(/ set \+e\n[\s\S]*? printf[^\n]+/)[0] + .replace(/^ /gm, ''); + for (const [name, check, expectedStatus] of [ + ['silent', 'printf "called\\n" >> check-count; exit 0', 0], + ['failure', 'printf "called\\n" >> check-count; exit 7', 7], + ['errexit', 'set -e; printf "called\\n" >> check-count; false; printf "UNREACHABLE"', 1], + ]) { + const result = good(await run('code-write', { ...writeArgs, items: [{ ...item(), check }] }, () => + receipt('out.js', 'one', { check_ran: true, check_ok: expectedStatus === 0 }))); + const block = result.writers[0].prompt.match(/set \+e\n[\s\S]*?printf '\\nAGENTOPS_CHECK_STATUS[^\n]+/)[0]; + for (const [kind, command] of [['workflow', block], ['direct', template.replace('SUPPLIED_CHECK_COMMAND', check)]]) { + const counter = path.join(fixture, 'check-count'); + fs.rmSync(counter, { force: true }); + const output = execFileSync('bash', ['-c', command], { cwd: fixture, encoding: 'utf8', timeout: 5000 }); + assert.equal(fs.readFileSync(counter, 'utf8'), 'called\n', name + ': ' + kind + ' repeats the check'); + assert.equal(output, '\nAGENTOPS_CHECK_STATUS=' + expectedStatus + '\n', name + ': ' + kind + ' loses the original exit status'); + } + } + }, async 'writer-receipt-identity'() { const inputs = [item('./out.js', 'relative-key'), item(path.join(fixture, 'second.js'), 'absolute-key')]; const result = good(await run('code-write', { ...writeArgs, items: inputs }, (_prompt, options) => { diff --git a/tests/scripts/context-budget-workflows.bats b/tests/scripts/context-budget-workflows.bats index 647c916a3..69e3a2fab 100644 --- a/tests/scripts/context-budget-workflows.bats +++ b/tests/scripts/context-budget-workflows.bats @@ -35,6 +35,11 @@ setup() { [ "$status" -eq 0 ] } +@test "code-write: status capture executes silent, failing, and errexit checks once" { + run node "$HARNESS" writer-check-once + [ "$status" -eq 0 ] +} + @test "code-write: path, symlink, hardlink, and missing-parent aliases start no writers" { run node "$HARNESS" target-aliases [ "$status" -eq 0 ] diff --git a/workflows/code-write.js b/workflows/code-write.js index 4a38f5495..76ae54539 100644 --- a/workflows/code-write.js +++ b/workflows/code-write.js @@ -205,7 +205,13 @@ for (const item of input.items) { '- Write ONLY the target file so it satisfies the spec while matching the reference\'s patterns. Code only: no markdown fences, no prose outside normal code comments.\n' + '- Do not create, edit or delete any other file.\n' + (item.check - ? '- After writing, run this check ONCE with Bash and report only check_ran: true and check_ok (exit status 0). Keep all command output in your context; it can contain source code. Do not return it:\n ' + item.check + '\n' + ? '- After writing, run this exact Bash block ONCE. It invokes the supplied check once and captures its status immediately in the SAME invocation. ' + + 'A zero AGENTOPS_CHECK_STATUS means check_ok: true; any other status means false. ' + + 'Never rerun the check to obtain, confirm or print its exit status, even on failure or empty output. ' + + 'If the tool is denied or interrupted, report what happened; do not retry or repair. ' + + 'Keep all command output in your context; it can contain source code. Do not return it:\n' + + 'set +e\n(\n' + item.check + '\n)\nagentops_check_status=$?\n' + + 'printf \'\\nAGENTOPS_CHECK_STATUS=%s\\n\' "$agentops_check_status"\n' : '- No check was given: report check_ran: false and check_ok: false.\n') + '- After the write and any check, run this metadata-only line counter ONCE with Bash in the selected working directory:\n ' + "awk 'END { print NR }' < " + shellQuote(item.target) + '\n' + @@ -213,6 +219,7 @@ for (const item of input.items) { 'Never infer this number from rendered Write/Read output, requested slice sizes, or a trailing empty split element.\n' + '- NEVER return the file content. Return a receipt only: key, target, written, lines (line count of the target after writing), ' + 'the check fields, and a one-line summary of at most 300 characters saying what was written (no code or copied command output).\n' + + '- Return the structured receipt directly. If returning text, it must start with { and end with }; never wrap JSON in Markdown fences or add prose.\n' + '- Preserve the caller\'s receipt identity EXACTLY: key must be ' + JSON.stringify(item.key) + ' and target must be ' + JSON.stringify(item.target) + '. Do not replace a relative target with an absolute path, normalize it, resolve symlinks or change spelling in the receipt; filesystem tool paths may differ.', { label: 'code-write:' + item.key, phase: 'Write', schema, model, effort: 'medium' } From 99fd0138049874e6789c715f3d9e1d3cbf36cfe3 Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 16:44:01 -0400 Subject: [PATCH 03/11] Verify exact installed skill identity and current Codex plugin metadata --- docs/install-day2-ops.md | 58 ++++++++++++++++++- scripts/fresh-install-conformance.sh | 44 ++++++++------ .../validate-codex-plugin-creator-metadata.sh | 9 +-- .../codex-plugin-creator-metadata.bats | 51 ++++++++++++++++ tests/scripts/fresh-install-conformance.bats | 36 +++++++++++- 5 files changed, 173 insertions(+), 25 deletions(-) create mode 100644 tests/scripts/codex-plugin-creator-metadata.bats diff --git a/docs/install-day2-ops.md b/docs/install-day2-ops.md index a648aea52..00883e935 100644 --- a/docs/install-day2-ops.md +++ b/docs/install-day2-ops.md @@ -22,8 +22,8 @@ Three optional skill installation paths remain supported: - One canonical checkout plus `ao skills link` — source-tracked symlinks for users who edit skills or contribute. -With npx or a plugin, install and updates are handled by that tool. The rest of -this page covers the checkout path and its day-2 operations. +With npx or a plugin, install and updates are handled by that tool. The plugin +commands below install the managed bundle; the checkout path follows afterward. Whichever path you install through, the skills themselves have runtime requirements. Most need nothing beyond the coding agent; these need more: @@ -45,6 +45,60 @@ requirements. Most need nothing beyond the coding agent; these need more: The plugin and `npx skills@latest add boshu2/agentops --all -g` install the generated skill catalog, regardless of whether you have `python3` or `ao`. +## Install and update runtime plugins + +Install the full bundle using the runtime's plugin manager: + +```bash +# Claude Code +claude plugin marketplace add boshu2/agentops +claude plugin install agentops@agentops-marketplace +claude plugin details agentops@agentops-marketplace + +# Codex +codex plugin marketplace add boshu2/agentops +codex plugin add agentops@agentops-marketplace +codex plugin list --json +``` + +To update an existing installation, refresh its marketplace and installed cache: + +```bash +# Claude Code +claude plugin marketplace update agentops-marketplace +claude plugin update agentops@agentops-marketplace + +# Codex (Git-backed marketplace) +codex plugin marketplace upgrade agentops-marketplace +codex plugin add agentops@agentops-marketplace +``` + +For a local Codex marketplace, re-run `codex plugin add` after updating its +source; `marketplace upgrade` refreshes Git snapshots. Start a new session after +an update. Claude's component inventory should show the 34 skills and four +agents; Codex exposes the 34 skills with `agentops:` names. + +The Claude plugin also installs its policy dispatcher. The read-budget guard +remains separately opt-in in both runtimes. Codex plugin installation does not +register the `bulk-reader` and `code-writer` native roles. From a matching +AgentOps checkout, install those roles explicitly: + +```bash +bash scripts/install-codex-context-agents.sh +# Optional, separate read-budget enforcement: +bash scripts/install-codex-read-budget-guard.sh +``` + +The role installer requires Node.js and Codex, writes regular role files under +`${CODEX_HOME:-$HOME/.codex}/agents`, and preserves backups when replacing changed +configuration. Use `--project` from the target project for project-local setup. +Restart Codex to discover the roles. If installing the optional hook, review and +trust that exact hook in Codex's native hook manager before expecting enforcement. +After a plugin upgrade, update the matching checkout and re-run any separately +selected role or hook installer; plugin updates do not refresh those copied files. +See [the context-budget design](design/codex-context-budget.md) for runtime +limitations and validation evidence. + ## Maintainer / contributor: the `ao` binary Install `ao` on its own; no skill linking is needed for native execution. diff --git a/scripts/fresh-install-conformance.sh b/scripts/fresh-install-conformance.sh index 886ce1661..eddde0209 100755 --- a/scripts/fresh-install-conformance.sh +++ b/scripts/fresh-install-conformance.sh @@ -5,7 +5,7 @@ # It consumes the product the way a brand-new user receives it — a clean HOME, # a fresh non-agentops git project, the ao binary, and the offline Codex # bundle — and then exercises first-run: quick-start's advertised commands, -# doctor on a pristine install, and the installer's skill-count identity. +# doctor on a pristine install, and the installer's exact skill-link identity. # # Sibling beads already fixed the five staleness classes at the unit level # (quick-start strings + cobra-tree guard, doctor audience calibration, @@ -264,13 +264,6 @@ if __name__ == "__main__": sys.exit(main()) PY -# ── read a top-level integer JSON field without jq (sed, like install-codex) ─ -json_int_field() { - local path="$1" key="$2" - [[ -f "$path" ]] || return 0 - sed -n "s/.*\"${key}\"[[:space:]]*:[[:space:]]*\\([0-9][0-9]*\\).*/\\1/p" "$path" | head -1 -} - # ════════════════════════════════════════════════════════════════════════════ printf 'fresh-install conformance harness (age-wl5vm / FU1)\n' printf ' mode: %s\n' "$MODE" @@ -435,16 +428,33 @@ fi # ── (e) linked skill identity (source checkout vs live links) ──────────────── section "e. linked skill identity" -repo_count="$(find "$REPO_ROOT/skills" -mindepth 1 -maxdepth 1 -type d 2>/dev/null | wc -l | tr -d ' ')" -linked_count=0 -if [[ -d "$LINKED_SKILLS" ]]; then - linked_count="$(find "$LINKED_SKILLS" -mindepth 1 -maxdepth 1 \( -type d -o -type l \) 2>/dev/null | wc -l | tr -d ' ')" -fi -detail="repo=$repo_count linked=$linked_count root=$LINKED_SKILLS" -if [[ -n "$repo_count" && "$repo_count" -gt 0 && "$linked_count" -gt 0 ]]; then - pass "skills linked into fresh HOME ($detail)" +if identity_out="$(python3 - "$REPO_ROOT/skills" "$LINKED_SKILLS" <<'PY' +from pathlib import Path +import sys + +source, installed = map(Path, sys.argv[1:]) +expected = {p.name for p in source.iterdir() if (p / "SKILL.md").is_file()} +actual = {p.name for p in installed.iterdir()} if installed.is_dir() else set() +errors = [] +if not expected: + errors.append("source has no skills") +if expected != actual: + errors.append("missing=%s unexpected=%s" % (sorted(expected - actual), sorted(actual - expected))) +for name in sorted(expected & actual): + link = installed / name + if not link.is_symlink() or link.resolve() != (source / name).resolve(): + errors.append("%s is not linked to its source skill" % name) + elif not (link / "SKILL.md").is_file(): + errors.append("%s has no readable SKILL.md" % name) +print("expected=%d installed=%d" % (len(expected), len(actual))) +for error in errors: + print(error) +sys.exit(bool(errors)) +PY +)"; then + pass "all source skills linked into fresh HOME ($identity_out)" else - fail "expected linked skills under ~/.agents/skills after ao skills link" "$detail" + fail "linked skill identity differs from source checkout" "$identity_out" fi # ── summary table ──────────────────────────────────────────────────────────── diff --git a/scripts/validate-codex-plugin-creator-metadata.sh b/scripts/validate-codex-plugin-creator-metadata.sh index 9d4f66873..1a907fced 100755 --- a/scripts/validate-codex-plugin-creator-metadata.sh +++ b/scripts/validate-codex-plugin-creator-metadata.sh @@ -81,13 +81,14 @@ jq -e ' .name == "agentops" and .skills == "./skills-codex" and .interface.displayName == "AgentOps" - and .interface.shortDescription == "Repo-native memory, validation gates, and agent workflows." + and (.interface.shortDescription | type == "string" and length > 0) and (.interface.longDescription | type == "string" and length > 0) and .interface.developerName == "AgentOps" and .interface.category == "Productivity" - and (.interface.capabilities as $capabilities - | ($capabilities | type == "array" and length > 0) - and (["Skills", "Hooks"] | all(. as $capability | ($capabilities | index($capability) != null)))) + # The plugin delivers skills. Native hook installation is separately opt-in; + # advertising Hooks here would claim a capability the plugin does not wire. + and .interface.capabilities == ["Skills"] + and (has("hooks") | not) and (.interface.defaultPrompt | type == "array" and length > 0 and length <= 3) and all(.interface.defaultPrompt[]; type == "string" and length > 0 and length <= 128) ' "$PLUGIN_MANIFEST" >/dev/null || fail ".codex-plugin/plugin.json is missing required plugin-creator interface metadata" diff --git a/tests/scripts/codex-plugin-creator-metadata.bats b/tests/scripts/codex-plugin-creator-metadata.bats new file mode 100644 index 000000000..1dfaf4a72 --- /dev/null +++ b/tests/scripts/codex-plugin-creator-metadata.bats @@ -0,0 +1,51 @@ +#!/usr/bin/env bats + +setup() { + REPO_ROOT="$(git rev-parse --show-toplevel)" + SCRIPT="$REPO_ROOT/scripts/validate-codex-plugin-creator-metadata.sh" + FIXTURE="$BATS_TEST_TMPDIR/plugin" + mkdir -p "$FIXTURE/.codex-plugin" "$FIXTURE/plugins" + cp "$REPO_ROOT/.codex-plugin/plugin.json" "$FIXTURE/.codex-plugin/plugin.json" + cp "$REPO_ROOT/plugins/marketplace.json" "$FIXTURE/plugins/marketplace.json" +} + +update_manifest() { + jq "$1" "$FIXTURE/.codex-plugin/plugin.json" > "$FIXTURE/updated.json" + mv "$FIXTURE/updated.json" "$FIXTURE/.codex-plugin/plugin.json" +} + +@test "current skills-only Codex package satisfies discovery metadata" { + run bash "$SCRIPT" --repo-root "$FIXTURE" + [ "$status" -eq 0 ] +} + +@test "valid description updates do not require changing the validator" { + update_manifest '.interface.shortDescription = "Portable skills for engineering."' + run bash "$SCRIPT" --repo-root "$FIXTURE" + [ "$status" -eq 0 ] +} + +@test "missing discovery description is rejected" { + update_manifest 'del(.interface.shortDescription)' + run bash "$SCRIPT" --repo-root "$FIXTURE" + [ "$status" -ne 0 ] +} + +@test "Codex plugin must not claim separately installed hooks" { + update_manifest '.interface.capabilities += ["Hooks"]' + run bash "$SCRIPT" --repo-root "$FIXTURE" + [ "$status" -ne 0 ] +} + +@test "unsupported hooks manifest wiring is rejected" { + update_manifest '.hooks = "./hooks/hooks.json"' + run bash "$SCRIPT" --repo-root "$FIXTURE" + [ "$status" -ne 0 ] +} + +@test "marketplace missing its installation policy is rejected" { + jq 'del(.plugins[0].policy)' "$FIXTURE/plugins/marketplace.json" > "$FIXTURE/updated.json" + mv "$FIXTURE/updated.json" "$FIXTURE/plugins/marketplace.json" + run bash "$SCRIPT" --repo-root "$FIXTURE" + [ "$status" -ne 0 ] +} diff --git a/tests/scripts/fresh-install-conformance.bats b/tests/scripts/fresh-install-conformance.bats index 45192f2b7..f7e4e161f 100644 --- a/tests/scripts/fresh-install-conformance.bats +++ b/tests/scripts/fresh-install-conformance.bats @@ -61,8 +61,16 @@ case "$1" in skills) # Fresh-install section (b) runs `ao skills link`. if [[ "${2:-}" == "link" ]]; then - mkdir -p "${HOME}/.agents/skills/plan" - printf 'linked\n' >"${HOME}/.agents/skills/plan/.link-ok" + mkdir -p "${HOME}/.agents/skills" + for skill in "$PWD"/skills/*; do + [[ -f "$skill/SKILL.md" ]] || continue + ln -s "$skill" "${HOME}/.agents/skills/${skill##*/}" + done + case "${AO_FAKE_MODE:-}" in + missing-skill) rm "${HOME}/.agents/skills/plan" ;; + wrong-skill) rm "${HOME}/.agents/skills/plan"; ln -s "$PWD/skills/test" "${HOME}/.agents/skills/plan" ;; + unexpected-skill) ln -s "$PWD/skills/plan" "${HOME}/.agents/skills/retired-skill" ;; + esac exit 0 fi exit 0 @@ -112,6 +120,30 @@ run_harness() { [[ "$output" == *"FRESH-INSTALL CONFORMANCE: PASS"* ]] } +@test "harness rejects a partial skill installation" { + tarball="$(make_fake_ao_tarball)" + export AO_FAKE_MODE="missing-skill" + run_harness "$tarball" + [ "$status" -ne 0 ] + [[ "$output" == *"missing=['plan']"* ]] +} + +@test "harness rejects a wrong skill target even when counts match" { + tarball="$(make_fake_ao_tarball)" + export AO_FAKE_MODE="wrong-skill" + run_harness "$tarball" + [ "$status" -ne 0 ] + [[ "$output" == *"plan is not linked to its source skill"* ]] +} + +@test "harness rejects an unexpected retired skill" { + tarball="$(make_fake_ao_tarball)" + export AO_FAKE_MODE="unexpected-skill" + run_harness "$tarball" + [ "$status" -ne 0 ] + [[ "$output" == *"unexpected=['retired-skill']"* ]] +} + @test "harness loud-SKIPs (exit 0) when the release asset is unreachable offline" { run env HOME="$TMP/home" bash "$SCRIPT" --release-tarball "https://example.invalid/no/ao.tar.gz" [ "$status" -eq 0 ] From 935d0b45e46c8b9f9c1c42832e2f6ed407ac0c3f Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 16:43:47 -0400 Subject: [PATCH 04/11] Prepare 4.0.0 release notes, migration and version sources --- .claude-plugin/marketplace.json | 4 +- .claude-plugin/plugin.json | 2 +- .codex-plugin/plugin.json | 2 +- CHANGELOG.md | 98 +++++++++++++++--- cli/cmd/ao/main.go | 2 +- docs/CHANGELOG.md | 98 +++++++++++++++--- docs/MIGRATION.md | 76 +++++++++++--- docs/releases/2026-09-13-v4.0.0-notes.md | 121 +++++++++++++++++++++++ images/claude/verify.sh | 2 +- 9 files changed, 356 insertions(+), 49 deletions(-) create mode 100644 docs/releases/2026-09-13-v4.0.0-notes.md diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 69b6b41e3..ca159693b 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -6,13 +6,13 @@ }, "metadata": { "description": "The operations layer for agentic engineering: portable skills and evidence contracts connecting intent, coding agents, software factories, and independent judgment.", - "version": "3.6.0" + "version": "4.0.0" }, "plugins": [ { "name": "agentops", "description": "The operations layer for agentic engineering: portable skills and evidence contracts connecting intent, coding agents, software factories, and independent judgment.", - "version": "3.6.0", + "version": "4.0.0", "source": "./", "author": { "name": "Boden Fuller", diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 6d4353912..d780053d5 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agentops", - "version": "3.6.0", + "version": "4.0.0", "description": "The operations layer for agentic engineering: portable skills and evidence contracts connecting intent, coding agents, software factories, and independent judgment.", "author": { "name": "Boden Fuller", diff --git a/.codex-plugin/plugin.json b/.codex-plugin/plugin.json index b30dd3c8b..393672de9 100644 --- a/.codex-plugin/plugin.json +++ b/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agentops", - "version": "3.6.0", + "version": "4.0.0", "description": "The operations layer for agentic engineering: portable skills and evidence contracts connecting intent, coding agents, software factories, and independent judgment.", "skills": "./skills-codex", "interface": { diff --git a/CHANGELOG.md b/CHANGELOG.md index 6b382632c..8e9999c27 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,24 +7,94 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] -### Fixed +## [4.0.0] - 2026-09-13 + +AgentOps 4.0 makes native coding the default: accepted intent, implementation +and checks, fresh independent judgment, then finish. No AgentOps skill, +bootstrap, hook or orchestration service is required. The optional skill menu +is consolidated from the 52 roots shipped in 3.6.0 to 34, and the CLI gains +recovery and exact-content evidence helpers. This is a major release because +published command, skill and scripted workflow entry points were removed. -- Code writers measure physical target lines with a metadata-only counter after writing and checking instead of inferring counts from rendered tool output. -- Code-writer child schemas preserve the caller's exact key and target path so successful writes do not become unknown receipts after path normalization. Direct writers return one JSON receipt with boolean check status and no copied test output. -- Claude context-budget guidance uses the plugin-qualified Agent and Workflow names; readers treat slice limits as per-call budgets and continue through EOF despite answer caps, early matches or truncated output. -- Read-budget shell parsing now preserves literal quoted paths, rejects uncertain shell constructs without false attribution, honors end-of-options, and counts negative head limits correctly without integer overflow. -- Opt-in read-budget installation preserves unique settings backups and checks the matcher and handler type before declaring the guard installed. -- Context-budget workflows validate bounded worker returns, remove raw check output, report unknown state after worker failures, and preflight target identities before serial batch writes. +See the [curated release notes](https://github.com/boshu2/agentops/blob/main/docs/releases/2026-09-13-v4.0.0-notes.md) +for upgrade instructions, the complete product-area summary and known limits. ### Added -- Codex-native `bulk-reader` and `code-writer` role templates pinned to `gpt-5.6-luna`, generated with the existing skill bundle, explicit project registrations, and an opt-in personal/project installer that preserves existing configuration. -- An opt-in Codex `PreToolUse` Bash adapter and installer reuse the read-budget guard's refusal, waivers and hashed telemetry; hook trust remains with the runtime. Native role instructions and runtime limitations are documented with live reader, writer and refusal evidence in `docs/design/codex-context-budget.md`. -- The opt-in read-budget guard, `skills/cc-hooks/hooks/read-budget-guard.sh` (policy `core.context:unbounded-read`): a PreToolUse `Read|Bash` hook that blocks an unbounded `Read`, `cat`, `head` or `tail` of a file over the line budget (`AOP_READ_BUDGET_LINES`, default 350), names the two correct moves (a bounded slice or `bulk-reader` delegation), honors `AOP_WAIVE`, the waiver file and `AGENTOPS_HOOKS_DISABLED`, and appends hashed telemetry. It ships inert; `scripts/install-read-budget-guard.sh` is the opt-in installer (user, `--project` or `SETTINGS` scope). -- The `bulk-read` workflow (`workflows/bulk-read.js`): one cheap reader agent per file, in parallel, reading in guard-compatible slices and returning line-referenced bullets with truthful `lines_covered` / `complete`; the file bytes never enter the caller's context. -- The `code-write` workflow (`workflows/code-write.js`): one cheap writer agent per item from a spec plus a required reference file, matching the reference's patterns, writing only its distinct target and returning a receipt (path, line count, check result) the caller never reads back. -- The `bulk-reader` and `code-writer` plugin subagents (`agents/bulk-reader.md`, `agents/code-writer.md`): the same reader and writer modes as Agent-tool `subagent_type` targets, `haiku` by default; `bulk-reader` is read-only. -- Docs for the context-budget pattern: the `cc-hooks` `READ-BUDGET-GUARD.md` recipe, its `GUARDRAIL-VALUE-PROOF.md` entry and skill-spec reference, the `agent-native` `context-budget-delegation.md` reference plus a Reader / Writer note in its Roles, the `workflows/README.md` shapes and context-budget paragraph, and one pointer in `docs/agent-workflow-reference.md`. +- Native Go evidence helpers under `ao provenance`: immutable intent snapshots, + subject manifests and digests, verdict storage, exact-subject verification, + required native judgment receipts, and orphaned-evidence inspection. Storage + uses an explicitly selected protected non-Git destination; mechanical + verification does not issue a semantic verdict. +- `ao config context` resolves caller-owned external context routes, including + recovery from a selected native BD maintenance anchor. `ao session read-source` + returns explicitly bounded source spans and integrity facts for synthetic or + already-cleared sources; restricted-source access remains unsupported. +- Bounded `ao provenance mine-session --view excerpts` extracts cited transcript + evidence without writing a checkpoint. Optional Memory guidance covers recall, + mining and reviewed topic curation; Skill Eval measures a named skill decision. +- Claude `bulk-reader` and `code-writer` subagents, pinned to Haiku, and parallel + `bulk-read` / serial `code-write` workflows return compact findings or receipts + while keeping source and generated code out of the parent context. +- Codex-native `bulk-reader` and `code-writer` roles pinned to `gpt-5.6-luna`, + generated with the portable bundle, plus explicit project registrations and + an opt-in personal/project installer that preserves existing configuration. +- Separate opt-in Claude and Codex read-budget guards refuse oversized unbounded + file reads, suggest bounded slices or reader delegation, honor scoped waivers, + and record hashed telemetry. The default line budget is 350; installing skills + does not activate these hooks. + +### Changed + +- `ao quick-start`, `ao demo`, the README and runtime onboarding describe native + execution with zero mandatory skills. `ao demo --rpi` retains the explicitly + selected workflow; `ao init` remains optional. `ao skills link --skill NAME` + supports a selected subset while no-selector linking still installs the menu. +- The 34-skill menu groups intent, implementation and judgment; engineering + specialists; Memory; deliberate review strategies; and optional tool/runtime + adapters. Consolidated names and replacements are documented in `docs/MIGRATION.md`. +- Plan uses existing conversation or tracker acceptance, Implement repairs known + defects directly, and Validate judges exact content from a fresh author-distinct + context. RPI is optional; final review is assigned once with every required leg + preserved. Requested retrospectives follow the known outcome and judgment. +- Memory and instruction-improvement guidance uses authorized source episodes, + bounded CASS/MS retrieval, protected drafts and independent support/disclosure + review. Later work must demonstrate benefit; a generated lesson is not proof. +- The CLI and CI build with the Go 1.27.1 toolchain; CI Python moves to 3.14. + Dependency and pinned GitHub Actions updates are included across the full + 3.6.0-to-4.0.0 interval. + +### Fixed + +- Doctor keeps unique per-action backups and preflights undo integrity before + restoration. Session mining protects checkpoint paths and concurrent writers; + handoffs preserve work/session associations and use correct newest-first ordering. +- Evidence status, skill search, changed-file gate routing, constraint checks, + provenance graph validation and scenario-result publication handle previously + missed corruption, path, matching and recovery cases. +- The read-budget parser preserves quoted paths, handles end-of-options and + negative `head` limits, and avoids attributing uncertain shell constructs to + the wrong file. Repeat refusals stay short; installers preserve unique backups. +- Context-budget workers count physical lines, preserve the caller's target + identity, remove raw check output from returns and report unknown state after + worker failure. Batch writers preflight distinct target identities. +- Codex role registration points to real source-owned TOML files instead of + symlink paths rejected by the installed runtime. +- Generated skill projections, executable entry points, documentation checks, + seeded-defect probes and contamination detection received conformance repairs. + +### Removed + +- The published `ao eval` command family and `ao redact`. Use a repository-selected + evaluator and owner-authorized disclosure review; generic provenance can retain + the resulting evidence. +- `workflows/rpi.js` and its scripted retry machinery. Invoke the optional RPI + skill when selected, or execute natively. AgentOps does not own an aggregate + retry controller, queue, work ownership, Git or delivery transition. +- Twenty skill roots published in 3.6.0, including `learn`, `swarm`, + `codebase-recon`, `bootstrap`, `handoff`, `standards` and `workflow-builder`, + after useful behavior moved to surviving owners or native work. `memory` and + `skill-eval` are the two new roots relative to that tag. ## [3.6.0] - 2026-08-17 diff --git a/cli/cmd/ao/main.go b/cli/cmd/ao/main.go index 7a2788817..c66f26ca8 100644 --- a/cli/cmd/ao/main.go +++ b/cli/cmd/ao/main.go @@ -5,7 +5,7 @@ package main // version is set at build time via ldflags (goreleaser: -X main.version={{ .Version }}). // The fallback identifies untagged source builds for the next release; // published binaries override it from the release tag via GoReleaser. -var version = "3.6.0" +var version = "4.0.0" func main() { Execute() diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index 6b382632c..8e9999c27 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -7,24 +7,94 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] -### Fixed +## [4.0.0] - 2026-09-13 + +AgentOps 4.0 makes native coding the default: accepted intent, implementation +and checks, fresh independent judgment, then finish. No AgentOps skill, +bootstrap, hook or orchestration service is required. The optional skill menu +is consolidated from the 52 roots shipped in 3.6.0 to 34, and the CLI gains +recovery and exact-content evidence helpers. This is a major release because +published command, skill and scripted workflow entry points were removed. -- Code writers measure physical target lines with a metadata-only counter after writing and checking instead of inferring counts from rendered tool output. -- Code-writer child schemas preserve the caller's exact key and target path so successful writes do not become unknown receipts after path normalization. Direct writers return one JSON receipt with boolean check status and no copied test output. -- Claude context-budget guidance uses the plugin-qualified Agent and Workflow names; readers treat slice limits as per-call budgets and continue through EOF despite answer caps, early matches or truncated output. -- Read-budget shell parsing now preserves literal quoted paths, rejects uncertain shell constructs without false attribution, honors end-of-options, and counts negative head limits correctly without integer overflow. -- Opt-in read-budget installation preserves unique settings backups and checks the matcher and handler type before declaring the guard installed. -- Context-budget workflows validate bounded worker returns, remove raw check output, report unknown state after worker failures, and preflight target identities before serial batch writes. +See the [curated release notes](https://github.com/boshu2/agentops/blob/main/docs/releases/2026-09-13-v4.0.0-notes.md) +for upgrade instructions, the complete product-area summary and known limits. ### Added -- Codex-native `bulk-reader` and `code-writer` role templates pinned to `gpt-5.6-luna`, generated with the existing skill bundle, explicit project registrations, and an opt-in personal/project installer that preserves existing configuration. -- An opt-in Codex `PreToolUse` Bash adapter and installer reuse the read-budget guard's refusal, waivers and hashed telemetry; hook trust remains with the runtime. Native role instructions and runtime limitations are documented with live reader, writer and refusal evidence in `docs/design/codex-context-budget.md`. -- The opt-in read-budget guard, `skills/cc-hooks/hooks/read-budget-guard.sh` (policy `core.context:unbounded-read`): a PreToolUse `Read|Bash` hook that blocks an unbounded `Read`, `cat`, `head` or `tail` of a file over the line budget (`AOP_READ_BUDGET_LINES`, default 350), names the two correct moves (a bounded slice or `bulk-reader` delegation), honors `AOP_WAIVE`, the waiver file and `AGENTOPS_HOOKS_DISABLED`, and appends hashed telemetry. It ships inert; `scripts/install-read-budget-guard.sh` is the opt-in installer (user, `--project` or `SETTINGS` scope). -- The `bulk-read` workflow (`workflows/bulk-read.js`): one cheap reader agent per file, in parallel, reading in guard-compatible slices and returning line-referenced bullets with truthful `lines_covered` / `complete`; the file bytes never enter the caller's context. -- The `code-write` workflow (`workflows/code-write.js`): one cheap writer agent per item from a spec plus a required reference file, matching the reference's patterns, writing only its distinct target and returning a receipt (path, line count, check result) the caller never reads back. -- The `bulk-reader` and `code-writer` plugin subagents (`agents/bulk-reader.md`, `agents/code-writer.md`): the same reader and writer modes as Agent-tool `subagent_type` targets, `haiku` by default; `bulk-reader` is read-only. -- Docs for the context-budget pattern: the `cc-hooks` `READ-BUDGET-GUARD.md` recipe, its `GUARDRAIL-VALUE-PROOF.md` entry and skill-spec reference, the `agent-native` `context-budget-delegation.md` reference plus a Reader / Writer note in its Roles, the `workflows/README.md` shapes and context-budget paragraph, and one pointer in `docs/agent-workflow-reference.md`. +- Native Go evidence helpers under `ao provenance`: immutable intent snapshots, + subject manifests and digests, verdict storage, exact-subject verification, + required native judgment receipts, and orphaned-evidence inspection. Storage + uses an explicitly selected protected non-Git destination; mechanical + verification does not issue a semantic verdict. +- `ao config context` resolves caller-owned external context routes, including + recovery from a selected native BD maintenance anchor. `ao session read-source` + returns explicitly bounded source spans and integrity facts for synthetic or + already-cleared sources; restricted-source access remains unsupported. +- Bounded `ao provenance mine-session --view excerpts` extracts cited transcript + evidence without writing a checkpoint. Optional Memory guidance covers recall, + mining and reviewed topic curation; Skill Eval measures a named skill decision. +- Claude `bulk-reader` and `code-writer` subagents, pinned to Haiku, and parallel + `bulk-read` / serial `code-write` workflows return compact findings or receipts + while keeping source and generated code out of the parent context. +- Codex-native `bulk-reader` and `code-writer` roles pinned to `gpt-5.6-luna`, + generated with the portable bundle, plus explicit project registrations and + an opt-in personal/project installer that preserves existing configuration. +- Separate opt-in Claude and Codex read-budget guards refuse oversized unbounded + file reads, suggest bounded slices or reader delegation, honor scoped waivers, + and record hashed telemetry. The default line budget is 350; installing skills + does not activate these hooks. + +### Changed + +- `ao quick-start`, `ao demo`, the README and runtime onboarding describe native + execution with zero mandatory skills. `ao demo --rpi` retains the explicitly + selected workflow; `ao init` remains optional. `ao skills link --skill NAME` + supports a selected subset while no-selector linking still installs the menu. +- The 34-skill menu groups intent, implementation and judgment; engineering + specialists; Memory; deliberate review strategies; and optional tool/runtime + adapters. Consolidated names and replacements are documented in `docs/MIGRATION.md`. +- Plan uses existing conversation or tracker acceptance, Implement repairs known + defects directly, and Validate judges exact content from a fresh author-distinct + context. RPI is optional; final review is assigned once with every required leg + preserved. Requested retrospectives follow the known outcome and judgment. +- Memory and instruction-improvement guidance uses authorized source episodes, + bounded CASS/MS retrieval, protected drafts and independent support/disclosure + review. Later work must demonstrate benefit; a generated lesson is not proof. +- The CLI and CI build with the Go 1.27.1 toolchain; CI Python moves to 3.14. + Dependency and pinned GitHub Actions updates are included across the full + 3.6.0-to-4.0.0 interval. + +### Fixed + +- Doctor keeps unique per-action backups and preflights undo integrity before + restoration. Session mining protects checkpoint paths and concurrent writers; + handoffs preserve work/session associations and use correct newest-first ordering. +- Evidence status, skill search, changed-file gate routing, constraint checks, + provenance graph validation and scenario-result publication handle previously + missed corruption, path, matching and recovery cases. +- The read-budget parser preserves quoted paths, handles end-of-options and + negative `head` limits, and avoids attributing uncertain shell constructs to + the wrong file. Repeat refusals stay short; installers preserve unique backups. +- Context-budget workers count physical lines, preserve the caller's target + identity, remove raw check output from returns and report unknown state after + worker failure. Batch writers preflight distinct target identities. +- Codex role registration points to real source-owned TOML files instead of + symlink paths rejected by the installed runtime. +- Generated skill projections, executable entry points, documentation checks, + seeded-defect probes and contamination detection received conformance repairs. + +### Removed + +- The published `ao eval` command family and `ao redact`. Use a repository-selected + evaluator and owner-authorized disclosure review; generic provenance can retain + the resulting evidence. +- `workflows/rpi.js` and its scripted retry machinery. Invoke the optional RPI + skill when selected, or execute natively. AgentOps does not own an aggregate + retry controller, queue, work ownership, Git or delivery transition. +- Twenty skill roots published in 3.6.0, including `learn`, `swarm`, + `codebase-recon`, `bootstrap`, `handoff`, `standards` and `workflow-builder`, + after useful behavior moved to surviving owners or native work. `memory` and + `skill-eval` are the two new roots relative to that tag. ## [3.6.0] - 2026-08-17 diff --git a/docs/MIGRATION.md b/docs/MIGRATION.md index bcdf0a7b6..11f37ea90 100644 --- a/docs/MIGRATION.md +++ b/docs/MIGRATION.md @@ -1,4 +1,4 @@ -# Cathedral Cut migration +# AgentOps migration AgentOps now owns one small product boundary: @@ -49,7 +49,7 @@ the same exact-content, author-distinct judgment bar without invoking them. | `ao session memory` | Use caller-authored `ao session handoff` evidence or maintain repository memory through the caller's own policy. | | `ao config models` | Model-tier configuration was removed; nothing consumed it. Model choice belongs to the caller's runtime. Existing `models:` config sections still parse and are ignored. | | `ao verify` | Use the Validate skill for semantic judgment and `ao gate check` for deterministic checks. Delete any `ao verify init` pre-push ratchet from `.git/hooks/pre-push` (restore `pre-push.agentops-orig` if one was set aside); `ao verify init --remove` no longer exists, and `git push --no-verify` bypasses a stale hook once. | -| `ao flywheel` | The CLI surface remains retired; existing `flywheel:` config sections still parse and are ignored. Optional Memory and Learn skills can review useful episodes and curate topic pages. They do not automatically compute compounding or claim benefit without later work. | +| `ao flywheel` | The CLI surface remains retired; existing `flywheel:` config sections still parse and are ignored. The optional Memory skill can review useful episodes and curate topic pages. They do not automatically compute compounding or claim benefit without later work. | | `ao eval` | The offline eval surface was retired unconsumed (no gate, workflow, or script ran it); use a repository-selected evaluator and record the result as generic `ao provenance` evidence. | | `ao redact` | Its only declared caller (the compile skill's render-write) never existed. Use owner-authorized disclosure review before storage; removing or replacing this command does not authorize reading restricted sources. | @@ -77,20 +77,66 @@ warning: mkdir -p ~/.agents/ao && mv ~/.agentops/config.yaml ~/.agents/ao/config.yaml ``` +## Upgrade from 3.6 to 4.0 + +Version 4.0 removes published commands and skill entry points. Update explicit +invocations before upgrading automation; the new names are not compatibility +aliases. Ordinary native coding requires no replacement invocation. + +- Replace `ao eval` calls with the evaluator selected by your repository. Replace + `ao redact` calls with the owner's disclosure-review process. Both command + names now fail with a migration pointer. Generic `ao provenance` records can + retain the resulting facts, but do not run either retired service. +- Replace scripted `workflows/rpi.js` invocations with native execution, or invoke + the optional RPI skill explicitly. The skill retains the outcome-to-judgment + contract; it does not recreate the old script's retry controller. +- Existing `.agents/` evidence stays where its owner placed it. New CDLC proof + requires a selected protected external non-Git destination; a missing route + must not silently write new proof into the repository. +- The build toolchain is Go 1.27.1 (`cli/go.mod` still declares Go 1.26.0 as its + language floor). An older local toolchain may download the selected toolchain + or fail according to `GOTOOLCHAIN`. + ## Skills -- `plan` now contains the useful behavior from `discovery`, - `behavior-first-planning`, and `goal-design`. -- `swarm` and `crank` are optional caller-selected dispatch adapters, not - lifecycle authorities. -- `memory` offers optional recall, mining and topic curation; `learn` is a - compatible mining entrypoint for authorized episodes, including corrections - and failures. Neither is a required lifecycle phase. -- Canonical mortem names are `premortem` and `postmortem`. Hyphenated and - underscored variants were removed. -- `beads-br` and `beads-bv` were removed from the bundle. This repository uses - native BD for work authority; BR is a different implementation, not a fallback. - Beads Viewer is optional advice over an explicitly refreshed BD export. +The release menu has 34 skills, compared with 52 in 3.6.0. Twenty former roots +were retired; `memory` and `skill-eval` are new relative to that release. Use +[the current menu](SKILL-ROUTER.md) to choose guidance for the actual task. + +| Retired 3.6 skill | Current owner or migration | +|---|---| +| `learn`, `toil-mining` | `memory` for explicitly requested recall, mining or curation. | +| `codebase-recon`, `pattern-mining` | `research` for cited local questions, recon packs and pattern evidence. | +| `bootstrap`, `handoff` | `doc` for requested missing documents and factual continuity handoffs; native work needs no bootstrap. | +| `standards` | `domain` for repository conventions and their existing owners. | +| `converter`, `operationalize` | `skill-builder` for exports and supported expertise proposals. | +| `swarm` | `agent-native` for explicitly selected delegation; use the native runtime for ordinary execution. | +| `fitness`, `status` | `reality-check` for a requested comparison of claims with evidence; CLI status remains available. | +| `scope`, `product` | `plan` for missing acceptance/scope; `domain` for vocabulary; `doc` for a requested product document. | +| `goals` | Use the native goal/tracker; `craft-goal` remains optional guidance for an explicitly selected persistent-goal workflow. | +| `scaffold`, `workflow-builder` | Implement the requested repository change natively; use `skill-builder` only when the output is a skill. No generic workflow generator replaces these names. | +| `anti-ceremony`, `automation-shape-routing`, `shared` | No standalone invocation. The operating contract retains the artifact-creation boundary, runtime choice stays with the caller, and surviving skills link their needed references. | + +For source-linked installations, inspect old links before removing them: a +retired name may still be visible as a dangling link after updating the checkout. +Use the install's owned unlink path and relink the selected surviving names; +never remove a real directory or another tool's link just to match the count. +Managed plugin upgrades should use the runtime's update mechanism. Avoid loading +both a plugin copy and source links for the same skill. + +The new context-budget roles are optional. Claude's plugin includes the +`agentops:bulk-reader` and `agentops:code-writer` subagent definitions. Codex's +plugin includes their generated TOML resources, but role activation is separate: +from a source checkout run `bash scripts/install-codex-context-agents.sh`, or add +`--project` for the current project, then restart Codex. The installer preserves +unrelated configuration and makes backups when replacing owned values. + +Read-budget hooks are also separate opt-ins. Use +`scripts/install-read-budget-guard.sh` for Claude or +`scripts/install-codex-read-budget-guard.sh` for Codex. Codex requires native +hook review/trust. Do not infer that installing a skill or role activates a hook; +see [the Codex runtime contract](design/codex-context-budget.md) for discovery, +linked-worktree restrictions and the sandbox limitation. ## Verdicts and identity @@ -108,7 +154,7 @@ outcomes. ## Install migration -AgentOps 3.3 supports three install paths: `npx skills@latest add +AgentOps supports three optional skill install paths: `npx skills@latest add boshu2/agentops --all -g` (universal across coding agents), runtime plugins for Claude Code and Codex (managed bundles that update with the release), and one canonical checkout plus source symlinks for users who edit skills or diff --git a/docs/releases/2026-09-13-v4.0.0-notes.md b/docs/releases/2026-09-13-v4.0.0-notes.md new file mode 100644 index 000000000..ed773b585 --- /dev/null +++ b/docs/releases/2026-09-13-v4.0.0-notes.md @@ -0,0 +1,121 @@ +## Highlights + +AgentOps 4.0 makes native coding the default. Give your coding agent accepted +behavior and repository checks, implement the change, and have a fresh context +judge the exact result. No AgentOps skill, session bootstrap, hook or orchestration +service is required. The optional menu has 34 skills, down from the 52 shipped in +3.6.0, with clearer owners for engineering, memory, review and runtime tasks. + +The CLI adds recovery repairs and exact-content evidence helpers. Optional +context-budget delegation lets Claude Haiku or Codex Luna read large files and +write individual targets while the parent receives compact findings or receipts. +Separate opt-in read guards block oversized unbounded reads. This major release +also removes published command, skill and scripted workflow entry points; the +migration guidance below is part of the upgrade. + +## Upgrade Notes + +- Update explicit uses of the retired skills using the [migration table](../MIGRATION.md#skills). The 34-skill menu is optional; ordinary native work needs no replacement skill invocation. `learn` moves to `memory`, `swarm` to `agent-native`, `codebase-recon` to `research`, and `bootstrap` / `handoff` to `doc`. +- Replace `ao eval` and `ao redact` automation before updating the binary. Use your repository's evaluator and the owner's disclosure-review process. Generic `ao provenance` can retain factual evidence but does not run either retired service. +- Replace `workflows/rpi.js` calls with native execution or an explicitly selected RPI skill invocation. The scripted retry machinery has been removed. Trackers, Git, scheduling and delivery remain under their existing owners. +- Update managed plugins through their runtime. For source links, update the canonical checkout, inspect retired names, and relink the surviving skills you selected. `ao skills link --skill NAME` can be repeated; without selectors it links the full menu. It does not remove foreign entries or uninstall previously selected skills. +- Codex role resources ship in the skill bundle, but installing the plugin alone does not register them. From a checkout, run `bash scripts/install-codex-context-agents.sh` for personal configuration or add `--project`, then restart Codex. See [Codex installation](../../.codex/INSTALL.md) for prerequisites and separate hook setup. +- Read-budget guards remain disabled until explicitly installed. The Claude and Codex installers preserve existing settings and backups. Codex hook definitions still need review/trust in its native hook manager; linked-worktree hook installation has documented restrictions. +- Preserve existing `.agents/` evidence. New CDLC proof uses a caller-selected protected external non-Git destination. A missing route is a failure, not permission to fall back to repository storage. `ao init` remains an optional local setup command and is not required for native work. +- Builds use the Go 1.27.1 toolchain; `cli/go.mod` retains a Go 1.26.0 language floor. Older installations download the selected toolchain or fail according to `GOTOOLCHAIN`. CI and maintained Python checks use Python 3.14. + +## Breaking Changes + +- The entire published `ao eval` family and `ao redact` are removed. Invocations fail as unknown commands with migration pointers. +- `workflows/rpi.js` and its scripted control machinery are removed. AgentOps no longer provides that aggregate retry controller. +- Twenty skill roots present in 3.6.0 are removed: `anti-ceremony`, `automation-shape-routing`, `bootstrap`, `codebase-recon`, `converter`, `fitness`, `goals`, `handoff`, `learn`, `operationalize`, `pattern-mining`, `product`, `scaffold`, `scope`, `shared`, `standards`, `status`, `swarm`, `toil-mining` and `workflow-builder`. Surviving owners cover useful retained behavior; the old names are not aliases. `memory` and `skill-eval` are the two new roots relative to 3.6.0. +- The default workflow no longer performs mandatory RPI, recall, bootstrap or learning stages. Callers that need those selected behaviors must request them explicitly. Acceptance, exact content and fresh independent judgment remain the completion bar. + +## At a Glance + +| Area | What changes for the user | +|---|---| +| Native work | Zero mandatory skills; optional guidance when a task needs it. | +| Skills | 34 current roots with explicit migration from the 52-root 3.6.0 bundle. | +| CLI | Recovery repairs, external context routing and exact-content evidence helpers. | +| Context use | Optional Claude/Codex readers, writers and separately installed read guards. | +| Validation | Fresh author-distinct judgment; structural and runtime checks report their actual scope. | +| Ownership | The caller's tracker, Git, runtime, memory destination and delivery policy stay authoritative. | + +## Product Areas + +### Install, Upgrade, and Distribution + +- Added: Optional context-role and read-guard installers preserve unrelated configuration, use unique backups and refuse malformed or unsupported destinations before publishing changes. +- Changed: Native onboarding leads with the agent and shell; managed full plugins and selected source links remain optional distribution paths. Release version defaults are aligned at 4.0.0. +- Changed: Release and installation CI use Go 1.27.1 and Python 3.14. Snapshot release builds remain a no-publish check of the distribution path. + +### CLI and Operator Commands + +- Added: `ao provenance snapshot-intent`, `manifest`, `digest`, `store-verdict`, `verify-manifest`, `verify-verdict` and `verify-subject` support exact-content evidence without invoking a tracker, Git or a semantic judge. `evidence-orphans` inspects references whose bound subjects changed. +- Added: `ao provenance verify-judgments` checks caller-required review profiles against native receipts, independent acceptance and exact subject identity. Requested model names or assistant self-descriptions cannot establish runtime identity. +- Added: `ao config context` resolves explicit external routes and can recover the same route from a selected native BD maintenance anchor. `ao session read-source` emits bounded source spans, continuation hashes and file identity facts under an explicit policy. +- Added: `ao provenance mine-session --view excerpts` provides bounded cited transcript excerpts without checkpoint writes. The existing event view remains available. +- Changed: `ao demo` and `ao quick-start` describe native execution; `ao demo --rpi` opts into the workflow example. `ao status --evidence-root` reads an explicitly selected evidence store. `ao skills link --skill` supports selected skills. +- Fixed: Doctor uses unique action backups, validates undo inputs before restoration and protects rename conflicts. Mining checkpoints reject unsafe paths and overlapping writers. Handoffs preserve work/session associations and select the newest valid artifact correctly. +- Fixed: Skill search, evidence status, provenance graph validation, scenario-result publication and changed-path gate routing handle previously missed malformed, corrupt and path-dependent cases. +- Removed: The `ao eval` family, its unused runtime adapters and `ao redact`. + +### Skills and Workflows + +- Changed: The 34-skill menu consolidates retained behavior into existing owners. Plan shapes missing acceptance in the conversation or tracker; Implement repairs known defects; Validate judges exact content from fresh context. RPI remains explicitly selectable. +- Changed: Final judgment is assigned once while preserving every required review leg. A requested retrospective consumes the known outcome and judgment, or names an interim cutoff; it does not become an extra code-acceptance gate. +- Added: Memory combines optional recall, bounded episode mining and reviewed topic curation. CASS and MS retrieve supporting evidence on demand; later work must demonstrate whether adopted guidance helped. +- Added: Skill Eval provides bounded routing and coding evaluations for a named skill decision. Structural conformance, repeated agreement and completed runtime sessions are not evidence of universal skill benefit. +- Added: Claude `bulk-read` and `code-write` workflows delegate to cheap workers and return bounded findings or receipts. Readers continue to EOF despite answer caps; writers require a reference, preserve caller target identities and preflight distinct batch targets. +- Removed: The retired skill roots and scripted RPI loop listed under Breaking Changes. Anti-ceremony obligations survive in the operating contract without requiring a standalone skill. + +### Codex and Runtime Integrations + +- Added: Codex-native `bulk-reader` and `code-writer` TOML roles pin `gpt-5.6-luna` with low/medium effort and fresh child contexts. The generated bundle carries their resources; the optional installer registers the names. +- Fixed: Project registrations point at the real source-owned role TOML files. The installed Codex runtime rejected registrations that pointed through symlinked role paths. +- Added: Claude plugin reader/writer subagents default to Haiku. Native agent dispatch is the default executor path; headless and factory adapters remain selected alternatives. +- Changed: Runtime guidance distinguishes actual native model/session identity, completion, deterministic checks and semantic judgment. Read-only reader instructions do not establish sandbox confinement. + +### Hooks and Lifecycle + +- Added: Separate opt-in Claude `Read|Bash` and Codex `Bash` read-budget guards block supported oversized unbounded reads above a configurable budget, default 350 lines. Refusals suggest a bounded slice or reader delegation; waivers and hashed telemetry remain supported. +- Fixed: Shell parsing handles quoted literal paths, end-of-options and negative `head` limits without overflow, and avoids false file attribution for uncertain syntax. Repeated refusals are compact. +- Fixed: Installer checks include matcher and handler type, preserve unique settings backups and leave hook trust with the native runtime. + +### Eval, Validation, and Release Gates + +- Added: Source-span integrity tests, native judgment-receipt verification, evidence-orphan checks, seeded-defect probes and bounded installed-skill coding trials strengthen evidence about specific behaviors. +- Fixed: Probe contamination and incomplete binding cannot silently count as proven coverage. Skill-trial reporting separates independently accepted work from control artifacts and runtime completion. +- Fixed: Component-owned checks, changed-path routing, generated projections, executable script modes, documentation checks and the local aggregate runner received conformance repairs. +- Changed: Required Go checks include the repository lint contract as well as build, vet and race/shuffled tests. CI and release checks remain mechanical evidence, separate from fresh semantic judgment. + +### Docs and Onboarding + +- Changed: README, runtime installation pages, migration guidance, the workflow reference and the generated skill router describe native execution and the current menu consistently. +- Added: Context-budget design and runtime evidence document invocation, installation, refused reads, role discovery and observed limitations. Memory references document authorized sources, protected drafts and supported topic curation. +- Changed: The release summary covers the full v3.6.0-to-v4.0.0 interval, including CLI removals, recovery, evidence helpers, skill consolidation and runtime changes. + +### Security, Privacy, and Supply Chain + +- Security: Evidence helpers require explicit protected non-Git storage, reject malformed identities and conflicting JSON fields, and publish immutable content-addressed artifacts atomically. Declared Git storage boundaries must resolve before writes. +- Security: Session-mining state protects source files and rejects unsafe ownership, permission and path configurations. Bounded excerpt extraction labels provenance and preserves source/destination authorization boundaries. +- Changed: Source routing checks owner, task, model, destination and native maintenance facts. They do not implement restricted-source access enforcement; the supported source-reading path remains limited to synthetic or already-cleared inputs. +- Changed: Pinned GitHub Actions and dependencies were refreshed, including `golang.org/x/text` 0.42.0, NumPy 2.5.3, SciPy 1.18.1 and Harbor 0.23.0. + +### Contributor/Internal Refactors + +- Refactored: Consumer-free evaluation machinery and retired workflow scripts were removed while supported bounded evaluation tooling stayed with its declared consumers. +- Changed: Canonical skill sources own behavior, and integrated regeneration owns portable Codex, skill inventory, image and documentation projections. The menu no longer carries retired roots as live skills. +- Fixed: Architecture-check temporary-directory races, Go lint compatibility and source-equivalence checks received targeted repairs. + +## Known Issues + +- Claude writer validation previously observed repeated execution of the supplied check and one fenced JSON receipt. Those failures remain open at this draft's cutoff; deterministic workflow checks alone do not establish the live fix. +- Codex's reader followed read-only instructions in live testing, but inherited its writable parent's sandbox. Enforced reader write confinement remains unproven; do not treat the role as a security boundary. +- Installing a Codex plugin does not activate custom agent registrations or trust hooks. The separate role installer and native hook review are required for those optional features. +- The guard covers supported native read/shell paths, not every possible way a program or hosted tool can read data. Bounded output and a reader's completion receipt do not prove semantic comprehension. +- Structural validation of all 34 packages and live tests of selected paths do not establish behavioral efficacy for every skill or task. No generalized token-cost reduction percentage is claimed. +- Restricted-source enforcement remains unavailable for the context/source helpers. Native freshness facts are attestations with documented trust boundaries, not cryptographic proof of process isolation. + +[Full changelog](https://github.com/boshu2/agentops/compare/v3.6.0...v4.0.0) diff --git a/images/claude/verify.sh b/images/claude/verify.sh index ed94d09f2..3689fa852 100755 --- a/images/claude/verify.sh +++ b/images/claude/verify.sh @@ -58,7 +58,7 @@ fi # Version guard: the Claude marketplace plugin manifest is the install entrypoint # for this image. Assert .claude-plugin/plugin.json declares the expected version # so a stale-version drift (plugin.json behind the release) fails the gate. -EXPECTED_VERSION="${AGENTOPS_EXPECTED_VERSION:-3.6.0}" +EXPECTED_VERSION="${AGENTOPS_EXPECTED_VERSION:-4.0.0}" plugin_manifest="$repo_root/.claude-plugin/plugin.json" if [ ! -f "$plugin_manifest" ]; then echo "FAIL: Claude plugin manifest not found: $plugin_manifest" >&2 From 35292607e3cadb8e832fef86858021edca51d5d8 Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 16:47:02 -0400 Subject: [PATCH 05/11] Integrate 4.0.0 projections and verified writer release notes --- CHANGELOG.md | 2 ++ docs/CHANGELOG.md | 2 ++ docs/design/codex-context-budget.md | 28 ++++++++++++++++++++++-- docs/releases/2026-09-13-v4.0.0-notes.md | 2 +- images/gemini/plugin.json | 2 +- 5 files changed, 32 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8e9999c27..791454e05 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -66,6 +66,8 @@ for upgrade instructions, the complete product-area summary and known limits. ### Fixed +- Claude writers capture a supplied check's status in its original invocation, avoiding the observed status-confirmation rerun. Native success, failing-check and direct-writer trials each ran the check once; the direct child returned plain JSON. +- Plugin conformance verifies exact skill membership and link destinations. Clean Claude and Codex plugin installs and upgrades are exercised separately from source linking; Codex metadata checks use the current Skills-only manifest contract. - Doctor keeps unique per-action backups and preflights undo integrity before restoration. Session mining protects checkpoint paths and concurrent writers; handoffs preserve work/session associations and use correct newest-first ordering. diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index 8e9999c27..791454e05 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -66,6 +66,8 @@ for upgrade instructions, the complete product-area summary and known limits. ### Fixed +- Claude writers capture a supplied check's status in its original invocation, avoiding the observed status-confirmation rerun. Native success, failing-check and direct-writer trials each ran the check once; the direct child returned plain JSON. +- Plugin conformance verifies exact skill membership and link destinations. Clean Claude and Codex plugin installs and upgrades are exercised separately from source linking; Codex metadata checks use the current Skills-only manifest contract. - Doctor keeps unique per-action backups and preflights undo integrity before restoration. Session mining protects checkpoint paths and concurrent writers; handoffs preserve work/session associations and use correct newest-first ordering. diff --git a/docs/design/codex-context-budget.md b/docs/design/codex-context-budget.md index a2a4740bd..8685e761f 100644 --- a/docs/design/codex-context-budget.md +++ b/docs/design/codex-context-budget.md @@ -1,7 +1,7 @@ # Codex context-budget design and evidence Status: implemented; observed runtime behavior and remaining limits below. -Evidence cutoff: 2026-09-12. Final gate results and the remaining Claude writer failure are recorded below. +Original evidence cutoff: 2026-09-12. The 4.0.0 writer follow-up below records the 2026-09-13 repair and live results. Acceptance is the two-job caller request recorded in private BD `age-z25n`. The exact e32e88c verdict was written before this job began and posted at https://github.com/boshu2/agentops/pull/1137#issuecomment-5648513520. @@ -294,7 +294,7 @@ complete child write payload, raw check output or source/check sentinel in the native parent context. The shared source in this run is fixes commit `53bcfec1480c205290f286b4a7ccd582216eb6f9`. -**Remaining observed failure:** eta and iota each invoked the supplied check +**Observed failure at this cutoff:** eta and iota each invoked the supplied check twice, while theta invoked it once. The fixture's independent invocation ledger and native child tool IDs agree. This violates the source instruction to run the check once; measured line counts and passing tests do not clear that failure. @@ -314,6 +314,30 @@ post-write line-count command. A fenced JSON receipt observed in a direct agent reply illustrates that direct-role formatting remains an instruction. No byte-filtering or strict direct-agent output parser is claimed. +### 4.0.0 writer follow-up — 2026-09-13 + +The duplicate invocation recovered a check's status by running the check again. +Both writer prompts now capture that status immediately in the original Bash +invocation, including silent checks and checks that enable `set -e` internally. +The direct role also supplies an explicit raw-JSON receipt example. + +Source repair `eae6f00c236ef88e38a9de43b8cdf289c3779123` passed 20 focused +tests. Native Opus parent `7910a1c7-7670-4a11-8a18-89fe4b46c4b9` delegated to +three Haiku 4.5 writers: workflow success, workflow deliberate check failure, +and direct writer success. Each supplied check ran once; receipts reported +the respective true, false and true status. The direct child's raw native +receipt was JSON without Markdown fences. All three targets had seven physical +lines and passed independent Bats checks. Only assigned targets were added. +The parent used delegation/wait tools and received neither generated contents +nor the check-output sentinel. The bounded run exited 0 after 49.97 seconds; +owned-process cleanup found no remaining processes. + +These observations close the reproduced check-count and direct-child receipt +failures for the tested cases. The coordinating parent's final presentation +still added fences, so parent presentation is not evidence of raw child format. +Direct-role formatting and target confinement remain instructions rather than +an output filter or per-file sandbox. + ## Checks and delivery The repair PR is [#1139](https://github.com/boshu2/agentops/pull/1139), separate diff --git a/docs/releases/2026-09-13-v4.0.0-notes.md b/docs/releases/2026-09-13-v4.0.0-notes.md index ed773b585..547f79555 100644 --- a/docs/releases/2026-09-13-v4.0.0-notes.md +++ b/docs/releases/2026-09-13-v4.0.0-notes.md @@ -63,6 +63,7 @@ migration guidance below is part of the upgrade. ### Skills and Workflows +- Fixed: Claude writers capture check status without a second invocation. Live success and deliberately failing checks ran once, and the direct child returned a plain JSON receipt. - Changed: The 34-skill menu consolidates retained behavior into existing owners. Plan shapes missing acceptance in the conversation or tracker; Implement repairs known defects; Validate judges exact content from fresh context. RPI remains explicitly selectable. - Changed: Final judgment is assigned once while preserving every required review leg. A requested retrospective consumes the known outcome and judgment, or names an interim cutoff; it does not become an extra code-acceptance gate. - Added: Memory combines optional recall, bounded episode mining and reviewed topic curation. CASS and MS retrieve supporting evidence on demand; later work must demonstrate whether adopted guidance helped. @@ -111,7 +112,6 @@ migration guidance below is part of the upgrade. ## Known Issues -- Claude writer validation previously observed repeated execution of the supplied check and one fenced JSON receipt. Those failures remain open at this draft's cutoff; deterministic workflow checks alone do not establish the live fix. - Codex's reader followed read-only instructions in live testing, but inherited its writable parent's sandbox. Enforced reader write confinement remains unproven; do not treat the role as a security boundary. - Installing a Codex plugin does not activate custom agent registrations or trust hooks. The separate role installer and native hook review are required for those optional features. - The guard covers supported native read/shell paths, not every possible way a program or hosted tool can read data. Bounded output and a reader's completion receipt do not prove semantic comprehension. diff --git a/images/gemini/plugin.json b/images/gemini/plugin.json index b5abea6ce..d40416080 100644 --- a/images/gemini/plugin.json +++ b/images/gemini/plugin.json @@ -13,5 +13,5 @@ "name": "agentops-core-gemini", "rules": "./rules", "skills": "./skills", - "version": "3.6.0" + "version": "4.0.0" } From 9f9812fc0f19cc9cb773c2f3f028e78cbc39d782 Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 16:58:35 -0400 Subject: [PATCH 06/11] Scan full release scope and fail pytest collection errors --- scripts/security-gate.sh | 4 +- scripts/toolchain-validate.sh | 60 ++++++---- tests/scripts/test-security-gate.sh | 26 +++++ tests/scripts/test-toolchain-validate.sh | 137 ++++++++++++++++++++++- 4 files changed, 199 insertions(+), 28 deletions(-) diff --git a/scripts/security-gate.sh b/scripts/security-gate.sh index 4b412342c..777682a67 100755 --- a/scripts/security-gate.sh +++ b/scripts/security-gate.sh @@ -16,7 +16,7 @@ Usage: scripts/security-gate.sh [--mode quick|full] [--json] [--require-tools] Runs the unified security gate using scripts/toolchain-validate.sh. Options: - --mode quick|full quick = skip slow tests (default), full = full suite + --mode quick|full quick = changed scope, skip slow tests (default); full = repository-wide suite --json output machine-readable summary JSON --require-tools fail if any scanner reports not_installed/error -h, --help show this help @@ -66,7 +66,7 @@ SECURITY_BASE="${SECURITY_GATE_OUTPUT_DIR:-${TMPDIR:-/tmp}/agentops-security}" SECURITY_DIR="$SECURITY_BASE/$RUN_ID" mkdir -p "$SECURITY_DIR" -TOOLCHAIN_ARGS=(--gate --json) +TOOLCHAIN_ARGS=(--all --gate --json) if [[ "$MODE" == "quick" ]]; then TOOLCHAIN_ARGS=(--quick --gate --json) fi diff --git a/scripts/toolchain-validate.sh b/scripts/toolchain-validate.sh index 549a44d9e..9b453b425 100755 --- a/scripts/toolchain-validate.sh +++ b/scripts/toolchain-validate.sh @@ -10,6 +10,7 @@ set -euo pipefail # --quick Skip slow tools (tests, comprehensive scans) # --json Output summary as JSON to stdout # --gate Exit non-zero on CRITICAL or HIGH findings +# --all Scan the full repository even with --gate (default: changed scope) # # Exit Codes: # 0 - Pass (no critical/high findings, or --gate not specified) @@ -26,12 +27,14 @@ OUTPUT_DIR="${TOOLCHAIN_OUTPUT_DIR:-${TMPDIR:-/tmp}/agentops-tooling}" QUICK=false JSON_OUTPUT=false GATE=false +ALL_FILES=false for arg in "$@"; do case $arg in --quick) QUICK=true ;; --json) JSON_OUTPUT=true ;; --gate) GATE=true ;; + --all) ALL_FILES=true ;; --help|-h) head -20 "$0" | grep "^#" | sed 's/^# *//' exit 0 @@ -46,7 +49,12 @@ done # Initialize output directory mkdir -p "$OUTPUT_DIR" -# Determine scope (for --gate, default to changed files only) +# Determine scope independently of whether findings should fail the command. +# Pre-commit/post-commit --gate callers retain their changed-file default. +SCOPE="all" +if [[ "$GATE" == "true" && "$ALL_FILES" != "true" ]]; then + SCOPE="changed" +fi TARGET_FILES=() in_git_repo() { @@ -84,7 +92,7 @@ collect_target_files() { return 0 } -if [[ "$GATE" == "true" ]]; then +if [[ "$SCOPE" == "changed" ]]; then while IFS= read -r f; do [[ -z "$f" ]] && continue TARGET_FILES+=("$REPO_ROOT/$f") @@ -202,7 +210,7 @@ ensure_json_or_error() { run_ruff() { local output_file="$OUTPUT_DIR/ruff.txt" - if [[ "$GATE" == "true" ]] && ! target_has_ext "py"; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_ext "py"; then echo "NO_PYTHON_FILES_IN_TARGET" > "$output_file" TOOL_STATUS["ruff"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -212,7 +220,7 @@ run_ruff() { if ! run_tool "ruff" ruff; then return 0; fi # Check if there are Python files - if ! find "$REPO_ROOT" -name "*.py" -type f | head -1 | grep -q .; then + if ! find "$REPO_ROOT" -name "*.py" -type f -print -quit | grep -q .; then echo "NO_PYTHON_FILES" > "$output_file" TOOL_STATUS["ruff"]="skipped" return 0 @@ -241,7 +249,7 @@ run_golangci() { local output_file="$OUTPUT_DIR/golangci-lint.txt" local golangci_cmd="$REPO_ROOT/scripts/golangci-lint-v2.sh" - if [[ "$GATE" == "true" ]] && ! target_has_any_ext go mod sum; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_any_ext go mod sum; then echo "NO_GO_CHANGES_IN_TARGET" > "$output_file" TOOL_STATUS["golangci-lint"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -389,7 +397,7 @@ run_gitleaks() { run_shellcheck() { local output_file="$OUTPUT_DIR/shellcheck.txt" - if [[ "$GATE" == "true" ]] && ! target_has_ext "sh"; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_ext "sh"; then echo "NO_SHELL_FILES_IN_TARGET" > "$output_file" TOOL_STATUS["shellcheck"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -400,7 +408,7 @@ run_shellcheck() { # Find all shell scripts local scripts - if [[ "$GATE" == "true" ]] && [[ "${#TARGET_FILES[@]}" -gt 0 ]]; then + if [[ "$SCOPE" == "changed" ]] && [[ "${#TARGET_FILES[@]}" -gt 0 ]]; then scripts="$(printf "%s\n" "${TARGET_FILES[@]}" | grep -E '\\.sh$' || true)" else scripts="$(find "$REPO_ROOT" -name "*.sh" -type f ! -path "*/.git/*" ! -path "*/.claude/worktrees/*" 2>/dev/null || true)" @@ -445,7 +453,7 @@ run_shellcheck() { run_radon() { local output_file="$OUTPUT_DIR/radon.txt" - if [[ "$GATE" == "true" ]] && ! target_has_ext "py"; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_ext "py"; then echo "NO_PYTHON_FILES_IN_TARGET" > "$output_file" TOOL_STATUS["radon"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -455,7 +463,7 @@ run_radon() { if ! run_tool "radon" radon; then return 0; fi # Check if there are Python files - if ! find "$REPO_ROOT" -name "*.py" -type f | head -1 | grep -q .; then + if ! find "$REPO_ROOT" -name "*.py" -type f -print -quit | grep -q .; then echo "NO_PYTHON_FILES" > "$output_file" TOOL_STATUS["radon"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -501,14 +509,17 @@ run_pytest() { if ! run_tool "pytest" pytest; then return 0; fi # Check if there are test files - if ! find "$REPO_ROOT" -name "test_*.py" -o -name "*_test.py" | head -1 | grep -q .; then + if ! find "$REPO_ROOT" -type f \( -name "test_*.py" -o -name "*_test.py" \) -print -quit | grep -q .; then echo "NO_TEST_FILES" > "$output_file" TOOL_STATUS["pytest"]="skipped" return 0 fi - # Run pytest with minimal output - if pytest "$REPO_ROOT" --tb=short -q > "$output_file" 2>&1; then + # Source skills and their generated projections can share test basenames. + # Importlib collects both without Python module-name collisions. + local pytest_rc=0 + pytest "$REPO_ROOT" --import-mode=importlib --tb=short -q > "$output_file" 2>&1 || pytest_rc=$? + if [[ "$pytest_rc" -eq 0 ]]; then echo "PASS" >> "$output_file" TOOL_STATUS["pytest"]="pass" else @@ -516,8 +527,16 @@ run_pytest() { failures=$(grep -cE "^FAILED" "$output_file" 2>/dev/null || true) failures=${failures:-0} failures=$(echo "$failures" | tr -d '[:space:]') + # Collection errors, interrupted runs, and usage/internal errors may + # contain no FAILED lines. A nonzero run still blocks the gate. + if [[ "$failures" -lt 1 ]]; then failures=1; fi CRITICAL_COUNT=$((CRITICAL_COUNT + failures)) - TOOL_STATUS["pytest"]="findings" + if [[ "$pytest_rc" -eq 1 ]]; then + TOOL_STATUS["pytest"]="findings" + else + TOOL_STATUS["pytest"]="error" + fi + printf '\nPYTEST_EXIT_CODE=%s\n' "$pytest_rc" >> "$output_file" fi } @@ -535,7 +554,7 @@ run_gotest() { local output_file="$OUTPUT_DIR/gotest.txt" - if [[ "$GATE" == "true" ]] && ! target_has_any_ext go mod sum; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_any_ext go mod sum; then echo "NO_GO_CHANGES_IN_TARGET" > "$output_file" TOOL_STATUS["go-test"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -594,7 +613,7 @@ run_semgrep() { local output_file="$OUTPUT_DIR/semgrep.txt" local stderr_file="$OUTPUT_DIR/semgrep.stderr.txt" - if [[ "$GATE" == "true" ]] && ! target_has_any_ext go py js ts tsx jsx java rb php cs; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_any_ext go py js ts tsx jsx java rb php cs; then echo "NO_CODE_FILES_IN_TARGET" > "$output_file" TOOL_STATUS["semgrep"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -658,7 +677,7 @@ run_trivy() { local output_file="$OUTPUT_DIR/trivy.txt" local stderr_file="$OUTPUT_DIR/trivy.stderr.txt" - if [[ "$GATE" == "true" ]] && ! target_has_any_ext go mod sum json lock yaml yml; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_any_ext go mod sum json lock yaml yml; then echo "NO_DEPENDENCY_CHANGES_IN_TARGET" > "$output_file" TOOL_STATUS["trivy"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -736,7 +755,7 @@ run_gosec() { local output_file="$OUTPUT_DIR/gosec.txt" local stderr_file="$OUTPUT_DIR/gosec.stderr.txt" - if [[ "$GATE" == "true" ]] && ! target_has_any_ext go mod sum; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_any_ext go mod sum; then echo "NO_GO_CHANGES_IN_TARGET" > "$output_file" TOOL_STATUS["gosec"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -846,7 +865,7 @@ run_govulncheck() { local output_file="$OUTPUT_DIR/govulncheck.txt" local stderr_file="$OUTPUT_DIR/govulncheck.stderr.txt" - if [[ "$GATE" == "true" ]] && ! target_has_any_ext go mod sum; then + if [[ "$SCOPE" == "changed" ]] && ! target_has_any_ext go mod sum; then echo "NO_GO_CHANGES_IN_TARGET" > "$output_file" TOOL_STATUS["govulncheck"]="skipped" TOOLS_SKIPPED=$((TOOLS_SKIPPED + 1)) @@ -970,9 +989,7 @@ log "Toolchain Validation" log "====================" log "Target: $REPO_ROOT" log "Output: $OUTPUT_DIR" -if [[ "$GATE" == "true" ]] && [[ "${#TARGET_FILES[@]}" -gt 0 ]]; then - log "Scope: changed files only" -fi +log "Scope: $SCOPE" log "" # Run all tools @@ -1022,6 +1039,7 @@ SUMMARY=$(cat <"$MOCK_TOOLCHAIN" <<'MOCK' #!/bin/bash +printf '%s\n' "$@" > "$MOCK_ARGS" cat <<'JSON' { "timestamp": "2026-02-19T00:00:00Z", @@ -55,12 +63,29 @@ cat <<'JSON' "gate_status": "PASS", "output_dir": "/tmp/agentops-tooling" } + JSON exit 0 MOCK chmod +x "$MOCK_TOOLCHAIN" } +test_scope_arguments() { + create_mock_toolchain + SECURITY_GATE_TOOLCHAIN_SCRIPT="$MOCK_TOOLCHAIN" scripts/security-gate.sh --mode full --json >/dev/null + if grep -qx -- '--all' "$MOCK_ARGS" && grep -qx -- '--gate' "$MOCK_ARGS"; then + pass "full security explicitly requests full-repository gate scope" + else + fail "full security did not request --all --gate" + fi + SECURITY_GATE_TOOLCHAIN_SCRIPT="$MOCK_TOOLCHAIN" scripts/security-gate.sh --mode quick --json >/dev/null + if grep -qx -- '--quick' "$MOCK_ARGS" && grep -qx -- '--gate' "$MOCK_ARGS" && ! grep -qx -- '--all' "$MOCK_ARGS"; then + pass "quick security preserves ordinary changed-scope gate arguments" + else + fail "quick security scope was broadened" + fi +} + test_executable() { if [[ -x "scripts/security-gate.sh" ]]; then pass "security-gate.sh is executable" @@ -145,6 +170,7 @@ test_help test_invalid_mode test_json_output test_artifacts +test_scope_arguments echo "" echo "================================" diff --git a/tests/scripts/test-toolchain-validate.sh b/tests/scripts/test-toolchain-validate.sh index d99ea4ea7..d2ae33683 100755 --- a/tests/scripts/test-toolchain-validate.sh +++ b/tests/scripts/test-toolchain-validate.sh @@ -7,7 +7,8 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" cd "$REPO_ROOT" -MOCK_DIR="/tmp/toolchain-test-$$" +MOCK_DIR="$(mktemp -d "${TMPDIR:-/tmp}/toolchain-test.XXXXXX")" +REAL_PYTEST="$(command -v pytest || true)" PASS_COUNT=0 FAIL_COUNT=0 @@ -16,6 +17,53 @@ cleanup() { } trap cleanup EXIT +# Exercise scanner selection and result handling in a tiny real Git repository. +# Scanner doubles keep these contract tests independent of downloads, host tool +# inventories, and findings in unrelated working trees. Release CI runs the real +# scanners separately. +mkdir -p "$MOCK_DIR/repo/scripts" "$MOCK_DIR/bin" "$MOCK_DIR/repo/cli" +cp "${TOOLCHAIN_TEST_SCRIPT:-$REPO_ROOT/scripts/toolchain-validate.sh}" "$MOCK_DIR/repo/scripts/toolchain-validate.sh" +cat > "$MOCK_DIR/bin/scanner" <<'MOCK' +#!/usr/bin/env bash +name="${0##*/}" +printf '%s %s\n' "$name" "$*" >> "$SCANNER_LOG" +case "$name" in + semgrep) printf '{"results":[]}\n' ;; + trivy) printf '{"Results":[]}\n' ;; + gosec) printf '{"Issues":[]}\n' ;; + hadolint) printf '[]\n' ;; + pytest) + if [[ "${PYTEST_EXIT:-0}" -eq 1 ]]; then + printf 'FAILED test_example.py::test_example - AssertionError\n' + elif [[ "${PYTEST_EXIT:-0}" -ne 0 ]]; then + printf 'ERROR collecting test_example.py\nInterrupted: 1 error during collection\n' + fi + exit "${PYTEST_EXIT:-0}" + ;; +esac +MOCK +chmod +x "$MOCK_DIR/bin/scanner" +for tool in ruff gitleaks shellcheck radon semgrep trivy gosec govulncheck hadolint pytest go; do + ln -s scanner "$MOCK_DIR/bin/$tool" +done +cp "$MOCK_DIR/bin/scanner" "$MOCK_DIR/repo/scripts/golangci-lint-v2.sh" +printf 'module example.invalid/fixture\n\ngo 1.25\n' > "$MOCK_DIR/repo/cli/go.mod" +printf 'package fixture\n' > "$MOCK_DIR/repo/cli/example.go" +printf 'def test_example():\n assert True\n' > "$MOCK_DIR/repo/test_example.py" +printf 'FROM scratch\n' > "$MOCK_DIR/repo/Dockerfile" +printf '# Fixture\n' > "$MOCK_DIR/repo/README.md" +export PATH="$MOCK_DIR/bin:$PATH" +export SCANNER_LOG="$MOCK_DIR/scanners.log" +export TOOLCHAIN_OUTPUT_DIR="$MOCK_DIR/tooling" +cd "$MOCK_DIR/repo" +git init -q +git config core.hooksPath /dev/null +git add . +git -c user.name=Fixture -c user.email=fixture@example.invalid commit -qm 'Add code fixture' +printf '\nDocumentation-only change.\n' >> README.md +git add README.md +git -c user.name=Fixture -c user.email=fixture@example.invalid commit -qm 'Update only documentation' + pass() { echo "PASS: $1" PASS_COUNT=$((PASS_COUNT + 1)) @@ -108,7 +156,7 @@ test_exit_no_gate() { fi } -# Test 7: Tool count matches expected (11 tools) +# Test 7: Tool count matches the shipped inventory (12 tools) test_tool_count() { local output output=$(./scripts/toolchain-validate.sh --json 2>/dev/null || true) @@ -116,10 +164,85 @@ test_tool_count() { local tool_count tool_count=$(echo "$output" | jq '.tools | keys | length' 2>/dev/null || echo 0) - if [[ "$tool_count" -eq 11 ]]; then - pass "Tool count is 11" + if [[ "$tool_count" -eq 12 ]]; then + pass "Tool count is 12" + else + fail "Expected 12 tools, got $tool_count" + fi +} + +test_gate_scope() { + local changed full + changed=$(./scripts/toolchain-validate.sh --gate --json) + full=$(./scripts/toolchain-validate.sh --all --gate --json) + if jq -e '.scope == "changed" and .tools.ruff == "skipped" and .tools.gosec == "skipped" and .tools["go-test"] == "skipped"' <<< "$changed" >/dev/null; then + pass "ordinary --gate retains docs-only HEAD changed scope" + else + fail "ordinary --gate broadened docs-only HEAD scope" + fi + if jq -e '.scope == "all" and .tools.ruff == "pass" and .tools.gosec == "pass" and .tools["go-test"] == "pass" and .tools.shellcheck == "pass" and .tools.semgrep == "pass" and .tools.trivy == "pass" and .tools.govulncheck == "pass"' <<< "$full" >/dev/null; then + pass "--all --gate scans code despite docs-only HEAD" + else + fail "full scope skipped code scanners after docs-only HEAD" + fi + printf '\n# Staged Python change\n' >> test_example.py + git add test_example.py + changed=$(./scripts/toolchain-validate.sh --gate --json) + if jq -e '.scope == "changed" and .tools.ruff == "pass" and .tools.gosec == "skipped"' <<< "$changed" >/dev/null; then + pass "ordinary --gate still prefers staged changes" + else + fail "staged Python scope changed" + fi +} + +test_pytest_failure_exits() { + local rc output status + for rc in 1 2 3 4 5; do + status=0 + output=$(PYTEST_EXIT="$rc" ./scripts/toolchain-validate.sh --all --gate --json) || status=$? + if [[ "$status" -eq 2 ]] && jq -e '.gate_status == "BLOCKED_CRITICAL" and .findings.critical >= 1' <<< "$output" >/dev/null; then + pass "pytest exit $rc blocks even without FAILED summary lines" + else + fail "pytest exit $rc was incorrectly green (gate exit $status)" + fi + if [[ "$rc" -ne 1 ]] && ! jq -e '.tools.pytest == "error"' <<< "$output" >/dev/null; then + fail "pytest exit $rc must report a tool error" + fi + done + if grep -q 'pytest .*--import-mode=importlib' "$SCANNER_LOG"; then + pass "pytest uses importlib collection for duplicate module basenames" + else + fail "pytest importlib collection flag missing" + fi +} + +test_duplicate_test_modules() { + if [[ -z "$REAL_PYTEST" ]]; then + fail "pytest is required to verify duplicate-module collection" + return + fi + mkdir -p "$MOCK_DIR/duplicate/source" "$MOCK_DIR/duplicate/projection" + printf 'def test_source():\n assert True\n' > "$MOCK_DIR/duplicate/source/test_same.py" + printf 'def test_projection():\n assert True\n' > "$MOCK_DIR/duplicate/projection/test_same.py" + if "$REAL_PYTEST" "$MOCK_DIR/duplicate" --import-mode=importlib -q > "$MOCK_DIR/duplicate.log" 2>&1 && + grep -q '2 passed' "$MOCK_DIR/duplicate.log"; then + pass "real pytest collects both source and projection modules" + else + fail "duplicate test modules were not both collected" + fi +} + +test_large_python_inventory() { + local index output + mkdir -p many-python-files + for ((index = 0; index < 2000; index++)); do + : > "many-python-files/test_fixture_with_a_long_name_to_exceed_the_pipe_buffer_${index}.py" + done + output=$(./scripts/toolchain-validate.sh --all --gate --json) + if jq -e '.tools.ruff == "pass" and .tools.radon == "pass" and .tools.pytest == "pass"' <<< "$output" >/dev/null; then + pass "large Python inventories do not turn SIGPIPE into false no-files skips" else - fail "Expected 11 tools, got $tool_count" + fail "large Python inventory was incorrectly treated as absent" fi } @@ -151,6 +274,10 @@ test_quick_skips_tests test_exit_no_gate test_tool_count test_output_dir +test_gate_scope +test_pytest_failure_exits +test_duplicate_test_modules +test_large_python_inventory echo "" echo "================================" From 6e18c7201d97a921207b7d8aa41bd0fa2cf90f0c Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 17:01:44 -0400 Subject: [PATCH 07/11] Install evaluator dependencies for nightly full security checks --- .github/workflows/nightly.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index f3e9e33ee..dbd50dd32 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -85,7 +85,9 @@ jobs: - name: Install scanner tools run: | python -m pip install --upgrade pip - python -m pip install semgrep==1.169.0 ruff==0.15.21 radon==6.0.1 pytest==9.1.1 + python -m pip install semgrep==1.169.0 ruff==0.15.21 radon==6.0.1 pytest==9.1.1 \ + -r evals/skills-rpi/requirements.txt \ + -r evals/skills-rpi/requirements-readout.txt GOBIN=/usr/local/bin go install github.com/securego/gosec/v2/cmd/gosec@v2.27.1 GOBIN=/usr/local/bin go install github.com/zricethezav/gitleaks/v8@v8.30.1 From 0713d5e732fa111b80d1c2376117cd04322429bc Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 17:04:18 -0400 Subject: [PATCH 08/11] Complete release security scope and Python collection prerequisites --- CHANGELOG.md | 1 + docs/CHANGELOG.md | 1 + docs/releases/2026-09-13-v4.0.0-notes.md | 1 + evals/skills-rpi/test_receipts.py | 4 +++ scripts/ci-local-release.sh | 4 ++- tests/scripts/ci-local-release.bats | 31 ++++++++++++++++++++++++ 6 files changed, 41 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 791454e05..05558e5a6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -66,6 +66,7 @@ for upgrade instructions, the complete product-area summary and known limits. ### Fixed +- Full release security checks scan the whole repository and treat Python collection or runtime errors as blocking failures. Nightly checks install the declared evaluator dependencies; test collection supports duplicate source/projection module names. - Claude writers capture a supplied check's status in its original invocation, avoiding the observed status-confirmation rerun. Native success, failing-check and direct-writer trials each ran the check once; the direct child returned plain JSON. - Plugin conformance verifies exact skill membership and link destinations. Clean Claude and Codex plugin installs and upgrades are exercised separately from source linking; Codex metadata checks use the current Skills-only manifest contract. - Doctor keeps unique per-action backups and preflights undo integrity before diff --git a/docs/CHANGELOG.md b/docs/CHANGELOG.md index 791454e05..05558e5a6 100644 --- a/docs/CHANGELOG.md +++ b/docs/CHANGELOG.md @@ -66,6 +66,7 @@ for upgrade instructions, the complete product-area summary and known limits. ### Fixed +- Full release security checks scan the whole repository and treat Python collection or runtime errors as blocking failures. Nightly checks install the declared evaluator dependencies; test collection supports duplicate source/projection module names. - Claude writers capture a supplied check's status in its original invocation, avoiding the observed status-confirmation rerun. Native success, failing-check and direct-writer trials each ran the check once; the direct child returned plain JSON. - Plugin conformance verifies exact skill membership and link destinations. Clean Claude and Codex plugin installs and upgrades are exercised separately from source linking; Codex metadata checks use the current Skills-only manifest contract. - Doctor keeps unique per-action backups and preflights undo integrity before diff --git a/docs/releases/2026-09-13-v4.0.0-notes.md b/docs/releases/2026-09-13-v4.0.0-notes.md index 547f79555..603407f1d 100644 --- a/docs/releases/2026-09-13-v4.0.0-notes.md +++ b/docs/releases/2026-09-13-v4.0.0-notes.md @@ -46,6 +46,7 @@ migration guidance below is part of the upgrade. ### Install, Upgrade, and Distribution +- Fixed: Full release security checks scan repository-wide and block on Python collection or runtime failures. Nightly installs the declared evaluator dependencies, and importlib collection handles duplicate source/projection test names. - Added: Optional context-role and read-guard installers preserve unrelated configuration, use unique backups and refuse malformed or unsupported destinations before publishing changes. - Changed: Native onboarding leads with the agent and shell; managed full plugins and selected source links remain optional distribution paths. Release version defaults are aligned at 4.0.0. - Changed: Release and installation CI use Go 1.27.1 and Python 3.14. Snapshot release builds remain a no-publish check of the distribution path. diff --git a/evals/skills-rpi/test_receipts.py b/evals/skills-rpi/test_receipts.py index aa86378a7..62242e41c 100644 --- a/evals/skills-rpi/test_receipts.py +++ b/evals/skills-rpi/test_receipts.py @@ -3,8 +3,12 @@ import io import tarfile from pathlib import Path +import sys import tempfile import unittest + +# Keep sibling CLI modules importable under pytest's importlib collection. +sys.path.insert(0, str(Path(__file__).resolve().parent)) from prepare import extract_cli, sha, tree_hash, write_json from receipts import collect diff --git a/scripts/ci-local-release.sh b/scripts/ci-local-release.sh index b930ea788..d1c8bbe61 100755 --- a/scripts/ci-local-release.sh +++ b/scripts/ci-local-release.sh @@ -656,11 +656,13 @@ run_security_gate() { local output_file="$ARTIFACT_DIR/security-gate-${SECURITY_MODE}.json" local security_dir="$SECURITY_TMP_BASE/security" local tooling_dir="$SECURITY_TMP_BASE/tooling" + local gitleaks_mode="range" + [[ "$SECURITY_MODE" != "full" ]] || gitleaks_mode="full" mkdir -p "$security_dir" "$tooling_dir" SECURITY_GATE_OUTPUT_DIR="$security_dir" \ TOOLCHAIN_OUTPUT_DIR="$tooling_dir" \ - TOOLCHAIN_GITLEAKS_MODE="${TOOLCHAIN_GITLEAKS_MODE:-range}" \ + TOOLCHAIN_GITLEAKS_MODE="${TOOLCHAIN_GITLEAKS_MODE:-$gitleaks_mode}" \ TOOLCHAIN_GITLEAKS_RANGE="${TOOLCHAIN_GITLEAKS_RANGE:-origin/main..HEAD}" \ TOOLCHAIN_GITLEAKS_GOMAXPROCS="${TOOLCHAIN_GITLEAKS_GOMAXPROCS:-2}" \ ./scripts/security-gate.sh --mode "$SECURITY_MODE" --json > "$output_file" diff --git a/tests/scripts/ci-local-release.bats b/tests/scripts/ci-local-release.bats index 8896ba46e..6ec550efe 100644 --- a/tests/scripts/ci-local-release.bats +++ b/tests/scripts/ci-local-release.bats @@ -127,6 +127,37 @@ teardown() { [ "$status" -eq 0 ] } +@test "full security scans all secrets while quick mode and explicit overrides preserve their scope" { + mkdir -p "$TMP_DIR/scripts" "$TMP_DIR/artifacts" + cat > "$TMP_DIR/scripts/security-gate.sh" <<'EOF' +#!/usr/bin/env bash +printf '{"gate_status":"PASS","gitleaks_mode":"%s"}\n' "$TOOLCHAIN_GITLEAKS_MODE" +EOF + chmod +x "$TMP_DIR/scripts/security-gate.sh" + + run env SCRIPT_UNDER_TEST="$SCRIPT" FIXTURE_DIR="$TMP_DIR" bash -c ' + set -euo pipefail + set -- + export AGENTOPS_CI_LOCAL_RELEASE_SOURCE_ONLY=1 + source "$SCRIPT_UNDER_TEST" + cd "$FIXTURE_DIR" + ARTIFACT_DIR="$FIXTURE_DIR/artifacts" + SECURITY_TMP_BASE="$FIXTURE_DIR/security-temp" + unset TOOLCHAIN_GITLEAKS_MODE + SECURITY_MODE=full + run_security_gate + jq -e '\''.gitleaks_mode == "full"'\'' "$ARTIFACT_DIR/security-gate-full.json" + SECURITY_MODE=quick + run_security_gate + jq -e '\''.gitleaks_mode == "range"'\'' "$ARTIFACT_DIR/security-gate-quick.json" + SECURITY_MODE=full + TOOLCHAIN_GITLEAKS_MODE=staged + run_security_gate + jq -e '\''.gitleaks_mode == "staged"'\'' "$ARTIFACT_DIR/security-gate-full.json" + ' + [ "$status" -eq 0 ] +} + @test "--release-version rejects garbage values" { run bash "$SCRIPT" --release-version not-a-version [ "$status" -eq 1 ] From db9428eef6377bc413fd06aa8035585cfe6a4d6b Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 17:05:41 -0400 Subject: [PATCH 09/11] Require complete scanner availability for full release rehearsals --- scripts/ci-local-release.sh | 8 ++++++-- tests/scripts/ci-local-release.bats | 5 +++++ 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/scripts/ci-local-release.sh b/scripts/ci-local-release.sh index d1c8bbe61..84758283c 100755 --- a/scripts/ci-local-release.sh +++ b/scripts/ci-local-release.sh @@ -657,7 +657,11 @@ run_security_gate() { local security_dir="$SECURITY_TMP_BASE/security" local tooling_dir="$SECURITY_TMP_BASE/tooling" local gitleaks_mode="range" - [[ "$SECURITY_MODE" != "full" ]] || gitleaks_mode="full" + local -a tool_requirement=() + if [[ "$SECURITY_MODE" == "full" ]]; then + gitleaks_mode="full" + tool_requirement=(--require-tools) + fi mkdir -p "$security_dir" "$tooling_dir" SECURITY_GATE_OUTPUT_DIR="$security_dir" \ @@ -665,7 +669,7 @@ run_security_gate() { TOOLCHAIN_GITLEAKS_MODE="${TOOLCHAIN_GITLEAKS_MODE:-$gitleaks_mode}" \ TOOLCHAIN_GITLEAKS_RANGE="${TOOLCHAIN_GITLEAKS_RANGE:-origin/main..HEAD}" \ TOOLCHAIN_GITLEAKS_GOMAXPROCS="${TOOLCHAIN_GITLEAKS_GOMAXPROCS:-2}" \ - ./scripts/security-gate.sh --mode "$SECURITY_MODE" --json > "$output_file" + ./scripts/security-gate.sh --mode "$SECURITY_MODE" --json "${tool_requirement[@]}" > "$output_file" jq -e '.gate_status' "$output_file" >/dev/null echo "Security report: $output_file" echo "Security artifacts: $security_dir" diff --git a/tests/scripts/ci-local-release.bats b/tests/scripts/ci-local-release.bats index 6ec550efe..a409e4143 100644 --- a/tests/scripts/ci-local-release.bats +++ b/tests/scripts/ci-local-release.bats @@ -131,6 +131,11 @@ teardown() { mkdir -p "$TMP_DIR/scripts" "$TMP_DIR/artifacts" cat > "$TMP_DIR/scripts/security-gate.sh" <<'EOF' #!/usr/bin/env bash +if [[ "$2" == "full" ]]; then + [[ "${4:-}" == "--require-tools" ]] || exit 9 +else + [[ "$#" == 3 ]] || exit 10 +fi printf '{"gate_status":"PASS","gitleaks_mode":"%s"}\n' "$TOOLCHAIN_GITLEAKS_MODE" EOF chmod +x "$TMP_DIR/scripts/security-gate.sh" From 991f485facdc83cafe8fd8bd7889b72f0b1f397b Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 17:05:35 -0400 Subject: [PATCH 10/11] Repair Python test imports and canonical schema lookup --- .../skills/validate/scripts/test_validate.py | 9 ++++- tests/python/test_scan_descriptions.py | 38 +++++++++++++++---- 2 files changed, 38 insertions(+), 9 deletions(-) diff --git a/packs/agentops-executor/agents/validator/skills/validate/scripts/test_validate.py b/packs/agentops-executor/agents/validator/skills/validate/scripts/test_validate.py index b94494f9d..cc28664c4 100755 --- a/packs/agentops-executor/agents/validator/skills/validate/scripts/test_validate.py +++ b/packs/agentops-executor/agents/validator/skills/validate/scripts/test_validate.py @@ -35,7 +35,14 @@ def draft(self): } def assert_schema_valid(self, artifact): - schema = json.loads((Path(__file__).parents[3] / "schemas" / "verdict.v2.schema.json").read_text()) + # This retained pack copy is deeper than the original skills tree. + # Both tests must validate against the repository's canonical schema. + schema_path = next( + ancestor / "schemas" / "verdict.v2.schema.json" + for ancestor in Path(__file__).resolve().parents + if (ancestor / "schemas" / "verdict.v2.schema.json").is_file() + ) + schema = json.loads(schema_path.read_text()) jsonschema.Draft202012Validator(schema).validate(artifact) def runtime_facts(self): diff --git a/tests/python/test_scan_descriptions.py b/tests/python/test_scan_descriptions.py index 88b2218d3..aaacd1fca 100644 --- a/tests/python/test_scan_descriptions.py +++ b/tests/python/test_scan_descriptions.py @@ -12,8 +12,9 @@ import sys import tempfile import unittest -from contextlib import redirect_stdout +from contextlib import redirect_stderr, redirect_stdout from pathlib import Path +from unittest.mock import patch REPO_ROOT = Path(__file__).resolve().parents[2] SCRIPT = REPO_ROOT / "skills" / "skill-builder" / "scripts" / "scan_descriptions.py" @@ -27,7 +28,10 @@ def _load_module(): # Register before exec so dataclass introspection can resolve the module # (required on Python 3.14+ for importlib-loaded modules with dataclasses). sys.modules[spec.name] = module - spec.loader.exec_module(module) + # Match direct script execution: its sibling modules are importable even + # when pytest's importlib mode does not modify sys.path for test files. + with patch.object(sys, "path", [str(SCRIPT.parent), *sys.path]): + spec.loader.exec_module(module) return module @@ -55,11 +59,11 @@ def test_explicit_marker_in_description_detected(self): md = _write_skill( self.root, "alpha", - 'name: alpha\ndescription: Does a thing. Triggers: "do thing", "alpha".', + "name: alpha\ndescription: 'Does a thing. Triggers: \"do thing\", \"alpha\".'", ) result = scan.scan_skill(md) self.assertTrue(result.has_trigger) - self.assertIn("explicit-marker", result.forms) + self.assertIn("inline-marker", result.forms) def test_block_scalar_use_when_detected(self): md = _write_skill( @@ -69,7 +73,7 @@ def test_block_scalar_use_when_detected(self): ) result = scan.scan_skill(md) self.assertTrue(result.has_trigger) - self.assertIn("block-scalar", result.forms) + self.assertIn("block-marker", result.forms) def test_triggers_list_with_three_items_detected(self): md = _write_skill( @@ -80,7 +84,7 @@ def test_triggers_list_with_three_items_detected(self): ) result = scan.scan_skill(md) self.assertTrue(result.has_trigger) - self.assertIn("triggers-list", result.forms) + self.assertIn("metadata-list", result.forms) def test_two_item_triggers_list_not_enough(self): md = _write_skill( @@ -111,6 +115,15 @@ def test_suggestion_has_no_duplicate_words(self): # "compile compile" must not appear — verb equals the only name token. self.assertNotIn("compile compile", result.suggestion) + def test_invalid_frontmatter_is_rejected(self): + md = _write_skill( + self.root, + "invalid", + "name: invalid\ndescription: Unquoted colon: is invalid YAML.", + ) + with self.assertRaisesRegex(scan.ProfileError, "frontmatter configuration error"): + scan.scan_skill(md) + class TestCli(unittest.TestCase): def setUp(self): @@ -130,7 +143,7 @@ def test_strict_exits_zero_when_all_have_triggers(self): _write_skill( self.root, "ok", - 'name: ok\ndescription: Does X. Triggers: "ok", "do x".', + "name: ok\ndescription: 'Does X. Triggers: \"ok\", \"do x\".'", ) with redirect_stdout(io.StringIO()): code = scan.main([str(self.root), "--strict", "--quiet"]) @@ -140,6 +153,15 @@ def test_missing_dir_exits_two(self): code = scan.main([str(self.root / "does-not-exist")]) self.assertEqual(code, 2) + def test_unknown_profile_exits_two(self): + _write_skill(self.root, "plain", "name: plain\ndescription: Plain description.") + error = io.StringIO() + with patch.dict(scan.os.environ, {"SKILL_CONFORMANCE_PROFILE_ID": "unknown-test-profile"}): + with redirect_stderr(error): + code = scan.main([str(self.root), "--strict", "--quiet"]) + self.assertEqual(code, 2) + self.assertIn("unknown profile", error.getvalue()) + def test_probe_flow_form_allows_quoted_commas(self): _write_skill( self.root, @@ -152,7 +174,7 @@ def test_probe_flow_form_allows_quoted_commas(self): def test_json_output_reports_counts(self): _write_skill(self.root, "a", "name: a\ndescription: Plain.") - _write_skill(self.root, "b", 'name: b\ndescription: X. Triggers: "b", "x".') + _write_skill(self.root, "b", "name: b\ndescription: 'X. Triggers: \"b\", \"x\".'") buf = io.StringIO() with redirect_stdout(buf): scan.main([str(self.root), "--json"]) From b721d02559e1495be6095ad97b820e88ceb4a049 Mon Sep 17 00:00:00 2001 From: Bo Date: Sun, 13 Sep 2026 17:06:06 -0400 Subject: [PATCH 11/11] Provision full security gates and build pytest candidate CLI --- .github/workflows/nightly.yml | 4 +- .github/workflows/release.yml | 27 +++++++++- scripts/toolchain-validate.sh | 22 +++++++- tests/scripts/test-toolchain-validate.sh | 69 ++++++++++++++++++++++++ 4 files changed, 118 insertions(+), 4 deletions(-) diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index dbd50dd32..f25289324 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -84,6 +84,8 @@ jobs: - name: Install scanner tools run: | + sudo apt-get update -qq + sudo apt-get install -y shellcheck python -m pip install --upgrade pip python -m pip install semgrep==1.169.0 ruff==0.15.21 radon==6.0.1 pytest==9.1.1 \ -r evals/skills-rpi/requirements.txt \ @@ -115,7 +117,7 @@ jobs: - name: Run security gate (full) run: | chmod +x scripts/security-gate.sh - ./scripts/security-gate.sh --mode full + ./scripts/security-gate.sh --mode full --require-tools env: SECURITY_GATE_OUTPUT_DIR: ${{ runner.temp }}/agentops-security TOOLCHAIN_OUTPUT_DIR: ${{ runner.temp }}/agentops-tooling diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 3d2aed8db..3fcdaa72c 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -54,11 +54,34 @@ jobs: - name: Install scanner tools run: | + sudo apt-get update -qq + sudo apt-get install -y shellcheck python -m pip install --upgrade pip - python -m pip install semgrep==1.169.0 + python -m pip install semgrep==1.169.0 ruff==0.15.21 radon==6.0.1 pytest==9.1.1 \ + -r evals/skills-rpi/requirements.txt \ + -r evals/skills-rpi/requirements-readout.txt + GOBIN=/usr/local/bin go install github.com/securego/gosec/v2/cmd/gosec@v2.27.1 GOBIN=/usr/local/bin go install github.com/zricethezav/gitleaks/v8@v8.30.1 + # golangci-lint + GOBIN=/usr/local/bin go install github.com/golangci/golangci-lint/v2/cmd/golangci-lint@v2.13.1 + + # Use the same scanner pins as the nightly full-security lane. + # govulncheck covers known-CVE reachability in dependencies and stdlib. + GOBIN=/usr/local/bin go install golang.org/x/vuln/cmd/govulncheck@v1.6.0 + + # trivy — pinned tag, download-then-execute. Piping a mutable-branch + # install script into sh was the semgrep gha-curl-pipe-shell CRITICAL + # that blocked the v3.2.0 publisher (supply-chain: a hijacked script + # would run unread). + curl -sSfL -o /tmp/trivy-install.sh https://raw.githubusercontent.com/aquasecurity/trivy/v0.72.0/contrib/install.sh + sh /tmp/trivy-install.sh -b /usr/local/bin v0.72.0 + + # hadolint + curl -sSfL -o /usr/local/bin/hadolint "https://github.com/hadolint/hadolint/releases/download/v2.14.0/hadolint-Linux-x86_64" + chmod +x /usr/local/bin/hadolint + - name: Generate pre-publish release evidence run: | mkdir -p release-artifacts @@ -71,7 +94,7 @@ jobs: # but fail before GoReleaser when the gate returns findings. chmod +x scripts/security-gate.sh set +e - ./scripts/security-gate.sh --mode full --json > release-artifacts/security-gate-summary.json + ./scripts/security-gate.sh --mode full --json --require-tools > release-artifacts/security-gate-summary.json SECURITY_RC=$? set -e jq empty release-artifacts/security-gate-summary.json diff --git a/scripts/toolchain-validate.sh b/scripts/toolchain-validate.sh index 9b453b425..3815c868f 100755 --- a/scripts/toolchain-validate.sh +++ b/scripts/toolchain-validate.sh @@ -515,10 +515,30 @@ run_pytest() { return 0 fi + # Evidence CLI tests require the exact candidate, never an ambient ao on + # PATH. Honor an explicit caller candidate; otherwise build this checkout. + local ao_bin="${AO_BIN:-}" + local ao_build_dir="" + if [[ -z "$ao_bin" ]]; then + ao_build_dir="$(mktemp -d "${TMPDIR:-/tmp}/agentops-pytest-ao.XXXXXX")" + ao_bin="$ao_build_dir/ao" + local build_rc=0 + (cd "$REPO_ROOT/cli" && go build -o "$ao_bin" ./cmd/ao) > "$OUTPUT_DIR/pytest-build.txt" 2>&1 || build_rc=$? + if [[ "$build_rc" -ne 0 || ! -x "$ao_bin" ]]; then + printf 'ERROR: candidate ao build failed (exit %s); pytest was not run\n' "$build_rc" > "$output_file" + cat "$OUTPUT_DIR/pytest-build.txt" >> "$output_file" + TOOL_STATUS["pytest"]="error" + CRITICAL_COUNT=$((CRITICAL_COUNT + 1)) + rm -rf "$ao_build_dir" + return 0 + fi + fi + # Source skills and their generated projections can share test basenames. # Importlib collects both without Python module-name collisions. local pytest_rc=0 - pytest "$REPO_ROOT" --import-mode=importlib --tb=short -q > "$output_file" 2>&1 || pytest_rc=$? + AO_BIN="$ao_bin" pytest "$REPO_ROOT" --import-mode=importlib --tb=short -q > "$output_file" 2>&1 || pytest_rc=$? + if [[ -n "$ao_build_dir" ]]; then rm -rf "$ao_build_dir"; fi if [[ "$pytest_rc" -eq 0 ]]; then echo "PASS" >> "$output_file" TOOL_STATUS["pytest"]="pass" diff --git a/tests/scripts/test-toolchain-validate.sh b/tests/scripts/test-toolchain-validate.sh index d2ae33683..e2dbdc3c7 100755 --- a/tests/scripts/test-toolchain-validate.sh +++ b/tests/scripts/test-toolchain-validate.sh @@ -32,7 +32,25 @@ case "$name" in trivy) printf '{"Results":[]}\n' ;; gosec) printf '{"Issues":[]}\n' ;; hadolint) printf '[]\n' ;; + go) + if [[ "${1:-}" == "build" ]]; then + printf 'go-build-cwd %s\n' "$PWD" >> "$SCANNER_LOG" + if [[ "${AO_BUILD_EXIT:-0}" -ne 0 ]]; then + printf 'fixture build error\n' >&2 + exit "$AO_BUILD_EXIT" + fi + [[ "${2:-}" == "-o" && "${4:-}" == "./cmd/ao" ]] || exit 9 + printf '#!/bin/sh\nprintf "candidate-ready\\n"\n' > "$3" + chmod +x "$3" + fi + ;; pytest) + printf 'pytest-ao %s\n' "${AO_BIN:-}" >> "$SCANNER_LOG" + if [[ ! -x "${AO_BIN:-}" ]]; then + printf 'ERROR: AO_BIN is not an executable candidate\n' + exit 2 + fi + "$AO_BIN" >> "$SCANNER_LOG" || exit 2 if [[ "${PYTEST_EXIT:-0}" -eq 1 ]]; then printf 'FAILED test_example.py::test_example - AssertionError\n' elif [[ "${PYTEST_EXIT:-0}" -ne 0 ]]; then @@ -246,6 +264,56 @@ test_large_python_inventory() { fi } +test_pytest_candidate_binary() { + local output candidate build_dir status + : > "$SCANNER_LOG" + output=$(AO_BIN='' ./scripts/toolchain-validate.sh --all --gate --json) + candidate=$(sed -n 's/^pytest-ao //p' "$SCANNER_LOG") + build_dir="${candidate%/*}" + if jq -e '.tools.pytest == "pass"' <<< "$output" >/dev/null && + grep -Fxq "go-build-cwd $PWD/cli" "$SCANNER_LOG" && + grep -Fxq 'candidate-ready' "$SCANNER_LOG" && [[ -n "$candidate" && ! -e "$build_dir" ]]; then + pass "pytest executes this checkout's temporary ao and cleans it afterward" + else + fail "pytest did not receive or clean the candidate ao" + fi + + candidate="$MOCK_DIR/caller-ao" + printf '#!/bin/sh\nprintf "caller-candidate\\n"\n' > "$candidate" + chmod +x "$candidate" + : > "$SCANNER_LOG" + output=$(AO_BIN="$candidate" ./scripts/toolchain-validate.sh --all --gate --json) + if jq -e '.tools.pytest == "pass"' <<< "$output" >/dev/null && + grep -Fxq "pytest-ao $candidate" "$SCANNER_LOG" && + grep -Fxq 'caller-candidate' "$SCANNER_LOG" && + ! grep -q '^go build ' "$SCANNER_LOG" && [[ -x "$candidate" ]]; then + pass "explicit AO_BIN is preserved without building or deleting it" + else + fail "explicit AO_BIN was replaced or removed" + fi + + : > "$SCANNER_LOG" + status=0 + output=$(AO_BIN='' AO_BUILD_EXIT=7 ./scripts/toolchain-validate.sh --all --gate --json) || status=$? + candidate=$(sed -n 's/^go build -o \(.*\) \.\/cmd\/ao$/\1/p' "$SCANNER_LOG") + if [[ "$status" -eq 2 ]] && jq -e '.tools.pytest == "error" and .gate_status == "BLOCKED_CRITICAL"' <<< "$output" >/dev/null && + ! grep -q '^pytest ' "$SCANNER_LOG" && [[ -n "$candidate" && ! -e "${candidate%/*}" ]]; then + pass "candidate build failure blocks before pytest and cleans temporary output" + else + fail "failed candidate build reached pytest or left a green gate" + fi + + : > "$SCANNER_LOG" + status=0 + output=$(AO_BIN='' PYTEST_EXIT=2 ./scripts/toolchain-validate.sh --all --gate --json) || status=$? + candidate=$(sed -n 's/^pytest-ao //p' "$SCANNER_LOG") + if [[ "$status" -eq 2 && -n "$candidate" && ! -e "${candidate%/*}" ]]; then + pass "pytest failure also cleans the temporary candidate binary" + else + fail "pytest failure leaked its temporary candidate binary" + fi +} + # Test 8: Output directory is created test_output_dir() { local test_dir @@ -278,6 +346,7 @@ test_gate_scope test_pytest_failure_exits test_duplicate_test_modules test_large_python_inventory +test_pytest_candidate_binary echo "" echo "================================"