diff --git a/.claude/SWARM.md b/.claude/SWARM.md deleted file mode 100644 index b5629bb..0000000 --- a/.claude/SWARM.md +++ /dev/null @@ -1,53 +0,0 @@ -# CPZAI Agent Swarm - -Swarm infrastructure per Alexandr Wang's formula: the right agentic loop plus the right evaluation system and metric for the agents to optimize. Installed identically across all CPZ repos; each repo adds its own `.claude/swarm-context.md` with local invariants and traps that every swarm agent reads first. - -## The evaluation system (why this beats one big agent) - -Nothing an agent claims is trusted. Every layer is graded by an independent adversary: - -- **Finders** are scored on confirmed findings; every candidate faces a 3-refuter panel that defaults to "refuted" when uncertain. Only 2-of-3 survivors count. The audit loop keeps spawning finder rounds until two consecutive rounds surface nothing new (converged), and reports precision (confirmed/candidates) as its metric. -- **Implementers** face a ship gate: a reviewer reads the actual `git diff` (never the implementer's report), re-runs verification itself, and blocks with concrete blockers. Blocked work loops back for repair, max 3 attempts. Metric: ship verdict + 0-10 score. -- **Designs** compete: three angles (minimal-diff, robustness-first, product-first) are drafted in parallel and a judge picks the winner, grafting the losers' best ideas. - -## How to run it - -Say any of these in a Claude Code session in the repo: - -- "run the swarm-audit workflow" — full-repo hunt for confirmed defects. Scoped: "run swarm-audit with scope: the billing edge functions". -- "run the swarm-fix workflow with the confirmed findings" — fixes an audit's output one finding at a time, each fix independently verified. Changes stay uncommitted for your review. -- "run the swarm-build workflow: " — feature/fix built end to end through design competition and the ship gate. - -Or turn on **ultracode** (say "ultracode" in the prompt) to make Claude orchestrate swarms for every substantive task by default. - -## Self-improvement loop - -Every workflow ends with a **Retro phase** that makes the next run measurably better: - -1. Appends this run's metrics (audit precision/rounds, fix rate, gate attempts/score) as one JSON line to `.claude/swarm-metrics.jsonl` — the permanent ledger; trends across runs are the swarm's report card. -2. Mines the run's failures: refuted candidates become false-positive patterns finders must not repeat; gate blockers become implementer guidance; confirmed-defect clusters become hot spots for extra depth. -3. Writes those lessons to `.claude/swarm-tuning.md`, which every swarm agent reads at the start of the next run — so precision and fix rate compound over time. -4. Promotes durable new invariants into `.claude/swarm-context.md`, and proposes (never applies) workflow-script changes under "Proposed script changes" in the tuning file for a human or the main session to adopt. - -5. Emails Chris a run report: the full report (metrics + trend, findings/fixes, every improvement made, proposed script changes) is saved to `.claude/swarm-reports/` and dropped as a **Gmail draft** to chris@cpz-lab.com with subject `[Swarm] report: `. Drafts only — the connected Gmail has no send capability, so nothing auto-sends; check the Drafts folder after a run. - -The Retro agent may touch ONLY those files (ledger, tuning, context, reports). Check `swarm-metrics.jsonl` occasionally: falling precision or rising gate attempts means the tuning file needs pruning. - -## Pieces - -- `.claude/agents/` — swarm-scout, swarm-finder, swarm-verifier, swarm-fixer, swarm-reviewer. Also usable individually via the Agent tool. -- `.claude/workflows/` — swarm-audit.js, swarm-build.js, swarm-fix.js. -- `.claude/swarm-context.md` — this repo's stack, invariants, and known traps. Keep it current; every swarm agent reads it first, so a stale entry misleads the whole swarm. - -## Cost controls (built in) - -- Verifier panels are severity-adaptive: critical/high findings get 3 refuters (2-of-3), medium/low get 2 (unanimous). Panels run on Sonnet — refutation is a focused code-trace task, and the panel is the workflow's token multiplier. -- Finders cap at 8 findings per lens per round with tight evidence; verification caps at the top 12 candidates per round by severity (dropped ones are logged, never silent). -- Retro agents run on Sonnet. -- Every loop honors a token budget: say "+300k" in your prompt to hard-cap a run; workflows stop cleanly and report what they skipped. -- Cheapest lever is scope: "run swarm-audit with scope: the billing edge functions" costs a fraction of a full-repo sweep. - -## Notes - -- Swarms are token-hungry (an unscoped audit can run 30+ agents). Scope audits when you don't need the full repo. -- Fixers and workflows never commit, push, or deploy. You keep the "git add commit push all" ritual. diff --git a/.claude/agents/swarm-finder.md b/.claude/agents/swarm-finder.md deleted file mode 100644 index 843867a..0000000 --- a/.claude/agents/swarm-finder.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -name: swarm-finder -description: Read-only defect hunter for swarm workflows. Given a scope and a lens, finds real, evidenced defects and returns them as structured findings. Never edits files. -tools: Read, Grep, Glob, Bash ---- - -You are a defect finder in an agent swarm. Many finders run in parallel, each with a different lens; an adversarial verification panel will try to refute everything you report. Your score is confirmed findings, and false positives cost you, so report only what you can prove. - -Method: -1. If `.claude/swarm-context.md` exists in the repo root, read it first. It lists this repo's stack, invariants, and known traps. -2. Hunt strictly within your assigned lens and scope. Depth beats breadth: trace real code paths end to end rather than skimming many files. -3. A finding must be evidenced by code you actually read. Cite the file, the line, and the exact mechanism of failure. -4. Every finding needs a concrete failure scenario: specific inputs or state that lead to wrong output, data loss, security exposure, or a crash. -5. Do not re-report anything on the already-known list you are given. - -Rules: -- No style nits, no hypotheticals, no "could be cleaner", no TODO archaeology. -- Bash is for read-only investigation only (git log, ls, grep, reading configs). Never modify files. Never run anything with side effects. -- Severity: critical = money loss, data loss, or security breach; high = user-visible breakage; medium = incorrect behavior with a workaround; low = latent trap. -- Your final output is consumed by a machine. Return findings exactly in the requested structure, nothing else. diff --git a/.claude/agents/swarm-fixer.md b/.claude/agents/swarm-fixer.md deleted file mode 100644 index d6abaf6..0000000 --- a/.claude/agents/swarm-fixer.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -name: swarm-fixer -description: Implementer for swarm workflows. Takes a confirmed defect or an approved design and produces the minimal correct change, runs the relevant tests, and reports honestly. ---- - -You are the implementer in an agent swarm. A reviewer gate will score your work against the task; changes that don't actually solve the task, break something else, or fake their way past verification get bounced back to you. - -Method: -1. If `.claude/swarm-context.md` exists in the repo root, read it first. It lists this repo's invariants and known traps; violating them is an automatic gate failure. -2. Read the code you are about to change AND its callers before editing. Match the surrounding style, naming, and idiom. -3. Make the minimal change that correctly solves the task. No drive-by refactors, no scope creep. -4. Run the narrowest relevant test/build/typecheck command that proves the change works. If the repo has no way to verify, say so explicitly. -5. Report: what changed (files + why), what you ran, and the ACTUAL results. - -Hard rules (these repos trade real money): -- Fail loudly. Never add silent fallbacks, mock data, placeholder values, or catch-and-ignore blocks. -- Never weaken auth, RLS, validation, or money-safety checks to make something pass. -- Never fabricate test results or claim success you did not verify. A report of "tests fail because X" is acceptable; a false "all green" is not. -- Do not commit, push, or deploy. Leave changes uncommitted in the working tree. diff --git a/.claude/agents/swarm-reviewer.md b/.claude/agents/swarm-reviewer.md deleted file mode 100644 index cf25b3e..0000000 --- a/.claude/agents/swarm-reviewer.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -name: swarm-reviewer -description: Ship gate for swarm workflows. Reviews the current uncommitted changes against a task, runs verification, and returns a ship/no-ship verdict with concrete blockers. -tools: Read, Grep, Glob, Bash ---- - -You are the ship gate of an agent swarm. The implementer's report is a claim, not evidence; your verdict is the evaluation metric the swarm optimizes, so a soft gate makes the whole swarm worthless. - -Method: -1. If `.claude/swarm-context.md` exists in the repo root, read it. A change violating a listed invariant is an automatic blocker. -2. Run `git diff` (and `git status`) to see exactly what changed. Review the diff, not the implementer's description of it. -3. Check the change against the task: does it fully solve it? Does it handle the edge cases the task implies? -4. Hunt for regressions: read the callers of changed code, check for broken contracts, weakened validation, silent fallbacks, or faked data. -5. Independently run the verification the implementer claims to have run (tests, typecheck, build) when feasible, and trust your own results over their report. - -Verdict rules: -- ship=true only if you would merge this to a money-handling production system. score 0-10 reflects quality beyond mere correctness. -- ship=false requires concrete, actionable blockers, each pointing at a file and the specific problem. No vague "needs more tests". -- An empty diff when the task required changes is ship=false. -- You may run tests/builds via Bash, but never edit files: report blockers, do not fix them. -- Return exactly the requested structure. diff --git a/.claude/agents/swarm-scout.md b/.claude/agents/swarm-scout.md deleted file mode 100644 index 7a6f8b7..0000000 --- a/.claude/agents/swarm-scout.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -name: swarm-scout -description: Read-only codebase mapper for swarm workflows. Given a task, maps the relevant subsystems, files, data flows, and invariants so designers and fixers start grounded. Never edits files. -tools: Read, Grep, Glob, Bash ---- - -You are the scout of an agent swarm. Downstream agents (designers, implementers) will act on your map without re-reading the whole repo, so wrong or missing entries cause wrong implementations. - -Method: -1. If `.claude/swarm-context.md` exists in the repo root, read it first and fold its contents into your map. -2. Locate every file the task will touch or depend on: entry points, the code to change, its callers, shared types/schemas, config, and the tests that cover it. -3. Trace the data flow end to end (request → handler → storage → response, or the repo's equivalent) and note where the task intersects it. -4. Record invariants and traps: naming conventions, error-handling patterns, auth/permission checks, schema constraints, anything that a naive change would break. -5. Note how this repo is tested and how a change is verified locally. - -Rules: -- Read-only. Never modify files; Bash only for investigation (git log, ls, grep). -- Report facts you verified by reading code, and clearly mark anything you inferred but did not verify. -- Output a compact structured map: relevant files with one-line roles, the data flow, invariants/traps, and verification steps. No prose padding. diff --git a/.claude/agents/swarm-verifier.md b/.claude/agents/swarm-verifier.md deleted file mode 100644 index 839acee..0000000 --- a/.claude/agents/swarm-verifier.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -name: swarm-verifier -description: Adversarial skeptic for swarm workflows. Given a claimed defect, tries hard to refute it by reading the actual code. Read-only. -tools: Read, Grep, Glob, Bash ---- - -You are the evaluation layer of an agent swarm. You receive one claimed defect. Your job is to REFUTE it. Assume the finder was wrong, lazy, or hallucinating until the code forces you to conclude otherwise. - -Method: -1. If `.claude/swarm-context.md` exists in the repo root, read it. Known traps listed there may explain behavior the finder misread. -2. Open the cited file at the cited location. Verify the quoted code actually exists and says what the finding claims. -3. Trace the full path: callers, guards, validation upstream, error handling downstream. Most false positives die here because a guard the finder never read already prevents the scenario. -4. Attempt the failure scenario mentally with concrete values. If the scenario cannot actually occur, the finding is refuted. -5. Check git log/blame for context if intent is unclear. - -Verdict rules: -- refuted=true if the code does not exist as claimed, a guard prevents the scenario, the behavior is intentional and correct, or the impact is not real. -- refuted=false ONLY when you traced the path yourself and the failure scenario genuinely occurs. -- If uncertain after honest effort, default to refuted=true. Unverifiable claims must not survive. -- Bash is read-only investigation only. Never modify anything. -- Return exactly the requested structure. Put your trace evidence in the reasoning field. diff --git a/.claude/swarm-context.md b/.claude/swarm-context.md deleted file mode 100644 index 553404e..0000000 --- a/.claude/swarm-context.md +++ /dev/null @@ -1,12 +0,0 @@ -# Swarm context: aquila-mcp-server - -## Stack -- MCP server exposing the CPZ platform to AI clients, deployed at mcp.cpz-lab.com. - -## Hard invariants -- Known trap: `CPZ_API_BASE_URL` misconfiguration silently points tools at the wrong environment; verify env resolution when touching request paths. -- Tools can place trades and read portfolios; auth and confirmation gates are money-safety features, never weaken or bypass them. -- No fallbacks, no mock data; a tool that fabricates a response on upstream failure is a critical defect. - -## Verification -- Run the test suite; exercise tools against the paper environment only. diff --git a/.claude/swarm-tuning.md b/.claude/swarm-tuning.md deleted file mode 100644 index 7e4ac93..0000000 --- a/.claude/swarm-tuning.md +++ /dev/null @@ -1,15 +0,0 @@ -# Swarm tuning (self-updated) - -Living lessons from past swarm runs in this repo. Written by each workflow's Retro agent; read by every swarm agent at the start of a run. The metrics ledger is `.claude/swarm-metrics.jsonl` (one JSON line per run). Humans may edit, but keep entries evidence-grounded: wrong guidance here compounds across every future run. - -## False-positive patterns (finders: do not repeat) -(none recorded yet) - -## Hot spots (extra depth here next run) -(none recorded yet) - -## Fixer/implementer lessons (what gate reviews keep rejecting) -(none recorded yet) - -## Proposed script changes (Retro proposes; a human or main session applies) -(none yet) diff --git a/.claude/workflows/swarm-audit.js b/.claude/workflows/swarm-audit.js deleted file mode 100644 index 44105bf..0000000 --- a/.claude/workflows/swarm-audit.js +++ /dev/null @@ -1,181 +0,0 @@ -export const meta = { - name: 'swarm-audit', - description: 'Multi-lens defect hunt with adversarial 3-vote verification, looping until two dry rounds', - whenToUse: 'Audit a repo or subsystem for real, confirmed defects. Optional args: {scope: "...", maxRounds: N}. Feed the confirmed output into swarm-fix.', - phases: [ - { title: 'Find', detail: '4 lenses per round: correctness, security, data-integrity, integration' }, - { title: 'Verify', detail: '3 adversarial refuters per finding; survives on 2+ non-refutals' }, - { title: 'Retro', detail: 'append run metrics to the ledger; update tuning and context so the next run is sharper' }, - ], -} - -const FINDINGS = { - type: 'object', - required: ['findings'], - properties: { - findings: { - type: 'array', - items: { - type: 'object', - required: ['file', 'title', 'severity', 'evidence', 'failure_scenario'], - properties: { - file: { type: 'string' }, - line: { type: 'integer' }, - title: { type: 'string' }, - severity: { enum: ['critical', 'high', 'medium', 'low'] }, - evidence: { type: 'string' }, - failure_scenario: { type: 'string' }, - }, - }, - }, - }, -} - -const VERDICT = { - type: 'object', - required: ['refuted', 'reasoning'], - properties: { - refuted: { type: 'boolean' }, - reasoning: { type: 'string' }, - }, -} - -// Role preambles are inlined (mirroring .claude/agents/swarm-*.md) so the -// workflow works even in sessions where those agent types are not registered. -const FINDER_ROLE = [ - 'You are a defect finder in an agent swarm. An adversarial verification panel will try to refute everything you report; false positives cost you, so report only what you can prove.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both first: context lists this repo stack, invariants, and known traps; tuning lists lessons from previous swarm runs, including false-positive patterns you must not repeat and hot spots worth extra depth.', - 'Hunt strictly within your assigned lens and scope. Depth beats breadth: trace real code paths end to end.', - 'A finding must be evidenced by code you actually read: cite file, line, and the exact mechanism. Every finding needs a concrete failure scenario (specific inputs or state leading to wrong output, data loss, security exposure, or a crash).', - 'No style nits, no hypotheticals, no TODO archaeology. Do not re-report anything on the already-known list.', - 'Report at most 8 findings, your best by severity. Keep evidence fields tight: the cited code plus the mechanism, not essays.', - 'Severity: critical = money loss, data loss, or security breach; high = user-visible breakage; medium = incorrect behavior with a workaround; low = latent trap.', - 'You are read-only: never modify files; Bash only for read-only investigation (git log, ls, grep).', -].join('\n') - -const VERIFIER_ROLE = [ - 'You are the adversarial evaluation layer of an agent swarm. You receive one claimed defect. Your job is to REFUTE it: assume the finder was wrong until the code forces you to conclude otherwise.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both; known traps and past-run lessons there may explain behavior the finder misread.', - 'Open the cited file, verify the quoted code exists and says what is claimed, then trace the full path: callers, guards, upstream validation, downstream error handling. Attempt the failure scenario mentally with concrete values.', - 'refuted=true if the code is not as claimed, a guard prevents the scenario, the behavior is intentional, or the impact is not real. refuted=false ONLY when you traced the path yourself and the failure genuinely occurs.', - 'If uncertain after honest effort, default to refuted=true. You are read-only: never modify anything. Put your trace evidence in the reasoning field.', -].join('\n') - -let parsedArgs = args -if (typeof parsedArgs === 'string') { - try { parsedArgs = JSON.parse(parsedArgs) } catch (e) { parsedArgs = { scope: args } } -} -const scope = (parsedArgs && parsedArgs.scope) || - 'the whole repository, prioritizing money paths, auth and permission checks, data writes, and recently changed code (check git log)' -const maxRounds = (parsedArgs && parsedArgs.maxRounds) || 3 - -const LENSES = [ - 'correctness: logic errors, inverted or wrong conditions, off-by-one, broken state machines, wrong variable used', - 'security: auth/permission bypass, RLS gaps, injection, secret exposure, unsafe defaults, trust of client input', - 'data-integrity: silent failures, swallowed errors, fallbacks that fake data, race conditions, partial writes', - 'integration: API contract mismatches, schema drift, wrong column or table names, env/config drift, stale types', -] - -const seen = new Set() -const confirmed = [] -const refutedLog = [] -const key = f => ((f.file || '') + ':' + (f.title || '')).toLowerCase() - -let round = 0 -let dry = 0 -while (dry < 2 && round < maxRounds) { - if (budget.total && budget.remaining() < 40000) { - log('Token budget nearly exhausted (' + Math.round(budget.remaining() / 1000) + 'k left): stopping before round ' + (round + 1)) - break - } - round += 1 - const known = Array.from(seen).join('\n') || '(none yet)' - const found = (await parallel(LENSES.map(lens => () => - agent( - FINDER_ROLE + '\n\nAudit scope: ' + scope + '\nYour lens: ' + lens + - '\nAlready-known findings, do NOT re-report these (file:title):\n' + known, - { label: 'find:' + lens.split(':')[0] + ':r' + round, phase: 'Find', schema: FINDINGS } - ) - ))).filter(Boolean).flatMap(r => r.findings || []) - - let fresh = found.filter(f => !seen.has(key(f))) - if (fresh.length === 0) { - dry += 1 - log('Round ' + round + ': dry (' + dry + '/2)') - continue - } - dry = 0 - fresh.forEach(f => seen.add(key(f))) - const sevOrder = { critical: 0, high: 1, medium: 2, low: 3 } - fresh.sort((a, b) => (sevOrder[a.severity] || 9) - (sevOrder[b.severity] || 9)) - if (fresh.length > 12) { - log('Round ' + round + ': capping verification at top 12 of ' + fresh.length + ' candidates by severity; dropped: ' + - fresh.slice(12).map(f => f.severity + ':' + f.title).join(' | ')) - fresh = fresh.slice(0, 12) - } - log('Round ' + round + ': ' + fresh.length + ' new candidate(s), sending to adversarial verification') - - // Cost-adaptive panels: critical/high get 3 refuters (2-of-3 to survive); - // medium/low get 2 refuters (unanimous non-refutal to survive). Verifiers run - // on sonnet: refutation is a focused code-trace task, and the panel is the - // token multiplier of the whole workflow. - const judged = await parallel(fresh.map(f => () => { - const panel = (f.severity === 'critical' || f.severity === 'high') ? 3 : 2 - const needed = 2 - return parallel(Array.from({ length: panel }, (_, i) => () => - agent( - VERIFIER_ROLE + '\n\nYou are refuter #' + (i + 1) + ' of ' + panel + '. Claimed defect:\n' + JSON.stringify(f, null, 2), - { label: 'verify:' + (f.file || '?'), phase: 'Verify', schema: VERDICT, model: 'sonnet' } - ) - )).then(votes => { - const valid = votes.filter(Boolean) - return { f, valid, survives: valid.filter(v => !v.refuted).length >= needed } - }) - })) - const survivors = judged.filter(Boolean).filter(j => j.survives) - survivors.forEach(j => confirmed.push(Object.assign({}, j.f, { verifier_notes: j.valid.map(v => v.reasoning) }))) - judged.filter(Boolean).filter(j => !j.survives).forEach(j => refutedLog.push({ - file: j.f.file, - title: j.f.title, - severity: j.f.severity, - why_refuted: (j.valid.filter(v => v.refuted)[0] || {}).reasoning || 'no valid verdicts', - })) - log('Round ' + round + ': ' + survivors.length + '/' + fresh.length + ' survived verification') -} - -const bySeverity = {} -confirmed.forEach(f => { bySeverity[f.severity] = (bySeverity[f.severity] || 0) + 1 }) -const order = { critical: 0, high: 1, medium: 2, low: 3 } -confirmed.sort((a, b) => (order[a.severity] || 9) - (order[b.severity] || 9)) - -const metric = { - candidates: seen.size, - confirmed: confirmed.length, - precision: seen.size ? Math.round(100 * confirmed.length / seen.size) + '%' : 'n/a', - rounds: round, - converged: dry >= 2, - bySeverity, -} - -phase('Retro') -const RETRO_ROLE = [ - 'You are the self-improvement layer of an agent swarm. Your job: make the NEXT swarm run measurably better than this one. You may create or edit ONLY these files, nothing else: .claude/swarm-metrics.jsonl, .claude/swarm-tuning.md, .claude/swarm-context.md, and report files under .claude/swarm-reports/.', - '1. Append exactly one JSON line for this run to .claude/swarm-metrics.jsonl containing: ts (ISO timestamp from running `date -u +%Y-%m-%dT%H:%M:%SZ`), workflow name, and the metric object you were given.', - '2. Read the whole ledger and compare against past runs: is precision falling, are rounds rising, are the same false-positive patterns recurring? State the trend.', - '3. Update .claude/swarm-tuning.md (create it from the existing template if missing): distill the refuted candidates into concrete false-positive patterns finders must not repeat; note hot spots (files/subsystems where confirmed defects clustered) deserving extra depth next run; under "## Proposed script changes", propose (never apply) any workflow-script improvements the evidence supports. Prune guidance that past runs prove stale; keep the file under 120 lines.', - '4. If a confirmed finding reveals a DURABLE repo invariant or trap not yet in .claude/swarm-context.md, add it there in the existing style. Only durable facts; run-specific noise goes in tuning, not context.', - '5. Email the report to Chris Preuss: write the full run report as markdown to .claude/swarm-reports/-swarm-audit.md (create the directory if needed) covering what ran, the metrics with trend vs the ledger, the confirmed findings, every improvement you made to tuning/context, and any proposed script changes. Then use ToolSearch (query "create_draft") to load the Gmail draft tool and create a draft addressed to chris@cpz-lab.com with subject "[Swarm] audit report: " and the report as the body. DRAFT ONLY, never send. If no Gmail tool is available in this session, note that in your summary and continue; do not fail the retro over it.', - 'Be conservative: every line you write is read by every future swarm agent, so wrong guidance compounds. Ground every claim in the run record you were given. Report back a short summary of the trend and what you changed.', -].join('\n') -const retro = await agent( - RETRO_ROLE + '\n\nRun record for this swarm-audit:\n' + JSON.stringify({ - workflow: 'swarm-audit', - scope, - metric, - confirmed: confirmed.map(f => ({ file: f.file, title: f.title, severity: f.severity })), - refuted: refutedLog, - }, null, 2), - { label: 'retro', phase: 'Retro', model: 'sonnet' } -) - -return { confirmed, metric, retro } diff --git a/.claude/workflows/swarm-build.js b/.claude/workflows/swarm-build.js deleted file mode 100644 index f2f3e0d..0000000 --- a/.claude/workflows/swarm-build.js +++ /dev/null @@ -1,155 +0,0 @@ -export const meta = { - name: 'swarm-build', - description: 'Scout, competing designs judged by a panel, winning design implemented, then a review-gated repair loop until ship', - whenToUse: 'Build a feature or fix end to end with design competition and a hard ship gate. Required args: {task: "..."}.', - phases: [ - { title: 'Scout', detail: 'map the relevant code' }, - { title: 'Design', detail: '3 competing designs + judge' }, - { title: 'Implement', detail: 'winning design implemented' }, - { title: 'Gate', detail: 'review-gated repair loop, max 3 attempts' }, - { title: 'Retro', detail: 'append run metrics to the ledger; update tuning so the next run is sharper' }, - ], -} - -const DESIGN = { - type: 'object', - required: ['plan', 'files', 'risks', 'verification'], - properties: { - plan: { type: 'string' }, - files: { type: 'array', items: { type: 'string' } }, - risks: { type: 'string' }, - verification: { type: 'string' }, - }, -} - -const JUDGMENT = { - type: 'object', - required: ['winner', 'why'], - properties: { - winner: { type: 'integer' }, - why: { type: 'string' }, - merge_ideas: { type: 'string' }, - }, -} - -const REVIEW = { - type: 'object', - required: ['ship', 'score', 'blockers'], - properties: { - ship: { type: 'boolean' }, - score: { type: 'integer' }, - blockers: { type: 'array', items: { type: 'string' } }, - }, -} - -// Role preambles inlined (mirroring .claude/agents/swarm-*.md) so the workflow -// works even in sessions where those agent types are not registered. -const SCOUT_ROLE = [ - 'You are the scout of an agent swarm. Downstream designers and an implementer will act on your map without re-reading the repo, so wrong or missing entries cause wrong implementations.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both first and fold them into your map.', - 'Locate every file the task will touch or depend on: entry points, code to change, callers, shared types/schemas, config, and covering tests. Trace the data flow end to end. Record invariants and traps a naive change would break. Note how a change is verified locally.', - 'You are read-only: never modify files; Bash only for investigation. Mark anything inferred but not verified. Output a compact structured map, no prose padding.', -].join('\n') - -const FIXER_ROLE = [ - 'You are the implementer in an agent swarm. A reviewer gate will score your work against the task; changes that do not solve the task, break something else, or fake verification get bounced back.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both first; violating a listed invariant is an automatic gate failure, and tuning lists what past gate reviews kept rejecting.', - 'Read the code you are about to change AND its callers before editing. Match surrounding style. Make the minimal change that correctly solves the task; no scope creep.', - 'Run the narrowest relevant test/build/typecheck that proves the change works, and report what you ran with the ACTUAL results.', - 'Hard rules: fail loudly, never add silent fallbacks, mock data, or catch-and-ignore; never weaken auth, RLS, validation, or money-safety checks; never fabricate results. Do not commit, push, or deploy; leave changes uncommitted.', -].join('\n') - -const REVIEWER_ROLE = [ - 'You are the ship gate of an agent swarm. The implementer report is a claim, not evidence.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both; a change violating a listed invariant is an automatic blocker.', - 'Run git diff and git status; review the diff itself, not the description of it. Check the change fully solves the task and its implied edge cases. Hunt regressions: read callers of changed code, look for broken contracts, weakened validation, silent fallbacks, faked data. Re-run claimed verification yourself when feasible and trust your own results.', - 'ship=true only if you would merge this to a money-handling production system; score 0-10. ship=false requires concrete blockers each pointing at a file and the specific problem. An empty diff when changes were required is ship=false. Never edit files yourself.', -].join('\n') - -const task = typeof args === 'string' ? args : (args && args.task) -if (!task) return { error: 'Usage: Workflow({name: "swarm-build", args: {task: ""}})' } - -phase('Scout') -const map = await agent( - SCOUT_ROLE + '\n\nMap the code relevant to this task so designers and an implementer can work from your map alone.\nTask: ' + task, - { label: 'scout' } -) - -phase('Design') -const ANGLES = [ - 'minimal-diff: the smallest correct change that fully solves the task', - 'robustness-first: enumerate failure modes and edge cases; no silent fallbacks, fail loudly', - 'product-first: the best user-facing outcome within reasonable scope', -] -const designs = (await parallel(ANGLES.map((angle, i) => () => - agent( - 'Produce an implementation design. You are competing against two other designs; a judge picks one winner.\nAngle: ' + angle + - '\nTask: ' + task + '\nCodebase map:\n' + map, - { label: 'design:' + angle.split(':')[0], phase: 'Design', schema: DESIGN } - ) -))).filter(Boolean) -if (!designs.length) return { error: 'All design agents failed' } - -const judgment = await agent( - 'Judge these competing implementation designs for the task. Return the 0-based index of the winner and any ideas from the losers worth merging in.' + - '\nTask: ' + task + '\nCodebase map:\n' + map + '\nDesigns:\n' + JSON.stringify(designs, null, 2), - { label: 'judge', phase: 'Design', schema: JUDGMENT } -) -const winner = designs[judgment && judgment.winner] || designs[0] - -phase('Implement') -let report = await agent( - FIXER_ROLE + '\n\nImplement this approved design. Leave changes uncommitted.\nTask: ' + task + - '\nDesign:\n' + JSON.stringify(winner, null, 2) + - '\nIdeas to merge from competing designs: ' + ((judgment && judgment.merge_ideas) || 'none') + - '\nRun the verification the design specifies and report actual results.', - { label: 'implement', phase: 'Implement' } -) - -phase('Gate') -let verdict = null -let attempts = 0 -for (let attempt = 1; attempt <= 3; attempt++) { - attempts = attempt - verdict = await agent( - REVIEWER_ROLE + '\n\nReview the current uncommitted changes against this task and return your verdict.\nTask: ' + task + - '\nImplementer report (treat as a claim, not evidence):\n' + report, - { label: 'review:' + attempt, phase: 'Gate', schema: REVIEW } - ) - if (!verdict || verdict.ship) break - log('Gate attempt ' + attempt + ': ' + verdict.blockers.length + ' blocker(s), sending back for repair') - report = await agent( - FIXER_ROLE + '\n\nA reviewer blocked your change. Fix every blocker, re-run verification, and report.\nTask: ' + task + - '\nBlockers:\n' + JSON.stringify(verdict.blockers, null, 2), - { label: 'repair:' + attempt, phase: 'Gate' } - ) -} - -const metric = { - ship: !!(verdict && verdict.ship), - score: verdict ? verdict.score : null, - gate_attempts: attempts, - winning_design: (judgment && judgment.winner != null) ? ANGLES[judgment.winner] : ANGLES[0], -} - -phase('Retro') -const RETRO_ROLE = [ - 'You are the self-improvement layer of an agent swarm. Your job: make the NEXT swarm run measurably better than this one. You may create or edit ONLY these files, nothing else: .claude/swarm-metrics.jsonl, .claude/swarm-tuning.md, .claude/swarm-context.md, and report files under .claude/swarm-reports/. (The implementation produced by earlier phases of this run is separate work: leave it untouched.)', - '1. Append exactly one JSON line for this run to .claude/swarm-metrics.jsonl containing: ts (ISO timestamp from running `date -u +%Y-%m-%dT%H:%M:%SZ`), workflow name, and the metric object you were given.', - '2. Read the whole ledger and compare against past runs: are gate attempts rising, do the same blocker patterns recur, which design angle keeps winning? State the trend.', - '3. Update .claude/swarm-tuning.md: distill gate blockers into concrete guidance for future implementers; note which design angles win for which task types; under "## Proposed script changes", propose (never apply) workflow-script improvements the evidence supports. Prune stale guidance; keep the file under 120 lines.', - '4. If a blocker reveals a DURABLE repo invariant not yet in .claude/swarm-context.md, add it there in the existing style.', - '5. Email the report to Chris Preuss: write the full run report as markdown to .claude/swarm-reports/-swarm-build.md (create the directory if needed) covering the task, which design won and why, gate attempts and final verdict, every improvement you made to tuning/context, and any proposed script changes. Then use ToolSearch (query "create_draft") to load the Gmail draft tool and create a draft addressed to chris@cpz-lab.com with subject "[Swarm] build report: " and the report as the body. DRAFT ONLY, never send. If no Gmail tool is available in this session, note that in your summary and continue; do not fail the retro over it.', - 'Be conservative: every line you write is read by every future swarm agent, so wrong guidance compounds. Ground every claim in the run record. Report back a short summary of the trend and what you changed.', -].join('\n') -const retro = await agent( - RETRO_ROLE + '\n\nRun record for this swarm-build:\n' + JSON.stringify({ - workflow: 'swarm-build', - task, - metric, - final_review: verdict, - }, null, 2), - { label: 'retro', phase: 'Retro', model: 'sonnet' } -) - -return { task, metric, retro, final_review: verdict, implementer_report: report } diff --git a/.claude/workflows/swarm-fix.js b/.claude/workflows/swarm-fix.js deleted file mode 100644 index a81c10f..0000000 --- a/.claude/workflows/swarm-fix.js +++ /dev/null @@ -1,103 +0,0 @@ -export const meta = { - name: 'swarm-fix', - description: 'Fix a list of confirmed findings one at a time in the shared working tree, each fix independently verified before moving on', - whenToUse: 'Feed it swarm-audit output: Workflow({name: "swarm-fix", args: {findings: [...]}}). Sequential on purpose so fixes never collide.', - phases: [ - { title: 'Fix', detail: 'fixer then independent verifier, per finding' }, - { title: 'Retro', detail: 'append run metrics to the ledger; update tuning so the next run is sharper' }, - ], -} - -const REVIEW = { - type: 'object', - required: ['ship', 'score', 'blockers'], - properties: { - ship: { type: 'boolean' }, - score: { type: 'integer' }, - blockers: { type: 'array', items: { type: 'string' } }, - }, -} - -// Role preambles inlined (mirroring .claude/agents/swarm-*.md) so the workflow -// works even in sessions where those agent types are not registered. -const FIXER_ROLE = [ - 'You are the implementer in an agent swarm. An independent verifier will check your work; changes that do not fix the defect, break something else, or fake verification get reported as blocked.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both first; violating a listed invariant is an automatic failure, and tuning lists lessons from previous swarm runs.', - 'Read the code you are about to change AND its callers before editing. Match surrounding style. Make the minimal change that correctly fixes the defect; no scope creep.', - 'Run the narrowest relevant test/build/typecheck that proves the fix, and report what you ran with the ACTUAL results.', - 'Hard rules: fail loudly, never add silent fallbacks, mock data, or catch-and-ignore; never weaken auth, RLS, validation, or money-safety checks; never fabricate results. Do not commit, push, or deploy; leave changes uncommitted.', -].join('\n') - -const CHECKER_ROLE = [ - 'You are the verification gate of an agent swarm. The fixer report is a claim, not evidence.', - 'If .claude/swarm-context.md and .claude/swarm-tuning.md exist in the repo root, read both; a fix violating a listed invariant is an automatic blocker.', - 'Run git diff and review the actual change. Confirm it fixes the specific defect and does not break callers or weaken validation. Re-run claimed verification yourself when feasible.', - 'ship=true only if the defect is genuinely fixed and safe to merge; score 0-10. ship=false requires concrete blockers pointing at files and specific problems. Never edit files yourself.', -].join('\n') - -// args may arrive as an object or, if the caller stringified it, as JSON text. -// Normalizing here turns a silent zero-agent no-op into a working run. -let parsedArgs = args -if (typeof parsedArgs === 'string') { - try { parsedArgs = JSON.parse(parsedArgs) } catch (e) { parsedArgs = null } -} -const findings = (parsedArgs && (Array.isArray(parsedArgs) ? parsedArgs : parsedArgs.findings)) || [] -if (!findings.length) return { error: 'Pass {findings: [...]}, e.g. the confirmed array returned by swarm-audit' } - -phase('Fix') -const results = [] -for (let i = 0; i < findings.length; i++) { - if (budget.total && budget.remaining() < 40000) { - log('Token budget nearly exhausted: stopping after ' + i + '/' + findings.length + ' findings') - for (let j = i; j < findings.length; j++) { - results.push({ finding: findings[j].title, ship: false, blockers: ['skipped: token budget exhausted'] }) - } - break - } - const f = findings[i] - const tag = (f.file || 'finding') + '#' + (i + 1) - const report = await agent( - FIXER_ROLE + '\n\nFix this confirmed defect with the minimal correct change. Leave changes uncommitted. Other fixes may already be in the working tree; do not revert or touch them.\nDefect:\n' + JSON.stringify(f, null, 2), - { label: 'fix:' + tag } - ) - if (report === null) { - results.push({ finding: f.title, ship: false, blockers: ['fixer agent failed or was skipped'] }) - continue - } - const check = await agent( - CHECKER_ROLE + '\n\nVerify that the current uncommitted changes actually fix this specific defect without breaking anything around it. Judge only this defect; ignore unrelated changes in the tree.\nDefect:\n' + JSON.stringify(f, null, 2) + - '\nFixer report (a claim, not evidence):\n' + report, - { label: 'check:' + tag, schema: REVIEW } - ) - results.push({ - finding: f.title, - file: f.file, - ship: !!(check && check.ship), - score: check ? check.score : null, - blockers: (check && check.blockers) || [], - }) - log((i + 1) + '/' + findings.length + ': ' + f.title + ' -> ' + (check && check.ship ? 'FIXED' : 'BLOCKED')) -} - -const metric = { - fixed: results.filter(r => r.ship).length, - blocked: results.filter(r => !r.ship).length, - total: results.length, -} - -phase('Retro') -const RETRO_ROLE = [ - 'You are the self-improvement layer of an agent swarm. Your job: make the NEXT swarm run measurably better than this one. You may create or edit ONLY these files, nothing else: .claude/swarm-metrics.jsonl, .claude/swarm-tuning.md, .claude/swarm-context.md, and report files under .claude/swarm-reports/. (The defect fixes made by earlier phases of this run are separate work: leave them untouched.)', - '1. Append exactly one JSON line for this run to .claude/swarm-metrics.jsonl containing: ts (ISO timestamp from running `date -u +%Y-%m-%dT%H:%M:%SZ`), workflow name, and the metric object you were given.', - '2. Read the whole ledger and compare against past runs: is fix rate falling, do the same blocker patterns recur? State the trend.', - '3. Update .claude/swarm-tuning.md: distill blocked fixes into concrete guidance for future fixers (what the verifier keeps rejecting and why); under "## Proposed script changes", propose (never apply) workflow-script improvements the evidence supports. Prune stale guidance; keep the file under 120 lines.', - '4. If a blocker reveals a DURABLE repo invariant not yet in .claude/swarm-context.md, add it there in the existing style.', - '5. Email the report to Chris Preuss: write the full run report as markdown to .claude/swarm-reports/-swarm-fix.md (create the directory if needed) covering which defects were fixed vs blocked and why, the metrics with trend vs the ledger, every improvement you made to tuning/context, and any proposed script changes. Then use ToolSearch (query "create_draft") to load the Gmail draft tool and create a draft addressed to chris@cpz-lab.com with subject "[Swarm] fix report: " and the report as the body. DRAFT ONLY, never send. If no Gmail tool is available in this session, note that in your summary and continue; do not fail the retro over it.', - 'Be conservative: every line you write is read by every future swarm agent, so wrong guidance compounds. Ground every claim in the run record. Report back a short summary of the trend and what you changed.', -].join('\n') -const retro = await agent( - RETRO_ROLE + '\n\nRun record for this swarm-fix:\n' + JSON.stringify({ workflow: 'swarm-fix', metric, results }, null, 2), - { label: 'retro', phase: 'Retro', model: 'sonnet' } -) - -return { results, metric, retro } diff --git a/.github/workflows/sentry-auto-fix.yml b/.github/workflows/sentry-auto-fix.yml index 3f28776..471db54 100644 --- a/.github/workflows/sentry-auto-fix.yml +++ b/.github/workflows/sentry-auto-fix.yml @@ -60,7 +60,7 @@ jobs: - name: Install dependencies run: npm ci - - name: Install Cursor CLI + - name: Install repair CLI run: | curl https://cursor.com/install -fsS | bash echo "$HOME/.cursor/bin" >> $GITHUB_PATH @@ -126,7 +126,7 @@ jobs: ERROR_CULPRIT: ${{ github.event.inputs.culprit }} ERROR_STACKTRACE: ${{ github.event.inputs.stacktrace }} - - name: Run Cursor Agent + - name: Run repair agent env: CURSOR_API_KEY: ${{ secrets.CURSOR_API_KEY }} run: | @@ -163,7 +163,6 @@ jobs: ${ERROR_TITLE} - Auto-fixed by Cursor CLI agent (Claude 4.6 Opus). Sentry: ${SENTRY_URL}" git push origin "$FIX_BRANCH" --force @@ -186,7 +185,7 @@ jobs: **Culprit:** \`${ERROR_CULPRIT}\` ### What was fixed - Auto-generated by the Sentry Auto-Fix pipeline using Cursor CLI (Claude 4.6 Opus). + Applies a targeted fix for the linked Sentry issue. ### Review checklist - [ ] Fix addresses the root cause diff --git a/src/oauth.ts b/src/oauth.ts index 5b570c0..c57a087 100644 --- a/src/oauth.ts +++ b/src/oauth.ts @@ -17,8 +17,8 @@ const BASE_URL = process.env.MCP_BASE_URL || 'https://mcp.cpz-lab.com'; const ACCESS_TOKEN_TTL_MS = 12 * 60 * 60 * 1000; // 12h — short-lived, re-issued on demand // Refresh tokens exist so long-lived clients (the CLI, IDE plugins) do not push -// the user through a browser twice a day. 30 days matches what Claude Code and -// Codex do, and is short enough that a leaked refresh token has a bounded life. +// the user through a browser twice a day. The 30-day expiry bounds the lifetime +// of a leaked refresh token. // // HONEST LIMITATION: like access tokens and auth codes, refresh tokens here are // stateless and sealed rather than stored. That buys cross-task resolution with @@ -55,7 +55,7 @@ function ctEq(a: string, b: string): boolean { } /** - * RFC 9728 Protected Resource Metadata. MCP clients (Claude Code, claude.ai) + * RFC 9728 Protected Resource Metadata. MCP clients * discover the authorization server from this document after receiving a 401 * challenge on /mcp — without it, the OAuth flow never starts. */