From 35bc8b3463868b3ed793c62e056c3ee6ea564a92 Mon Sep 17 00:00:00 2001 From: "Vincent (Wen Yu) Ge" Date: Fri, 18 Sep 2026 13:15:05 -0400 Subject: [PATCH 1/3] feat(audit): stream audit progress as areas An audit resolved tens of ledger checks and none of it left the machine: `tasks` holds the program's step rows, which a headless or CI run does not have, so a five-minute audit pushed `tasks: []`. Shipping every check would be noise, and each row carries a file path and a details string derived from the user's code. - `rollUpAuditAreas` folds the ledger into one row per audit area, appended after the program rows. No wire-schema change, so no backend change. A finding is progress, not a failure, so an area never reports `failed`. - `startAuditLedgerWatcher` mirrors the ledger into the session through the UI's frameworkContext seam, owned by `runAgent` so every path gets it, and started before the run recipe seeds the ledger. `HeadlessUI` tees that setter into the store, without which a headless run mirrors nothing. - pi gets native mirrors of the three `audit_*` tools. They existed only on the MCP facade, and pi mounts no MCP server, so an audit on the harness the switchboard binds it to could not move a check off pending. Descriptions and write semantics move to shared helpers so the facades cannot drift. - `--task-stream-log` dumps every attempted sync as JSONL. `--ci` always dumps and never pushes, since a synthetic run would create a session row in a real project. Consent now gates the push, not the dump. - Seeds `init-not-duplicated`, which the skill resolves in a batch with `init-correct`; a batch resolve rejects atomically, so its absence discarded both. - The audit program declares an e2e profile, and the snapshot harness signs framework-context by value so a screen rerendering from the ledger is captured more than once. Generated-By: PostHog Desktop Task-Id: 6490a5fd-5ee9-4203-b9c4-0b4ff180c45b --- e2e-harness/action-registry.ts | 13 +- e2e-harness/e2e-profile.ts | 1 + e2e-harness/profiles.ts | 3 + scripts/tui-host.no-jest.ts | 38 +++++- src/__tests__/cli.test.ts | 32 ++++- .../harness/pi/__tests__/task-tools.test.ts | 14 ++ .../runner/harness/pi/__tests__/tools.test.ts | 47 +++++++ src/lib/agent/runner/harness/pi/task.ts | 10 +- src/lib/agent/runner/harness/pi/tools.ts | 120 ++++++++++++++++++ src/lib/agent/runner/index.ts | 22 +++- src/lib/programs/__tests__/audit-seed.test.ts | 7 + src/lib/programs/audit/index.ts | 15 ++- src/lib/programs/audit/ledger-watcher.ts | 38 ++++++ src/lib/programs/audit/seed.ts | 9 ++ src/lib/programs/audit/test/e2e.json | 44 +++++++ src/lib/programs/dispatch-family.ts | 25 +++- src/lib/programs/events-audit/index.ts | 10 +- src/lib/programs/program-step.ts | 2 + src/lib/runners/run-non-interactive.ts | 54 +++++--- src/lib/runners/run-wizard.ts | 29 ++++- .../task-stream/__tests__/audit-areas.test.ts | 66 ++++++++++ .../__tests__/file-destination.test.ts | 80 ++++++++++++ .../__tests__/task-stream-push.test.ts | 67 +++++++++- src/lib/task-stream/audit-areas.ts | 61 +++++++++ src/lib/task-stream/destinations/file.ts | 86 +++++++++++++ src/lib/task-stream/index.ts | 3 + src/lib/task-stream/task-stream-push.ts | 13 +- src/lib/wizard-tools/mcp.ts | 105 ++++----------- src/lib/wizard-tools/tools.ts | 74 +++++++++++ src/ui/headless-ui.ts | 13 +- src/ui/tui/screens/audit/AuditRunScreen.tsx | 17 +-- src/utils/paths.ts | 5 + src/wizard.ts | 7 + 33 files changed, 980 insertions(+), 150 deletions(-) create mode 100644 src/lib/programs/audit/ledger-watcher.ts create mode 100644 src/lib/programs/audit/test/e2e.json create mode 100644 src/lib/task-stream/__tests__/audit-areas.test.ts create mode 100644 src/lib/task-stream/__tests__/file-destination.test.ts create mode 100644 src/lib/task-stream/audit-areas.ts create mode 100644 src/lib/task-stream/destinations/file.ts diff --git a/e2e-harness/action-registry.ts b/e2e-harness/action-registry.ts index 58e5f7e0c..63c48870c 100644 --- a/e2e-harness/action-registry.ts +++ b/e2e-harness/action-registry.ts @@ -67,19 +67,18 @@ function requireString( * - the runner or agent advances them: auth (runner sets credentials), run * (agent sets runPhase), ai-opt-in (org approval / ci auto-consent), exit, * and the no-dismiss terminal overlays. - * - screens of programs the integration e2e profile never enters (audit, - * doctor). + * - screens of programs the integration e2e profile never enters (doctor). */ export const NO_ACTION_SCREENS: ReadonlySet = new Set([ ScreenId.Auth, ScreenId.Run, ScreenId.AiOptIn, ScreenId.Exit, + // The agent advances the audit run, the same way it advances `run`. ScreenId.AuditRun, ScreenId.DoctorReport, // The detector + picker are interactive; no headless e2e drives this screen. ScreenId.SelfDrivingIntegrationDetect, - ScreenId.AuditOutro, ScreenId.SelfDrivingIntegrationCheck, ScreenId.SelfDrivingIntegrationDetect, ScreenId.SelfDrivingHandoff, @@ -210,6 +209,14 @@ export const ACTION_REGISTRY: Partial> = { apply: (store) => store.setOutroDismissed(), }, ], + [ScreenId.AuditOutro]: [ + { + id: 'dismiss_outro', + description: + 'Dismiss the audit outro, which carries the report, dashboard, and notebook links.', + apply: (store) => store.setOutroDismissed(), + }, + ], [ScreenId.MintFailure]: [ { id: 'continue_setup', diff --git a/e2e-harness/e2e-profile.ts b/e2e-harness/e2e-profile.ts index a09fd96bc..051fab496 100644 --- a/e2e-harness/e2e-profile.ts +++ b/e2e-harness/e2e-profile.ts @@ -321,6 +321,7 @@ export function decideE2eAction( case ScreenId.Outro: case ScreenId.SourceMapsOutro: + case ScreenId.AuditOutro: return { action: { id: 'dismiss_outro' } }; case ScreenId.Mcp: diff --git a/e2e-harness/profiles.ts b/e2e-harness/profiles.ts index 9134faba3..3cc126d33 100644 --- a/e2e-harness/profiles.ts +++ b/e2e-harness/profiles.ts @@ -27,6 +27,7 @@ import selfDrivingE2e from '@lib/programs/self-driving/test/e2e.json'; import sourceMapsE2e from '@lib/programs/error-tracking-upload-source-maps/test/e2e.json'; import errorTrackingE2e from '@lib/programs/error-tracking/test/e2e.json'; import warehouseSourceE2e from '@lib/programs/warehouse-source/test/e2e.json'; +import auditE2e from '@lib/programs/audit/test/e2e.json'; const PROFILES: Partial> = { [Program.PostHogIntegration]: @@ -39,6 +40,7 @@ const PROFILES: Partial> = { sourceMapsE2e.profile as WizardE2eProfile, [Program.ErrorTracking]: errorTrackingE2e.profile as WizardE2eProfile, [Program.WarehouseSource]: warehouseSourceE2e.profile as WizardE2eProfile, + [Program.Audit]: auditE2e.profile as WizardE2eProfile, }; const VARIATIONS: Partial> = { @@ -51,6 +53,7 @@ const VARIATIONS: Partial> = { [Program.ErrorTracking]: errorTrackingE2e.variations as WizardE2eVariation[], [Program.WarehouseSource]: warehouseSourceE2e.variations as WizardE2eVariation[], + [Program.Audit]: auditE2e.variations as WizardE2eVariation[], }; /** The e2e profile for a program, or the happy-path default if none is set. */ diff --git a/scripts/tui-host.no-jest.ts b/scripts/tui-host.no-jest.ts index 4ae89aeea..ded2d55a2 100644 --- a/scripts/tui-host.no-jest.ts +++ b/scripts/tui-host.no-jest.ts @@ -28,6 +28,8 @@ import { buildSession } from '@lib/wizard-session'; import { initLocalDev } from '@lib/local-dev'; import { configureGatewayFromCIEnvironment } from '@lib/gateway-session'; import { runAgent } from '@lib/agent/agent-runner'; +import { TaskStreamPush, createFileDestination } from '@lib/task-stream/index'; +import { getAuditChecks } from '@lib/programs/audit/types'; import { authenticate } from '@lib/agent/runner/shared/authenticate'; import { getOrAskForProjectData } from '@utils/setup-utils'; import { logToFile } from '@utils/debug'; @@ -55,6 +57,16 @@ import { readReportFile, } from '@e2e-harness/e2e-result'; +/** Cheap 32-bit FNV-1a, to fold framework-context values into a signature. */ +function digest(s: string): string { + let h = 0x811c9dc5; + for (let i = 0; i < s.length; i++) { + h ^= s.charCodeAt(i); + h = Math.imul(h, 0x01000193); + } + return (h >>> 0).toString(36); +} + const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms)); const mark = (m: string) => logToFile(`[tui-host] ${m}`); @@ -233,6 +245,25 @@ async function main() { sequence: (process.env.SNAP_SEQUENCE || undefined) as Sequence | undefined, model: process.env.SNAP_MODEL || undefined, }); + // Dumped, never pushed: an e2e run is synthetic, like `--ci`. + const streamLog = createFileDestination(process.env.TASK_STREAM_LOG ?? ''); + if (streamLog) { + const stream = new TaskStreamPush({ + store, + programId, + destinations: [streamLog], + eventPlanPath: programConfig.eventPlanFile + ? join(store.session.installDir, programConfig.eventPlanFile) + : undefined, + auditChecks: programConfig.auditLedgerFile + ? () => getAuditChecks(store.session) + : undefined, + }); + stream.attach(); + process.on('exit', () => void stream.shutdown(0)); + mark(`task stream dump → ${streamLog.path}`); + } + // Optional skip-ahead: pre-resolve the self-driving integration check so its // screen never shows (INTEGRATE=true integrates first; false = already set up). if (process.env.INTEGRATE === 'true' || process.env.INTEGRATE === 'false') { @@ -429,10 +460,9 @@ async function main() { overlay: store.router.hasOverlay, tasks: store.tasks.map((t) => [t.label, t.status, t.done]), phase: store.session.runPhase, - // Snap on within-screen state too: when a screen publishes new - // framework-context (e.g. the detector's projects), so the picker frame - // is captured, not just the loading state. Generic — keys, not values. - ctx: Object.keys(store.session.frameworkContext).sort().join(','), + // Values, not just keys: a screen rerendering from an artifact updated + // in place (the audit ledger) keeps its key and would snap once, empty. + ctx: digest(JSON.stringify(store.session.frameworkContext)), }); const snap = (): Promise => { const sig = signature(); diff --git a/src/__tests__/cli.test.ts b/src/__tests__/cli.test.ts index dac07d95e..2b8a723f7 100644 --- a/src/__tests__/cli.test.ts +++ b/src/__tests__/cli.test.ts @@ -10,15 +10,22 @@ const { mockBuildSessionCli, mockProvisionNewAccountCli } = vi.hoisted(() => ({ // Headless-only machinery, stubbed so the headless path doesn't construct a // real WizardStore (which would re-call the mocked buildSession) or open a real // network stream. The spies assert the stream is wired in headless and not CI. -const { mockStreamAttach, mockStreamShutdown } = vi.hoisted(() => ({ - mockStreamAttach: vi.fn(), - mockStreamShutdown: vi.fn(), -})); +const { mockStreamAttach, mockStreamShutdown, mockStreamDestinations } = + vi.hoisted(() => ({ + mockStreamAttach: vi.fn(), + mockStreamShutdown: vi.fn(), + // Which destinations each run wired up, by name. The CI contract is about + // destinations, not about whether a stream exists. + mockStreamDestinations: vi.fn(), + })); vi.mock('../lib/task-stream/index', () => ({ // shutdown() hardcodes a resolved Promise (not a bare vi.fn) so the // interactive runWizard's dangling SIGTERM handler — which calls // shutdown().catch() and outlives these tests — never hits undefined.catch. TaskStreamPush: class { + constructor(opts: { destinations: Array<{ name: string }> }) { + mockStreamDestinations(opts.destinations.map((d) => d.name)); + } attach() { mockStreamAttach(); } @@ -27,7 +34,13 @@ vi.mock('../lib/task-stream/index', () => ({ return Promise.resolve(); } }, - PostHogDestination: class {}, + PostHogDestination: class { + readonly name = 'posthog'; + }, + createFileDestination: (value: unknown) => + value === undefined || value === null || value === false + ? null + : { name: 'file', path: '/tmp/task-stream.jsonl' }, })); vi.mock('../ui/tui/store', async (importOriginal) => ({ ...(await importOriginal()), @@ -435,7 +448,9 @@ describe('CLI argument parsing', () => { expect(analytics.setTag).toHaveBeenCalledWith('build', 'ci'); }); - test('does not stream wizard-session state in CI', async () => { + // CI dumps the stream to a local file and never pushes: a CI run is + // synthetic, so a push would create a session row in a real project. + test('dumps the wizard-session stream locally and never pushes in CI', async () => { await runCLI([ '--ci', '--api-key', @@ -444,7 +459,8 @@ describe('CLI argument parsing', () => { '/tmp/test', ]); - expect(mockStreamAttach).not.toHaveBeenCalled(); + expect(mockStreamAttach).toHaveBeenCalled(); + expect(mockStreamDestinations).toHaveBeenCalledWith(['file']); }); // The CI bot authenticates with a wizard-app pha_ token, the same @@ -544,6 +560,8 @@ describe('CLI argument parsing', () => { expect(mockStreamAttach).toHaveBeenCalled(); expect(mockStreamShutdown).toHaveBeenCalled(); + // Headless is the surface the web app watches, so it pushes. + expect(mockStreamDestinations).toHaveBeenCalledWith(['posthog']); }); test('does not require --region when headless is set', async () => { diff --git a/src/lib/agent/runner/harness/pi/__tests__/task-tools.test.ts b/src/lib/agent/runner/harness/pi/__tests__/task-tools.test.ts index 4a19bac3d..f55a98995 100644 --- a/src/lib/agent/runner/harness/pi/__tests__/task-tools.test.ts +++ b/src/lib/agent/runner/harness/pi/__tests__/task-tools.test.ts @@ -115,3 +115,17 @@ describe('fenceDisallowList', () => { expect(fenceDisallowList(undefined)).toEqual([]); }); }); + +describe('audit ledger tools on pi', () => { + it('grants them only to a task that asked, under either name form', () => { + expect(allowedPiWizardTools(['Read']).has('audit_resolve_checks')).toBe( + false, + ); + const granted = allowedPiWizardTools([ + 'mcp__wizard-tools__audit_seed_checks', + 'audit_add_checks', + 'mcp__wizard-tools__audit_resolve_checks', + ]); + expect([...granted].filter((t) => t.startsWith('audit_'))).toHaveLength(3); + }); +}); diff --git a/src/lib/agent/runner/harness/pi/__tests__/tools.test.ts b/src/lib/agent/runner/harness/pi/__tests__/tools.test.ts index 29ad20a1b..018e85adf 100644 --- a/src/lib/agent/runner/harness/pi/__tests__/tools.test.ts +++ b/src/lib/agent/runner/harness/pi/__tests__/tools.test.ts @@ -5,6 +5,7 @@ */ import { mkdtempSync } from 'node:fs'; import { mkdir, readFile, writeFile } from 'node:fs/promises'; +import { readFileSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; import { describe, it, expect, vi } from 'vitest'; @@ -657,3 +658,49 @@ describe('pi task tool grant — the names the inventory shows', () => { ]); }); }); + +/** pi mounts no MCP server, so these native tools are the only ledger writer. */ +describe('audit ledger tools', () => { + it('seeds, resolves by id, and appends, at the path the watcher reads', async () => { + const workingDirectory = mkdtempSync(join(tmpdir(), 'pi-audit-ledger-')); + const tools = createWizardPiTools({ + workingDirectory, + skillsBaseUrl: 'http://localhost:0', + triageProvider: undefined, + }); + const tool = (name: string) => { + const found = tools.find((t) => t.name === name); + if (!found) throw new Error(`${name} not registered`); + return found; + }; + const ledger = () => + JSON.parse( + readFileSync( + join(workingDirectory, '.posthog-audit-checks.json'), + 'utf8', + ), + ) as Array<{ id: string; status: string }>; + const check = (id: string) => ({ + id, + area: 'Installation', + label: id, + status: 'pending' as const, + }); + + await tool('audit_seed_checks').execute('1', { + checks: [check('sdk-installed'), check('init-correct')], + } as never); + await tool('audit_resolve_checks').execute('2', { + updates: [{ id: 'sdk-installed', status: 'pass' }], + } as never); + await tool('audit_add_checks').execute('3', { + checks: [check('live-data-source-maps')], + } as never); + + expect(ledger().map((c) => [c.id, c.status])).toEqual([ + ['sdk-installed', 'pass'], + ['init-correct', 'pending'], + ['live-data-source-maps', 'pending'], + ]); + }); +}); diff --git a/src/lib/agent/runner/harness/pi/task.ts b/src/lib/agent/runner/harness/pi/task.ts index a00a8569e..37deec62d 100644 --- a/src/lib/agent/runner/harness/pi/task.ts +++ b/src/lib/agent/runner/harness/pi/task.ts @@ -107,7 +107,15 @@ const ALWAYS_ON_WIZARD_TOOLS = [ 'publish_handoff', ]; -const OPT_IN_WIZARD_TOOLS = ['wizard_ask', 'load_skill_menu', 'install_skill']; +const OPT_IN_WIZARD_TOOLS = [ + 'wizard_ask', + 'load_skill_menu', + 'install_skill', + // Audit programs only: they declare the ledger tools on `allowedTools`. + 'audit_seed_checks', + 'audit_add_checks', + 'audit_resolve_checks', +]; export function allowedPiWizardTools( allowedTools: readonly string[] | undefined, diff --git a/src/lib/agent/runner/harness/pi/tools.ts b/src/lib/agent/runner/harness/pi/tools.ts index 0647e3c01..b194cc79a 100644 --- a/src/lib/agent/runner/harness/pi/tools.ts +++ b/src/lib/agent/runner/harness/pi/tools.ts @@ -19,12 +19,22 @@ import type { ToolDefinition } from '@earendil-works/pi-coding-agent'; import { analytics } from '@utils/analytics'; import { logToFile } from '@utils/debug'; import { + AUDIT_ADD_CHECKS_DESCRIPTION, + AUDIT_ADD_CHECKS_PARAM_DESCRIPTION, + AUDIT_RESOLVE_CHECKS_DESCRIPTION, + AUDIT_RESOLVE_CHECKS_PARAM_DESCRIPTION, + AUDIT_SEED_CHECKS_DESCRIPTION, + AUDIT_SEED_CHECKS_PARAM_DESCRIPTION, + AUDIT_STATUSES, CHECK_ENV_KEYS_DESCRIPTION, CHECK_ENV_KEYS_FILE_PATH_DESCRIPTION, DEFAULT_ASK_MAX_QUESTIONS, ENV_FILE_PATH_DESCRIPTION, WIZARD_TOOL_NAMES, + addAuditChecks, checkEnvKeys as checkEnvKeysCore, + resolveAuditChecks, + seedAuditChecks, createAskAccounting, describeAskCancellation, ensureGitignoreCoverage, @@ -52,6 +62,9 @@ import { publishHandoff, } from '@lib/wizard-tools/handoff'; import { createSecretVault } from '@lib/secret-vault'; +import { AUDIT_CHECKS_FILE } from '@lib/programs/audit/types'; +import type { AuditCheck, AuditStatus } from '@lib/programs/audit/types'; +import { makeMutex } from '@utils/atomic-ledger'; import { withMode } from './index'; import { detectNodePackageManagers, @@ -257,6 +270,109 @@ export function createWizardPiTools(ctx: PiToolsContext): ToolDefinition[] { }, }); + // ── Audit ledger ──────────────────────────────────────────────────── + // Native mirror of the three MCP audit tools; pi mounts no MCP server, so + // without these an audit cannot move a check off pending. Descriptions and + // write semantics come from the shared helpers, so the facades cannot drift. + const auditLedgerPath = path.join(workingDirectory, AUDIT_CHECKS_FILE); + const auditMutex = makeMutex(); + const auditStatus = Type.Union( + AUDIT_STATUSES.map((s) => Type.Literal(s)), + { description: 'Check outcome' }, + ); + const auditCheck = Type.Object({ + id: Type.String({ description: 'Stable kebab-case check id' }), + area: Type.String({ description: 'Short group name, e.g. Installation' }), + label: Type.String({ description: 'Short human name for the check' }), + status: auditStatus, + file: Type.Optional(Type.String({ description: 'Optional path:line' })), + details: Type.Optional( + Type.String({ description: 'Optional one-line explanation' }), + ), + }); + + const auditSeedChecks = defineTool({ + name: 'audit_seed_checks', + label: 'Seed audit checks', + description: AUDIT_SEED_CHECKS_DESCRIPTION, + promptSnippet: + 'audit_seed_checks(checks) — write the full pending checklist to the audit ledger', + parameters: Type.Object({ + checks: Type.Array(auditCheck, { + description: AUDIT_SEED_CHECKS_PARAM_DESCRIPTION, + }), + }), + execute: (_id, args) => + auditMutex(() => { + const result = seedAuditChecks( + auditLedgerPath, + args.checks as AuditCheck[], + ); + logToFile(`[pi] audit_seed_checks: ${result.message}`); + return text(result.message); + }), + }); + + const auditAddChecks = defineTool({ + name: 'audit_add_checks', + label: 'Add audit checks', + description: AUDIT_ADD_CHECKS_DESCRIPTION, + promptSnippet: + 'audit_add_checks(checks) — append runtime-discovered checks to the ledger', + parameters: Type.Object({ + checks: Type.Array(auditCheck, { + minItems: 1, + description: AUDIT_ADD_CHECKS_PARAM_DESCRIPTION, + }), + }), + execute: (_id, args) => + auditMutex(() => { + const result = addAuditChecks( + auditLedgerPath, + args.checks as AuditCheck[], + ); + logToFile(`[pi] audit_add_checks: ${result.message}`); + return text(result.message); + }), + }); + + const auditResolveChecks = defineTool({ + name: 'audit_resolve_checks', + label: 'Resolve audit checks', + description: AUDIT_RESOLVE_CHECKS_DESCRIPTION, + promptSnippet: + 'audit_resolve_checks(updates) — patch each check by id as you finish it', + parameters: Type.Object({ + updates: Type.Array( + Type.Object({ + id: Type.String({ description: 'Existing check id' }), + status: auditStatus, + file: Type.Optional( + Type.String({ description: 'Optional path:line' }), + ), + details: Type.Optional( + Type.String({ description: 'Optional one-line explanation' }), + ), + }), + { minItems: 1, description: AUDIT_RESOLVE_CHECKS_PARAM_DESCRIPTION }, + ), + }), + execute: (_id, args) => + auditMutex(() => { + const result = resolveAuditChecks( + auditLedgerPath, + args.updates as Array<{ + id: string; + status: AuditStatus; + file?: string; + details?: string; + }>, + ); + logToFile(`[pi] audit_resolve_checks: ${result.message}`); + return text(result.message); + }), + }); + const detectPm = defineTool({ name: 'detect_package_manager', label: 'Detect package manager', @@ -449,6 +565,10 @@ export function createWizardPiTools(ctx: PiToolsContext): ToolDefinition[] { checkEnvKeys, setEnvValues, detectPm, + // Parallel: the mutex serializes the writes, so a subagent fan-out is safe. + withMode(auditSeedChecks, 'parallel'), + withMode(auditAddChecks, 'parallel'), + withMode(auditResolveChecks, 'parallel'), // Sequential: it mutates the store's handoff state. withMode(publishHandoffTool, 'sequential'), ]; diff --git a/src/lib/agent/runner/index.ts b/src/lib/agent/runner/index.ts index 84daec102..f11bf73c8 100644 --- a/src/lib/agent/runner/index.ts +++ b/src/lib/agent/runner/index.ts @@ -35,6 +35,7 @@ import { type SwitchboardCtx, } from './switchboard'; import { flushScanReport } from '../../yara-hooks'; +import { startAuditLedgerWatcher } from '../../programs/audit/ledger-watcher'; import { registerCleanup } from '../../../utils/wizard-abort'; export type { @@ -59,12 +60,23 @@ export async function runAgent( throw new Error(`Program "${programConfig.id}" has no run configuration.`); } - const runDef = - typeof programConfig.run === 'function' - ? await programConfig.run(session) - : programConfig.run; + // Before `run()` resolves: an audit seeds the ledger from inside its recipe, + // and a watcher started later would ignore that write as pre-existing. + const ledger = programConfig.auditLedgerFile + ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) + : null; + if (ledger) registerCleanup(() => ledger.stop()); - await runProgram(session, runDef, programConfig, options); + try { + const runDef = + typeof programConfig.run === 'function' + ? await programConfig.run(session) + : programConfig.run; + + await runProgram(session, runDef, programConfig, options); + } finally { + ledger?.stop(); + } } /** diff --git a/src/lib/programs/__tests__/audit-seed.test.ts b/src/lib/programs/__tests__/audit-seed.test.ts index 56e3dd6d9..1897a2b19 100644 --- a/src/lib/programs/__tests__/audit-seed.test.ts +++ b/src/lib/programs/__tests__/audit-seed.test.ts @@ -28,6 +28,13 @@ describe('AUDIT_SEED_CHECKS', () => { expect(AUDIT_SEED_CHECKS[sweep].area).toBe('Live Data'); }); + it('seeds every id the audit skill resolves', () => { + // A batch resolve rejects atomically, so one missing row discards the call. + expect(ids(AUDIT_SEED_CHECKS)).toEqual( + expect.arrayContaining(['init-correct', 'init-not-duplicated']), + ); + }); + it('fits every area in the checks viewer column', () => { // Area is the one hard constraint: computeLayout pins it to a fixed // COL_AREA_WIDTH that never flexes, so a longer area name is truncated at diff --git a/src/lib/programs/audit/index.ts b/src/lib/programs/audit/index.ts index ce784d82b..fc5f60297 100644 --- a/src/lib/programs/audit/index.ts +++ b/src/lib/programs/audit/index.ts @@ -9,7 +9,11 @@ import { OutroKind } from '@lib/wizard-session'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import { headlessOption, regionOption } from '@lib/headless-mode'; import { AUDIT_ABORT_CASES } from './detect.js'; -import { AUDIT_CHECKS_KEY, AUDIT_REPORT_FILE } from './types.js'; +import { + AUDIT_CHECKS_FILE, + AUDIT_CHECKS_KEY, + AUDIT_REPORT_FILE, +} from './types.js'; import { AUDIT_SEED_CHECKS, seedAuditLedger } from './seed.js'; /** Audit-specific screens for the shared agent-skill pipeline. */ @@ -96,7 +100,14 @@ export const auditConfig: ProgramConfig = { ...baseConfig, steps: auditSteps, run: auditRun, - allowedTools: ['Agent'], + auditLedgerFile: AUDIT_CHECKS_FILE, + // Ledger tools are opt-in per program; pi matches on the short name. + allowedTools: [ + 'Agent', + WIZARD_TOOL_NAMES.auditSeedChecks, + WIZARD_TOOL_NAMES.auditAddChecks, + WIZARD_TOOL_NAMES.auditResolveChecks, + ], disallowedTools: [WIZARD_TOOL_NAMES.wizardAsk], // The experimental headless flag — declared on `audit` (and basic // integration) rather than globally. mergeCommandOptions lands it on the diff --git a/src/lib/programs/audit/ledger-watcher.ts b/src/lib/programs/audit/ledger-watcher.ts new file mode 100644 index 000000000..f7a312543 --- /dev/null +++ b/src/lib/programs/audit/ledger-watcher.ts @@ -0,0 +1,38 @@ +/** + * Mirrors the agent's `.posthog-audit-checks.json` into the session, so the TUI + * screens and the task stream read one value. `runAgent` owns the lifecycle, so + * every path gets it — including the e2e host, which builds no task stream. + */ + +import path from 'path'; +import { getUI } from '@ui'; +import { + startFileWatcher, + type FileWatcherHandle, + type FileWatcherOptions, +} from '@lib/file-watcher'; +import { logToFile } from '@utils/debug'; +import { AUDIT_CHECKS_KEY, coerceAuditChecks } from './types.js'; + +const MAX_LEDGER_FILE_BYTES = 256 * 1024; + +export function startAuditLedgerWatcher( + installDir: string, + file: string, + options: FileWatcherOptions = {}, +): FileWatcherHandle { + const target = path.join(installDir, file); + logToFile(`[audit-ledger] watching ${target}`); + + return startFileWatcher( + target, + (parsed) => + getUI().setFrameworkContext(AUDIT_CHECKS_KEY, coerceAuditChecks(parsed)), + { + // A ledger an earlier run left behind stays ignored until this run writes. + ignoreInitialFile: true, + maxFileSizeBytes: MAX_LEDGER_FILE_BYTES, + ...options, + }, + ); +} diff --git a/src/lib/programs/audit/seed.ts b/src/lib/programs/audit/seed.ts index 017055f1d..47a673c11 100644 --- a/src/lib/programs/audit/seed.ts +++ b/src/lib/programs/audit/seed.ts @@ -10,6 +10,9 @@ import { AUDIT_CHECKS_FILE, type AuditCheck } from './types.js'; * succeed — the skill writes the report to disk, then mirrors it into a * PostHog notebook as its final step). * + * Every id the skill resolves needs a row here: a batch resolve rejects + * atomically, so one missing id discards the whole call. + * * `posthog-side-findings` is a sweep row, not a rule: the skill reads what * PostHog itself already computed for this project (error-tracking * recommendations, health issues) and appends one row per open finding via @@ -36,6 +39,12 @@ export const AUDIT_SEED_CHECKS: AuditCheck[] = [ label: 'Initialization is correct', status: 'pending', }, + { + id: 'init-not-duplicated', + area: 'Installation', + label: 'One initialization per runtime', + status: 'pending', + }, { id: 'identify-stable-distinct-id', area: 'Identification', diff --git a/src/lib/programs/audit/test/e2e.json b/src/lib/programs/audit/test/e2e.json new file mode 100644 index 000000000..ed409362a --- /dev/null +++ b/src/lib/programs/audit/test/e2e.json @@ -0,0 +1,44 @@ +{ + "program": "audit", + "summary": "Happy path for `wizard audit all`: confirm the intro, let the agent run the read-only audit, dismiss the outro that carries the report, dashboard, and notebook links. The run seeds the audit ledger before the agent starts; the agent resolves each check, and the task stream carries one row per area. The audit asks nothing of its own \u2014 wizard_ask is disallowed for this program.", + "profile": { + "setup": "first", + "healthCheck": "dismiss", + "mcp": "skip", + "slack": "skip", + "skills": "delete", + "ask": "first" + }, + "variations": [ + { + "name": "default", + "summary": "the program's resolved binding, with no switchboard override" + } + ], + "path": [ + { + "screen": "audit-intro", + "auto": "confirm & continue" + }, + { + "screen": "health-check", + "auto": "dismiss a flagged outage" + }, + { + "screen": "auth", + "auto": "(external) \u2014 the runner resolves credentials from the phx key" + }, + { + "screen": "audit-run", + "auto": "(external) \u2014 the agent runs the audit skill against the ledger" + }, + { + "screen": "audit-outro", + "auto": "dismiss" + }, + { + "screen": "keep-skills", + "auto": "delete the installed skills" + } + ] +} diff --git a/src/lib/programs/dispatch-family.ts b/src/lib/programs/dispatch-family.ts index 571bcf24c..75105f875 100644 --- a/src/lib/programs/dispatch-family.ts +++ b/src/lib/programs/dispatch-family.ts @@ -1,6 +1,8 @@ import type { Arguments } from 'yargs'; import { auditConfig } from '@lib/programs/audit/index'; +import { AUDIT_CHECKS_FILE } from '@lib/programs/audit/types'; +import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import { agentSkillConfig } from '@lib/programs/program-registry'; import { webAnalyticsDoctorConfig } from '@lib/programs/web-analytics-doctor/index'; import type { ProgramConfig } from '@lib/programs/program-step'; @@ -57,10 +59,27 @@ const NATIVE_HANDLERS: Record> = { * `skillId` injected. The comprehensive `audit all` is the one exception — * skillId 'audit' triggers the specialized auditConfig (custom hooks, * content blocks, screens). + * + * This is the one place that knows a subcommand belongs to `audit`, so the + * generic skill program picks up the ledger here rather than for every skill. */ -function configForCliEntry(entry: CliEntry): ProgramConfig { +function configForCliEntry(entry: CliEntry, family: string): ProgramConfig { if (entry.skillId === 'audit') return auditConfig; - return { ...agentSkillConfig, skillId: entry.skillId }; + return { + ...agentSkillConfig, + skillId: entry.skillId, + ...(family === 'audit' + ? { + auditLedgerFile: AUDIT_CHECKS_FILE, + allowedTools: [ + ...(agentSkillConfig.allowedTools ?? []), + WIZARD_TOOL_NAMES.auditSeedChecks, + WIZARD_TOOL_NAMES.auditAddChecks, + WIZARD_TOOL_NAMES.auditResolveChecks, + ], + } + : {}), + }; } function familyEntries(family: string, entries: CliEntry[]): CliEntry[] { @@ -114,7 +133,7 @@ export async function dispatchFamily( const entries = menu.cliEntries ?? []; const entry = familyEntries(family, entries).find((e) => e.command === sub); if (entry) { - dispatchProgram(configForCliEntry(entry), argv); + dispatchProgram(configForCliEntry(entry, family), argv); return; } diff --git a/src/lib/programs/events-audit/index.ts b/src/lib/programs/events-audit/index.ts index b80fad1ee..080e59d69 100644 --- a/src/lib/programs/events-audit/index.ts +++ b/src/lib/programs/events-audit/index.ts @@ -6,7 +6,7 @@ import { SPINNER_MESSAGE } from '@lib/framework-config'; import { isUsingTypeScript } from '@utils/setup-utils'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import { EVENTS_AUDIT_PROGRAM } from './steps.js'; -import { AUDIT_CHECKS_KEY } from '@lib/programs/audit/types'; +import { AUDIT_CHECKS_FILE, AUDIT_CHECKS_KEY } from '@lib/programs/audit/types'; import { seedAuditLedger } from '@lib/programs/audit/seed'; import { EVENTS_AUDIT_SEED_CHECKS } from './seed.js'; @@ -33,7 +33,13 @@ export const eventsAuditConfig: ProgramConfig = { // Top-level reportFile so AuditRunScreen can resolve the report path // synchronously without unwrapping the deferred `run` function. reportFile: SETUP_REPORT_FILE, - allowedTools: ['Agent'], + auditLedgerFile: AUDIT_CHECKS_FILE, + allowedTools: [ + 'Agent', + WIZARD_TOOL_NAMES.auditSeedChecks, + WIZARD_TOOL_NAMES.auditAddChecks, + WIZARD_TOOL_NAMES.auditResolveChecks, + ], disallowedTools: [WIZARD_TOOL_NAMES.wizardAsk], run: (session: WizardSession): Promise => { diff --git a/src/lib/programs/program-step.ts b/src/lib/programs/program-step.ts index 064bbb2d9..8383e891d 100644 --- a/src/lib/programs/program-step.ts +++ b/src/lib/programs/program-step.ts @@ -283,6 +283,8 @@ export interface ProgramConfig { * stale or unrelated `.posthog-events.json` file. */ eventPlanFile?: string; + /** Audit ledger to mirror into the session, relative to `installDir`. */ + auditLedgerFile?: string; /** * LearnCard deck rendered in the shared `RunScreen` while the agent * runs. Lives at `/content/index.tsx` by convention. diff --git a/src/lib/runners/run-non-interactive.ts b/src/lib/runners/run-non-interactive.ts index 5fb00abca..7d0cca839 100644 --- a/src/lib/runners/run-non-interactive.ts +++ b/src/lib/runners/run-non-interactive.ts @@ -8,6 +8,7 @@ import type { CloudRegion } from '@utils/types'; import { getUI, setUI } from '@ui'; import { LoggingUI } from '@ui/logging-ui'; import type { ProgramConfig } from '@lib/programs/program-step'; +import { getAuditChecks } from '@lib/programs/audit/types'; import { analytics } from '@utils/analytics'; import { resolveNoTelemetry } from './resolve-no-telemetry'; import type { WizardStore } from '@ui/tui/store'; @@ -173,35 +174,54 @@ export function runNonInteractive( // Headless streams run state to the PostHog backend so the web app can show // live progress. Reuses the interactive TaskStreamPush + WizardStore (no Ink // render): HeadlessUI keeps LoggingUI's output and feeds task updates into - // the store; this runner drives the phase transitions. CI does not stream. + // the store; this runner drives the phase transitions. Headless pushes to + // PostHog (the web app is that run's only UI); `--ci` is synthetic, so it + // dumps locally and pushes nothing. Telemetry consent gates the push only. let store: WizardStore | null = null; let taskStream: TaskStreamPush | null = null; - if (mode === 'headless') { + { const { WizardStore } = await import('@ui/tui/store'); const { HeadlessUI } = await import('@ui/headless-ui'); - const { TaskStreamPush, PostHogDestination } = await import( - '@lib/task-stream/index' - ); + const { TaskStreamPush, PostHogDestination, createFileDestination } = + await import('@lib/task-stream/index'); - store = new WizardStore(config.id); - store.session = session; - setUI(new HeadlessUI(store)); + // `''` resolves to the default path, so `--ci` always dumps. + const logTarget = + mode === 'ci' ? options.taskStreamLog ?? '' : options.taskStreamLog; + const fileDestination = createFileDestination(logTarget); + const posthogDestination = + mode === 'headless' && !session.noTelemetry + ? new PostHogDestination({ + getCredentials: () => session.credentials, + onError: (e) => logToFile('[headless task-stream]', e.message), + }) + : null; + const destinations = [ + ...(posthogDestination ? [posthogDestination] : []), + ...(fileDestination ? [fileDestination] : []), + ]; + + const headlessStore = new WizardStore(config.id); + store = headlessStore; + headlessStore.session = session; + setUI(new HeadlessUI(headlessStore)); taskStream = new TaskStreamPush({ - store, + store: headlessStore, programId: config.id, - destinations: [ - new PostHogDestination({ - getCredentials: () => session.credentials, - onError: (e) => logToFile('[headless task-stream]', e.message), - }), - ], + destinations, eventPlanPath: config.eventPlanFile ? join(session.installDir, config.eventPlanFile) : undefined, - enabled: !session.noTelemetry, + auditChecks: config.auditLedgerFile + ? () => getAuditChecks(headlessStore.session) + : undefined, + enabled: destinations.length > 0, }); taskStream.attach(); - store.setRunPhase(RunPhase.Running); + headlessStore.setRunPhase(RunPhase.Running); + if (fileDestination) { + logToFile(`[task-stream] ${mode} dump: ${fileDestination.path}`); + } } // wizardAbort exits via process.exit, so flush the terminal phase before any diff --git a/src/lib/runners/run-wizard.ts b/src/lib/runners/run-wizard.ts index 435cfb9d4..86a497f27 100644 --- a/src/lib/runners/run-wizard.ts +++ b/src/lib/runners/run-wizard.ts @@ -3,6 +3,7 @@ import { logToFile, getLogFilePath } from '@utils/debug'; import { runAgent } from '@lib/agent/agent-runner'; import { authenticate } from '@lib/agent/runner/shared/authenticate'; import { getProgramConfig } from '@lib/programs/program-registry'; +import { getAuditChecks } from '@lib/programs/audit/types'; import { maybeStampAiSdkDetected } from '@lib/programs/posthog-integration/detect'; import type { ProgramConfig } from '@lib/programs/program-step'; import type { Harness, Sequence } from '@lib/constants'; @@ -89,6 +90,9 @@ export function runWizard( const { PostHogDestination } = await import( '@lib/task-stream/destinations/posthog' ); + const { createFileDestination } = await import( + '@lib/task-stream/destinations/file' + ); // Before the TUI mounts: once Ink owns the alt screen, anything written // to it is wiped on unmount (see the catch block below), so an abort here @@ -194,19 +198,30 @@ export function runWizard( // for the launch program would report the whole run under a program the // user left on the intro screen. Nothing before this point produces a // task to push. - const taskStreamEnabled = !session.noTelemetry; + // Consent gates the push, not the dump: `--no-telemetry` still logs. + const fileDestination = createFileDestination(options.taskStreamLog); + const destinations = [ + ...(session.noTelemetry + ? [] + : [ + new PostHogDestination({ + getCredentials: () => activeTui.store.session.credentials, + onError: (err) => logToFile('[task-stream-push]', err.message), + }), + ]), + ...(fileDestination ? [fileDestination] : []), + ]; + const taskStreamEnabled = destinations.length > 0; const activeStream = new TaskStreamPush({ store: activeTui.store, programId: config.id, - destinations: [ - new PostHogDestination({ - getCredentials: () => activeTui.store.session.credentials, - onError: (err) => logToFile('[task-stream-push]', err.message), - }), - ], + destinations, eventPlanPath: config.eventPlanFile ? join(session.installDir, config.eventPlanFile) : undefined, + auditChecks: config.auditLedgerFile + ? () => getAuditChecks(activeTui.store.session) + : undefined, enabled: taskStreamEnabled, }); taskStream = activeStream; diff --git a/src/lib/task-stream/__tests__/audit-areas.test.ts b/src/lib/task-stream/__tests__/audit-areas.test.ts new file mode 100644 index 000000000..4996c1c19 --- /dev/null +++ b/src/lib/task-stream/__tests__/audit-areas.test.ts @@ -0,0 +1,66 @@ +import { + rollUpAuditAreas, + MAX_AUDIT_AREAS, +} from '@lib/task-stream/audit-areas'; +import { StreamTaskStatus } from '@lib/task-stream/types'; +import type { AuditCheck } from '@lib/programs/audit/types'; + +const check = (over: Partial): AuditCheck => ({ + id: 'id', + area: 'Installation', + label: 'label', + status: 'pending', + ...over, +}); + +describe('rollUpAuditAreas', () => { + it('emits one row per area, in first-seen order, from its checks', () => { + expect( + rollUpAuditAreas( + [ + check({ id: 'a', status: 'pass' }), + check({ id: 'b', status: 'pending' }), + check({ id: 'c', area: 'Live Data', status: 'pass' }), + check({ id: 'd', area: 'Write report' }), + ], + 2, + ), + ).toEqual([ + { id: '2', title: 'Installation', status: StreamTaskStatus.InProgress }, + { id: '3', title: 'Live Data', status: StreamTaskStatus.Completed }, + { id: '4', title: 'Write report', status: StreamTaskStatus.Pending }, + ]); + }); + + it('treats every finding as progress, never as a failed task', () => { + for (const status of ['pass', 'error', 'warning', 'suggestion'] as const) { + expect(rollUpAuditAreas([check({ status })])[0].status).toBe( + StreamTaskStatus.Completed, + ); + } + }); + + it('survives an agent-written ledger, and carries no check detail', () => { + expect(rollUpAuditAreas(null)).toEqual([]); + expect(rollUpAuditAreas({ checks: [] })).toEqual([]); + expect( + rollUpAuditAreas([ + check({ status: 'nonsense' as AuditCheck['status'] }), + ])[0].status, + ).toBe(StreamTaskStatus.Pending); + + const rows = rollUpAuditAreas([ + null, + 'not an object', + check({ area: ' ' }), + check({ area: ` ${'x'.repeat(200)} ` }), + check({ file: 'src/secret.ts:12', details: 'quotes user code' }), + ...Array.from({ length: MAX_AUDIT_AREAS + 5 }, (_, i) => + check({ area: `Area ${i}` }), + ), + ]); + expect(rows).toHaveLength(MAX_AUDIT_AREAS); + expect(rows[0].title).toHaveLength(80); + expect(JSON.stringify(rows)).not.toMatch(/secret|user code|label/); + }); +}); diff --git a/src/lib/task-stream/__tests__/file-destination.test.ts b/src/lib/task-stream/__tests__/file-destination.test.ts new file mode 100644 index 000000000..ecb97622d --- /dev/null +++ b/src/lib/task-stream/__tests__/file-destination.test.ts @@ -0,0 +1,80 @@ +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { + FileDestination, + createFileDestination, +} from '@lib/task-stream/destinations/file'; +import { StreamEvent, type TaskStreamUpdate } from '@lib/task-stream/types'; +import { WIZARD_TASK_STREAM_FILE } from '@utils/paths'; +import { RunPhase } from '@lib/wizard-session'; + +const payload = (over: Partial = {}): TaskStreamUpdate => ({ + session_id: 'audit-audit-2026-01-01T00:00:00Z', + workflow_id: 'audit', + skill_id: 'audit', + started_at: '2026-01-01T00:00:00Z', + run_phase: RunPhase.Running, + tasks: [], + timestamp: '2026-01-01T00:00:01Z', + ...over, +}); + +describe('FileDestination', () => { + let dir: string; + + beforeEach(() => { + dir = mkdtempSync(join(tmpdir(), 'wizard-stream-log-')); + }); + + afterEach(() => rmSync(dir, { recursive: true, force: true })); + + it('appends one line per attempt, and holds one run per file', async () => { + const path = join(dir, 'nested', 'stream.jsonl'); + writeFileSync(join(dir, 'stale.jsonl'), 'from an earlier run\n'); + + const dest = new FileDestination({ path, now: () => 'AT' }); + await dest.send(StreamEvent.Create, payload()); + await dest.send( + StreamEvent.Complete, + payload({ skill_id: 'audit-events' }), + ); + + const lines = readFileSync(path, 'utf8').trim().split('\n'); + expect(lines).toHaveLength(2); + expect(JSON.parse(lines[0])).toEqual({ + at: 'AT', + event: StreamEvent.Create, + payload: payload(), + }); + + // A second run over the same path truncates rather than appending. + const second = new FileDestination({ path }); + await second.send(StreamEvent.Create, payload()); + expect(readFileSync(path, 'utf8').trim().split('\n')).toHaveLength(1); + }); + + it('never throws when the path is unwritable', async () => { + const dest = new FileDestination({ path: dir }); + await expect( + dest.send(StreamEvent.Create, payload()), + ).resolves.toBeUndefined(); + }); +}); + +describe('createFileDestination', () => { + it('is off unless asked, and never writes in a published build', () => { + expect(createFileDestination(undefined)).toBeNull(); + expect(createFileDestination(false)).toBeNull(); + expect( + createFileDestination('/tmp/x.jsonl', { productionBuild: true }), + ).toBeNull(); + }); + + it('resolves the flag value to a path', () => { + expect(createFileDestination('')?.path).toBe(WIZARD_TASK_STREAM_FILE); + expect( + createFileDestination('logs/run.jsonl', { cwd: '/tmp/project' })?.path, + ).toBe('/tmp/project/logs/run.jsonl'); + }); +}); diff --git a/src/lib/task-stream/__tests__/task-stream-push.test.ts b/src/lib/task-stream/__tests__/task-stream-push.test.ts index 6a4067006..52272ffd5 100644 --- a/src/lib/task-stream/__tests__/task-stream-push.test.ts +++ b/src/lib/task-stream/__tests__/task-stream-push.test.ts @@ -1,10 +1,11 @@ import { TaskStreamPush } from '@lib/task-stream/task-stream-push'; -import { StreamEvent } from '@lib/task-stream/types'; +import { StreamEvent, StreamTaskStatus } from '@lib/task-stream/types'; import type { TaskStreamDestination, TaskStreamUpdate, } from '@lib/task-stream/types'; import type { WizardStore, TaskItem } from '@ui/tui/store'; +import { TaskStatus } from '@ui/wizard-ui'; import { RunPhase, type PendingQuestion } from '@lib/wizard-session'; import { mkdtempSync, rmSync, writeFileSync } from 'node:fs'; import { tmpdir } from 'node:os'; @@ -34,6 +35,8 @@ function createMockStore(overrides: Partial = {}) { ...overrides, }; + const frameworkContext: Record = {}; + const store = { get session() { return { @@ -42,8 +45,13 @@ function createMockStore(overrides: Partial = {}) { outroData: null, installDir: state.installDir, pendingQuestion: state.pendingQuestion ?? null, + frameworkContext, }; }, + setFrameworkContext(key: string, value: unknown) { + frameworkContext[key] = value; + for (const cb of listeners) cb(); + }, get tasks() { return state.tasks; }, @@ -106,6 +114,7 @@ function createPush( dest?: ReturnType; enabled?: boolean; eventPlanPath?: string; + auditChecks?: () => unknown; } = {}, ) { const dest = opts.dest ?? createMockDestination(); @@ -114,6 +123,7 @@ function createPush( programId: 'test-program', destinations: [dest], eventPlanPath: opts.eventPlanPath, + auditChecks: opts.auditChecks, enabled: opts.enabled, }); return { push, dest }; @@ -257,6 +267,61 @@ describe('TaskStreamPush', () => { expect(dest.calls[0][1].event_plan).toEqual({ events: plan }); }); + it('appends one task row per audit area, after the program rows', async () => { + const checks = [ + { + id: 'sdk-installed', + area: 'Installation', + label: 'SDK installed', + status: 'pass', + file: 'src/app.tsx:3', + details: 'quotes the user code', + }, + { + id: 'init-correct', + area: 'Installation', + label: 'init correct', + status: 'pending', + }, + { + id: 'write-report', + area: 'Write report', + label: 'report', + status: 'pending', + }, + ]; + const store = createMockStore({ + tasks: [ + { label: 'Welcome', status: TaskStatus.Completed, done: true }, + { label: 'Running', status: TaskStatus.InProgress, done: false }, + ], + }); + const { push, dest } = createPush(store, { auditChecks: () => checks }); + + await push.push(); + + const payload = dest.calls.at(-1)?.[1]; + expect(payload?.tasks).toEqual([ + { id: '0', title: 'Welcome', status: StreamTaskStatus.Completed }, + { id: '1', title: 'Running', status: StreamTaskStatus.InProgress }, + { id: '2', title: 'Installation', status: StreamTaskStatus.InProgress }, + { id: '3', title: 'Write report', status: StreamTaskStatus.Pending }, + ]); + // The check labels, files, and details stay on the machine. + expect(JSON.stringify(payload)).not.toContain('SDK installed'); + expect(JSON.stringify(payload)).not.toContain('src/app.tsx'); + expect(JSON.stringify(payload)).not.toContain('quotes the user code'); + }); + + it('sends no area rows for a program without a ledger', async () => { + const store = createMockStore(); + const { push, dest } = createPush(store); + + await push.push(); + + expect(dest.calls[0][1].tasks).toEqual([]); + }); + it('omits eventPlan when empty', async () => { const store = createMockStore({ eventPlan: [] }); const { push, dest } = createPush(store); diff --git a/src/lib/task-stream/audit-areas.ts b/src/lib/task-stream/audit-areas.ts new file mode 100644 index 000000000..db77679cb --- /dev/null +++ b/src/lib/task-stream/audit-areas.ts @@ -0,0 +1,61 @@ +/** + * Audit ledger → stream tasks: one row per audit *area*, never per check. + * + * A run resolves tens of checks, and each row carries a `file` path and a + * `details` string derived from the user's code. Neither belongs on the wire. + * A finding is progress, not a failure, so an area never reports `failed`. + */ + +import type { AuditCheck, AuditStatus } from '@lib/programs/audit/types'; +import { StreamTaskStatus, type StreamTask } from './types'; + +export const MAX_AUDIT_AREAS = 24; +const MAX_AREA_TITLE_LENGTH = 80; + +const RESOLVED: readonly AuditStatus[] = [ + 'pass', + 'error', + 'warning', + 'suggestion', +]; + +function areaTitle(value: unknown): string | null { + if (typeof value !== 'string' || !value.trim()) return null; + return value.trim().slice(0, MAX_AREA_TITLE_LENGTH); +} + +/** `idOffset` continues the program's task ids, which are stringified integers. */ +export function rollUpAuditAreas(checks: unknown, idOffset = 0): StreamTask[] { + if (!Array.isArray(checks)) return []; + + // Map keeps insertion order, which is the order the ledger seeds its areas in. + const areas = new Map(); + + for (const entry of checks) { + if (!entry || typeof entry !== 'object') continue; + const title = areaTitle((entry as Partial).area); + if (!title) continue; + + let counts = areas.get(title); + if (!counts) { + if (areas.size >= MAX_AUDIT_AREAS) continue; + counts = { total: 0, resolved: 0 }; + areas.set(title, counts); + } + counts.total += 1; + // The ledger is agent-written: anything unrecognized counts as unresolved. + if (RESOLVED.includes((entry as Partial).status as AuditStatus)) + counts.resolved += 1; + } + + return [...areas.entries()].map(([title, { total, resolved }], index) => ({ + id: String(idOffset + index), + title, + status: + resolved === 0 + ? StreamTaskStatus.Pending + : resolved < total + ? StreamTaskStatus.InProgress + : StreamTaskStatus.Completed, + })); +} diff --git a/src/lib/task-stream/destinations/file.ts b/src/lib/task-stream/destinations/file.ts new file mode 100644 index 000000000..966024ce6 --- /dev/null +++ b/src/lib/task-stream/destinations/file.ts @@ -0,0 +1,86 @@ +/** + * Dumps every *attempted* sync to a local JSONL file, one line per push. A line + * means the run published that payload, not that a backend accepted it — `--ci` + * swaps the PostHog destination for this one, so a synthetic run never creates a + * session row in a real project. + * + * Dev builds only: the payload holds the run's handoff report. + */ + +import { appendFileSync, mkdirSync, writeFileSync } from 'node:fs'; +import { dirname, isAbsolute, resolve } from 'node:path'; +import { IS_PRODUCTION_BUILD } from '@env'; +import { WIZARD_TASK_STREAM_FILE } from '@utils/paths'; +import { logToFile } from '@utils/debug'; +import type { + StreamEvent, + TaskStreamDestination, + TaskStreamUpdate, +} from '@lib/task-stream/types'; + +export interface FileDestinationOptions { + path: string; + /** Override for tests. */ + now?: () => string; +} + +export class FileDestination implements TaskStreamDestination { + readonly name = 'file'; + readonly path: string; + + private readonly now: () => string; + private disabled = false; + + constructor(opts: FileDestinationOptions) { + this.path = opts.path; + this.now = opts.now ?? (() => new Date().toISOString()); + // Truncate up front: one run, one file. + this.write(() => writeFileSync(this.path, '', { mode: 0o600 })); + } + + send(event: StreamEvent, payload: TaskStreamUpdate): Promise { + if (!this.disabled) { + const line = `${JSON.stringify({ at: this.now(), event, payload })}\n`; + this.write(() => appendFileSync(this.path, line, { mode: 0o600 })); + } + return Promise.resolve(); + } + + /** Never throws, never writes to stdio; the first failure ends the dump. */ + private write(op: () => void): void { + try { + mkdirSync(dirname(this.path), { recursive: true }); + op(); + } catch (err) { + this.disabled = true; + logToFile( + `[task-stream] file destination disabled (${ + err instanceof Error ? err.message : String(err) + }): ${this.path}`, + ); + } + } +} + +/** + * Resolve a `--task-stream-log` value, or null when the run did not ask for a + * dump. `''` and `true` mean the default path — how `--ci` asks for one. + */ +export function createFileDestination( + value: unknown, + opts: { productionBuild?: boolean; cwd?: string } = {}, +): FileDestination | null { + if (opts.productionBuild ?? IS_PRODUCTION_BUILD) return null; + if (value === undefined || value === null || value === false) return null; + + const raw = typeof value === 'string' ? value.trim() : ''; + const path = + raw === '' + ? WIZARD_TASK_STREAM_FILE + : isAbsolute(raw) + ? raw + : resolve(opts.cwd ?? process.cwd(), raw); + + logToFile(`[task-stream] file destination enabled: ${path}`); + return new FileDestination({ path }); +} diff --git a/src/lib/task-stream/index.ts b/src/lib/task-stream/index.ts index c04367269..a8dc7625e 100644 --- a/src/lib/task-stream/index.ts +++ b/src/lib/task-stream/index.ts @@ -6,6 +6,9 @@ export { TaskStreamPush } from './task-stream-push'; export type { TaskStreamPushOptions } from './task-stream-push'; export { PostHogDestination } from './destinations/posthog'; +export { FileDestination, createFileDestination } from './destinations/file'; + +export { rollUpAuditAreas, MAX_AUDIT_AREAS } from './audit-areas'; export { StreamTaskStatus, StreamEvent } from './types'; export type { diff --git a/src/lib/task-stream/task-stream-push.ts b/src/lib/task-stream/task-stream-push.ts index a4396bd3a..e7c2515d2 100644 --- a/src/lib/task-stream/task-stream-push.ts +++ b/src/lib/task-stream/task-stream-push.ts @@ -34,6 +34,7 @@ import { StreamEvent, } from './types'; import { EventPlanWatcher } from './event-plan-watcher'; +import { rollUpAuditAreas } from './audit-areas'; import { logToFile } from '@utils/debug'; import { sanitizeErrorDetail } from '@lib/errors'; @@ -115,6 +116,8 @@ export interface TaskStreamPushOptions { destinations: TaskStreamDestination[]; /** Optional absolute event-plan path to load into the store once. */ eventPlanPath?: string; + /** The run's audit ledger, when it has one. The runner owns the watcher. */ + auditChecks?: () => unknown; /** When false, destination subscription/delivery remains disabled. */ enabled?: boolean; } @@ -126,6 +129,7 @@ export class TaskStreamPush { private readonly programId: string; private readonly sessionId: string; private readonly eventPlanWatcher: EventPlanWatcher | null; + private readonly auditChecks: (() => unknown) | null; private enabled: boolean; private created = false; @@ -147,6 +151,7 @@ export class TaskStreamPush { this.eventPlanWatcher = opts.eventPlanPath ? new EventPlanWatcher(this.store, opts.eventPlanPath) : null; + this.auditChecks = opts.auditChecks ?? null; this.startedAt = secondPrecisionIso(startedAt); // skillId may not be set yet — fall back to programId so the // session_id is stable for the whole run regardless of when the @@ -297,13 +302,19 @@ export class TaskStreamPush { const skillId = sanitizeChannelId(session.skillId ?? this.programId); const phase = session.runPhase; + // Program rows carry the phase; the area rows carry the audit's progress. + const programTasks = buildTasks(tasks); + const auditAreas = this.auditChecks + ? rollUpAuditAreas(this.auditChecks(), programTasks.length) + : []; + const payload: TaskStreamUpdate = { session_id: this.sessionId, workflow_id: this.programId, skill_id: skillId, started_at: this.startedAt, run_phase: phase, - tasks: buildTasks(tasks), + tasks: [...programTasks, ...auditAreas], event_plan: eventPlan.length > 0 ? { events: eventPlan } : undefined, error: buildError(phase, session.outroData), pending_input: buildPendingInput(session.pendingQuestion), diff --git a/src/lib/wizard-tools/mcp.ts b/src/lib/wizard-tools/mcp.ts index 925308701..4ad43e235 100644 --- a/src/lib/wizard-tools/mcp.ts +++ b/src/lib/wizard-tools/mcp.ts @@ -39,8 +39,7 @@ import { CHECK_ENV_KEYS_FILE_PATH_DESCRIPTION, ENV_FILE_PATH_DESCRIPTION, SERVER_NAME, - appendAuditChecksToLedger, - applyAuditUpdates, + addAuditChecks, downloadSkill, ensureGitignoreCoverage, createAskAccounting, @@ -49,15 +48,21 @@ import { checkEnvKeys as checkEnvKeysCore, mergeEnvValues, normaliseAskSubject, - readLedger, resolveAskQuestionKinds, + resolveAuditChecks, resolveEnvPath, resolveEnvSecretRefs, + seedAuditChecks, templateEnvWriteRefusal, legacyKeyNameRefusal, vaultSensitiveAnswers, - writeLedgerAtomic, type SkillEntry, + AUDIT_ADD_CHECKS_DESCRIPTION, + AUDIT_ADD_CHECKS_PARAM_DESCRIPTION, + AUDIT_RESOLVE_CHECKS_DESCRIPTION, + AUDIT_RESOLVE_CHECKS_PARAM_DESCRIPTION, + AUDIT_SEED_CHECKS_DESCRIPTION, + AUDIT_SEED_CHECKS_PARAM_DESCRIPTION, AUDIT_STATUSES, WIZARD_ASK_KIND_DESCRIPTION, WIZARD_ASK_SENSITIVE_DESCRIPTION, @@ -472,23 +477,18 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { const auditSeedChecks = tool( 'audit_seed_checks', - 'Seed the audit ledger at .posthog-audit-checks.json with the full set of pending checks. Call this once at the start of the audit. Atomically replaces any existing ledger.', + AUDIT_SEED_CHECKS_DESCRIPTION, { checks: z .array(auditCheckSchema) - .describe('Full pending checklist to write to the ledger'), + .describe(AUDIT_SEED_CHECKS_PARAM_DESCRIPTION), }, async (args: { checks: AuditCheck[] }) => { return auditMutex(() => { - writeLedgerAtomic(auditLedgerPath, args.checks); + const result = seedAuditChecks(auditLedgerPath, args.checks); logToFile(`audit_seed_checks: wrote ${args.checks.length} entries`); return { - content: [ - { - type: 'text' as const, - text: `Seeded ${args.checks.length} audit checks.`, - }, - ], + content: [{ type: 'text' as const, text: result.message }], }; }); }, @@ -498,51 +498,20 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { const auditAddChecks = tool( 'audit_add_checks', - 'Append one or more pending checks to the existing audit ledger at .posthog-audit-checks.json. Call audit_seed_checks first. Atomically rejects duplicate ids without changing the ledger.', + AUDIT_ADD_CHECKS_DESCRIPTION, { checks: z .array(auditCheckSchema) .min(1) - .describe('Additional checks to append to the existing ledger'), + .describe(AUDIT_ADD_CHECKS_PARAM_DESCRIPTION), }, async (args: { checks: AuditCheck[] }) => { return auditMutex(() => { - const result = appendAuditChecksToLedger(auditLedgerPath, args.checks); - - if (!result.ok) { - if (result.reason === 'missing-ledger') { - return { - content: [ - { - type: 'text' as const, - text: 'Error: audit ledger does not exist. Run audit_seed_checks first.', - }, - ], - isError: true, - }; - } - - return { - content: [ - { - type: 'text' as const, - text: `Error: duplicate check id(s): ${result.ids.join( - ', ', - )}. Check ids must be unique.`, - }, - ], - isError: true, - }; - } - - logToFile(`audit_add_checks: added ${result.added} entries`); + const result = addAuditChecks(auditLedgerPath, args.checks); + logToFile(`audit_add_checks: ${result.message}`); return { - content: [ - { - type: 'text' as const, - text: `Added ${result.added} audit check(s).`, - }, - ], + content: [{ type: 'text' as const, text: result.message }], + ...(result.ok ? {} : { isError: true }), }; }); }, @@ -552,12 +521,12 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { const auditResolveChecks = tool( 'audit_resolve_checks', - "Resolve one or more audit checks by id. Patches each entry's status (and optional file/details) and writes the ledger back atomically. Concurrent calls serialize.", + AUDIT_RESOLVE_CHECKS_DESCRIPTION, { updates: z .array(auditUpdateSchema) .min(1) - .describe('Patches to apply, keyed by check id'), + .describe(AUDIT_RESOLVE_CHECKS_PARAM_DESCRIPTION), }, async (args: { updates: Array<{ @@ -568,35 +537,11 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { }>; }) => { return auditMutex(() => { - const current = readLedger(auditLedgerPath); - const { next, unknown } = applyAuditUpdates(current, args.updates); - - if (unknown.length > 0) { - return { - content: [ - { - type: 'text' as const, - text: `Error: unknown check id(s): ${unknown.join( - ', ', - )}. Run audit_seed_checks first or check the id.`, - }, - ], - isError: true, - }; - } - - writeLedgerAtomic(auditLedgerPath, next); - logToFile( - `audit_resolve_checks: applied ${args.updates.length} update(s)`, - ); - + const result = resolveAuditChecks(auditLedgerPath, args.updates); + logToFile(`audit_resolve_checks: ${result.message}`); return { - content: [ - { - type: 'text' as const, - text: `Resolved ${args.updates.length} check(s).`, - }, - ], + content: [{ type: 'text' as const, text: result.message }], + ...(result.ok ? {} : { isError: true }), }; }); }, diff --git a/src/lib/wizard-tools/tools.ts b/src/lib/wizard-tools/tools.ts index 38fb03a88..06b35c55c 100644 --- a/src/lib/wizard-tools/tools.ts +++ b/src/lib/wizard-tools/tools.ts @@ -1129,6 +1129,80 @@ export type AppendAuditChecksResult = | { ok: false; reason: 'missing-ledger' } | { ok: false; reason: 'duplicate-ids'; ids: string[] }; +// Shared by both facades (MCP server, pi native tools) so neither can drift. +export const AUDIT_SEED_CHECKS_DESCRIPTION = + 'Seed the audit ledger at .posthog-audit-checks.json with the full set of pending checks. Call this once at the start of the audit. Atomically replaces any existing ledger.'; +export const AUDIT_SEED_CHECKS_PARAM_DESCRIPTION = + 'Full pending checklist to write to the ledger'; +export const AUDIT_ADD_CHECKS_DESCRIPTION = + 'Append one or more pending checks to the existing audit ledger at .posthog-audit-checks.json. Call audit_seed_checks first. Atomically rejects duplicate ids without changing the ledger.'; +export const AUDIT_ADD_CHECKS_PARAM_DESCRIPTION = + 'Additional checks to append to the existing ledger'; +export const AUDIT_RESOLVE_CHECKS_DESCRIPTION = + "Resolve one or more audit checks by id. Patches each entry's status (and optional file/details) and writes the ledger back atomically. Concurrent calls serialize."; +export const AUDIT_RESOLVE_CHECKS_PARAM_DESCRIPTION = + 'Patches to apply, keyed by check id'; + +/** Outcome of a ledger mutation, with the agent-facing message. */ +export interface AuditLedgerResult { + ok: boolean; + message: string; +} + +/** The three ledger mutations. The caller owns serialization (a mutex). */ +export function seedAuditChecks( + targetPath: string, + checks: AuditCheck[], +): AuditLedgerResult { + writeLedgerAtomic(targetPath, checks); + return { ok: true, message: `Seeded ${checks.length} audit checks.` }; +} + +export function addAuditChecks( + targetPath: string, + checks: AuditCheck[], +): AuditLedgerResult { + const result = appendAuditChecksToLedger(targetPath, checks); + if (result.ok) { + return { ok: true, message: `Added ${result.added} audit check(s).` }; + } + if (result.reason === 'missing-ledger') { + return { + ok: false, + message: + 'Error: audit ledger does not exist. Run audit_seed_checks first.', + }; + } + return { + ok: false, + message: `Error: duplicate check id(s): ${result.ids.join( + ', ', + )}. Check ids must be unique.`, + }; +} + +export function resolveAuditChecks( + targetPath: string, + updates: Array<{ + id: string; + status: AuditStatus; + file?: string; + details?: string; + }>, +): AuditLedgerResult { + const { next, unknown } = applyAuditUpdates(readLedger(targetPath), updates); + if (unknown.length > 0) { + return { + ok: false, + message: `Error: unknown check id(s): ${unknown.join( + ', ', + )}. Run audit_seed_checks first or check the id.`, + }; + } + writeLedgerAtomic(targetPath, next); + return { ok: true, message: `Resolved ${updates.length} check(s).` }; +} + export function appendAuditChecksToLedger( targetPath: string, additions: AuditCheck[], diff --git a/src/ui/headless-ui.ts b/src/ui/headless-ui.ts index 897dfc69d..7029865f1 100644 --- a/src/ui/headless-ui.ts +++ b/src/ui/headless-ui.ts @@ -6,9 +6,8 @@ import type { WizardStore } from './tui/store'; * wizard-session sync (`TaskStreamPush`) can observe a headless run. We extend * `LoggingUI` (not `InkUI`) because its blocking/gate methods would wait on a * TUI that never renders; the runner drives phase transitions on the store - * directly, so only UI-originated per-run updates tee through here. Runner - * machinery loads agent-authored artifacts such as the event plan directly - * into the store. + * directly, so only UI-originated per-run updates tee through here. The audit + * ledger arrives through `setFrameworkContext`, the seam every UI implements. */ export class HeadlessUI extends LoggingUI { constructor(private readonly store: WizardStore) { @@ -25,4 +24,12 @@ export class HeadlessUI extends LoggingUI { setHandoffText(text: string): void { this.store.setHandoffText(text); } + + setFrameworkContext(key: string, value: unknown): void { + this.store.setFrameworkContext(key, value); + } + + getFrameworkContext(key: string): unknown { + return this.store.session.frameworkContext[key]; + } } diff --git a/src/ui/tui/screens/audit/AuditRunScreen.tsx b/src/ui/tui/screens/audit/AuditRunScreen.tsx index 1f5e22b7f..b9cbc4f87 100644 --- a/src/ui/tui/screens/audit/AuditRunScreen.tsx +++ b/src/ui/tui/screens/audit/AuditRunScreen.tsx @@ -1,5 +1,4 @@ import { useSyncExternalStore } from 'react'; -import { join } from 'node:path'; import { Box } from 'ink'; import type { WizardStore } from '@ui/tui/store'; import { @@ -9,19 +8,12 @@ import { HNViewer, } from '@ui/tui/primitives/index'; import { useStdoutDimensions } from '@ui/tui/hooks/useStdoutDimensions'; -import { useFileWatcher } from '@ui/tui/hooks/file-watcher'; import { AuditChecksViewer } from './AuditChecksViewer/AuditChecksViewer.js'; import { AuditAreaPane } from './AuditAreaPane.js'; import { AUDIT_AREA_SLIDES } from './slides/index.js'; import { EVENTS_AUDIT_AREA_SLIDES } from './slides/events-audit/index.js'; import { PendingChecksList } from './PendingChecksList.js'; -import { - AUDIT_CHECKS_FILE, - AUDIT_CHECKS_KEY, - AUDIT_REPORT_FILE, - coerceAuditChecks, - getAuditChecks, -} from '@lib/programs/audit/types'; +import { AUDIT_REPORT_FILE, getAuditChecks } from '@lib/programs/audit/types'; import { getProgramConfig } from '@lib/programs/program-registry'; import { WIZARD_LOG_FILE } from '@utils/paths'; @@ -35,11 +27,8 @@ export const AuditRunScreen = ({ store }: AuditRunScreenProps) => { () => store.getSnapshot(), ); - // Mirror the agent's audit ledger into the store. - useFileWatcher(join(store.session.installDir, AUDIT_CHECKS_FILE), (parsed) => - store.setFrameworkContext(AUDIT_CHECKS_KEY, coerceAuditChecks(parsed)), - ); - + // The ledger reaches the store through `AuditLedgerWatcher`, which runs for + // headless runs too. This screen only renders what the store holds. const statuses = store.statusMessages.length > 0 ? store.statusMessages : undefined; diff --git a/src/utils/paths.ts b/src/utils/paths.ts index c8253b6ac..76f0bbb5a 100644 --- a/src/utils/paths.ts +++ b/src/utils/paths.ts @@ -6,6 +6,11 @@ const TMP = process.platform === 'win32' ? tmpdir() : '/tmp'; export const WIZARD_LOG_FILE = join(TMP, 'posthog-wizard.log'); export const WIZARD_BENCHMARK_FILE = join(TMP, 'posthog-wizard-benchmark.json'); +/** Default target of `--task-stream-log`. Dev builds only. */ +export const WIZARD_TASK_STREAM_FILE = join( + TMP, + 'posthog-wizard-task-stream.jsonl', +); export const WIZARD_YARA_REPORT_FILE = join( TMP, 'posthog-wizard-yara-report.json', diff --git a/src/wizard.ts b/src/wizard.ts index e8e497320..8671b8892 100644 --- a/src/wizard.ts +++ b/src/wizard.ts @@ -133,6 +133,13 @@ export class Wizard { type: 'boolean', hidden: true, }) + // Rides beside the PostHog destination rather than replacing it, so a + // logged run is the same run the backend sees. + .option('task-stream-log', { + describe: + 'Append every task-stream payload to a local JSONL file. Pass a path, or pass the flag alone for /tmp/posthog-wizard-task-stream.jsonl\nenv: POSTHOG_WIZARD_TASK_STREAM_LOG', + type: 'string', + }) // ── Local dev targets (see docs/local-dev.md) ────────────────── // Not `hidden`: the build gate already keeps them from users, so // hiding them would only cost the dev-build help row. Deliberately not From d66add933df563b6a6dbb8ba8936b6924cf4ce07 Mon Sep 17 00:00:00 2001 From: "Vincent (Wen Yu) Ge" <29069505+gewenyu99@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:44:14 -0400 Subject: [PATCH 2/3] docs: correct the default audit leaf, document --task-stream-log (#1266) --- AGENTS.md | 4 ++-- docs/local-dev.md | 9 +++++++++ 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 913ac35f5..bf6620e5c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -97,8 +97,8 @@ aliases. | Subcommand | What it audits | | ----------------------------- | ---------------------------------------------------- | -| `wizard audit events` | event capture quality + cost (**default** leaf) | -| `wizard audit all` | comprehensive audit across every area | +| `wizard audit events` | event capture quality + cost | +| `wizard audit all` | comprehensive audit across every area (**default**) | | `wizard audit autocapture` | autocapture setup + cost | | `wizard audit feature-flags` | feature flag usage + cost | | `wizard audit identify` | `$identify` implementation | diff --git a/docs/local-dev.md b/docs/local-dev.md index 5050b5c77..10e11ba7b 100644 --- a/docs/local-dev.md +++ b/docs/local-dev.md @@ -76,10 +76,19 @@ These flags are available in dev/test builds. Published builds reject them. | `--local-context-mill` | `POSTHOG_WIZARD_LOCAL_CONTEXT_MILL` | skills → `:8765` | | `--local-mcp` | `POSTHOG_WIZARD_LOCAL_MCP` | MCP → `:8787` | | `--local-posthog` | `POSTHOG_WIZARD_LOCAL_POSTHOG` | PostHog origins → `:8010` | +| `--task-stream-log[=path]` | `POSTHOG_WIZARD_TASK_STREAM_LOG` | dump every attempted task-stream sync as JSONL (default `/tmp/posthog-wizard-task-stream.jsonl`) | `--local-posthog` is sugar over `--base-url`. It pins the API host, app host, OAuth server, and the LLM gateway derived from them. +`--task-stream-log` records what the run published, one JSON line per push, +truncated per run. It rides beside the PostHog destination rather than +replacing it, so a logged run is the same run the backend sees. A line means +the payload was attempted, not accepted — `[task-stream] wizard/sessions push +ok: 201` in the debug log is the delivery signal. `--ci` dumps to the default +path on every run and never pushes, since a synthetic run would otherwise +create a session row in a real project. + ### Precedence Most specific wins: From 9827897d0dc08ac798bb5a8344bc4f042eb882df Mon Sep 17 00:00:00 2001 From: "Vincent (Wen Yu) Ge" <29069505+gewenyu99@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:44:24 -0400 Subject: [PATCH 3/3] feat(audit): report which audit ran (#1265) --- src/__tests__/programs-cli.test.ts | 19 +++++++++++++++++++ src/lib/programs/dispatch-family.ts | 1 + src/lib/programs/program-step.ts | 6 ++++++ src/lib/runners/run-non-interactive.ts | 2 +- src/lib/runners/run-wizard.ts | 2 +- src/utils/analytics.ts | 3 +++ 6 files changed, 31 insertions(+), 2 deletions(-) diff --git a/src/__tests__/programs-cli.test.ts b/src/__tests__/programs-cli.test.ts index 7b891e1ca..89a972809 100644 --- a/src/__tests__/programs-cli.test.ts +++ b/src/__tests__/programs-cli.test.ts @@ -127,6 +127,25 @@ describe('dispatchFamily', () => { expect(opts).toMatchObject({ debug: true }); }); + test('an audit leaf publishes under the family, not under agent-skill', async () => { + // A leaf runs on the generic skill program, whose id is `agent-skill`, so + // without the override every audit would share one indistinguishable + // channel with every other `wizard skill` run. skill_id discriminates. + mockMenu([ + entry({ + skillId: 'audit-events', + command: 'events', + parentCommand: 'audit', + }), + ]); + await dispatchFamily('audit', makeArgv({ skill: 'events' })); + const [config] = mockRunWizard.mock.calls[0] as [ + { id?: string; streamWorkflowId?: string }, + ]; + expect(config.id).toBe('agent-skill'); + expect(config.streamWorkflowId).toBe('audit'); + }); + test('routes through runWizardCI when --ci is set', async () => { mockMenu([ entry({ diff --git a/src/lib/programs/dispatch-family.ts b/src/lib/programs/dispatch-family.ts index 75105f875..a730d8198 100644 --- a/src/lib/programs/dispatch-family.ts +++ b/src/lib/programs/dispatch-family.ts @@ -71,6 +71,7 @@ function configForCliEntry(entry: CliEntry, family: string): ProgramConfig { ...(family === 'audit' ? { auditLedgerFile: AUDIT_CHECKS_FILE, + streamWorkflowId: family, allowedTools: [ ...(agentSkillConfig.allowedTools ?? []), WIZARD_TOOL_NAMES.auditSeedChecks, diff --git a/src/lib/programs/program-step.ts b/src/lib/programs/program-step.ts index 8383e891d..9388bee79 100644 --- a/src/lib/programs/program-step.ts +++ b/src/lib/programs/program-step.ts @@ -285,6 +285,12 @@ export interface ProgramConfig { eventPlanFile?: string; /** Audit ledger to mirror into the session, relative to `installDir`. */ auditLedgerFile?: string; + /** + * Channel the task stream publishes this run under, when it differs from the + * program id. A family leaf runs on the generic skill program, so without + * this every `wizard audit ` would report as `agent-skill`. + */ + streamWorkflowId?: string; /** * LearnCard deck rendered in the shared `RunScreen` while the agent * runs. Lives at `/content/index.tsx` by convention. diff --git a/src/lib/runners/run-non-interactive.ts b/src/lib/runners/run-non-interactive.ts index 7d0cca839..7230c3ec0 100644 --- a/src/lib/runners/run-non-interactive.ts +++ b/src/lib/runners/run-non-interactive.ts @@ -207,7 +207,7 @@ export function runNonInteractive( setUI(new HeadlessUI(headlessStore)); taskStream = new TaskStreamPush({ store: headlessStore, - programId: config.id, + programId: config.streamWorkflowId ?? config.id, destinations, eventPlanPath: config.eventPlanFile ? join(session.installDir, config.eventPlanFile) diff --git a/src/lib/runners/run-wizard.ts b/src/lib/runners/run-wizard.ts index 86a497f27..1ee2e7120 100644 --- a/src/lib/runners/run-wizard.ts +++ b/src/lib/runners/run-wizard.ts @@ -214,7 +214,7 @@ export function runWizard( const taskStreamEnabled = destinations.length > 0; const activeStream = new TaskStreamPush({ store: activeTui.store, - programId: config.id, + programId: config.streamWorkflowId ?? config.id, destinations, eventPlanPath: config.eventPlanFile ? join(session.installDir, config.eventPlanFile) diff --git a/src/utils/analytics.ts b/src/utils/analytics.ts index d6500a015..f494fc439 100644 --- a/src/utils/analytics.ts +++ b/src/utils/analytics.ts @@ -51,6 +51,9 @@ export function sessionProperties( return { integration: session.integration, + // Discriminates a family leaf: `command` is the first positional, so every + // `wizard audit ` reports the same `audit`. + skill_id: session.skillId, detected_framework: session.detectedFrameworkLabel, typescript: session.typescript, project_id: session.credentials?.projectId,