diff --git a/scripts/tui-host.no-jest.ts b/scripts/tui-host.no-jest.ts index ded2d55a2..d20bfc8cb 100644 --- a/scripts/tui-host.no-jest.ts +++ b/scripts/tui-host.no-jest.ts @@ -27,7 +27,7 @@ import type { Harness, Sequence } from '@lib/constants'; import { buildSession } from '@lib/wizard-session'; import { initLocalDev } from '@lib/local-dev'; import { configureGatewayFromCIEnvironment } from '@lib/gateway-session'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runProgramAgent } from '@lib/programs/run-agent-legacy'; import { TaskStreamPush, createFileDestination } from '@lib/task-stream/index'; import { getAuditChecks } from '@lib/programs/audit/types'; import { authenticate } from '@lib/agent/runner/shared/authenticate'; @@ -334,13 +334,13 @@ async function main() { await step.run(await runSessionFor(step)); store.completeRunStep(step.id); } else if (step.screenId === 'run') { - await runAgent(programConfig, await runSessionFor(step)); + await runProgramAgent(programConfig, await runSessionFor(step)); } else if (step.isComplete) { await store.waitUntil(step.isComplete); } } } else { - await runAgent(programConfig, store.session); + await runProgramAgent(programConfig, store.session); } }; diff --git a/src/__tests__/architecture/known-violations.json b/src/__tests__/architecture/known-violations.json index d523b2a73..d83527e0a 100644 --- a/src/__tests__/architecture/known-violations.json +++ b/src/__tests__/architecture/known-violations.json @@ -27,7 +27,6 @@ "src/lib/programs/migration/index.ts -> src/lib/programs/migration/content/index.tsx", "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/agent-interface.ts", "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/runner/shared/bootstrap.ts", "src/lib/programs/posthog-integration/index.ts -> src/lib/programs/posthog-integration/content/index.tsx", "src/lib/programs/program-registry.ts -> src/lib/programs/agent-skill/content/index.tsx", "src/lib/programs/program-step.ts -> src/lib/agent/agent-runner.ts", @@ -37,6 +36,12 @@ "src/lib/programs/replay-vision/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/revenue-analytics/detect.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/revenue-analytics/index.ts -> src/lib/programs/revenue-analytics/content/index.tsx", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/agent-interface.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/claude-settings.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/index.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/shared/authenticate.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/switchboard/index.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/yara-hooks.ts", "src/lib/programs/self-driving/detect.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/self-driving/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/index.tsx", @@ -57,6 +62,7 @@ "src/steps/add-or-update-environment-variables.ts -> src/telemetry.ts", "src/steps/run-prettier.ts -> src/telemetry.ts", "src/steps/upload-environment-variables/index.ts -> src/telemetry.ts", + "src/ui/agent-progress.ts -> src/lib/agent/progress.ts", "src/ui/index.ts -> src/ui/logging-ui.ts", "src/ui/logging-ui.ts -> src/lib/agent/claude-settings.ts", "src/ui/tui/components/PhaseVisuals.tsx -> src/lib/agent/agent-phase.ts", diff --git a/src/__tests__/cli.test.ts b/src/__tests__/cli.test.ts index 2b8a723f7..4633470ee 100644 --- a/src/__tests__/cli.test.ts +++ b/src/__tests__/cli.test.ts @@ -112,8 +112,8 @@ vi.mock('../utils/wizard-abort', async (importOriginal) => ({ ...(await importOriginal()), wizardAbort: vi.fn(), })); -vi.mock('../lib/agent/agent-runner', () => ({ - runAgent: vi.fn().mockResolvedValue(undefined), +vi.mock('../lib/programs/run-agent-legacy', () => ({ + runProgramAgent: vi.fn().mockResolvedValue(undefined), })); describe('CLI argument parsing', () => { diff --git a/src/__tests__/provision-cli.test.ts b/src/__tests__/provision-cli.test.ts index d1b561ba4..95d547bd0 100644 --- a/src/__tests__/provision-cli.test.ts +++ b/src/__tests__/provision-cli.test.ts @@ -63,8 +63,8 @@ vi.mock('../utils/wizard-abort', async (importOriginal) => ({ ...(await importOriginal()), wizardAbort: vi.fn(), })); -vi.mock('../lib/agent/agent-runner', () => ({ - runAgent: vi.fn().mockResolvedValue(undefined), +vi.mock('../lib/programs/run-agent-legacy', () => ({ + runProgramAgent: vi.fn().mockResolvedValue(undefined), })); import { provisionCommand } from '../commands/provision'; diff --git a/src/lib/__tests__/agent-interface.test.ts b/src/lib/__tests__/agent-interface.test.ts index 0eb9ce49d..cfe1110c3 100644 --- a/src/lib/__tests__/agent-interface.test.ts +++ b/src/lib/__tests__/agent-interface.test.ts @@ -13,7 +13,6 @@ import { import { AgentOutputSignals } from '@lib/agent/output-signals'; import { RESUME_INSTRUCTION } from '@lib/agent/signals'; import { analytics } from '@utils/analytics'; -import { wizardAbort } from '@utils/wizard-abort'; import { Sequence } from '@lib/constants'; import type { WizardRunOptions } from '@utils/types'; import type { SpinnerHandle } from '@ui'; @@ -25,11 +24,6 @@ import { // Mock dependencies vi.mock('../../utils/analytics'); vi.mock('../../utils/debug'); -// wizardAbort exits the process; the 401 tests below need it to just reject. -vi.mock('@utils/wizard-abort', async (importOriginal) => ({ - ...(await importOriginal()), - wizardAbort: vi.fn(), -})); // Mock the SDK module const mockQuery = vi.fn(); @@ -739,6 +733,10 @@ describe('subprocess gateway credentials', () => { describe('gateway re-mint on 401', () => { const spinner = { start: vi.fn(), stop: vi.fn(), message: vi.fn() }; + // Where the run reports the auth screen; stands where getUI() used to. + const emit = vi.fn(); + const authErrors = () => + emit.mock.calls.filter(([e]) => e.kind === 'authError'); const options: WizardRunOptions = { debug: false, installDir: '/test/dir', @@ -766,6 +764,7 @@ describe('gateway re-mint on 401', () => { triageProvider: () => Promise.resolve('false_positive'), gatewayAuth, refreshGatewayAuth, + emit, }); const run = (cfg: ReturnType) => runAgent(cfg, 'test prompt', options, spinner as unknown as SpinnerHandle, { @@ -822,7 +821,7 @@ describe('gateway re-mint on 401', () => { beforeEach(() => { vi.clearAllMocks(); mockUIInstance.spinner.mockReturnValue(spinner); - vi.mocked(wizardAbort).mockRejectedValue(new Error('wizardAbort: exit')); + emit.mockReset(); }); it('mints once and resumes the session when an aged bearer is rejected', async () => { @@ -838,7 +837,7 @@ describe('gateway re-mint on 401', () => { expect(result).toEqual({}); expect(refresh).toHaveBeenCalledTimes(1); - expect(wizardAbort).not.toHaveBeenCalled(); + expect(authErrors()).toHaveLength(0); expect(mockQuery).toHaveBeenCalledTimes(2); const [first, second] = mockQuery.mock.calls.map((c) => c[0]); expect(first.options.resume).toBeUndefined(); @@ -877,11 +876,12 @@ describe('gateway re-mint on 401', () => { expect(refresh).toHaveBeenCalledTimes(1); expect(mockQuery).toHaveBeenCalledTimes(2); - expect(mockUIInstance.showAuthError).toHaveBeenCalledTimes(1); - expect(wizardAbort).toHaveBeenCalledTimes(1); - // In production wizardAbort exits; the mocked rejection surfaces as the - // run's API error. - expect(result.error).toBe('WIZARD_API_ERROR'); + expect(authErrors()).toHaveLength(1); + // The auth screen is reported, and the decided failure goes back to the + // caller, which owns the exit. + expect(result.error).toBeUndefined(); + expect(result.failure?.message).toBe('Authentication failed (401)'); + expect(result.failure?.code).toBeDefined(); }); it('judges a failed resumed session on its own error, not the old 401', async () => { @@ -918,7 +918,7 @@ describe('gateway re-mint on 401', () => { expect(result.error).toBe('WIZARD_API_ERROR'); expect(result.message).toContain('500'); expect(result.message).not.toContain('401'); - expect(mockUIInstance.showAuthError).not.toHaveBeenCalled(); + expect(authErrors()).toHaveLength(0); }); it('does not re-mint when a fresh bearer is rejected', async () => { @@ -932,9 +932,8 @@ describe('gateway re-mint on 401', () => { // A fresh token the gateway rejects is a bad credential, not age. expect(refresh).not.toHaveBeenCalled(); expect(mockQuery).toHaveBeenCalledTimes(1); - expect(mockUIInstance.showAuthError).toHaveBeenCalledTimes(1); - expect(wizardAbort).toHaveBeenCalledTimes(1); - expect(result.error).toBe('WIZARD_API_ERROR'); + expect(authErrors()).toHaveLength(1); + expect(result.failure?.message).toBe('Authentication failed (401)'); }); }); diff --git a/src/lib/__tests__/wizard-ask-bridge.test.ts b/src/lib/__tests__/wizard-ask-bridge.test.ts index 1f12b9a4c..650f71b81 100644 --- a/src/lib/__tests__/wizard-ask-bridge.test.ts +++ b/src/lib/__tests__/wizard-ask-bridge.test.ts @@ -212,15 +212,17 @@ describe('createWizardAskBridge', () => { }); describe('timeout', () => { - it('resolves every field with the cancelled sentinel and dismisses the host overlay when the user does not answer in time', async () => { + it('resolves every field with the cancelled sentinel and aborts the question when the user does not answer in time', async () => { vi.useFakeTimers(); try { // showQuestion intentionally never resolves — the timeout has to win. - const cancelQuestion = vi.fn(); + const signals: AbortSignal[] = []; const bridge = createWizardAskBridge({ getSource: () => 'product-tours', - showQuestion: () => new Promise(() => undefined), - cancelQuestion, + showQuestion: (_question, { signal }) => { + signals.push(signal); + return new Promise(() => undefined); + }, timeoutMs: 1000, }); @@ -230,6 +232,7 @@ describe('createWizardAskBridge', () => { { id: 'audience', prompt: 'Who?', kind: 'text' }, ], }); + expect(signals[0].aborted).toBe(false); vi.advanceTimersByTime(1000); @@ -244,7 +247,7 @@ describe('createWizardAskBridge', () => { // Without this, the host's pending-question state survives the // timeout and every later wizard_ask in the run is rejected as a // duplicate request. - expect(cancelQuestion).toHaveBeenCalledTimes(1); + expect(signals[0].aborted).toBe(true); const cancelledCall = wizardCaptureMock.mock.calls.find( ([name]) => name === 'wizard_ask cancelled', @@ -255,14 +258,16 @@ describe('createWizardAskBridge', () => { } }); - it('does not dismiss the overlay when the user answers before the timeout', async () => { + it('does not abort the question when the user answers before the timeout', async () => { vi.useFakeTimers(); try { - const cancelQuestion = vi.fn(); + const signals: AbortSignal[] = []; const bridge = createWizardAskBridge({ getSource: () => 'product-tours', - showQuestion: () => Promise.resolve({ goal: 'ship it' }), - cancelQuestion, + showQuestion: (_question, { signal }) => { + signals.push(signal); + return Promise.resolve({ goal: 'ship it' }); + }, timeoutMs: 1000, }); @@ -271,7 +276,37 @@ describe('createWizardAskBridge', () => { }); vi.advanceTimersByTime(1000); - expect(cancelQuestion).not.toHaveBeenCalled(); + expect(signals[0].aborted).toBe(false); + } finally { + vi.useRealTimers(); + } + }); + + it('aborts only the question whose timeout fired', async () => { + vi.useFakeTimers(); + try { + const signals: AbortSignal[] = []; + const bridge = createWizardAskBridge({ + getSource: () => 'product-tours', + showQuestion: (_question, { signal }) => { + signals.push(signal); + return new Promise(() => undefined); + }, + timeoutMs: 1000, + }); + const questions = [ + { id: 'goal', prompt: 'Goal?', kind: 'text' as const }, + ]; + + const first = bridge.request({ questions }); + vi.advanceTimersByTime(500); + void bridge.request({ questions }); + vi.advanceTimersByTime(500); + + await expect(first).resolves.toMatchObject({ timedOut: true }); + // The second question is still open: the first one's timeout must not + // dismiss it. + expect(signals.map((signal) => signal.aborted)).toEqual([true, false]); } finally { vi.useRealTimers(); } diff --git a/src/lib/agent/__tests__/progress-collector.test.ts b/src/lib/agent/__tests__/progress-collector.test.ts new file mode 100644 index 000000000..061292550 --- /dev/null +++ b/src/lib/agent/__tests__/progress-collector.test.ts @@ -0,0 +1,39 @@ +import { OutroKind } from '@lib/wizard-session'; +import { createProgressCollector } from '../runner/shared/progress-collector'; +import { MAX_STATUS_MESSAGES } from '@lib/status-history'; + +it('retains a bounded FIFO of statuses and drops consecutive duplicates', () => { + const observer = vi.fn(); + const collector = createProgressCollector(observer); + for (let i = 0; i < MAX_STATUS_MESSAGES + 3; i++) { + collector.emit({ kind: 'status', message: `status ${i}` }); + collector.emit({ kind: 'status', message: `status ${i}` }); + } + const statuses = collector.snapshot().statusMessages; + expect(statuses).toHaveLength(MAX_STATUS_MESSAGES); + expect(statuses[0]).toBe('status 3'); + expect(statuses.at(-1)).toBe(`status ${MAX_STATUS_MESSAGES + 2}`); + expect(observer).toHaveBeenCalledTimes((MAX_STATUS_MESSAGES + 3) * 2); + statuses.push('external mutation'); + expect(collector.snapshot().statusMessages).toHaveLength(MAX_STATUS_MESSAGES); +}); + +it('isolates nested completion data from a mutating observer', () => { + const outro = { + kind: OutroKind.Success, + message: 'Done', + nextSteps: { heading: 'Next', items: ['Keep the report'] }, + }; + const collector = createProgressCollector((event) => { + if (event.kind === 'completion') { + event.outro.message = 'corrupted'; + event.outro.nextSteps?.items.push('corrupted'); + } + }); + collector.emit({ kind: 'completion', outro }); + expect(outro).toEqual({ + kind: OutroKind.Success, + message: 'Done', + nextSteps: { heading: 'Next', items: ['Keep the report'] }, + }); +}); diff --git a/src/lib/agent/__tests__/run-agent-standalone.test.ts b/src/lib/agent/__tests__/run-agent-standalone.test.ts new file mode 100644 index 000000000..3ce26d1c8 --- /dev/null +++ b/src/lib/agent/__tests__/run-agent-standalone.test.ts @@ -0,0 +1,687 @@ +/** + * `runAgent(config, input, options)` runs with no UI, no store, no session and + * no program registry: a fake harness stands in for the SDK, and everything + * the run reports arrives through `onProgress` or comes back in the result. + * + * The `@ui` mock below throws on use. It is never reached — that is the + * assertion the whole file rests on. + */ +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { Harness, Sequence, DEFAULT_AGENT_MODEL } from '@lib/constants'; +import { HostResolution } from '@lib/host-resolution'; +import { + OutroKind, + type AskAnswers, + type PendingQuestion, +} from '@lib/wizard-session'; +import { AGENT_ERROR_CODE, ErrorCodes } from '@lib/errors'; +import { AgentErrorType } from '@lib/agent/signals'; +import type { AgentFailure } from '@lib/agent/runner/shared/types'; +import type { AgentProgress } from '@lib/agent/progress'; +import type { + AgentHarness, + BackendRunInputs, + TaskRunInputs, +} from '@lib/agent/runner/harness/types'; + +vi.mock('@ui', () => ({ + getUI: () => { + throw new Error('the agent reached for getUI()'); + }, + setUI: () => { + throw new Error('the agent reached for setUI()'); + }, +})); +vi.mock('@utils/debug'); +vi.mock('@utils/terminal-bell'); +vi.mock('@lib/yara-hooks', async (original) => ({ + ...(await original()), + flushScanReport: vi.fn(), +})); +vi.mock('@utils/analytics', () => ({ + analytics: { + build: 'test', + runId: 'run-1', + wizardCapture: vi.fn(), + capture: vi.fn(), + captureException: vi.fn(), + setTag: vi.fn(), + shutdown: vi.fn().mockResolvedValue(undefined), + }, +})); +vi.mock('@lib/gateway-session', async (importOriginal) => ({ + ...(await importOriginal()), + gatewayAuth: vi.fn().mockResolvedValue({ + gatewayUrl: 'https://gateway.test', + token: 'phe_run', + teamId: 1, + refreshAtMs: Date.now() + 3_600_000, + }), +})); + +// The fake harness: reports a little of everything, then returns what the +// current test told it to. +const harnessState = vi.hoisted(() => ({ + result: {} as { error?: string; message?: string; failure?: unknown }, + throws: undefined as Error | undefined, + lastInputs: undefined as unknown, + tasks: [] as TaskRunInputs[], + selected: [] as Harness[], + taskFailure: undefined as AgentFailure | undefined, + seedFailure: undefined as AgentFailure | undefined, + askQuestions: undefined as PendingQuestion['questions'] | undefined, +})); +vi.mock('@lib/agent/runner/switchboard/harness', () => { + const askIfRequested = async (inputs: BackendRunInputs | TaskRunInputs) => { + if (!harnessState.askQuestions || !inputs.askBridge) return; + const { answers } = await inputs.askBridge.request({ + questions: harnessState.askQuestions, + }); + inputs.emit({ + kind: 'status', + message: `answered:${JSON.stringify(answers)}`, + }); + }; + const fake: AgentHarness = { + name: Harness.pi, + async runTask(inputs: TaskRunInputs) { + harnessState.tasks.push(inputs); + const { store, currentTaskId } = inputs.orchestrator; + if (!currentTaskId) { + if (harnessState.seedFailure) + return Promise.resolve({ failure: harnessState.seedFailure }); + store.enqueue({ type: 'install' }); + } else if (harnessState.taskFailure) { + return Promise.resolve({ failure: harnessState.taskFailure }); + } else { + await askIfRequested(inputs); + store.complete(currentTaskId, { + goals: 'install', + did: 'installed', + forNextAgent: 'done', + }); + } + return Promise.resolve({}); + }, + async run(inputs: BackendRunInputs) { + harnessState.lastInputs = inputs; + const { emit, spinner } = inputs; + emit({ kind: 'log', level: 'step', message: 'Initializing agent' }); + spinner.start('Working'); + emit({ kind: 'status', message: 'Installing the SDK' }); + emit({ + kind: 'tasks', + tasks: [{ content: 'Install', status: 'completed' }], + }); + emit({ kind: 'url', which: 'dashboard', url: 'https://d/1' }); + emit({ kind: 'url', which: 'notebook', url: 'https://n/1' }); + emit({ kind: 'stage', stage: 'Configure' }); + emit({ kind: 'finalCost', usd: 0.25 }); + emit({ + kind: 'usage', + delta: { + inputTokens: 10, + outputTokens: 5, + cacheReadTokens: 0, + cacheCreationTokens: 0, + cacheCreation5m: 0, + cacheCreation1h: 0, + }, + }); + await askIfRequested(inputs); + if (harnessState.throws) throw harnessState.throws; + spinner.stop('Done'); + return harnessState.result as never; + }, + }; + return { + HARNESS_OPTIONS: { [Harness.pi]: fake }, + getHarness: (name: Harness) => { + harnessState.selected.push(name); + return { ...fake, name }; + }, + resolveHarness: (ctx: { cliHarness?: Harness }) => ({ + harness: ctx.cliHarness ?? Harness.pi, + model: DEFAULT_AGENT_MODEL, + }), + }; +}); + +vi.mock('@lib/agent/agent-prompt-loader', async (original) => { + const actual = await original< + typeof import('@lib/agent/agent-prompt-loader') + >(); + return { + ...actual, + loadAgentRegistry: vi.fn(() => + Promise.resolve( + actual.buildRegistry( + [ + actual.parseAgentPrompt( + '---\ntype: seed\nseed: true\n---\nPlan work', + 'seed', + 'test-program', + ), + actual.parseAgentPrompt( + '---\ntype: install\nallowedTools: [wizard_ask]\n---\nInstall it', + 'install', + 'test-program', + ), + ], + 'test-program', + ), + ), + ), + }; +}); +vi.mock('@lib/wizard-tools', async (original) => ({ + ...(await original()), + fetchSkillMenu: vi.fn().mockResolvedValue({ categories: {} }), +})); + +import { runAgent, RunOutcome } from '@lib/agent/runner'; +import type { RunConfig, RunInput } from '@lib/agent/runner'; +import { analytics } from '@utils/analytics'; +import { initLogFile } from '@utils/debug'; +import { flushScanReport } from '@lib/yara-hooks'; +import { QUEUE_DIR_NAME } from '../runner/sequence/orchestrator/queue'; + +let tmp: string; + +const config = (over: Partial = {}): RunConfig => ({ + programId: 'test-program', + run: { + integrationLabel: 'test-integration', + spinnerMessage: 'Working...', + successMessage: 'Done!', + estimatedDurationMinutes: 1, + reportFile: 'report.md', + docsUrl: 'https://docs.test', + abortCases: [ + { + match: /no stripe/i, + message: 'No Stripe here', + body: 'Stripe is required.', + errorCode: ErrorCodes.AgentAbort, + }, + ], + }, + composed: false, + binding: { sequence: Sequence.linear, harness: Harness.pi, model: 'm' }, + switchboard: { program: 'test-program', flags: {} }, + skillsBaseUrl: 'https://skills.test', + wizardFlags: {}, + wizardFlagPayloads: {}, + wizardMetadata: {}, + ...over, +}); + +const input = (over: Partial = {}): RunInput => ({ + installDir: tmp, + credentials: { + accessToken: 'tok', + projectApiKey: 'phc_test', + host: HostResolution.fromApiHost('https://us.posthog.com'), + projectId: 1, + }, + project: null, + apiUser: null, + skillId: 'test-integration', + flags: { + ci: false, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + host: {}, + ...over, +}); + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'run-agent-standalone-')); + harnessState.result = {}; + harnessState.tasks = []; + harnessState.selected = []; + harnessState.taskFailure = undefined; + harnessState.seedFailure = undefined; + harnessState.throws = undefined; + harnessState.lastInputs = undefined; + harnessState.askQuestions = undefined; + vi.mocked(analytics.shutdown).mockClear(); + vi.mocked(initLogFile).mockClear(); + vi.mocked(flushScanReport).mockClear(); +}); + +afterEach(() => fs.rmSync(tmp, { recursive: true, force: true })); + +describe('runAgent standalone', () => { + it.each([ + [Harness.pi, Sequence.linear], + [Harness.anthropic, Sequence.linear], + [Harness.pi, Sequence.orchestrator], + [Harness.anthropic, Sequence.orchestrator], + ])( + 'waits for the answer before completing %s %s', + async (harness, sequence) => { + harnessState.askQuestions = [ + { id: 'q1', prompt: 'Continue?', kind: 'text' }, + ]; + let release: ((answers: AskAnswers) => void) | undefined; + const ask = vi.fn( + () => + new Promise((resolve) => { + release = resolve; + }), + ); + const events: AgentProgress[] = []; + let settled = false; + const running = runAgent( + config({ + binding: { harness, sequence, model: DEFAULT_AGENT_MODEL }, + switchboard: { + program: 'test-program', + flags: {}, + cliHarness: harness, + }, + }), + input(), + { interaction: { ask }, onProgress: (event) => events.push(event) }, + ).then((result) => { + settled = true; + return result; + }); + + try { + await vi.waitFor(() => expect(ask).toHaveBeenCalledTimes(1)); + await new Promise((resolve) => setImmediate(resolve)); + expect(settled).toBe(false); + expect(events.some((event) => event.kind === 'completion')).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + expect(flushScanReport).not.toHaveBeenCalled(); + } finally { + release?.({ q1: 'yes' }); + } + + const result = await running; + expect(result.outcome).toBe(RunOutcome.Success); + expect(result.snapshot.statusMessages).toContain('answered:{"q1":"yes"}'); + expect(analytics.shutdown).not.toHaveBeenCalled(); + expect(flushScanReport).toHaveBeenCalledTimes(1); + }, + ); + + it.each([Sequence.linear, Sequence.orchestrator])( + 'does not wait for disabled questions in %s', + async (sequence) => { + harnessState.askQuestions = [ + { id: 'q1', prompt: 'Continue?', kind: 'text' }, + ]; + const ask = vi.fn(() => + Promise.reject(new Error('disabled answerer called')), + ); + const runConfig = config({ + binding: { harness: Harness.pi, sequence, model: DEFAULT_AGENT_MODEL }, + }); + const runInput = input(); + runInput.flags.ci = true; + vi.stubEnv('WIZARD_ASK_AUTODRIVE', ''); + try { + const result = await runAgent(runConfig, runInput, { + interaction: { ask }, + }); + expect(result.outcome).toBe(RunOutcome.Success); + expect(ask).not.toHaveBeenCalled(); + const inputs = + sequence === Sequence.linear + ? (harnessState.lastInputs as BackendRunInputs) + : harnessState.tasks.at(-1); + expect(inputs?.askBridge).toBeUndefined(); + } finally { + vi.unstubAllEnvs(); + } + }, + ); + + it.each([Harness.pi, Harness.anthropic])( + 'dispatches %s through both sequence arms', + async (harness) => { + for (const sequence of [Sequence.linear, Sequence.orchestrator]) { + const result = await runAgent( + config({ + binding: { harness, sequence, model: DEFAULT_AGENT_MODEL }, + switchboard: { + program: 'test-program', + flags: {}, + cliHarness: harness, + }, + }), + input(), + ); + expect(result.outcome).toBe('success'); + expect(harnessState.selected.at(-1)).toBe(harness); + if (sequence === Sequence.linear) { + expect(harnessState.tasks).toHaveLength(0); + } else { + expect(harnessState.tasks).toHaveLength(2); + expect( + harnessState.tasks[1].orchestrator.currentTaskId, + ).toBeDefined(); + expect(result.snapshot.tasks).toEqual([ + { content: 'install', activeForm: 'install', status: 'completed' }, + ]); + } + } + // Terminal analytics belong to the process, so to the host. + expect(analytics.shutdown).not.toHaveBeenCalled(); + }, + ); + + it('cleans up when the seed fails before the drain starts', async () => { + const failure = { message: 'Authentication failed (401)' }; + harnessState.seedFailure = failure; + const result = await runAgent( + config({ + binding: { + harness: Harness.anthropic, + sequence: Sequence.orchestrator, + model: DEFAULT_AGENT_MODEL, + }, + switchboard: { + program: 'test-program', + flags: {}, + cliHarness: Harness.anthropic, + }, + }), + input(), + ); + expect(result.outcome).toBe('failed'); + expect(result.failure).toBe(failure); + expect(harnessState.tasks).toHaveLength(1); + expect(fs.existsSync(path.join(tmp, QUEUE_DIR_NAME))).toBe(false); + }); + + it('returns an anthropic orchestrator task failure and cleans up the queue', async () => { + const failure = { + code: ErrorCodes.AgentAbort, + message: 'Authentication failed (401)', + }; + harnessState.taskFailure = failure; + const result = await runAgent( + config({ + binding: { + harness: Harness.anthropic, + sequence: Sequence.orchestrator, + model: DEFAULT_AGENT_MODEL, + }, + switchboard: { + program: 'test-program', + flags: {}, + cliHarness: Harness.anthropic, + }, + }), + input(), + ); + expect(result.outcome).toBe('failed'); + expect(result.failure).toBe(failure); + expect(harnessState.tasks).toHaveLength(2); + expect(fs.existsSync(path.join(tmp, QUEUE_DIR_NAME))).toBe(false); + }); + + it.each([Sequence.linear, Sequence.orchestrator])( + 'preserves %s completion and scan-flush ordering and leaves the shutdown to the host', + async (sequence) => { + const order: string[] = []; + vi.mocked(analytics.shutdown).mockImplementationOnce(() => { + order.push('shutdown'); + return Promise.resolve(); + }); + vi.mocked(flushScanReport).mockImplementationOnce(() => { + order.push('scan-flush'); + }); + const result = await runAgent( + config({ + binding: { + harness: Harness.pi, + sequence, + model: DEFAULT_AGENT_MODEL, + }, + }), + input(), + { + onProgress: (event) => { + if (event.kind === 'completion') { + order.push( + fs.existsSync(path.join(tmp, QUEUE_DIR_NAME)) + ? 'queue-present' + : 'queue-clean', + ); + order.push('completion'); + } + if (event.kind === 'lifecycle' && event.phase === 'completed') + order.push('outro'); + }, + }, + ); + expect(result.outcome).toBe('success'); + expect(order).toEqual([ + 'queue-clean', + 'completion', + 'outro', + 'scan-flush', + ]); + }, + ); + + it('runs to success with an observer and an answerer', async () => { + const events: AgentProgress[] = []; + const ask = vi.fn(() => Promise.resolve({ q1: 'yes' })); + harnessState.askQuestions = [ + { id: 'q1', prompt: 'Continue?', kind: 'single' as const }, + ]; + + const result = await runAgent(config(), input(), { + onProgress: (e) => events.push(e), + interaction: { ask }, + }); + + expect(result.outcome).toBe('success'); + expect(result.skillId).toBe('test-integration'); + expect(result.outro).toEqual({ + kind: OutroKind.Success, + message: 'Done!', + reportFile: 'report.md', + docsUrl: 'https://docs.test', + continueUrl: undefined, + }); + expect(result.failure).toBeUndefined(); + + // The answerer saw the question the bridge built, with that question's own + // signal, and its answer came back. + expect(ask).toHaveBeenCalledTimes(1); + const [question, context] = ask.mock.calls[0] as unknown as [ + PendingQuestion, + { signal: AbortSignal }, + ]; + expect(question.questions[0].id).toBe('q1'); + expect(question.source).toBe('test-integration'); + expect(context.signal.aborted).toBe(false); + expect(events).toContainEqual({ + kind: 'status', + message: 'answered:{"q1":"yes"}', + }); + + // Emission order: started first, completion then completed last. + const kinds = events.map( + (e) => `${e.kind}${'phase' in e ? `:${e.phase}` : ''}`, + ); + expect(kinds[0]).toBe('lifecycle:started'); + expect(kinds.slice(-2)).toEqual(['completion', 'lifecycle:completed']); + + // The agent's own snapshot, independent of the observer. + expect(result.snapshot.dashboardUrl).toBe('https://d/1'); + expect(result.snapshot.notebookUrl).toBe('https://n/1'); + expect(result.snapshot.stage).toBe('Configure'); + expect(result.snapshot.finalCostUsd).toBe(0.25); + expect(result.snapshot.tasks).toEqual([ + { content: 'Install', status: 'completed' }, + ]); + expect(result.snapshot.statusMessages[0]).toBe('Installing the SDK'); + expect(result.snapshot.usage).toEqual({ + inputTokens: 10, + outputTokens: 5, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('runs to a complete result with no options at all', async () => { + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('success'); + expect(result.outro?.kind).toBe(OutroKind.Success); + expect(result.snapshot.dashboardUrl).toBe('https://d/1'); + // No answerer: no bridge is installed, so `wizard_ask` reports unavailable + // rather than hanging. + const inputs = harnessState.lastInputs as BackendRunInputs; + expect(inputs.askBridge).toBeUndefined(); + }); + + it('installs no bridge in CI even with an answerer, as before', async () => { + await runAgent( + config(), + input({ + flags: { + ci: true, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + }), + { interaction: { ask: () => Promise.resolve({}) } }, + ); + const inputs = harnessState.lastInputs as BackendRunInputs; + expect(inputs.askBridge).toBeUndefined(); + }); + + it('returns an agent abort as a decided failure with the matched case', async () => { + harnessState.result = { + error: AgentErrorType.ABORT, + message: 'No Stripe found', + }; + const events: AgentProgress[] = []; + + const result = await runAgent(config(), input(), { + onProgress: (e) => events.push(e), + }); + + expect(result.outcome).toBe('aborted'); + expect(result.failure?.code).toBe(ErrorCodes.AgentAbort); + expect(result.failure?.outroData).toMatchObject({ + kind: OutroKind.Error, + message: 'No Stripe here', + body: 'Stripe is required.', + errorDetail: { reason: 'No Stripe found' }, + }); + expect(result.failure?.error?.message).toBe( + 'Agent aborted: No Stripe found', + ); + expect(events.some((e) => e.kind === 'completion')).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('returns a coded failure for a harness error', async () => { + harnessState.result = { error: AgentErrorType.NO_PROGRESS }; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('failed'); + expect(result.failure?.code).toBe( + AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], + ); + expect(result.failure?.message).toContain('without changing your project'); + }); + + it('passes a harness-decided failure through untouched', async () => { + const failure = { code: ErrorCodes.AgentAbort, message: 'decided' }; + harnessState.result = { failure }; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('failed'); + expect(result.failure).toBe(failure); + }); + + it('skips the terminal outro for a composed sub-run', async () => { + const events: AgentProgress[] = []; + + const result = await runAgent(config({ composed: true }), input(), { + onProgress: (e) => events.push(e), + }); + + expect(result.outcome).toBe('success'); + expect(result.outro).toBeUndefined(); + expect(events.some((e) => e.kind === 'completion')).toBe(false); + expect( + events.some((e) => e.kind === 'lifecycle' && e.phase === 'completed'), + ).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('finishes when the observer throws on every event', async () => { + const result = await runAgent(config(), input(), { + onProgress: () => { + throw new Error('projection broke'); + }, + }); + + expect(result.outcome).toBe('success'); + expect(result.snapshot.tasks).toHaveLength(1); + }); + + it('calls the bound hooks with the run credentials', async () => { + const postRun = vi.fn().mockResolvedValue(undefined); + const buildOutroData = vi.fn(() => ({ + kind: OutroKind.Success, + message: 'custom', + })); + + const result = await runAgent( + config({ hooks: { postRun, buildOutroData } }), + input(), + ); + + expect(postRun).toHaveBeenCalledWith( + expect.objectContaining({ projectApiKey: 'phc_test' }), + ); + expect(buildOutroData).toHaveBeenCalledTimes(1); + expect(result.outro).toEqual({ + kind: OutroKind.Success, + message: 'custom', + }); + }); + + it('returns a crash as a result instead of rejecting', async () => { + const boom = new Error('SDK exploded'); + harnessState.throws = boom; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('crashed'); + expect(result.failure?.error).toBe(boom); + expect(result.failure?.message).toBe('SDK exploded'); + expect(result.failure?.code).toBe(ErrorCodes.InternalUnhandled); + // What was reported before the crash survives in the snapshot. + expect(result.snapshot.statusMessages).toContain('Installing the SDK'); + }); +}); diff --git a/src/lib/agent/agent-interface.ts b/src/lib/agent/agent-interface.ts index 18f3a6565..258153986 100644 --- a/src/lib/agent/agent-interface.ts +++ b/src/lib/agent/agent-interface.ts @@ -6,8 +6,11 @@ import path from 'path'; import * as os from 'os'; import { createRequire } from 'node:module'; -import { getUI, type SpinnerHandle } from '@ui'; -import type { TokenUsageDelta } from '@ui/wizard-ui'; +import type { + ProgressEmitter, + SpinnerHandle, + TokenUsageDelta, +} from './progress'; import { debug, logToFile, initLogFile, getLogFilePath } from '@utils/debug'; import type { WizardRunOptions } from '@utils/types'; import { analytics } from '@utils/analytics'; @@ -27,7 +30,8 @@ import { type AdditionalFeature, ADDITIONAL_FEATURE_PROMPTS, } from '@lib/wizard-session'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@utils/wizard-abort'; +import type { AgentFailure } from './runner/shared/types'; import { createCustomHeaders } from '@utils/custom-headers'; import type { HostResolution } from '@lib/host-resolution'; import { @@ -243,6 +247,12 @@ export type AgentConfig = { * `--capture-aio` is off. Constructed once per run by the harness. */ capture?: AioCapture; + /** + * Where the run reports: log lines, tasks, status, URLs, usage, the handoff + * document and the auth-error detail. Absent → the run reports nowhere and + * still completes. + */ + emit?: ProgressEmitter; }; /** @@ -354,8 +364,12 @@ type AgentRunConfig = { program?: string; /** Resolved sequence, for the sequence-axis commandments. */ sequence: Sequence; + /** Where the run reports. A no-op when the caller passed none. */ + emit?: ProgressEmitter; }; +const NO_PROGRESS: ProgressEmitter = () => undefined; + /** * Global identifiers attached to every LLM gateway trace for a run. They ride on * each `$ai_generation` the gateway emits (in the `X-PostHog-Properties` blob @@ -531,6 +545,7 @@ export async function initializeAgent( initLogFile(); logToFile('Agent initialization starting'); logToFile('Install directory:', options.installDir); + const emit = config.emit ?? NO_PROGRESS; try { // Configure model routing (inherited by the SDK subprocess). All model @@ -661,6 +676,7 @@ export async function initializeAgent( program: config.integrationLabel, // A queue context is present only on a task run; that is the sequence. sequence: config.orchestrator ? Sequence.orchestrator : Sequence.linear, + emit, }; logToFile('Agent config:', { @@ -689,9 +705,11 @@ export async function initializeAgent( return agentRunConfig; } catch (error) { - getUI().log.error( - `Failed to initialize agent: ${(error as Error).message}`, - ); + emit({ + kind: 'log', + level: 'error', + message: `Failed to initialize agent: ${(error as Error).message}`, + }); logToFile('Agent initialization error:', error); debug('Agent initialization error:', error); throw error; @@ -742,7 +760,12 @@ export async function runAgent( onMessage(message: any): void; finalize(resultMessage: any, totalDurationMs: number): any; }, -): Promise<{ error?: AgentErrorType; message?: string }> { +): Promise<{ + error?: AgentErrorType; + message?: string; + failure?: AgentFailure; +}> { + const emit = agentConfig.emit ?? NO_PROGRESS; const { spinnerMessage = 'Customizing your PostHog setup...', successMessage = 'PostHog integration complete', @@ -856,7 +879,7 @@ export async function runAgent( // CostTrackerPlugin.onFinalize uses to correct per-turn drift. const totalCostUsd = Number(lastResultMessage?.total_cost_usd ?? 0); if (totalCostUsd > 0) { - getUI().setFinalTokenCostUsd(totalCostUsd); + emit({ kind: 'finalCost', usd: totalCostUsd }); } try { middleware?.finalize(lastResultMessage, durationMs); @@ -882,6 +905,9 @@ export async function runAgent( let sessionId: string | undefined; let reminted = false; let remintRequested = false; + // A 401 on a fresh bearer: the auth screen was reported, and this is the + // failure the caller ends the run with. The query is aborted to unwind. + let authFailure: AgentFailure | undefined; const agentConfigDir = createIsolatedAgentConfigDir(); const timeoutMs = config?.timeoutMs; const timeoutId = timeoutMs @@ -1174,6 +1200,7 @@ export async function runAgent( options, spinner, signals, + emit, receivedSuccessResult, tasks, agentConfig.suppressTaskRender ?? false, @@ -1252,15 +1279,20 @@ export async function runAgent( ...authError, sessionExpired, }); - getUI().showAuthError({ - hasSettingsConflict: authError.hasSettingsConflict, - conflicts: authError.conflicts, - usingManagedLogin: authError.usingManagedLogin, - credentialPlaces: authError.credentialPlaces, - sessionExpired, - logFilePath: getLogFilePath(), + emit({ + kind: 'authError', + detail: { + hasSettingsConflict: authError.hasSettingsConflict, + conflicts: authError.conflicts, + usingManagedLogin: authError.usingManagedLogin, + credentialPlaces: authError.credentialPlaces, + sessionExpired, + logFilePath: getLogFilePath(), + }, }); - await wizardAbort({ + // The caller ends the run with this; the query is abandoned here + // where the process used to exit. + authFailure = { code: authCode, message: 'Authentication failed (401)', error: new WizardError( @@ -1276,7 +1308,9 @@ export async function runAgent( }, authCode, ), - }); + }; + abortController.abort(); + break; } try { @@ -1302,6 +1336,7 @@ export async function runAgent( } catch (error) { // The abort we asked for; anything else belongs to the outer catch. if (remintRequested) return 'remint'; + if (authFailure) return 'done'; throw error; } return remintRequested ? 'remint' : 'done'; @@ -1333,6 +1368,12 @@ export async function runAgent( await runQuery(sessionId); } + // A fresh bearer was rejected. The auth screen is already up; hand the + // decided failure to the caller, which owns the exit. + if (authFailure) { + return { failure: authFailure }; + } + // A YARA hook detected a terminal violation and aborted the run. if (yaraViolationReason) { logToFile('Agent error: YARA_VIOLATION'); @@ -1449,15 +1490,20 @@ export async function runAgent( // No API error found, re-throw the original exception spinner.stop(errorMessage); - getUI().log.error(`Error: ${(error as Error).message}`); + emit({ + kind: 'log', + level: 'error', + message: `Error: ${(error as Error).message}`, + }); logToFile('Agent run failed:', error); debug('Full error:', error); throw error; } finally { if (timeoutId) clearTimeout(timeoutId); // Always capture run duration, even on abort/error, so we can alert on - // long runs where the user gave up before completion. - if (!receivedSuccessResult) { + // long runs where the user gave up before completion. A 401 never reached + // this block before (the process exited first), so it still does not count. + if (!receivedSuccessResult && !authFailure) { const durationMs = Date.now() - startTime; analytics.wizardCapture('agent aborted', { duration_ms: durationMs, @@ -1766,6 +1812,7 @@ function handleSDKMessage( options: WizardRunOptions, spinner: SpinnerHandle, signals: AgentOutputSignals, + emit: ProgressEmitter, receivedSuccessResult = false, tasks?: Map, // The orchestrator owns the TUI task panel (it renders its queue). Suppress the @@ -1791,7 +1838,7 @@ function handleSDKMessage( const sorted = Array.from(tasks.values()).sort( (a, b) => rank(a.status) - rank(b.status), ); - getUI().syncTodos(sorted); + emit({ kind: 'tasks', tasks: sorted.map((t) => ({ ...t })) }); }; logToFile(`SDK Message: ${message.type}`, JSON.stringify(message, null, 2)); @@ -1806,7 +1853,7 @@ function handleSDKMessage( // dedup for SDK-retried turns — see addTokenUsage's doc comment), so // this stays live-updating for every run, not just `--benchmark`. const tokenUsageDelta = extractTokenUsageDelta(message); - if (tokenUsageDelta) getUI().addTokenUsage(tokenUsageDelta); + if (tokenUsageDelta) emit({ kind: 'usage', delta: tokenUsageDelta }); // Extract text content from assistant messages const content = message.message?.content; @@ -1826,7 +1873,7 @@ function handleSDKMessage( const statusMatch = block.text.match(statusRegex); if (statusMatch) { const statusText = statusMatch[1].trim(); - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } @@ -1840,7 +1887,11 @@ function handleSDKMessage( ); const dashboardMatch = block.text.match(dashboardRegex); if (dashboardMatch) { - getUI().setDashboardUrl(dashboardMatch[1].trim()); + emit({ + kind: 'url', + which: 'dashboard', + url: dashboardMatch[1].trim(), + }); } // Check for [NOTEBOOK_URL] markers @@ -1853,7 +1904,11 @@ function handleSDKMessage( ); const notebookMatch = block.text.match(notebookRegex); if (notebookMatch) { - getUI().setNotebookUrl(notebookMatch[1].trim()); + emit({ + kind: 'url', + which: 'notebook', + url: notebookMatch[1].trim(), + }); } } @@ -1876,7 +1931,7 @@ function handleSDKMessage( // Mirror the active tool into the Visualizer's "stage" indicator. if (block.type === 'tool_use') { const stage = classifyToolToStage((block as ToolUseBlock).name); - if (stage) getUI().setStage(stage); + if (stage) emit({ kind: 'stage', stage }); } } } @@ -1929,7 +1984,7 @@ function handleSDKMessage( // mode race conditions). Full message already logged above via JSON dump. if (message.errors && !receivedSuccessResult) { for (const err of message.errors) { - getUI().log.error(`Error: ${err}`); + emit({ kind: 'log', level: 'error', message: `Error: ${err}` }); logToFile('ERROR:', err); } } @@ -1944,7 +1999,7 @@ function handleSDKMessage( // Full message already logged above via JSON dump. if (message.errors && !receivedSuccessResult) { for (const err of message.errors) { - getUI().log.error(`Error: ${err}`); + emit({ kind: 'log', level: 'error', message: `Error: ${err}` }); logToFile('ERROR:', err); } } diff --git a/src/lib/agent/agent-runner.ts b/src/lib/agent/agent-runner.ts index 8b92c4ebe..2850afcc1 100644 --- a/src/lib/agent/agent-runner.ts +++ b/src/lib/agent/agent-runner.ts @@ -1,11 +1,12 @@ /** * Re-export shim. The runner has been split into agent/runner/. * Import from there directly; this shim keeps existing importers working. + * The session-driven `runProgramAgent(programConfig, session)` lives in + * `src/lib/programs/run-agent-legacy.ts`. */ export { runAgent, - runProgram, shouldDisableAsk, type ProgramRun, type BootstrapResult, diff --git a/src/lib/agent/progress.ts b/src/lib/agent/progress.ts new file mode 100644 index 000000000..5e246d40d --- /dev/null +++ b/src/lib/agent/progress.ts @@ -0,0 +1,98 @@ +/** + * The agent's public progress and interaction contracts. + * + * `runAgent` reports through one optional callback and asks through one + * optional set of capabilities. Neither reaches into a UI singleton, a store, + * or a session: every payload is copied data, every question is awaited on an + * injected answerer. The legacy adapter in `src/lib/programs/run-agent-legacy.ts` + * maps these back onto `WizardUI` one call per event, so the terminal output of + * every existing runner is unchanged. + */ + +import type { + AskAnswers, + OutroData, + PendingQuestion, + TaskNotice, +} from '@lib/wizard-session'; +import type { + AuthErrorDetail, + SpinnerHandle, + TokenUsageDelta, +} from '@ui/wizard-ui'; + +export type { AuthErrorDetail, SpinnerHandle, TokenUsageDelta }; + +/** One task as the host renders it. The same shape `WizardUI.syncTodos` takes. */ +export interface TaskSnapshot { + content: string; + status: string; + activeForm?: string; +} + +export type ProgressLogLevel = 'info' | 'warn' | 'error' | 'success' | 'step'; + +/** + * Everything the agent reports while it runs. One event per former + * `getUI()` call, in the same order, with the same payload, so a reducer that + * maps each case back onto `WizardUI` reproduces today's output exactly. + * + * Payloads are copies. Never a store, a setter, a function or a live + * collection. The callback returns nothing and the agent never branches on it. + */ +export type AgentProgress = + /** The run's main work has started (`WizardUI.startRun`). */ + | { kind: 'lifecycle'; phase: 'started' } + /** The run finished and the host may show its outro (`WizardUI.outro`). */ + | { kind: 'lifecycle'; phase: 'completed'; message: string } + /** The run spinner (`WizardUI.spinner()`), one handle per run. */ + | { + kind: 'spinner'; + action: 'start' | 'stop' | 'message'; + message?: string; + } + /** A log line (`WizardUI.log[level]`). */ + | { kind: 'log'; level: ProgressLogLevel; message: string } + /** A `[STATUS]` line the agent printed (`WizardUI.pushStatus`). */ + | { kind: 'status'; message: string } + /** The full task list, already sorted for display (`WizardUI.syncTodos`). */ + | { kind: 'tasks'; tasks: TaskSnapshot[] } + /** The stage of work derived from the active tool (`WizardUI.setStage`). */ + | { kind: 'stage'; stage: string } + /** A PostHog URL the agent created (`setDashboardUrl` / `setNotebookUrl`). */ + | { kind: 'url'; which: 'dashboard' | 'notebook'; url: string } + /** One assistant turn's token usage (`WizardUI.addTokenUsage`). */ + | { kind: 'usage'; delta: TokenUsageDelta } + /** The SDK's authoritative run cost (`WizardUI.setFinalTokenCostUsd`). */ + | { kind: 'finalCost'; usd: number } + /** The gateway returned 401; a failure follows (`WizardUI.showAuthError`). */ + | { kind: 'authError'; detail: AuthErrorDetail } + /** The run's final outro payload (`WizardUI.setOutroData`). */ + | { kind: 'completion'; outro: OutroData }; + +export type ProgressEmitter = (event: AgentProgress) => void; + +/** + * The questions the agent may need a person (or a script) to answer. Every + * capability is optional. With none supplied the agent installs no ask bridge, + * so `wizard_ask` returns its existing "not available" error, and an optional + * task notice is declined — the same path a `--ci` run takes today. + * On a request's `signal` abort, the host dismisses that request alone, and + * that dismissal must not throw: abort listeners run where the agent cannot + * catch them, so Node would rethrow the error as an uncaught exception. + */ +export interface AgentInteraction { + /** + * Open a question and resolve with the answers. The bridge that calls this + * owns the timeout, the `__cancelled__` sentinel and the analytics. + */ + ask?: ( + question: PendingQuestion, + context: { signal: AbortSignal }, + ) => Promise; + /** Offer an optional step and resolve with whether to keep it. */ + taskNotice?: ( + notice: TaskNotice, + context: { signal: AbortSignal }, + ) => Promise; +} diff --git a/src/lib/agent/runner/README.md b/src/lib/agent/runner/README.md index 86fee7f1b..5f43656cb 100644 --- a/src/lib/agent/runner/README.md +++ b/src/lib/agent/runner/README.md @@ -39,13 +39,18 @@ for the coordinated change checklist. Five layers, each with its own job. Nothing crosses layers unless it has to. -**The entry point** (`index.ts`) is the front door. It receives a program config -and a session, and orchestrates the run at the highest level. - -**Bootstrap** (`shared/bootstrap.ts`) is the on-ramp. Every run starts with the -same setup work — health checks, settings conflicts, OAuth, PostHog feature flag -fetch, MCP URL, AI opt-in gate. Whether the run turns out to be linear or -orchestrator, anthropic or pi, the setup is the same. +**The entry point** (`index.ts`) is the front door: +`runAgent(config, input, {onProgress?, interaction?}) → RunResult`. It takes +resolved execution data and an invocation snapshot (`shared/types.ts`), reports +through `onProgress` and asks through `interaction` (`../progress.ts`), and +returns every ending as a result. It never renders, reads a session or exits. +The gates, OAuth, flags and binding lookup that used to run here live in +`src/lib/programs/run-agent-legacy.ts`, which also maps progress back onto +`getUI()` for today's runners. + +**Prepare** (`shared/bootstrap.ts`) is the on-ramp inside the agent: logging +targets, the gateway mint and the scan-triage classifier. Whether the run turns +out to be linear or orchestrator, anthropic or pi, the setup is the same. **The switchboard** (`switchboard/`) is the router. Given a program id + the fetched flags + any CLI overrides, it returns a `ProgramBinding` — which query @@ -72,8 +77,7 @@ gateway. ## How they connect -- Bootstrap prepares shared services, including triage selected for the resolved - harness. +- Prepare mints the gateway token and builds triage for the resolved harness. - The switchboard knows which sequences and harnesses exist (via its two registries), but not what they do. - A sequence knows how to shape a conversation, but delegates the actual model @@ -84,12 +88,13 @@ Each layer is replaceable. ## Flow -1. Program picked → session built. -2. Bootstrap runs (shared setup, fetches PostHog flags). -3. Switchboard resolves a `ProgramBinding { sequence, harness, model }`. -4. Analytics tags the run with its bindings. -5. Sequence takes over — shapes the LLM's work into one conversation (linear) or - many (orchestrator). -6. Harness drives each conversation through its SDK, using the bound model, on +1. The caller runs its gates, authenticates, fetches PostHog flags and resolves + a `ProgramBinding { sequence, harness, model }`; analytics tags the run. +2. `runAgent(config, input, options)` prepares (mint, triage). +3. Sequence takes over — shapes the LLM's work into one conversation (linear) or + many (orchestrator), reporting through `onProgress`. +4. Harness drives each conversation through its SDK, using the bound model, on the PostHog LLM gateway. -7. Cleanup runs (scan report, settings restore, outro). +5. The scan report flushes; `runAgent` returns a `RunResult`. +6. The caller applies it: a decided failure goes to `wizardAbort`, a crash is + rethrown for the runner's own handling. diff --git a/src/lib/agent/runner/harness/anthropic/__tests__/pending-question.test.ts b/src/lib/agent/runner/harness/anthropic/__tests__/pending-question.test.ts new file mode 100644 index 000000000..966cbcb9d --- /dev/null +++ b/src/lib/agent/runner/harness/anthropic/__tests__/pending-question.test.ts @@ -0,0 +1,201 @@ +import { initializeAgent, wizardCanUseTool } from '@lib/agent/agent-interface'; +import { createAskBridge } from '../../../shared/ask'; +import { anthropicBackend } from '..'; +import type { BackendRunInputs, TaskRunInputs } from '../../types'; +import type { AskAnswers } from '@lib/wizard-session'; +import { Harness, Sequence } from '@lib/constants'; +import { HostResolution } from '@lib/host-resolution'; + +vi.mock('@utils/analytics'); +vi.mock('@utils/debug'); +vi.mock('@lib/agent/aio-capture', () => ({ createAioCapture: vi.fn() })); +vi.mock('@lib/agent/agent-interface', async (original) => ({ + ...(await original()), + initializeAgent: vi.fn().mockResolvedValue({}), + runAgent: vi.fn().mockResolvedValue({}), +})); + +const questions = [{ id: 'q', prompt: 'Continue?', kind: 'text' as const }]; + +async function initializeHarness( + mode: 'linear' | 'task', + askBridge: BackendRunInputs['askBridge'], +) { + const credentials = { + accessToken: 'test', + projectApiKey: 'test', + projectId: 1, + host: HostResolution.fromApiHost('https://us.posthog.com'), + }; + const inputs: BackendRunInputs = { + config: { + programId: 'test', + run: { + integrationLabel: 'test', + spinnerMessage: 'Working', + successMessage: 'Done', + estimatedDurationMinutes: 1, + reportFile: 'report.md', + docsUrl: 'https://docs.test', + }, + composed: false, + binding: { + harness: Harness.anthropic, + sequence: Sequence.linear, + model: 'test', + }, + switchboard: { program: 'test', flags: {} }, + skillsBaseUrl: 'https://skills.test', + wizardFlags: {}, + wizardFlagPayloads: {}, + wizardMetadata: {}, + }, + input: { + installDir: '/test', + flags: { + ci: false, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + host: {}, + credentials, + project: null, + apiUser: null, + }, + boot: { + programId: 'test', + skillsBaseUrl: 'https://skills.test', + credentials, + wizardFlags: {}, + wizardFlagPayloads: {}, + wizardMetadata: {}, + project: null, + triageProvider: undefined, + }, + emit: vi.fn(), + prompt: 'test', + spinner: { start: vi.fn(), stop: vi.fn(), message: vi.fn() }, + model: 'test', + askBridge, + }; + if (mode === 'linear') { + await anthropicBackend.run(inputs); + } else { + if (!anthropicBackend.runTask) throw new Error('Missing task harness'); + await anthropicBackend.runTask({ + ...inputs, + spinnerMessage: 'Working', + successMessage: 'Done', + additionalFeatureQueue: [], + requestRemark: false, + analyticsProperties: {}, + orchestrator: {} as TaskRunInputs['orchestrator'], + }); + } + const call = vi.mocked(initializeAgent).mock.calls.at(-1); + if (!call) throw new Error('Harness did not initialize the agent'); + const [config] = call; + return (behavior: 'allow' | 'deny') => { + for (const tool of ['Write', 'Edit']) { + expect( + wizardCanUseTool( + tool, + { file_path: 'src/app.ts', content: 'changed' }, + { wizardAskPending: config.getPendingQuestion?.() != null }, + ).behavior, + ).toBe(behavior); + } + }; +} + +afterEach(() => { + vi.useRealTimers(); + vi.clearAllMocks(); +}); + +describe.each(['linear', 'task'] as const)( + 'Anthropic %s pending-question permission guard', + (mode) => { + it.each(['answer', 'cancel', 'timeout', 'reject'] as const)( + 'blocks simultaneous Write/Edit and releases them after %s', + async (ending) => { + vi.useFakeTimers(); + let resolve!: (answers: AskAnswers) => void; + let reject!: (error: Error) => void; + let signal: AbortSignal | undefined; + const bridge = createAskBridge( + { + ask: (_question, context) => { + expectPermission('deny'); + signal = context.signal; + return new Promise((res, rej) => { + resolve = res; + reject = rej; + }); + }, + }, + { getSource: () => 'test', richLinks: false, timeoutMs: 1000 }, + ); + if (!bridge) throw new Error('Missing ask bridge'); + const expectPermission = await initializeHarness(mode, bridge); + expectPermission('allow'); + const pending = bridge.request({ questions }); + expectPermission('deny'); + + if (ending === 'reject') { + const rejected = expect(pending).rejects.toThrow('answerer failed'); + reject(new Error('answerer failed')); + await rejected; + } else { + if (ending === 'timeout') { + await vi.advanceTimersByTimeAsync(1000); + } else { + resolve({ q: ending === 'answer' ? 'yes' : '__cancelled__' }); + } + await expect(pending).resolves.toEqual({ + answers: { q: ending === 'answer' ? 'yes' : '__cancelled__' }, + timedOut: ending === 'timeout', + }); + } + + expectPermission('allow'); + expect(signal?.aborted).toBe(ending === 'timeout'); + expect(vi.getTimerCount()).toBe(0); + }, + ); + + it('keeps the guard when a concurrent question is rejected', async () => { + let resolve!: (answers: AskAnswers) => void; + const ask = vi + .fn() + .mockImplementationOnce( + () => new Promise((res) => (resolve = res)), + ) + .mockRejectedValueOnce(new Error('another request is pending')); + const bridge = createAskBridge( + { ask }, + { getSource: () => 'test', richLinks: false }, + ); + if (!bridge) throw new Error('Missing ask bridge'); + const expectPermission = await initializeHarness(mode, bridge); + const first = bridge.request({ questions }); + await expect(bridge.request({ questions })).rejects.toThrow( + 'another request is pending', + ); + expectPermission('deny'); + resolve({ q: 'yes' }); + await first; + expectPermission('allow'); + }); + + it('allows file mutations without an answerer', async () => { + const expectPermission = await initializeHarness(mode, undefined); + expectPermission('allow'); + }); + }, +); diff --git a/src/lib/agent/runner/harness/anthropic/index.ts b/src/lib/agent/runner/harness/anthropic/index.ts index 4f5e449ec..660c5808e 100644 --- a/src/lib/agent/runner/harness/anthropic/index.ts +++ b/src/lib/agent/runner/harness/anthropic/index.ts @@ -1,6 +1,5 @@ // Supported legacy SDK fallback; both this adapter and Pi implement run and runTask. -import { getUI } from '@ui'; import { Harness } from '@lib/constants'; import { initializeAgent, @@ -9,7 +8,8 @@ import { import { createAioCapture } from '@lib/agent/aio-capture'; import { getLogFilePath, logToFile } from '@utils/debug'; import { detectNodePackageManagers } from '@lib/detection/package-manager'; -import { sessionToOptions } from '@lib/agent/runner/shared/bootstrap'; +import { runOptions } from '@lib/agent/runner/shared/bootstrap'; +import { createEmitLog } from '@lib/agent/runner/shared/progress-collector'; import type { AgentResult, AgentHarness, @@ -22,30 +22,32 @@ export const anthropicBackend: AgentHarness = { async run(inputs: BackendRunInputs): Promise { const { - session, - config, - programConfig, + config: runConfig, + input, boot, + emit, prompt, spinner, askBridge, middleware, model, } = inputs; + const config = runConfig.run; const { skillsBaseUrl, credentials, wizardFlags, wizardMetadata } = boot; const { accessToken, host, projectApiKey } = credentials; + const log = createEmitLog(emit); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey, apiHost: host.apiHost, runTags: wizardMetadata, }); - getUI().log.step('Initializing Claude agent...'); + log.step('Initializing Claude agent...'); const agent = await initializeAgent( { - workingDirectory: session.installDir, + workingDirectory: input.installDir, posthogMcpUrl: host.mcpUrl, posthogApiKey: accessToken, host, @@ -58,23 +60,24 @@ export const anthropicBackend: AgentHarness = { programId: boot.programId, integrationLabel: config.integrationLabel, askBridge, + getPendingQuestion: askBridge?.getPendingQuestion, askMaxQuestions: config.maxQuestions, - allowedTools: programConfig.allowedTools, - disallowedTools: programConfig.disallowedTools, - getPendingQuestion: () => session.pendingQuestion, + allowedTools: runConfig.allowedTools, + disallowedTools: runConfig.disallowedTools, modelOverride: model, capture, + emit, }, - sessionToOptions(session), + runOptions(input), ); - getUI().log.step(`Verbose logs: ${getLogFilePath()}`); - getUI().log.success("Agent initialized. Let's get cooking!"); + log.step(`Verbose logs: ${getLogFilePath()}`); + log.success("Agent initialized. Let's get cooking!"); logToFile('[agent-runner] agent initialized'); return executeAgent( agent, prompt, - sessionToOptions(session), + runOptions(input), spinner, { estimatedDurationMinutes: config.estimatedDurationMinutes, @@ -94,9 +97,10 @@ export const anthropicBackend: AgentHarness = { async runTask(inputs: TaskRunInputs): Promise { const { - session, - programConfig, + config, + input, boot, + emit, prompt, spinner, model, @@ -111,10 +115,10 @@ export const anthropicBackend: AgentHarness = { requestRemark, analyticsProperties, } = inputs; - const options = sessionToOptions(session); + const options = runOptions(input); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, @@ -125,7 +129,7 @@ export const anthropicBackend: AgentHarness = { // enqueue_task attribute to the right agent when tasks run in parallel. const agent = await initializeAgent( { - workingDirectory: session.installDir, + workingDirectory: input.installDir, posthogMcpUrl: boot.credentials.host.mcpUrl, posthogApiKey: boot.credentials.accessToken, host: boot.credentials.host, @@ -134,12 +138,14 @@ export const anthropicBackend: AgentHarness = { programId: boot.programId, wizardFlags: boot.wizardFlags, wizardMetadata: boot.wizardMetadata, - integrationLabel: programConfig.id, + integrationLabel: config.programId, // Only a task allowed to ask carries a bridge, so the Write/Edit pause // that rides on a pending question stays inside that task's agent. askBridge, + getPendingQuestion: askBridge?.getPendingQuestion, orchestrator, capture, + emit, }, options, ); diff --git a/src/lib/agent/runner/harness/pi/index.ts b/src/lib/agent/runner/harness/pi/index.ts index 092e59718..82c705615 100644 --- a/src/lib/agent/runner/harness/pi/index.ts +++ b/src/lib/agent/runner/harness/pi/index.ts @@ -14,7 +14,6 @@ import fs from 'fs'; import path from 'path'; -import { getUI } from '@ui'; import { getLogFilePath, logToFile } from '@utils/debug'; import { Harness, @@ -41,6 +40,8 @@ import type { TaskRunInputs, } from '../types'; import type { BootstrapResult } from '@lib/agent/runner/shared/types'; +import type { ProgressEmitter } from '@lib/agent/progress'; +import { createEmitLog } from '@lib/agent/runner/shared/progress-collector'; import type { TaskStore } from './tasks'; import { completionFailure, runErrorType } from './completion'; @@ -147,10 +148,19 @@ export function extractText(message: unknown): string { * the MCP creates them) into the outro link, mirroring the anthropic path's * signal parsing (#9). The marker carries the URL the MCP returned. */ -export function applyOutroMarkers(textBlock: string): void { +export function applyOutroMarkers( + textBlock: string, + emit: ProgressEmitter, +): void { const markers: Array<[string, (url: string) => void]> = [ - [AgentSignals.DASHBOARD_URL, (url) => getUI().setDashboardUrl(url)], - [AgentSignals.NOTEBOOK_URL, (url) => getUI().setNotebookUrl(url)], + [ + AgentSignals.DASHBOARD_URL, + (url) => emit({ kind: 'url', which: 'dashboard', url }), + ], + [ + AgentSignals.NOTEBOOK_URL, + (url) => emit({ kind: 'url', which: 'notebook', url }), + ], ]; for (const [marker, apply] of markers) { const idx = textBlock.indexOf(marker); @@ -195,20 +205,22 @@ export const piBackend: AgentHarness = { name: Harness.pi, async run(inputs: BackendRunInputs): Promise { - const { session, boot, prompt, spinner, config, programConfig } = inputs; + const { config: runConfig, input, boot, emit, prompt, spinner } = inputs; + const config = runConfig.run; const modelId = inputs.model; + const log = createEmitLog(emit); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, }); // Init banner (parity #5). - getUI().log.step('Initializing Wizard agent...'); - getUI().log.step(`Verbose logs: ${getLogFilePath()}`); - getUI().log.success("Agent initialized. Let's get cooking!"); + log.step('Initializing Wizard agent...'); + log.step(`Verbose logs: ${getLogFilePath()}`); + log.success("Agent initialized. Let's get cooking!"); spinner.start(config.spinnerMessage ?? 'Customizing your PostHog setup...'); @@ -306,11 +318,11 @@ export const piBackend: AgentHarness = { const { createSecurityExtension } = await import('./security'); const security = createSecurityExtension({ - disallowedTools: programConfig.disallowedTools, + disallowedTools: runConfig.disallowedTools, getWizardAskPending: () => askState.pending, triageProvider: boot.triageProvider, // Where pi's bash runs; the rm allowance is confined to this tree. - workingDirectory: session.installDir, + workingDirectory: input.installDir, }); // Pay warlock's WASM-init + rule-compile cost now, off the tool-call @@ -360,11 +372,11 @@ export const piBackend: AgentHarness = { } const resourceLoader = new DefaultResourceLoader({ - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), systemPrompt: assembleCommandments({ - program: programConfig.id, + program: runConfig.programId, sequence: Sequence.linear, harness: Harness.pi, caps: { bash: true, posthogMcp }, @@ -390,12 +402,14 @@ export const piBackend: AgentHarness = { const { createWizardPiTaskTools } = await import('./tasks'); const { createDispatchAgentTool } = await import('./subagent'); // Created once so the run loop can read the store for the completion guard. - const wizardTaskTools = createWizardPiTaskTools(); + const wizardTaskTools = createWizardPiTaskTools((tasks) => + emit({ kind: 'tasks', tasks }), + ); // The one bash the agent (and its subagents) may use: every subprocess it // spawns gets a scrubbed env, so no secret or ambient variable reaches an // `npm install`. Shared with the subagent so the lockdown is inherited. const scrubbedBash = withMode( - createBashToolDefinition(session.installDir, { + createBashToolDefinition(input.installDir, { spawnHook: (ctx) => ({ ...ctx, env: buildScrubbedEnv() }), }), 'sequential', @@ -406,18 +420,18 @@ export const piBackend: AgentHarness = { // defaults so we can supply the env-scrubbed bash above; read/edit/write // are the stock definitions. Reads run in parallel so a batched turn of // independent reads executes at once; edit/write/bash stay sequential. - withMode(createReadToolDefinition(session.installDir), 'parallel'), - withMode(createEditToolDefinition(session.installDir), 'sequential'), - withMode(createWriteToolDefinition(session.installDir), 'sequential'), + withMode(createReadToolDefinition(input.installDir), 'parallel'), + withMode(createEditToolDefinition(input.installDir), 'sequential'), + withMode(createWriteToolDefinition(input.installDir), 'sequential'), scrubbedBash, // Native ls/find/grep so the agent explores with proper tools instead // of fence-blocked `bash {ls/find}` (the profiled retry-spirals came // from this gap). Parallel — exploration batches cleanly. - withMode(createLsToolDefinition(session.installDir), 'parallel'), - withMode(createFindToolDefinition(session.installDir), 'parallel'), - withMode(createGrepToolDefinition(session.installDir), 'parallel'), + withMode(createLsToolDefinition(input.installDir), 'parallel'), + withMode(createFindToolDefinition(input.installDir), 'parallel'), + withMode(createGrepToolDefinition(input.installDir), 'parallel'), ...createWizardPiTools({ - workingDirectory: session.installDir, + workingDirectory: input.installDir, skillsBaseUrl: boot.skillsBaseUrl, triageProvider: boot.triageProvider, detectPackageManager: config.detectPackageManager, @@ -431,7 +445,7 @@ export const piBackend: AgentHarness = { }, // Skip wizard_ask when the program disallows it (bare pi tool names // don't match the MCP-prefixed disallow list at the security gate). - disallowedTools: programConfig.disallowedTools, + disallowedTools: runConfig.disallowedTools, }), // Task/todo tools (#526): render the todo list live in the TUI, parity // with the anthropic path. @@ -442,7 +456,7 @@ export const piBackend: AgentHarness = { createDispatchAgentTool({ model, modelRegistry: registry, - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), securityFactory: security.factory as (pi: unknown) => void, bashTool: scrubbedBash, @@ -456,8 +470,8 @@ export const piBackend: AgentHarness = { // Reasoning effort from the switchboard capability matrix (undefined = // pi's default). Sent as `reasoning_effort` for openai-completions. thinkingLevel: caps.thinkingLevel, - cwd: session.installDir, - sessionManager: SessionManager.inMemory(session.installDir), + cwd: input.installDir, + sessionManager: SessionManager.inMemory(input.installDir), resourceLoader, // Disable the default built-in tools; `customTools` re-registers // read/edit/write + an env-scrubbed bash, so no subprocess inherits the @@ -508,12 +522,12 @@ export const piBackend: AgentHarness = { const assistant = extractText(event.message).trim(); if (assistant) { logToFile(`[pi] assistant: ${assistant.slice(0, 1000)}`); - applyOutroMarkers(assistant); + applyOutroMarkers(assistant, emit); // Surface [STATUS] lines into the live spinner + status history, // mirroring the anthropic path — pi otherwise drops them. const statusText = lastStatusLine(assistant); if (statusText) { - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } for (const line of assistant.split('\n')) signals.push(line); @@ -644,7 +658,7 @@ export const piBackend: AgentHarness = { // on completion; pi's `rm` is fence-blocked, so the agent can't — clean it // up host-side rather than leave a stale (often empty) artifact (#15). try { - const planFile = path.join(session.installDir, '.posthog-events.json'); + const planFile = path.join(input.installDir, '.posthog-events.json'); if (fs.existsSync(planFile)) await fs.promises.rm(planFile); } catch (err) { logToFile(`[pi] .posthog-events.json cleanup skipped: ${String(err)}`); @@ -668,7 +682,7 @@ export const piBackend: AgentHarness = { const message = err instanceof Error ? err.message : String(err); logToFile(`[pi] run error: ${message}`); spinner.stop(config.errorMessage ?? `${config.integrationLabel} failed`); - getUI().log.error(`pi backend error: ${message}`); + log.error(`pi backend error: ${message}`); const error = runErrorType(message); captureAborted(error); return { error, message }; diff --git a/src/lib/agent/runner/harness/pi/task.ts b/src/lib/agent/runner/harness/pi/task.ts index 37deec62d..9d7adc97c 100644 --- a/src/lib/agent/runner/harness/pi/task.ts +++ b/src/lib/agent/runner/harness/pi/task.ts @@ -16,7 +16,6 @@ * Loaded lazily from `index.ts` (typebox/ESM constraint, same as tools.ts). */ -import { getUI } from '@ui'; import { logToFile } from '@utils/debug'; import { analytics } from '@utils/analytics'; import { @@ -166,9 +165,10 @@ function isSettled(ctx: OrchestratorToolsContext): boolean { export async function runPiTask(inputs: TaskRunInputs): Promise { const { - session, - programConfig, + config, + input, boot, + emit, prompt, spinner, model: modelId, @@ -187,7 +187,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { if (spinnerMessage) spinner.start(spinnerMessage); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, @@ -319,10 +319,10 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { const orchestratorTools = allowedOrchestratorTools(disallowedTools); const resourceLoader = new DefaultResourceLoader({ - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), systemPrompt: assembleCommandments({ - program: programConfig.id, + program: config.programId, sequence: Sequence.orchestrator, harness: Harness.pi, caps: { bash: codingTools.has('bash'), posthogMcp }, @@ -339,7 +339,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { // The task's coding tools, gated by its allow list. Reads and searches run // in parallel; mutating tools stay sequential. Bash subprocesses get the // scrubbed env, same as the linear run. - const dir = session.installDir; + const dir = input.installDir; const codingToolFactories = { read: () => withMode(createReadToolDefinition(dir), 'parallel'), edit: () => withMode(createEditToolDefinition(dir), 'sequential'), @@ -436,10 +436,10 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { const assistant = extractText(event.message).trim(); if (assistant) { logToFile(`[pi-task] assistant: ${assistant.slice(0, 1000)}`); - applyOutroMarkers(assistant); + applyOutroMarkers(assistant, emit); const statusText = lastStatusLine(assistant); if (statusText) { - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } for (const line of assistant.split('\n')) signals.push(line); diff --git a/src/lib/agent/runner/harness/pi/tasks.ts b/src/lib/agent/runner/harness/pi/tasks.ts index e12f66e1e..0ee0f4462 100644 --- a/src/lib/agent/runner/harness/pi/tasks.ts +++ b/src/lib/agent/runner/harness/pi/tasks.ts @@ -1,15 +1,15 @@ /** * Task/todo parity for pi (#526). The same four Task tools the anthropic path * exposes (TaskCreate/Update/Get/List), as pi `defineTool` tools backed by a - * shared in-memory store. Every mutation pushes the list to the TUI via - * `getUI().syncTodos`, so the todo panel updates live under pi exactly like the - * anthropic path — the thing that was missing before. + * shared in-memory store. Every mutation reports the list through `onSync` + * (a `tasks` progress event), so the todo panel updates live under pi exactly + * like the anthropic path — the thing that was missing before. */ import { Type } from 'typebox'; import { defineTool } from '@earendil-works/pi-coding-agent'; import type { ToolDefinition } from '@earendil-works/pi-coding-agent'; -import { getUI } from '@ui'; +import type { TaskSnapshot } from '@lib/agent/progress'; export type TaskStatus = 'pending' | 'in_progress' | 'completed'; export interface TaskEntry { @@ -26,22 +26,23 @@ function text(s: string): { return { content: [{ type: 'text', text: s }], details: {} }; } -function syncToTui(store: TaskStore): void { - getUI().syncTodos( - Array.from(store.values()).map((t) => ({ - content: t.content, - status: t.status, - activeForm: t.activeForm, - })), - ); +function snapshot(store: TaskStore): TaskSnapshot[] { + return Array.from(store.values()).map((t) => ({ + content: t.content, + status: t.status, + activeForm: t.activeForm, + })); } -/** Build the four Task tools over a fresh store. */ -export function createWizardPiTaskTools(): { +/** Build the four Task tools over a fresh store. `onSync` gets the list after every mutation. */ +export function createWizardPiTaskTools( + onSync: (tasks: TaskSnapshot[]) => void = () => undefined, +): { tools: ToolDefinition[]; store: TaskStore; } { const store: TaskStore = new Map(); + const syncToTui = (): void => onSync(snapshot(store)); const taskCreate = defineTool({ name: 'TaskCreate', @@ -64,7 +65,7 @@ export function createWizardPiTaskTools(): { status: 'pending', activeForm: args.activeForm, }); - syncToTui(store); + syncToTui(); return text(`Created ${id}`); }, }); @@ -97,7 +98,7 @@ export function createWizardPiTaskTools(): { status: (args.status as TaskStatus) ?? existing.status, activeForm: args.activeForm ?? existing.activeForm, }); - syncToTui(store); + syncToTui(); return text(`Updated ${args.taskId}`); }, }); diff --git a/src/lib/agent/runner/harness/types.ts b/src/lib/agent/runner/harness/types.ts index 67c59e5c9..35c082c21 100644 --- a/src/lib/agent/runner/harness/types.ts +++ b/src/lib/agent/runner/harness/types.ts @@ -13,23 +13,27 @@ * per drained task. A harness without orchestrator support omits the method; * `orchestrator-runner.ts` checks for it at the call site and fails loudly * rather than silently downgrading. + * + * A harness reports through `emit` and never reaches for a UI. It returns an + * error classification, or a decided `failure` the sequence must return as + * the run's result. */ -import type { WizardSession } from '@lib/wizard-session'; import type { AdditionalFeature } from '@lib/wizard-session'; import type { Harness } from '@lib/constants'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { SpinnerHandle } from '@ui'; import type { WizardAskBridge } from '@lib/wizard-ask-bridge'; import type { AgentErrorType } from '@lib/agent/agent-interface'; +import type { ProgressEmitter, SpinnerHandle } from '@lib/agent/progress'; import type { OrchestratorToolsContext } from '@lib/agent/runner/sequence/orchestrator/queue-tools'; import type { EffortLevel, ThinkingLevel, } from '@lib/agent/runner/switchboard/models'; import type { - ProgramRun, + AgentFailure, BootstrapResult, + RunConfig, + RunInput, } from '@lib/agent/runner/shared/types'; /** The benchmark/telemetry hook threaded through a run, if enabled. */ @@ -40,14 +44,14 @@ export interface RunMiddleware { /** * Everything a runner needs to run one program. Assembled by `linear.ts` from - * the bootstrap result and the program config; the runner consumes it and never + * the prepared run and the run config; the runner consumes it and never * re-derives run context. */ export interface BackendRunInputs { - session: WizardSession; - config: ProgramRun; - programConfig: ProgramConfig; + config: RunConfig; + input: RunInput; boot: BootstrapResult; + emit: ProgressEmitter; /** The fully assembled prompt. */ prompt: string; /** Installed framework-skill path, when the program installs one. */ @@ -56,7 +60,7 @@ export interface BackendRunInputs { spinner: SpinnerHandle; /** Interactive question bridge; undefined in CI/headless (ask disabled). */ askBridge?: WizardAskBridge; - /** Benchmark middleware, when `session.benchmark` is set. */ + /** Benchmark middleware, when `--benchmark` is set. */ middleware?: RunMiddleware; /** Gateway model id resolved from the (runner, model) pair. */ model: string; @@ -64,8 +68,16 @@ export interface BackendRunInputs { thinkingLevel?: EffortLevel; } -/** What a runner reports back: an error classification, or nothing on success. */ -export type AgentResult = { error?: AgentErrorType; message?: string }; +/** + * What a runner reports back: an error classification, or nothing on success. + * `failure` is a fully decided abort the harness already reported to the host + * (the 401 auth screen); the sequence returns it as the run's result. + */ +export type AgentResult = { + error?: AgentErrorType; + message?: string; + failure?: AgentFailure; +}; /** * One orchestrator-mode unit of work — the seed plan, or one drained task. @@ -75,9 +87,10 @@ export type AgentResult = { error?: AgentErrorType; message?: string }; * them from the program-level config the linear pipeline assembles once. */ export interface TaskRunInputs { - session: WizardSession; - programConfig: ProgramConfig; + config: RunConfig; + input: RunInput; boot: BootstrapResult; + emit: ProgressEmitter; /** The fully assembled per-task or seed prompt. */ prompt: string; spinner: SpinnerHandle; diff --git a/src/lib/agent/runner/index.ts b/src/lib/agent/runner/index.ts index f11bf73c8..3bac11087 100644 --- a/src/lib/agent/runner/index.ts +++ b/src/lib/agent/runner/index.ts @@ -1,214 +1,130 @@ /** * Unified program runner — dispatcher. * - * Single configurable pipeline for all programs. Each program - * provides a ProgramRun (via the `run` field on ProgramConfig) - * that controls: - * - Whether a skill is pre-installed or discovered at runtime - * - How the agent prompt is built - * - What MCP servers and package manager detector to use - * - What happens after the agent completes + * One callable, `runAgent(config, input, options)`, runs a program's agent + * pipeline to a decided result. `config` is resolved execution data (which + * program, how it is routed, which flags apply); `input` is the invocation + * snapshot (directory, credentials, project); `options` carries an optional + * progress observer, an optional answerer. * - * The pipeline runs a shared bootstrap (logging, health check, settings, OAuth, - * flags, MCP url), then forks. The `orchestrator` variant routes to the - * experimental task-queue runner. Every other variant runs the fixed linear - * pipeline: + * The pipeline prepares the run (logging targets, gateway mint, scan triage), + * then forks. The `orchestrator` variant routes to the task-queue runner. + * Every other variant runs the fixed linear pipeline: * [skill install] → agent init → prompt → run → errors → [postRun] → outro + * + * The agent reports and asks, it never renders, never reads a session, never + * exits the process, never sends the process's terminal analytics and never + * rejects. A decided failure comes back in + * `RunResult.failure` with the same fields `wizardAbort` takes; an error the + * agent did not decide (a refused mint, an SDK crash) comes back as + * `outcome: RunOutcome.Crashed` with the original error attached, so a caller can keep + * handling it the way it always did. The legacy adapter in + * `src/lib/programs/run-agent-legacy.ts` rebuilds today's session-driven + * behavior on top of this call for every existing caller. */ -import type { WizardSession } from '../../wizard-session'; -import { analytics } from '@utils/analytics'; -import { - Sequence, - WIZARD_ORCHESTRATOR_FLAG_KEY, - WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY, -} from '@lib/constants'; +import { Sequence } from '@lib/constants'; +import { RunOutcome } from './shared/types'; +export { RunOutcome } from './shared/types'; +import { classifyRunFailure } from '@lib/errors'; import { logToFile } from '@utils/debug'; -import { getUI } from '../../../ui'; -import type { ProgramConfig } from '../../programs/program-step'; -import type { ProgramRun, BootstrapResult } from './shared/types'; -import { bootstrapProgram } from './shared/bootstrap'; -import { - getSequence, - resolveBinding, - type ProgramBinding, - type SwitchboardCtx, -} from './switchboard'; +import type { + RunAgentOptions, + RunConfig, + RunInput, + RunResult, +} from './shared/types'; +import { prepareRun } from './shared/bootstrap'; +import { createProgressCollector } from './shared/progress-collector'; +import { getSequence } from './switchboard'; import { flushScanReport } from '../../yara-hooks'; -import { startAuditLedgerWatcher } from '../../programs/audit/ledger-watcher'; -import { registerCleanup } from '../../../utils/wizard-abort'; export type { - ProgramRun, - BootstrapResult, AbortCase, - PromptContext, + AgentFailure, + BootstrapResult, Credentials, + ProgramRun, + PromptContext, + ResolvedBinding, + RunAgentOptions, + RunConfig, + RunFlags, + RunHooks, + RunInput, + RunResult, + RunSnapshot, + SeedTaskEntry, } from './shared/types'; +export type { + AgentInteraction, + AgentProgress, + ProgressEmitter, +} from '@lib/agent/progress'; export { shouldDisableAsk } from './shared/bootstrap'; - -/** - * Resolve a ProgramConfig's agent run definition and execute the pipeline. - * Entry point for bin.ts — handles buildRunConfig, bootstrap, and (future) run field. - */ -export async function runAgent( - programConfig: ProgramConfig, - session: WizardSession, - options: { composed?: boolean } = {}, -): Promise { - if (!programConfig.run) { - throw new Error(`Program "${programConfig.id}" has no run configuration.`); - } - - // Before `run()` resolves: an audit seeds the ledger from inside its recipe, - // and a watcher started later would ignore that write as pre-existing. - const ledger = programConfig.auditLedgerFile - ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) - : null; - if (ledger) registerCleanup(() => ledger.stop()); - - try { - const runDef = - typeof programConfig.run === 'function' - ? await programConfig.run(session) - : programConfig.run; - - await runProgram(session, runDef, programConfig, options); - } finally { - ledger?.stop(); - } -} +export { resolveBinding } from './switchboard'; +export type { ProgramBinding, SwitchboardCtx } from './switchboard'; +export { TASK_OUTCOMES_KEY } from './sequence/orchestrator/queue'; /** * Run a program's agent pipeline. * - * Bootstrap → bind the program via the switchboard (resolve which sequence - * and harness will run it, tag both axes) → dispatch to the resolved - * sequence's runner. + * Prepare → dispatch to the sequence the binding names → return its result + * with the agent's own snapshot of what it reported. Missing observers change + * nothing; a throwing observer is logged and the run continues. */ -export async function runProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - options: { composed?: boolean } = {}, -): Promise { - const boot = await bootstrapProgram(session, config, programConfig); +export async function runAgent( + config: RunConfig, + input: RunInput, + options: RunAgentOptions = {}, +): Promise { + const collector = createProgressCollector(options.onProgress); + const { emit } = collector; + const log = (message: string) => + emit({ kind: 'log', level: 'info', message }); // Flush the warlock scan report once, at this single seam, on every - // termination path and for every harness (linear, orchestrator, or future): - // - registerCleanup covers the abort/cancel path (wizardAbort runs the - // registered cleanups; the success path never calls them) - // - the finally covers normal completion and any direct throw that unwinds - // through here - // flushScanReport is idempotent (it zeroes scan state), so the overlap is a - // harmless no-op. No harness has to know reporting exists. - registerCleanup(() => flushScanReport(session)); + // termination path and for every harness (linear, orchestrator, or future). + // flushScanReport is idempotent (it zeroes scan state), so a caller that also + // flushes from its own cleanup path sees a harmless no-op. No harness has to + // know reporting exists. try { - const binding = resolveProgramRunner( - session, - programConfig, - boot, - options.composed ?? false, - ); - if (binding.sequence === Sequence.orchestrator) { - getUI().log.info('Task-queue orchestrator enabled.'); + const boot = await prepareRun(config, input); + if (config.binding.sequence === Sequence.orchestrator) { + log('Task-queue orchestrator enabled.'); } - return await getSequence(binding.sequence).run( - session, + logToFile( + `[agent-runner] run program=${config.programId} sequence=${config.binding.sequence}` + + ` harness=${config.binding.harness} composed=${config.composed}`, + ); + const result = await getSequence(config.binding.sequence).run({ config, - programConfig, + input, boot, - options.composed ?? false, - ); + emit, + interaction: options.interaction, + }); + return { + ...result, + skillId: input.skillId, + snapshot: collector.snapshot(), + }; + } catch (error) { + // Not a decision the agent made. Hand it back whole rather than throw, so + // every ending of a run is a result the caller reads the same way. + const failure = classifyRunFailure(error); + logToFile('[agent-runner] run crashed:', error); + return { + outcome: RunOutcome.Crashed, + skillId: input.skillId, + failure: { + code: failure.code, + message: failure.message, + error: error instanceof Error ? error : new Error(String(error)), + }, + snapshot: collector.snapshot(), + }; } finally { - flushScanReport(session); + flushScanReport({ yaraReport: input.flags.yaraReport }); } } - -/** - * Resolve which sequence and harness will run a program (CLI → PostHog flag → - * per-program binding → default), tag both axes onto analytics, and return the - * binding for downstream dispatch. - * - * The one place `runner/index.ts` reaches into the switchboard — every other - * concern (bootstrap, cleanup, dispatch, per-task per-role harness picks) is - * either upstream or downstream of this call. - */ -function resolveProgramRunner( - session: WizardSession, - programConfig: ProgramConfig, - boot: BootstrapResult, - composed: boolean, -): ProgramBinding { - const ctx = { - program: programConfig.id, - composed, - flags: boot.wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliSequence: session.sequence, - cliModel: session.model, - }; - const binding = resolveBinding(ctx); - tagBinding(boot, binding); - captureSwitchboardDecision(ctx, binding); - return binding; -} - -/** - * One event + one log line per run: what entered the switchboard, which - * precedence rung decided each axis, and the final pick. - */ -function captureSwitchboardDecision( - ctx: SwitchboardCtx, - binding: ProgramBinding, -): void { - const trace = ctx.trace ?? {}; - // Unpinned orchestrator runs choose a model per task from the context-mill agent prompts; the orchestrator logs that map once the prompts load. - const perTaskModel = - binding.sequence === Sequence.orchestrator && trace.model === 'binding'; - const model = perTaskModel ? 'chosen-per-task' : binding.model; - const modelSource = perTaskModel ? 'agent-prompts' : trace.model; - analytics.wizardCapture('switchboard resolved', { - program: ctx.program, - flag_self_driving_use_pi_harness: - ctx.flags[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY], - flag_self_driving_pi_payload: JSON.stringify( - ctx.flagPayloads?.[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY] ?? null, - ), - flag_orchestrator: ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY], - cli_harness: ctx.cliHarness, - cli_sequence: ctx.cliSequence, - cli_model: ctx.cliModel, - harness_source: trace.harness, - model_source: modelSource, - sequence_source: trace.sequence, - harness: binding.harness, - model, - thinking_level: binding.thinkingLevel, - sequence: binding.sequence, - }); - logToFile( - `[switchboard] decision: program=${ctx.program}` + - ` in(orchestrator=${ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY] ?? '-'},` + - ` cli=${ctx.cliHarness ?? '-'}/${ctx.cliSequence ?? '-'}/${ - ctx.cliModel ?? '-' - })` + - ` → harness=${binding.harness} (${trace.harness ?? '?'})` + - ` model=${model} (${modelSource ?? '?'})` + - ` sequence=${binding.sequence} (${trace.sequence ?? '?'})`, - ); -} - -/** - * Tag the run with its two routing axes. Sequence is stable for the whole - * run; harness reflects the run-level (default-role) resolution — orchestrator - * per-task calls emit their own `harness` property in their events so per-task - * aggregations attribute correctly. - */ -function tagBinding(boot: BootstrapResult, binding: ProgramBinding): void { - analytics.setTag('sequence', binding.sequence); - analytics.setTag('harness', binding.harness); - boot.wizardMetadata.SEQUENCE = binding.sequence; - boot.wizardMetadata.HARNESS = binding.harness; -} diff --git a/src/lib/agent/runner/sequence/linear.ts b/src/lib/agent/runner/sequence/linear.ts index 94ab86841..c452a4fa8 100644 --- a/src/lib/agent/runner/sequence/linear.ts +++ b/src/lib/agent/runner/sequence/linear.ts @@ -1,112 +1,92 @@ /** * The linear pipeline. Single execution path for all non-orchestrator programs, * both skill-based (revenue analytics) and framework-based (core integration). - * The `ProgramRun` controls what varies between them; `programConfig` carries the - * program-level static metadata (tool allow/disallow lists, etc.). + * The `ProgramRun` controls what varies between them; `RunConfig` + * carries the program-level static metadata (tool allow/disallow lists, etc.). + * + * Reports through `emit`, asks through `interaction`, and returns a decided + * `RunResult`. Every former `getUI()` call is one progress event in the same + * place; every former `wizardAbort` is a returned failure with the same + * arguments, so the caller's exit sequence is unchanged. */ -import type { WizardSession } from '../../../wizard-session'; -import { OutroKind } from '../../../wizard-session'; -import { getUI } from '../../../../ui'; +import { OutroKind, type OutroData } from '@lib/wizard-session'; import { AgentErrorType, AgentSignals } from '../../agent-interface'; -import { restoreClaudeSettings } from '../../claude-settings'; import { logToFile } from '../../../../utils/debug'; import { createBenchmarkPipeline } from '../../../middleware/benchmark'; -import { - wizardAbort, - WizardError, - registerCleanup, -} from '../../../../utils/wizard-abort'; +import { WizardError } from '../../../../utils/wizard-abort'; import { ErrorCodes, AGENT_ERROR_CODE } from '@lib/errors'; import { analytics } from '../../../../utils/analytics'; -import { - formatScanReport, - formatYaraAbortMessage, - writeScanReport, -} from '../../../yara-hooks'; +import { formatYaraAbortMessage } from '../../../yara-hooks'; import { installSkillById } from '../../../wizard-tools'; -import { createWizardAskBridge } from '../../../wizard-ask-bridge'; -import type { ProgramConfig } from '../../../programs/program-step'; import { assemblePrompt } from '../../agent-prompt'; -import type { ProgramRun, BootstrapResult } from '../shared/types'; -import { abortOnInstallFailure } from '../shared/errors'; -import { shouldDisableAsk, sessionToOptions } from '../shared/bootstrap'; -import { resolveHarness, getHarness } from '../switchboard'; +import type { SequenceResult, SequenceContext } from '../shared/types'; +import { failed, installFailure } from '../shared/errors'; +import { RunOutcome } from '../shared/types'; +import { shouldDisableAsk, runOptions } from '../shared/bootstrap'; +import { createEmitSpinner } from '../shared/progress-collector'; +import { createAskBridge } from '../shared/ask'; +import { getHarness } from '../switchboard'; -export async function runLinearProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, - composed = false, -): Promise { - const { skillsBaseUrl, credentials, wizardFlags, project } = boot; +export async function runLinearProgram({ + config, + input, + boot, + emit, + interaction, +}: SequenceContext): Promise { + const { run, composed } = config; + const { skillsBaseUrl, credentials, project } = boot; const { projectApiKey, host, projectId } = credentials; // 5. Skill install (if skillId provided) let skillPath: string | undefined; - if (config.skillId) { - logToFile(`[agent-runner] installing skill ${config.skillId}`); + if (run.skillId) { + logToFile(`[agent-runner] installing skill ${run.skillId}`); const installResult = await installSkillById( - config.skillId, - session.installDir, + run.skillId, + input.installDir, skillsBaseUrl, { triage: boot.triageProvider }, ); if (installResult.kind !== 'ok') { - await abortOnInstallFailure(config.integrationLabel, installResult); - return; + return failed(installFailure(run.integrationLabel, installResult)); } skillPath = installResult.path; logToFile(`[agent-runner] skill installed at ${skillPath}`); } // 6. Initialize agent - const spinner = getUI().spinner(); + const spinner = createEmitSpinner(emit); - const restoreSettings = () => restoreClaudeSettings(session.installDir); - getUI().onEnterScreen('outro', restoreSettings); - - if (session.yaraReport) { - registerCleanup(() => { - const reportPath = writeScanReport(); - if (reportPath) { - const summary = formatScanReport(); - getUI().log.info(`YARA scan report: ${reportPath}${summary ?? ''}`); - } - }); - } - - getUI().startRun(); + emit({ kind: 'lifecycle', phase: 'started' }); // wizard_ask needs an answerer. A human answers at the keyboard; the e2e // snapshot/MCP host answers via its driver and sets WIZARD_ASK_AUTODRIVE. // CI/signup with neither has no answerer, so we omit the bridge and the tool // returns an actionable error rather than hanging on a never-resolving prompt. const askDisabled = - shouldDisableAsk(session) && process.env.WIZARD_ASK_AUTODRIVE !== '1'; - const askBridge = askDisabled + shouldDisableAsk(input.flags) && process.env.WIZARD_ASK_AUTODRIVE !== '1'; + const ask = askDisabled ? undefined - : createWizardAskBridge({ - getSource: () => session.skillId ?? config.integrationLabel, - showQuestion: (q) => getUI().requestQuestion(q), - cancelQuestion: () => getUI().cancelPendingQuestion(), - richLinks: config.richLinks ?? false, - timeoutMs: config.askTimeoutMs, + : createAskBridge(interaction, { + getSource: () => input.skillId ?? run.integrationLabel, + richLinks: run.richLinks ?? false, + timeoutMs: run.askTimeoutMs, }); - const middleware = session.benchmark - ? createBenchmarkPipeline(spinner, sessionToOptions(session)) + const middleware = input.flags.benchmark + ? createBenchmarkPipeline(spinner, runOptions(input)) : undefined; // 7. Build prompt - const prompt = assemblePrompt(config, { + const prompt = assemblePrompt(run, { projectId, projectApiKey, host, skillPath, orgAiDataProcessingApproved: - session.apiUser?.organization?.is_ai_data_processing_approved ?? null, + input.apiUser?.organization?.is_ai_data_processing_approved ?? null, teamProductOptIns: project ? { sessionReplay: project.session_recording_opt_in ?? null, @@ -117,37 +97,34 @@ export async function runLinearProgram( }); logToFile(`[agent-runner] prompt assembled (${prompt.length} chars)`); - // 8. Resolve the (runner, model) pair from the central plan and run the agent - // through the selected runner. The runner owns the agent loop + model - // transport; everything around it (skill install, prompt, ask bridge, error - // routing, outro) stays here so every runner shares it. - const pick = resolveHarness({ - program: programConfig.id, - flags: wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliModel: session.model, - }); - const agentResult = await getHarness(pick.harness).run({ - session, + // 8. Run the agent through the run-level harness. The harness owns the agent + // loop + model transport; everything around it (skill install, prompt, ask + // bridge, error routing, outro) stays here so every harness shares it. + const { harness, model, thinkingLevel } = config.binding; + const agentResult = await getHarness(harness).run({ config, - programConfig, + input, boot, + emit, prompt, skillPath, spinner, - askBridge, + askBridge: ask, middleware, - model: pick.model, - thinkingLevel: pick.thinkingLevel, + model, + thinkingLevel, }); - // 9. Error handling (full set from both runners) + // 9. Error handling (full set from both harnesses) + if (agentResult.failure) { + return failed(agentResult.failure); + } + if (agentResult.error === AgentErrorType.ABORT) { const reason = agentResult.message ?? ''; - const matched = config.abortCases?.find((c) => c.match.test(reason)); + const matched = run.abortCases?.find((c) => c.match.test(reason)); const abortCode = matched?.errorCode ?? ErrorCodes.AgentAbort; - const outroData: WizardSession['outroData'] = matched + const outroData: OutroData = matched ? { kind: OutroKind.Error, message: matched.message, @@ -158,44 +135,47 @@ export async function runLinearProgram( } : { kind: OutroKind.Error, - message: `${config.integrationLabel} aborted`, + message: `${run.integrationLabel} aborted`, body: reason || 'The agent aborted the program.', - docsUrl: config.docsUrl, + docsUrl: run.docsUrl, errorCode: abortCode, errorDetail: { reason }, }; analytics.wizardCapture('agent aborted', { - integration: config.integrationLabel, + integration: run.integrationLabel, reason, matched: matched?.message ?? null, }); - await wizardAbort({ - outroData, - code: abortCode, - error: new WizardError( - `Agent aborted: ${reason}`, - { - integration: config.integrationLabel, - error_type: AgentErrorType.ABORT, - reason, - }, - abortCode, - ), - }); + return { + outcome: RunOutcome.Aborted, + failure: { + outroData, + code: abortCode, + error: new WizardError( + `Agent aborted: ${reason}`, + { + integration: run.integrationLabel, + error_type: AgentErrorType.ABORT, + reason, + }, + abortCode, + ), + }, + }; } if (agentResult.error === AgentErrorType.MCP_MISSING) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.MCP_MISSING], message: 'Could not access the PostHog MCP server\n\n' + 'The wizard was unable to connect to the PostHog MCP server.\n' + 'This could be due to a network issue or a configuration problem.\n\n' + - `Please try again, or check the documentation:\n${config.docsUrl}`, + `Please try again, or check the documentation:\n${run.docsUrl}`, error: new WizardError( 'Agent could not access PostHog MCP server', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.MCP_MISSING, signal: AgentSignals.ERROR_MCP_MISSING, }, @@ -205,16 +185,16 @@ export async function runLinearProgram( } if (agentResult.error === AgentErrorType.RESOURCE_MISSING) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.RESOURCE_MISSING], message: 'Could not access the setup resource\n\n' + 'This may indicate a version mismatch or a temporary service issue.\n\n' + - `Please try again, or check the documentation:\n${config.docsUrl}`, + `Please try again, or check the documentation:\n${run.docsUrl}`, error: new WizardError( 'Agent could not access setup resource', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.RESOURCE_MISSING, signal: AgentSignals.ERROR_RESOURCE_MISSING, }, @@ -224,13 +204,13 @@ export async function runLinearProgram( } if (agentResult.error === AgentErrorType.YARA_VIOLATION) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.YARA_VIOLATION], message: formatYaraAbortMessage(), error: new WizardError( 'YARA scanner terminated session', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.YARA_VIOLATION, }, AGENT_ERROR_CODE[AgentErrorType.YARA_VIOLATION], @@ -240,10 +220,10 @@ export async function runLinearProgram( if (agentResult.error === AgentErrorType.NO_PROGRESS) { analytics.wizardCapture('agent no progress', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.NO_PROGRESS, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], message: 'The Wizard exited without changing your project. Please contact the ' + @@ -251,7 +231,7 @@ export async function runLinearProgram( error: new WizardError( 'Agent made no progress', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.NO_PROGRESS, }, AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], @@ -261,10 +241,10 @@ export async function runLinearProgram( if (agentResult.error === AgentErrorType.INCOMPLETE_TASKS) { analytics.wizardCapture('agent incomplete tasks', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.INCOMPLETE_TASKS, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.INCOMPLETE_TASKS], message: 'The Wizard exited without completing its planned tasks. Please contact ' + @@ -272,7 +252,7 @@ export async function runLinearProgram( error: new WizardError( 'Agent left planned tasks incomplete', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.INCOMPLETE_TASKS, }, AGENT_ERROR_CODE[AgentErrorType.INCOMPLETE_TASKS], @@ -285,12 +265,12 @@ export async function runLinearProgram( agentResult.error === AgentErrorType.API_ERROR ) { analytics.wizardCapture('agent api error', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: agentResult.error, error_message: agentResult.message, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[agentResult.error], message: `API Error\n\n${ agentResult.message || 'Unknown error' @@ -298,7 +278,7 @@ export async function runLinearProgram( error: new WizardError( `API error: ${agentResult.message}`, { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: agentResult.error, }, AGENT_ERROR_CODE[agentResult.error], @@ -307,39 +287,32 @@ export async function runLinearProgram( } // 10. Post-run hooks - if (config.postRun) { - await config.postRun(session, credentials); + if (config.hooks?.postRun) { + await config.hooks.postRun(credentials); } - // A composed sub-run (integration inside self-driving) skips the terminal - // outro + analytics shutdown so the shared client survives the host's run. - if (composed) return; + // A composed sub-run leaves the terminal outro to its host. + if (composed) { + return { outcome: RunOutcome.Success }; + } // 11. Outro - // Push outro data through the UI (not via direct `session.outroData = ...` - // mutation) so the live store gets the value. agent-runner's `session` - // parameter is captured at runAgent() invocation time, and any `setKey` - // call between then and here (e.g. setDashboardUrl, setNotebookUrl) forks - // the session reference — direct mutation then lands on a stale snapshot - // that the screen never reads. UI.setOutroData() goes through the store - // and also merges in any post-snapshot URLs from the live session. - const outroData = config.buildOutroData - ? config.buildOutroData(session, credentials) + const outroData: OutroData | undefined = config.hooks?.buildOutroData + ? config.hooks.buildOutroData(credentials) : { kind: OutroKind.Success, - message: config.successMessage, - reportFile: config.reportFile, - docsUrl: config.docsUrl, - continueUrl: session.signup + message: run.successMessage, + reportFile: run.reportFile, + docsUrl: run.docsUrl, + continueUrl: input.flags.signup ? `${host.appHost}/products?source=wizard` : undefined, }; if (outroData) { - getUI().setOutroData(outroData); + emit({ kind: 'completion', outro: outroData }); } - getUI().outro(config.successMessage); + emit({ kind: 'lifecycle', phase: 'completed', message: run.successMessage }); - // 12. Analytics shutdown - await analytics.shutdown('success'); + return { outcome: RunOutcome.Success, outro: outroData }; } diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/executor.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/executor.test.ts index f83794ad8..c76f75e86 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/executor.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/executor.test.ts @@ -9,6 +9,7 @@ import { } from '@lib/agent/runner/sequence/orchestrator/queue'; import { drainQueue, + RunTaskFatal, type RunTask, } from '@lib/agent/runner/sequence/orchestrator/executor'; @@ -39,6 +40,49 @@ describe('drainQueue', () => { return Promise.resolve(); }; + it('waits for live siblings after a fatal error and starts no dependents', async () => { + const fatal = new RunTaskFatal({ message: 'Authentication failed' }); + let release!: () => void; + const blocked = new Promise((resolve) => { + release = resolve; + }); + q.enqueue({ type: 'fatal' }); + const sibling = q.enqueue({ type: 'sibling' }); + q.enqueue({ type: 'dependent', dependsOn: [sibling.id] }); + const started: string[] = []; + const drain = drainQueue(q, async (task) => { + started.push(task.type); + if (task.type === 'fatal') throw fatal; + await blocked; + q.complete(task.id, HANDOFF); + }); + let settled = false; + const result = drain.catch((error: unknown) => { + settled = true; + return error; + }); + await new Promise((resolve) => setImmediate(resolve)); + expect(settled).toBe(false); + expect(started).toEqual(['fatal', 'sibling']); + release(); + expect(await result).toBe(fatal); + expect(q.get(sibling.id)?.status).toBe(TaskStatus.Done); + expect(started).toEqual(['fatal', 'sibling']); + }); + + it('preserves a fatal failure when a sibling completes in the same turn', async () => { + q.enqueue({ type: 'success' }); + q.enqueue({ type: 'fatal' }); + const fatal = new RunTaskFatal({ message: 'Authentication failed' }); + await expect( + drainQueue(q, (task) => { + if (task.type === 'fatal') return Promise.reject(fatal); + q.complete(task.id, HANDOFF); + return Promise.resolve(); + }), + ).rejects.toBe(fatal); + }); + it('runs a single task to done and drains', async () => { const a = q.enqueue({ type: 'install' }); await drainQueue(q, completing, { maxStarts: 50 }); diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts index d2379b32c..9ee3fca2a 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts @@ -10,9 +10,6 @@ import * as fs from 'fs'; import * as os from 'os'; import * as path from 'path'; -vi.mock('@ui', () => ({ - getUI: () => ({ showTaskNotice: vi.fn(), cancelTaskNotice: vi.fn() }), -})); vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture: vi.fn(), diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts index c6acf1014..c0be82feb 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts @@ -11,17 +11,15 @@ import type { TaskNotice } from '@lib/wizard-session'; // Hoisted: `vi.mock` factories are lifted above the imports, so the analytics // factory would otherwise read these before they exist. -const { showTaskNotice, cancelTaskNotice, wizardCapture, captureException } = - vi.hoisted(() => ({ - showTaskNotice: vi.fn<(notice: TaskNotice) => Promise>(), - cancelTaskNotice: vi.fn(), - wizardCapture: vi.fn(), - captureException: vi.fn(), - })); - -vi.mock('@ui', () => ({ - getUI: () => ({ showTaskNotice, cancelTaskNotice }), +const { showTaskNotice, wizardCapture, captureException } = vi.hoisted(() => ({ + showTaskNotice: + vi.fn< + (notice: TaskNotice, context: { signal: AbortSignal }) => Promise + >(), + wizardCapture: vi.fn(), + captureException: vi.fn(), })); + vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture, @@ -38,6 +36,16 @@ import { TASK_NOTICE_TIMEOUT_MS, } from '@lib/agent/runner/sequence/orchestrator/orchestrator-runner'; +/** The answerer under test, standing where `getUI()` used to. */ +const interaction = { taskNotice: showTaskNotice }; + +/** The signal the offer handed the host with its one notice. */ +const noticeSignal = (): AbortSignal => { + const call = showTaskNotice.mock.calls[0]; + if (!call) throw new Error('No notice was shown'); + return call[1].signal; +}; + const NOTICE: TaskNotice = { title: 'Connect your data sources', body: ['We found some sources.'], @@ -49,7 +57,6 @@ const NOTICE: TaskNotice = { const resetMocks = () => { showTaskNotice.mockReset(); - cancelTaskNotice.mockReset(); wizardCapture.mockReset(); captureException.mockReset(); }; @@ -67,12 +74,12 @@ describe('task notice timeout', () => { // Nobody presses anything. showTaskNotice.mockReturnValue(new Promise(() => undefined)); - const promise = offerSeededTask(NOTICE, 1000); + const promise = offerSeededTask(NOTICE, { timeoutMs: 1000, interaction }); vi.advanceTimersByTime(1000); await expect(promise).resolves.toEqual({ keep: false, timedOut: true }); // Without this the modal stays on screen over the rest of the run. - expect(cancelTaskNotice).toHaveBeenCalledTimes(1); + expect(noticeSignal().aborted).toBe(true); } finally { vi.useRealTimers(); } @@ -83,13 +90,16 @@ describe('task notice timeout', () => { try { showTaskNotice.mockResolvedValue(true); - const result = await offerSeededTask(NOTICE, 1000); + const result = await offerSeededTask(NOTICE, { + timeoutMs: 1000, + interaction, + }); expect(result).toEqual({ keep: true, timedOut: false }); vi.advanceTimersByTime(5000); // The timer must not fire after an answer — it would close a modal that // is no longer there and stamp a second outcome on the step. - expect(cancelTaskNotice).not.toHaveBeenCalled(); + expect(noticeSignal().aborted).toBe(false); } finally { vi.useRealTimers(); } @@ -102,11 +112,13 @@ describe('task notice timeout', () => { // Both decline, but only one of them means "the user was not there" — // the run reports them differently, and now skips them differently too. - await expect(offerSeededTask(NOTICE, 1000)).resolves.toEqual({ + await expect( + offerSeededTask(NOTICE, { timeoutMs: 1000, interaction }), + ).resolves.toEqual({ keep: false, timedOut: false, }); - expect(cancelTaskNotice).not.toHaveBeenCalled(); + expect(noticeSignal().aborted).toBe(false); } finally { vi.useRealTimers(); } @@ -164,7 +176,9 @@ describe('askSeededConsent', () => { it('records an acceptance', async () => { showTaskNotice.mockResolvedValue(true); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, { timeoutMs: 1000, interaction }), + ).resolves.toEqual({ keep: true, timedOut: false, errored: false, @@ -174,7 +188,9 @@ describe('askSeededConsent', () => { it('records an explicit decline', async () => { showTaskNotice.mockResolvedValue(false); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, { timeoutMs: 1000, interaction }), + ).resolves.toEqual({ keep: false, timedOut: false, errored: false, @@ -186,7 +202,10 @@ describe('askSeededConsent', () => { try { showTaskNotice.mockReturnValue(new Promise(() => undefined)); - const promise = askSeededConsent('warehouse', NOTICE, 1000); + const promise = askSeededConsent('warehouse', NOTICE, { + timeoutMs: 1000, + interaction, + }); vi.advanceTimersByTime(1000); await expect(promise).resolves.toEqual({ @@ -202,7 +221,10 @@ describe('askSeededConsent', () => { it('reports the answer on one event per offer', async () => { showTaskNotice.mockResolvedValue(true); - await askSeededConsent('warehouse', NOTICE, 1000); + await askSeededConsent('warehouse', NOTICE, { + timeoutMs: 1000, + interaction, + }); expect(wizardCapture).toHaveBeenCalledTimes(1); expect(wizardCapture).toHaveBeenCalledWith( @@ -216,7 +238,9 @@ describe('askSeededConsent', () => { // that could not be put to the user must never be read as a yes. showTaskNotice.mockRejectedValue(new Error('UI blew up')); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, { timeoutMs: 1000, interaction }), + ).resolves.toEqual({ keep: false, timedOut: false, errored: true, @@ -247,7 +271,12 @@ describe('offering notices from the seed loop', () => { const answers: { keep: boolean; timedOut: boolean; errored: boolean }[] = []; for (const entry of entries) { - answers.push(await askSeededConsent(entry.type, NOTICE, 60_000)); + answers.push( + await askSeededConsent(entry.type, NOTICE, { + timeoutMs: 60_000, + interaction, + }), + ); } return answers; }; diff --git a/src/lib/agent/runner/sequence/orchestrator/executor.ts b/src/lib/agent/runner/sequence/orchestrator/executor.ts index 616d29442..e70e5723c 100644 --- a/src/lib/agent/runner/sequence/orchestrator/executor.ts +++ b/src/lib/agent/runner/sequence/orchestrator/executor.ts @@ -10,6 +10,7 @@ * injected: the real one spins up a fresh agent, the tests use a fake. */ import { analytics } from '@utils/analytics'; +import type { AgentFailure } from '../../shared/types'; import { logToFile } from '@utils/debug'; import { TaskStatus, type QueueStore, type QueuedTask } from './queue'; @@ -33,6 +34,19 @@ export type TaskResolver = ( * (via the task agent calling complete_task). */ export type RunTask = (task: QueuedTask) => Promise; +/** + * Thrown by a `RunTask` when its harness returned a decided failure that ends + * the whole run (a 401 the auth screen already reported), not just the task. + * `drainQueue` lets it through so the sequence can return the failure where + * the harness used to exit the process. + */ +export class RunTaskFatal extends Error { + constructor(public readonly failure: AgentFailure) { + super(failure.message ?? 'agent run failed'); + this.name = 'RunTaskFatal'; + } +} + export interface DrainOptions { /** Backstop against a pathological always-one-more-pending loop. */ maxStarts: number; @@ -51,6 +65,7 @@ async function runOne( try { await runTask(task); } catch (error) { + if (error instanceof RunTaskFatal) throw error; // The task threw rather than reporting. The outcome check below handles // the queue; the exception itself should never be silent. logToFile(`[executor] runTask threw for ${task.type}:`, error); @@ -97,19 +112,25 @@ export async function drainQueue( ): Promise { const running = new Map>(); let starts = 0; + let failure: { error: unknown } | undefined; - for (;;) { - for (const task of store.nextRunnable()) { - if (++starts > opts.maxStarts) break; - // runOne marks the task running synchronously, so the next - // nextRunnable() call no longer offers it. - const p = runOne(store, runTask, task).finally(() => - running.delete(task.id), - ); - running.set(task.id, p); + try { + for (;;) { + if (failure) throw failure.error; + for (const task of store.nextRunnable()) { + if (++starts > opts.maxStarts) break; + const p = runOne(store, runTask, task) + .catch((error: unknown) => { + failure ??= { error }; + }) + .finally(() => running.delete(task.id)); + running.set(task.id, p); + } + if (running.size === 0) break; + await Promise.race(running.values()); } - if (running.size === 0) break; - // Wake on the first finish; it may have unblocked dependents or requeued. - await Promise.race(running.values()); + } finally { + // No queue or skill cleanup may run while a sibling still uses them. + await Promise.allSettled(running.values()); } } diff --git a/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts b/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts index 69d61fd77..e4535c85c 100644 --- a/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts +++ b/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts @@ -10,6 +10,8 @@ * task resolve to a prompt fetched at startup into the registry. The wizard side * stays product-ignorant: it is the queue, the executor, and the loader. */ +import { failed } from '../../shared/errors'; +import { RunOutcome } from '../../shared/types'; import { randomUUID } from 'crypto'; import { cpSync, @@ -20,31 +22,28 @@ import { writeFileSync, } from 'fs'; import * as path from 'path'; -import { - OutroKind, - type TaskNotice, - type WizardSession, -} from '@lib/wizard-session'; -import { - POSTHOG_DOCS_URL, - WIZARD_CONTACT_EMAIL, - type Integration, -} from '@lib/constants'; -import { FRAMEWORK_REGISTRY } from '@lib/registry'; +import { OutroKind, type TaskNotice } from '@lib/wizard-session'; +import { POSTHOG_DOCS_URL, WIZARD_CONTACT_EMAIL } from '@lib/constants'; import { installSkillById, fetchSkillMenu, type SkillEntry, } from '@lib/wizard-tools'; -import { getUI } from '@ui'; import { analytics } from '@utils/analytics'; import { ciExcludedTaskTypes } from '@utils/ci-flag-overrides'; import { logToFile } from '@utils/debug'; import { ringTerminalBell } from '@utils/terminal-bell'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@utils/wizard-abort'; import { ErrorCodes } from '@lib/errors'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { BootstrapResult, ProgramRun } from '../../shared/types'; +import type { AgentInteraction } from '@lib/agent/progress'; +import type { + AgentFailure, + RunConfig, + SequenceResult, + SequenceContext, +} from '../../shared/types'; +import { createEmitSpinner } from '../../shared/progress-collector'; +import { createAskBridge } from '../../shared/ask'; import { areSeededTasksEnabled, getHarness, @@ -58,19 +57,15 @@ import { QueueStore, QUEUE_DIR_NAME, SkipReason, - TASK_OUTCOMES_KEY, TaskStatus, type QueuedTask, type TaskOutcome, } from './queue'; -import { drainQueue, type RunTask } from './executor'; +import { drainQueue, RunTaskFatal, type RunTask } from './executor'; import { RunMetrics } from './run-metrics'; import { dependencyClosure, uncoveredBySink } from './queue-tools'; import { deferSeededTasks } from './seeded-deps'; -import { - createWizardAskBridge, - LONGER_ASK_TIMEOUT_MS, -} from '@lib/wizard-ask-bridge'; +import { LONGER_ASK_TIMEOUT_MS } from '@lib/wizard-ask-bridge'; import { shouldDisableAsk } from '../../shared/bootstrap'; import { agentRunTools, @@ -185,7 +180,7 @@ export function resolveSkillVariantId( } /** - * The framework reference is the full `integration` skill. `session.skillId` is + * The framework reference is the full `integration` skill. `input.skillId` is * the bare framework (e.g. `django`), but the skill menu ids it as * `integration-`. */ @@ -213,6 +208,11 @@ function resolveReferenceSkillId( */ export const TASK_NOTICE_TIMEOUT_MS = 5 * 60 * 1000; +interface SeededTaskOptions { + timeoutMs?: number; + interaction?: AgentInteraction; +} + /** * Offer an optional step, defaulting to declining it if nobody answers. * @@ -223,21 +223,30 @@ export const TASK_NOTICE_TIMEOUT_MS = 5 * 60 * 1000; */ export async function offerSeededTask( notice: TaskNotice, - timeoutMs: number = TASK_NOTICE_TIMEOUT_MS, + { timeoutMs = TASK_NOTICE_TIMEOUT_MS, interaction }: SeededTaskOptions = {}, ): Promise<{ keep: boolean; timedOut: boolean }> { + // No one to show the notice to: a step nobody can answer for must not run. + // The same answer a non-interactive host gives today. + if (!interaction?.taskNotice) return { keep: false, timedOut: false }; + const { taskNotice } = interaction; + + const controller = new AbortController(); let timer: ReturnType | undefined; let timedOut = false; const timeout = new Promise((resolve) => { timer = setTimeout(() => { timedOut = true; - // Dismisses the overlay and settles the showTaskNotice promise too, so - // the losing side of the race cannot leave a modal on screen. - getUI().cancelTaskNotice(); + // The host dismisses this notice's overlay and settles its promise too, + // so the losing side of the race cannot leave a modal on screen. + controller.abort(); resolve(false); }, timeoutMs); }); try { - const keep = await Promise.race([getUI().showTaskNotice(notice), timeout]); + const keep = await Promise.race([ + taskNotice(notice, { signal: controller.signal }), + timeout, + ]); return { keep, timedOut }; } finally { if (timer) clearTimeout(timer); @@ -338,9 +347,9 @@ export function skipDeclinedSeededTasks( export async function askSeededConsent( type: string, notice: TaskNotice, - timeoutMs?: number, + options: SeededTaskOptions = {}, ): Promise { - const consent = await offerSeededTask(notice, timeoutMs).then( + const consent = await offerSeededTask(notice, options).then( (answer): SeededConsent => ({ ...answer, errored: false }), (err: unknown): SeededConsent => { logToFile( @@ -464,42 +473,63 @@ export function displayOrder( * program config — the registry and seed note both read this one list. */ export function effectiveExcludedTaskTypes( - programConfig: ProgramConfig, + source: Pick, flags: Record, ): string[] { return [ ...ciExcludedTaskTypes(), - ...(programConfig.excludedTaskTypes?.(flags) ?? []), + ...(source.excludedTaskTypes?.(flags) ?? []), ]; } export async function runOrchestrator( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, -): Promise { + context: SequenceContext, +): Promise { + let cleaned = false; + const cleanupQueue = (): void => { + if (cleaned) return; + cleaned = true; + try { + rmSync(path.join(context.input.installDir, QUEUE_DIR_NAME), { + recursive: true, + force: true, + }); + } catch (error) { + analytics.captureException( + error instanceof Error ? error : new Error(String(error)), + { step: 'orchestrator_cache_cleanup' }, + ); + } + }; + try { + return await executeOrchestrator(context, cleanupQueue); + } finally { + cleanupQueue(); + } +} + +async function executeOrchestrator( + { config, input, boot, emit, interaction }: SequenceContext, + cleanupQueue: () => void, +): Promise { const runId = randomUUID(); + const { run } = config; + const programId = config.programId; // Switchboard context — reused for every per-role harness resolution below. - const switchboardCtx = { - program: programConfig.id, - flags: boot.wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliSequence: session.sequence, - cliModel: session.model, - }; + // The caller resolved the run-level binding from it; per-task roles overlay + // `binding.contextMillOverride[role]` on the same inputs. + const switchboardCtx = { ...config.switchboard, trace: undefined }; // The WHAT (agent prompts) is served from context-mill. Fetch the registry // once up front: its types drive enqueue validation, and resolving a task to // its run config is then synchronous, with no mid-drain network latency. - const flow = programConfig.agentFlow ?? programConfig.id; + const flow = config.agentFlow ?? programId; const registry = await loadAgentRegistry(boot.skillsBaseUrl, flow, { - exclude: effectiveExcludedTaskTypes(programConfig, boot.wizardFlags), + exclude: effectiveExcludedTaskTypes(config, boot.wizardFlags), // Baked into the prompts at load, so enqueue, dispatch, and telemetry all read one effective spec. overrides: resolveStageOverrides( - programConfig.id, + programId, boot.wizardFlags, boot.wizardFlagPayloads, ), @@ -538,7 +568,7 @@ export async function runOrchestrator( ? Date.parse(t.finishedAt) - Date.parse(t.startedAt) : undefined; - const store = new QueueStore(session.installDir, runId, { + const store = new QueueStore(input.installDir, runId, { onTransition: (event, task) => { const pick = resolveHarness(switchboardCtx, task.type); // Mirror dispatch's allow-list fallback so attribution names the model that runs. @@ -611,20 +641,20 @@ export async function runOrchestrator( let commandmentsPath: string | undefined; let referenceInstallPath: string | undefined; const menuSkillEntries = await fetchSkillMenuEntries(boot.skillsBaseUrl); - // The framework key for reference + variant resolution. `session.integration` - // is the detected framework and always wins; `session.skillId` is the - // fallback for the basic-integration path, where bootstrap sets it to the + // The framework key for reference + variant resolution. `input.integration` + // is the detected framework and always wins; `input.skillId` is the + // fallback for the basic-integration path, where the caller sets it to the // framework label. Programs whose run config carries their own skill id // (agent-skill commands like replay-vision) would otherwise leak that id in - // here as a bogus framework after bootstrap overwrites the detect result. - const framework = session.integration ?? session.skillId ?? undefined; + // here as a bogus framework after the caller overwrites the detect result. + const framework = input.integration ?? input.skillId ?? undefined; const referenceSkillId = framework ? resolveReferenceSkillId(menuSkillEntries, framework) : undefined; if (referenceSkillId) { const ref = await installSkillById( referenceSkillId, - session.installDir, + input.installDir, boot.skillsBaseUrl, { skillsRoot: path.join(QUEUE_DIR_NAME, 'reference'), @@ -634,11 +664,11 @@ export async function runOrchestrator( if (ref.kind === 'ok') { referenceInstallPath = ref.path; const example = path.join(ref.path, 'references', 'EXAMPLE.md'); - if (existsSync(path.join(session.installDir, example))) { + if (existsSync(path.join(input.installDir, example))) { examplePath = example; } const commandments = path.join(ref.path, 'references', 'COMMANDMENTS.md'); - if (existsSync(path.join(session.installDir, commandments))) { + if (existsSync(path.join(input.installDir, commandments))) { commandmentsPath = commandments; } } else { @@ -673,11 +703,9 @@ export async function runOrchestrator( } } if (missingVariants.length > 0) { - // The framework's own docs page from its config; generic docs when detection found none. - const docsUrl = framework - ? FRAMEWORK_REGISTRY[framework as Integration]?.metadata.docsUrl - : undefined; - await wizardAbort({ + // The framework's own docs page, resolved by the caller; generic docs when detection found none. + const docsUrl = framework ? input.frameworkDocsUrl : undefined; + return failed({ code: ErrorCodes.AgentOrchestratorSkillVariantMissing, message: 'Setup instructions for this project failed to download.\n' + @@ -708,27 +736,28 @@ export async function runOrchestrator( }; logToFile( - `[orchestrator] START program=${programConfig.id} dir=${session.installDir} run=${runId}`, + `[orchestrator] START program=${programId} dir=${input.installDir} run=${runId}`, ); analytics.wizardCapture('orchestrator started', { - program_id: programConfig.id, + program_id: programId, }); - getUI().startRun(); + emit({ kind: 'lifecycle', phase: 'started' }); // Label precedence: what the orchestrator set at enqueue, then the agent // prompt's default, then the bare type. const labelFor = (t: { type: string; label?: string }) => t.label ?? registry.get(t.type)?.label ?? t.type; const renderQueue = () => - getUI().syncTodos( - displayOrder(store.list(), (t) => + emit({ + kind: 'tasks', + tasks: displayOrder(store.list(), (t) => registry.runnerSeededTypes.includes(t.type), ).map((t) => ({ content: labelFor(t), status: toTodoStatus(t.status), activeForm: labelFor(t), })), - ); + }); // Each task's run binds the wizard-tools MCP server to a per-task // orchestrator context so complete_task / enqueue_task attribute correctly @@ -757,7 +786,7 @@ export async function runOrchestrator( // Kill switch: off (or unset), the wizard queues nothing itself and the run // is byte-identical to a project with no detected sources. const seedEntries = areSeededTasksEnabled(boot.wizardFlags) - ? programConfig.seedTasks?.(session) ?? [] + ? config.seedTasks?.() ?? [] : []; const seededTypes: string[] = []; // Kept so their dependencies can be resolved once the planner has run — they @@ -806,7 +835,7 @@ export async function runOrchestrator( // not, which is why the offer lives here and not there. seededConsent.set( task.id, - await askSeededConsent(seeded.type, seeded.notice), + await askSeededConsent(seeded.type, seeded.notice, { interaction }), ); } logToFile(`[orchestrator] runner-seeded task ${seeded.type}`); @@ -838,11 +867,11 @@ export async function runOrchestrator( // One bridge for the run, handed only to a task whose prompt allows asking. // Absent in CI and signup, where nobody can answer. - const askBridge = shouldDisableAsk(session) + const askBridge = shouldDisableAsk(input.flags) ? undefined - : createWizardAskBridge({ - getSource: () => session.skillId ?? programConfig.id, - showQuestion: (q) => { + : createAskBridge(interaction, { + getSource: () => input.skillId ?? programId, + beforeShow: () => { // How late the first ask lands is the measure of this run shape: it // should follow the autonomous work, not interrupt it. metrics.recordAsk(Date.now()); @@ -852,16 +881,14 @@ export async function runOrchestrator( // unanswered ask still times out into the deep-link fallback — it just // gives a person who stepped away a chance to come back first. ringTerminalBell(); - return getUI().requestQuestion(q); }, - cancelQuestion: () => getUI().cancelPendingQuestion(), - richLinks: config.richLinks ?? false, + richLinks: run.richLinks ?? false, // A task ask waits on a person, and the drain waits it out — the // executor holds the task's promise — so this is the only real limit. timeoutMs: LONGER_ASK_TIMEOUT_MS, }); - const spinner = getUI().spinner(); + const spinner = createEmitSpinner(emit); // 1. Seed the queue with the orchestrator agent. It is itself an agent prompt // (the WHAT), so its model and tools come from its frontmatter. The seed @@ -874,9 +901,10 @@ export async function runOrchestrator( const seedHarness = requireTaskHarness(seedPick); const seedModel = promptModelFor(seedPrompt, seedPick.harness); const seedResult = await seedHarness.runTask({ - session, - programConfig, + config, + input, boot, + emit, // The exclusion note names `registry.excludedTypes` — the excluded types // this flow actually had — so the planner never hears about work that was // never available, and overlapping exclusion sources cannot double-list. @@ -897,6 +925,8 @@ export async function runOrchestrator( requestRemark: false, analyticsProperties: { task_type: 'seed', harness: seedPick.harness }, }); + // A decided seed failure ends the run and releases its queue artifacts. + if (seedResult.failure) return failed(seedResult.failure); if (seedResult.error) { logToFile( `[orchestrator] seed error: ${seedResult.error} ${ @@ -997,7 +1027,7 @@ export async function runOrchestrator( analytics.wizardCapture('orchestrator sink invariant violated', { uncovered_types: unwaited.map((t) => t.type), }); - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorSinkInvariant, message: `The wizard could not plan this setup: the final step would have skipped ${unwaited .map((t) => t.type) @@ -1021,7 +1051,7 @@ export async function runOrchestrator( // Task agents can install durable skills mid-run (load_skill), and only the // framework reference docs earn a place — snapshot what was already there so // the sweeps remove exactly what this run added. - const claudeSkillsDir = path.join(session.installDir, '.claude', 'skills'); + const claudeSkillsDir = path.join(input.installDir, '.claude', 'skills'); const preexistingSkills = new Set( existsSync(claudeSkillsDir) ? readdirSync(claudeSkillsDir) : [], ); @@ -1054,7 +1084,7 @@ export async function runOrchestrator( } const result = await installSkillById( variantId, - session.installDir, + input.installDir, boot.skillsBaseUrl, { skillsRoot: taskSkillsRoot, triage: boot.triageProvider }, ); @@ -1085,10 +1115,11 @@ export async function runOrchestrator( const taskPick = resolveHarness(switchboardCtx, task.type); const taskHarness = requireTaskHarness(taskPick); const taskModel = taskModelSpec(registry, task, taskPick.harness); - await taskHarness.runTask({ - session, - programConfig, + const taskResult = await taskHarness.runTask({ + config, + input, boot, + emit, prompt: assembleTaskPrompt(promptContext, resolved.prompt, skillPaths), spinner, model: requireKnownModel(taskModel.model, taskPick.model), @@ -1107,6 +1138,10 @@ export async function runOrchestrator( harness: taskPick.harness, }, }); + // A decided failure (a 401 the harness already reported) is the run's, + // not the task's: stop the drain and report it, where the harness used + // to exit the process. + if (taskResult.failure) throw new RunTaskFatal(taskResult.failure); } finally { // Durable skills a task installed are irrelevant to later tasks — and // the sdk harness auto-loads .claude/skills into every agent — so sweep @@ -1130,19 +1165,25 @@ export async function runOrchestrator( renderQueue(); } + let fatal: AgentFailure | undefined; try { await drainQueue(store, runTask); + } catch (error) { + if (!(error instanceof RunTaskFatal)) throw error; + fatal = error.failure; } finally { // The queue file is wiped below; the e2e harness reads outcomes from here. - session.frameworkContext[TASK_OUTCOMES_KEY] = store.list().map((t) => ({ - type: t.type, - status: t.status, - optional: t.optional === true, - })) satisfies TaskOutcome[]; + config.hooks?.recordTaskOutcomes?.( + store.list().map((t) => ({ + type: t.type, + status: t.status, + optional: t.optional === true, + })) satisfies TaskOutcome[], + ); try { if (referenceSkillId && referenceInstallPath) { promoteReferenceSkill( - path.join(session.installDir, referenceInstallPath), + path.join(input.installDir, referenceInstallPath), claudeSkillsDir, referenceSkillId, ); @@ -1153,20 +1194,7 @@ export async function runOrchestrator( { step: 'orchestrator_reference_promote' }, ); } - // Success or failure, no run artifact outlives the run — wipe the whole - // cache folder (queue, handoffs, reference example, installed task - // instructions). The .DELETE-ME.md inside is the fallback if we don't. - try { - rmSync(path.join(session.installDir, QUEUE_DIR_NAME), { - recursive: true, - force: true, - }); - } catch (err) { - analytics.captureException( - err instanceof Error ? err : new Error(String(err)), - { step: 'orchestrator_cache_cleanup' }, - ); - } + cleanupQueue(); try { sweepRunInstalledSkills( claudeSkillsDir, @@ -1181,6 +1209,8 @@ export async function runOrchestrator( } } + if (fatal) return failed(fatal); + renderQueue(); const summary = store.summary(); @@ -1246,7 +1276,7 @@ export async function runOrchestrator( ', ', )}.\n\nPlease try again, approving all permissions on the PostHog authorization screen. If it still fails, report it to: ${WIZARD_CONTACT_EMAIL}` : `The wizard was unable to set up PostHog: ${whatFailed}.\n\nPlease report this to: ${WIZARD_CONTACT_EMAIL}`; - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorTasksFailed, message, error: new WizardError( @@ -1267,7 +1297,7 @@ export async function runOrchestrator( // usable model response (e.g. the gateway returned empty completions). // "0/0 completed" is a dead run, not a success with an empty denominator. if (summary.total === 0) { - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorHollowRun, message: `The wizard was unable to set up PostHog: the planning step produced no work, so nothing ran.\n\nPlease try again — and if it happens again, report it to: ${WIZARD_CONTACT_EMAIL}`, error: new WizardError( @@ -1293,19 +1323,19 @@ export async function runOrchestrator( } steps completed${ stepNotes.length > 0 ? ` (${stepNotes.join(', ')})` : '' }.`; - getUI().setOutroData({ + const outro = { kind: OutroKind.Success, message, body: conflict ? `⚠ Build conflict: ${conflict}\nFull details are in the setup report.` : undefined, docsUrl: 'https://posthog.com/docs/ai-engineering/ai-wizard', - nextSteps: config.buildOutroNextSteps?.( - session, + nextSteps: config.hooks?.buildOutroNextSteps?.( boot.credentials, completedSeededTypes(store, seededTasks), ), - }); - getUI().outro(message); - await analytics.shutdown('success'); + }; + emit({ kind: 'completion', outro }); + emit({ kind: 'lifecycle', phase: 'completed', message }); + return { outcome: RunOutcome.Success, outro }; } diff --git a/src/lib/agent/runner/shared/ask.ts b/src/lib/agent/runner/shared/ask.ts new file mode 100644 index 000000000..56cae7874 --- /dev/null +++ b/src/lib/agent/runner/shared/ask.ts @@ -0,0 +1,40 @@ +/** + * The ask bridge over an injected answerer. + * + * `createWizardAskBridge` already owns request ids, the timeout race, the + * `__cancelled__` sentinel and the analytics; it only ever needed a + * `showQuestion` that honours each question's own signal. Here that comes from + * `AgentInteraction` instead of `getUI()`. With no answerer there is no bridge, + * so `wizard_ask` reports its existing "not available" error rather than + * hanging on a question nobody can see. + */ + +import { + createWizardAskBridge, + type WizardAskBridge, +} from '@lib/wizard-ask-bridge'; +import type { AgentInteraction } from '@lib/agent/progress'; + +export function createAskBridge( + interaction: AgentInteraction | undefined, + options: { + getSource: () => string; + richLinks: boolean; + timeoutMs?: number; + /** Runs before each question is shown (the orchestrator's bell and metric). */ + beforeShow?: () => void; + }, +): WizardAskBridge | undefined { + const ask = interaction?.ask; + if (!ask) return undefined; + + return createWizardAskBridge({ + getSource: options.getSource, + showQuestion: (question, context) => { + options.beforeShow?.(); + return ask(question, context); + }, + richLinks: options.richLinks, + timeoutMs: options.timeoutMs, + }); +} diff --git a/src/lib/agent/runner/shared/bootstrap.ts b/src/lib/agent/runner/shared/bootstrap.ts index dc12a2b60..7384c5f1a 100644 --- a/src/lib/agent/runner/shared/bootstrap.ts +++ b/src/lib/agent/runner/shared/bootstrap.ts @@ -1,46 +1,21 @@ /** - * Shared bootstrap for the runner pipeline. + * Shared preparation for the runner pipeline. * - * Runs before the fork into the linear or orchestrator arm: logging, health - * check, settings conflicts, OAuth and credentials, feature flags, variant - * metadata, and MCP url. Sets `session.credentials`, role, and user as side - * effects. Returns the values the arms still need. + * Runs before the fork into the linear or orchestrator arm: logging targets, + * the gateway mint and the scan-triage classifier built on it. Everything the + * caller must decide first — health gates, settings conflicts, authentication, + * the AI opt-in gate, post-auth gates, feature flags, run tags, token refresh — + * arrives already resolved in `RunConfig` and `RunInput`. */ -import type { WizardSession } from '@lib/wizard-session'; -import { analytics } from '@utils/analytics'; -import { getUI } from '@ui'; -import { authenticate, refreshAccessTokenIfNeeded } from './authenticate'; -import { maybeStampAiSdkDetected } from '@lib/programs/posthog-integration/detect'; import { createTriageLLMProvider } from '@lib/agent/triage-provider'; import { gatewayAuth } from '@lib/gateway-session'; -import { resolveHarness } from '../switchboard'; -import { buildRunTags } from '@lib/agent/agent-interface'; -import { - checkAllSettingsConflicts, - backupAndFixClaudeSettings, - classifySettingsConflicts, -} from '@lib/agent/claude-settings'; -import { - evaluateWizardReadiness, - WizardReadiness, - SIGNUP_WIZARD_READINESS_CONFIG, - getBlockingServiceKeys, - SERVICE_LABELS, -} from '@lib/health-checks/readiness'; -import { enableDebugLogs, logToFile, initLogFile } from '@utils/debug'; -import { wizardAbort } from '@utils/wizard-abort'; -import { ErrorCodes } from '@lib/errors'; -import { isNonInteractiveEnvironment } from '@utils/environment'; -import { CallType, getSkillsBaseUrl, IS_DEV } from '@lib/constants'; +import { logToFile } from '@utils/debug'; +import { CallType, IS_DEV } from '@lib/constants'; import { VERSION } from '@lib/version'; import { mcpUrlFor } from '@lib/host-resolution'; import type { WizardRunOptions } from '@utils/types'; -import { - postAuthGateSteps, - type ProgramConfig, -} from '@lib/programs/program-step'; -import type { ProgramRun, BootstrapResult } from './types'; +import type { BootstrapResult, RunConfig, RunFlags, RunInput } from './types'; // ── Helpers ────────────────────────────────────────────────────────── @@ -51,7 +26,7 @@ import type { ProgramRun, BootstrapResult } from './types'; * the program's `disallowedTools` so the SDK rejects calls outright. * Extracted so the policy can be unit-tested directly. * - * `session.e2eAsk` is the one escape hatch. The e2e harness runs a `ci` + * `e2eAsk` is the one escape hatch. The e2e harness runs a `ci` * session, but it does have an answerer — the driver loop answers each * `wizard_ask` batch from the program's e2e profile. Without the flag the * agent-in-the-loop layer (the ask bridge in both sequence arms, and the @@ -61,254 +36,55 @@ import type { ProgramRun, BootstrapResult } from './types'; * populates it, so plain `--ci` and `--signup` runs behave exactly as before. */ export function shouldDisableAsk( - session: Pick, + flags: Pick, ): boolean { - return (session.ci || session.signup) && !session.e2eAsk; + return (flags.ci || flags.signup) && !flags.e2eAsk; } -export function sessionToOptions(session: WizardSession): WizardRunOptions { +/** The option bag the agent interface and the middleware read. */ +export function runOptions(input: RunInput): WizardRunOptions { return { - installDir: session.installDir, - debug: session.debug, - signup: session.signup, - ci: session.ci, - benchmark: session.benchmark, - projectId: session.projectId, - apiKey: session.apiKey, - yaraReport: session.yaraReport, + installDir: input.installDir, + debug: input.flags.debug, + signup: input.flags.signup, + ci: input.flags.ci, + benchmark: input.flags.benchmark, + projectId: input.host.projectId, + apiKey: input.host.apiKey, + yaraReport: input.flags.yaraReport, }; } -// ── Bootstrap ───────────────────────────────────────────────────────── +// ── Prepare ─────────────────────────────────────────────────────────── /** - * Shared setup for both arms: logging, health check, settings conflicts, OAuth - * and credentials, then the feature flags, variant metadata, and MCP url. Sets - * `session.credentials`, role, and user as a side effect. Returns the values the - * arms still need. + * Shared setup for both arms: logging targets, then the gateway mint and the + * triage classifier. Throws when the mint is refused, so the run fails before + * any agent starts — the caller maps that the way it maps any unexpected error. */ -export async function bootstrapProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, +export async function prepareRun( + config: RunConfig, + input: RunInput, ): Promise { - // 1. Init logging + debug - initLogFile(); - session.skillId = config.skillId ?? config.integrationLabel; - logToFile( - `[agent-runner] START ${config.integrationLabel} build=${analytics.build}` + - `${session.ci ? ' (non-interactive)' : ''}`, - ); - - if (session.debug) { - enableDebugLogs(); - } - - const skillsBaseUrl = getSkillsBaseUrl(); + const { skillsBaseUrl } = config; // Where this run actually points. The three services switch independently, // so otherwise "why did it use prod skills?" means reading three call sites. logToFile( `[agent-runner] targets build=${VERSION}${IS_DEV ? '/dev' : ''} ` + `skills=${skillsBaseUrl} ` + - `mcp=${mcpUrlFor(session.localMcp)} ` + - `posthog=${session.baseUrl ?? 'region-resolved'}`, - ); - - // 2. Health check (guarded — skip if TUI already ran it). Only - // programs that declare a health-check screen get pre-flight checks; - // for everything else the checks never fire and never block. - const hasHealthCheckScreen = programConfig.steps.some( - (s) => s.screenId === 'health-check', + `mcp=${mcpUrlFor(input.flags.localMcp)} ` + + `posthog=${input.host.baseUrl ?? 'region-resolved'}`, ); - if (session.readinessResult) { - logToFile( - `[agent-runner] readiness pre-computed by TUI: decision=${session.readinessResult.decision}` + - `${ - session.outageDismissed ? ' (outage dismissed by user)' : '' - } — skipping re-check`, - ); - } - if (hasHealthCheckScreen && !session.readinessResult) { - logToFile('[agent-runner] evaluating wizard readiness'); - const readinessConfig = session.signup - ? SIGNUP_WIZARD_READINESS_CONFIG - : undefined; - const readiness = await evaluateWizardReadiness(readinessConfig); - logToFile(`[agent-runner] readiness=${readiness.decision}`); - if (readiness.decision === WizardReadiness.No) { - const blockingKeys = getBlockingServiceKeys( - readiness.health, - readinessConfig, - ); - const blockingLabels = blockingKeys.map( - (k) => `${SERVICE_LABELS[k]} (${readiness.health[k].status})`, - ); - logToFile(`[agent-runner] blocked by: ${blockingLabels.join(', ')}`); - - await getUI().showBlockingOutage(readiness); - - // The TUI lets the user continue past an outage; non-interactive runs - // (CI) do the same automatically — the degraded services are reported - // above, but we proceed rather than aborting on a transient upstream blip. - if (!isNonInteractiveEnvironment()) { - await wizardAbort({ - code: ErrorCodes.EnvServiceOutage, - message: - 'Cannot start — external services are down:\n' + - blockingLabels.map((l) => ` - ${l}`).join('\n') + - '\n\nPlease try again later.', - }); - } - } else if (readiness.decision === WizardReadiness.YesWithWarnings) { - getUI().setReadinessWarnings(readiness); - } - } - - // 3. Settings conflicts - const settingsConflicts = checkAllSettingsConflicts(session.installDir); - logToFile( - `[agent-runner] settings conflicts: ${ - settingsConflicts.length > 0 - ? settingsConflicts - .map((c) => `${c.source}(${c.keys.join(',')})`) - .join('; ') - : 'none' - }`, - ); - - if (settingsConflicts.length > 0) { - for (const conflict of settingsConflicts) { - const level = conflict.source === 'managed' ? 'org' : conflict.source; - analytics.wizardCapture('settings conflict detected', { - level, - keys: conflict.keys, - }); - } - - const { autoFix, failClosed, warnOnly } = - classifySettingsConflicts(settingsConflicts); - - // User-global and project-local files are already neutralized — the agent - // runs with settingSources:['project'], so the SDK never reads them. Record - // it and move on; don't make the user act on a setting that can't bite. - for (const conflict of warnOnly) { - logToFile( - `[agent-runner] settings conflict in ${conflict.source} (${conflict.path}) ` + - `neutralized by settingSources:['project'] — not blocking`, - ); - analytics.wizardCapture('settings conflict neutralized', { - level: conflict.source, - keys: conflict.keys, - }); - } - - // Writable project settings.json — the SDK *does* read it, but we can back - // it up and remove it (restored at outro). Neutralize without prompting. - let unfixable = failClosed; - if (autoFix.length > 0) { - const fixed = backupAndFixClaudeSettings(session.installDir); - if (fixed) { - logToFile('[agent-runner] auto-neutralized writable settings conflict'); - analytics.wizardCapture('settings conflict auto-neutralized', { - keys: autoFix.flatMap((c) => c.keys), - }); - } else { - // Couldn't remove it — don't run into the redirect; fail closed instead. - logToFile( - '[agent-runner] could not back up writable settings conflict — failing closed', - ); - unfixable = [...failClosed, ...autoFix]; - } - } - - // What we cannot neutralize (org-managed, always read by the SDK; or a - // writable file we failed to back up) must be fixed by the user. Fail - // closed: the screen names the file + keys and exits. - if (unfixable.length > 0) { - if (isNonInteractiveEnvironment()) { - await wizardAbort({ - code: ErrorCodes.SettingsUnfixableConflict, - message: - 'Cannot start — a Claude settings file redirects the agent away ' + - 'from the PostHog gateway and cannot be neutralized automatically:\n' + - unfixable - .map((c) => ` - ${c.source} (${c.path}): ${c.keys.join(', ')}`) - .join('\n') + - '\n\nRemove the conflicting keys and re-run the wizard.', - }); - } - await getUI().showSettingsOverride(unfixable, () => - backupAndFixClaudeSettings(session.installDir), - ); - logToFile('[agent-runner] settings override resolved'); - } - } - - analytics.wizardCapture('agent started', { - integration: config.integrationLabel, - program_id: programConfig.id, - skill_id: config.skillId ?? null, - }); - - // 4. Authenticate — idempotent within a run (see authenticate()). A second - // agent run in the same invocation (self-driving's integration phase) reuses - // the first login; it does not launch another OAuth. authenticate() also - // identifies the user and sets analytics groups. - await authenticate(session, programConfig.id); - maybeStampAiSdkDetected(session); - const project = session.apiProject; - - // 4.5. AI opt-in enforcement. Parks here while AiOptInRequiredScreen is - // up if the org hasn't approved third-party AI — BEFORE the skill - // install and agent start, so no source leaves the machine. The screen - // alone is cosmetic; this await is the actual gate. Resolves - // immediately when the program declared requiresAi: false or in CI. - // In bootstrapProgram so both the linear and orchestrator arms gate. - logToFile('[agent-runner] checking AI opt-in gate'); - await getUI().waitForAiOptIn(); - logToFile('[agent-runner] AI opt-in gate cleared'); - - // Park for any interactive step the user must complete AFTER authenticating - // but BEFORE the agent runs — e.g. the source-maps project picker, which - // needs credentials to scan and writes its choice to frameworkContext that - // the run prompt reads. Generic: await every gated step between auth and run. - for (const step of postAuthGateSteps(programConfig.steps)) { - logToFile(`[agent-runner] awaiting post-auth gate: ${step.id}`); - await getUI().waitForGate(step.id); - logToFile(`[agent-runner] post-auth gate cleared: ${step.id}`); - } - - // Feature flags. Both arms need these, and the fork decision reads the flags. - // This map is PostHog-side only — CLI `--harness` / `--sequence` precedence - // lives at the resolution sites (`runner/index.ts` for sequence, - // `resolveHarness` for harness), not here. - const wizardFlags = await analytics.getAllFlagsForWizard(); - const wizardFlagPayloads = analytics.getWizardFlagPayloads(); - - // Gateway trace tags for this run. The runner stamps its variant onto this - // after the fork (see runProgram), so the value reflects which arm ran. - const wizardMetadata = buildRunTags({ - programId: programConfig.id, - integration: config.integrationLabel, - runId: analytics.runId, - build: analytics.build, - skillId: config.skillId, - }); - - // The agent can't swap tokens mid-run, so freshness is measured after every park above, right before the mint. - await refreshAccessTokenIfNeeded(session); - // Credentials (incl. the resolved host family and its MCP url) live on - // `session.credentials`; narrow once at this boundary — `authenticate` above - // set them — so downstream readers get a non-null type without asserting. - const credentials = session.credentials!; + const { credentials } = input; + const { wizardFlags, wizardFlagPayloads, wizardMetadata, programId } = config; // Mint now so a refusal fails the boot before any agent starts. Later // readers re-resolve through the cache, which re-mints past the refresh // point. const currentGatewayAuth = () => - gatewayAuth(credentials.host, credentials.accessToken, programConfig.id); + gatewayAuth(credentials.host, credentials.accessToken, programId); await currentGatewayAuth(); return { @@ -316,33 +92,24 @@ export async function bootstrapProgram( credentials, // Carried so per-task sessions re-resolve against the same program the boot // minted for, rather than digging it back out of the metadata bag. - programId: programConfig.id, + programId, wizardFlags, wizardFlagPayloads, wizardMetadata, - project, - // Resolved once, here: the only place holding both the switchboard inputs + project: input.project, + // Resolved once, here: the only place holding both the run-level harness // and the gateway auth. Every skill install downstream reads it off boot. - triageProvider: createTriageLLMProvider( - async () => { - const auth = await currentGatewayAuth(); - return { - baseURL: auth.gatewayUrl, - authToken: auth.token, - teamId: auth.teamId, - // `call_type` splits scan spend out of the program's agent cost, - // the same tag the in-run triage provider carries. - wizardMetadata: { ...wizardMetadata, call_type: CallType.yaraTriage }, - wizardFlags, - }; - }, - resolveHarness({ - program: programConfig.id, - flags: wizardFlags, - flagPayloads: wizardFlagPayloads, - cliHarness: session.harness, - cliModel: session.model, - }).harness, - ), + triageProvider: createTriageLLMProvider(async () => { + const auth = await currentGatewayAuth(); + return { + baseURL: auth.gatewayUrl, + authToken: auth.token, + teamId: auth.teamId, + // `call_type` splits scan spend out of the program's agent cost, + // the same tag the in-run triage provider carries. + wizardMetadata: { ...wizardMetadata, call_type: CallType.yaraTriage }, + wizardFlags, + }; + }, config.binding.harness), }; } diff --git a/src/lib/agent/runner/shared/errors.ts b/src/lib/agent/runner/shared/errors.ts index 376cb45d4..8578d2a0c 100644 --- a/src/lib/agent/runner/shared/errors.ts +++ b/src/lib/agent/runner/shared/errors.ts @@ -4,14 +4,19 @@ import type { InstallSkillResult } from '@lib/wizard-tools'; import { skillErrorCode } from '@lib/errors'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@utils/wizard-abort'; +import { RunOutcome, type AgentFailure, type SequenceResult } from './types'; -export async function abortOnInstallFailure( - integrationLabel: string, - result: InstallSkillResult, -): Promise { - if (result.kind === 'ok') return; +export const failed = (failure: AgentFailure): SequenceResult => ({ + outcome: RunOutcome.Failed, + failure, +}); +/** The failure a skill install error decides. The caller reports and exits. */ +export function installFailure( + integrationLabel: string, + result: Exclude, +): AgentFailure { const code = skillErrorCode(result) ?? undefined; const message = (() => { @@ -25,7 +30,7 @@ export async function abortOnInstallFailure( } })(); - await wizardAbort({ + return { message, code, error: new WizardError( @@ -40,5 +45,5 @@ export async function abortOnInstallFailure( }, code, ), - }); + }; } diff --git a/src/lib/agent/runner/shared/progress-collector.ts b/src/lib/agent/runner/shared/progress-collector.ts new file mode 100644 index 000000000..6f493dbc3 --- /dev/null +++ b/src/lib/agent/runner/shared/progress-collector.ts @@ -0,0 +1,121 @@ +/** + * The agent's own record of what it reported. + * + * `runAgent` accumulates its final `RunSnapshot` here, independently of any + * observer, so a missing or throwing `onProgress` never makes the result + * incomplete. The emitter wraps the caller's callback: it applies the event + * here first, then hands a copy to the observer, and logs rather than + * propagates anything the observer throws. + */ + +import { logToFile } from '@utils/debug'; +import { appendStatus } from '@lib/status-history'; +import type { + AgentProgress, + ProgressEmitter, + SpinnerHandle, +} from '@lib/agent/progress'; +import type { RunSnapshot } from './types'; + +export interface ProgressCollector { + emit: ProgressEmitter; + snapshot(): RunSnapshot; +} + +export function createProgressCollector( + onProgress?: (event: AgentProgress) => void, +): ProgressCollector { + const snapshot: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, + }; + + const apply = (event: AgentProgress): void => { + switch (event.kind) { + case 'tasks': + snapshot.tasks = event.tasks.map((t) => ({ ...t })); + break; + case 'status': + snapshot.statusMessages = appendStatus( + snapshot.statusMessages, + event.message, + ); + break; + case 'stage': + snapshot.stage = event.stage; + break; + case 'url': + if (event.which === 'dashboard') snapshot.dashboardUrl = event.url; + else snapshot.notebookUrl = event.url; + break; + case 'usage': + snapshot.usage.inputTokens += event.delta.inputTokens; + snapshot.usage.outputTokens += event.delta.outputTokens; + snapshot.usage.cacheReadTokens += event.delta.cacheReadTokens; + snapshot.usage.cacheCreationTokens += event.delta.cacheCreationTokens; + break; + case 'finalCost': + snapshot.finalCostUsd = event.usd; + break; + default: + break; + } + }; + + const emit: ProgressEmitter = (event) => { + apply(event); + if (!onProgress) return; + try { + onProgress(structuredClone(event)); + } catch (error) { + // A broken projection is the host's problem, not the run's. Say so in + // the log and carry on; the snapshot above is the source of truth. + logToFile( + `[agent] progress observer threw on ${event.kind}:`, + error instanceof Error ? error.message : error, + ); + } + }; + + return { + emit, + snapshot: () => ({ + ...snapshot, + tasks: snapshot.tasks.map((t) => ({ ...t })), + statusMessages: [...snapshot.statusMessages], + usage: { ...snapshot.usage }, + }), + }; +} + +/** The run spinner as a progress emitter. One per run, like `getUI().spinner()`. */ +export function createEmitSpinner(emit: ProgressEmitter): SpinnerHandle { + return { + start: (message) => emit({ kind: 'spinner', action: 'start', message }), + stop: (message) => emit({ kind: 'spinner', action: 'stop', message }), + message: (message) => emit({ kind: 'spinner', action: 'message', message }), + }; +} + +/** `WizardUI.log` as a progress emitter. */ +export function createEmitLog(emit: ProgressEmitter): { + info(message: string): void; + warn(message: string): void; + error(message: string): void; + success(message: string): void; + step(message: string): void; +} { + return { + info: (message) => emit({ kind: 'log', level: 'info', message }), + warn: (message) => emit({ kind: 'log', level: 'warn', message }), + error: (message) => emit({ kind: 'log', level: 'error', message }), + success: (message) => emit({ kind: 'log', level: 'success', message }), + step: (message) => emit({ kind: 'log', level: 'step', message }), + }; +} diff --git a/src/lib/agent/runner/shared/types.ts b/src/lib/agent/runner/shared/types.ts index 4f4292f83..2f9b1df9f 100644 --- a/src/lib/agent/runner/shared/types.ts +++ b/src/lib/agent/runner/shared/types.ts @@ -1,16 +1,31 @@ /** - * Shared types for the runner pipeline. + * The agent's run contracts. + * + * `runAgent(config, input, options)` takes resolved execution data and an + * invocation snapshot, reports through `options.onProgress`, asks through + * `options.interaction`, and returns a `RunResult`. Nothing here names a UI, + * a store, a session or a program registry: the caller resolves those and + * hands over plain data. `src/lib/programs/run-agent-legacy.ts` is the caller + * that rebuilds today's session-driven behavior on top of this contract. */ import type { - Credentials, AdditionalFeature, + CloudRegion, + Credentials, + OutroData, + TaskNotice, WizardSession, } from '@lib/wizard-session'; import type { PromptContext } from '@lib/agent/agent-prompt'; import type { PackageManagerDetector } from '@lib/detection/package-manager'; -import type { ApiProject } from '@lib/api'; +import type { ApiProject, ApiUser } from '@lib/api'; +import type { Harness, Integration, Sequence } from '@lib/constants'; +import type { ErrorCode } from '@lib/errors'; import type { LLMProvider } from '@posthog/warlock'; +import type { AgentInteraction, ProgressEmitter } from '@lib/agent/progress'; +import type { EffortLevel } from '../switchboard/models'; +import type { SwitchboardCtx } from '../switchboard'; export type { PromptContext, Credentials }; @@ -23,7 +38,7 @@ export interface AbortCase { message: string; body: string; docsUrl?: string; - errorCode?: import('@lib/errors').ErrorCode; + errorCode?: ErrorCode; } /** @@ -32,6 +47,10 @@ export interface AbortCase { * Every program provides one of these — either as a static object * or via a function that builds one from the session. The runner * assembles the final prompt from `prompt` + `skillId`. + * + * The three session-taking hooks at the end are the caller's: the agent never + * calls them. `run-agent-legacy.ts` binds them and hands the agent + * `RunConfig.hooks` instead. */ export interface ProgramRun { /** Analytics label (e.g. 'revenue-analytics-setup', 'nextjs') */ @@ -116,15 +135,138 @@ export interface ProgramRun { resolveStepKey?: (stepName: string | undefined) => string | undefined; } +/** A task the caller queues itself before the orchestrator's planner runs. */ +export interface SeedTaskEntry { + type: string; + label?: string; + inputs?: Record; + /** Shown before the run starts, letting the user decline the task. */ + notice?: TaskNotice; +} + /** - * Result of the shared bootstrap, consumed by both the linear and the - * orchestrator arm. `bootstrapProgram` runs `authenticate` before returning, so - * `credentials` is guaranteed non-null here — the single narrowing point owns - * the invariant, and downstream readers get a properly non-null type for free. + * Completion hooks the caller binds for the agent. Each receives the run's + * resolved credentials, exactly what the linear and orchestrator sequences + * passed alongside the session before. + */ +export interface RunHooks { + /** Runs after the agent completes, before the outro (linear only). */ + postRun?: (credentials: Credentials) => Promise; + /** Custom outro data (linear only). Omit for the default outro. */ + buildOutroData?: (credentials: Credentials) => OutroData | undefined; + /** Outro bullets for the orchestrated sequence. */ + buildOutroNextSteps?: ( + credentials: Credentials, + completedSeededTypes: readonly string[], + ) => { heading: string; items: string[] } | undefined; + /** Receives the drained queue's final outcomes before the cache wipe (orchestrated only). */ + recordTaskOutcomes?: ( + outcomes: import('../sequence/orchestrator/queue').TaskOutcome[], + ) => void; +} + +/** The run-level routing decision the caller made. */ +export interface ResolvedBinding { + sequence: Sequence; + harness: Harness; + /** Gateway model id. */ + model: string; + /** Reasoning-effort override. Absent → the model's table default. */ + thinkingLevel?: EffortLevel; +} + +/** + * Resolved execution data for one agent run. The caller has already decided + * which program this is, how it is routed and which flags apply; the agent + * treats every label as opaque. + */ +export interface RunConfig { + /** Program id: gateway spend pin, analytics label, commandments axis. */ + programId: string; + /** The program's run definition. Its session-taking hooks are the caller's, see `hooks`. */ + run: ProgramRun; + /** A composed sub-run leaves the terminal outro to its host. */ + composed: boolean; + /** Run-level sequence, harness and model. */ + binding: ResolvedBinding; + /** + * The inputs the run-level binding was resolved from. The orchestrator + * re-resolves the harness per task role from these; nothing else reads them. + */ + switchboard: SwitchboardCtx; + /** Primary skills origin (context-mill dev or GitHub Releases). */ + skillsBaseUrl: string; + /** Feature flag key → variant, evaluated before the run. */ + wizardFlags: Record; + /** Flag payloads from the same snapshot. */ + wizardFlagPayloads: Record; + /** Gateway trace tags for this run, already stamped with sequence and harness. */ + wizardMetadata: Record; + /** Extra tools added on top of BASE_ALLOWED_TOOLS for this run. */ + allowedTools?: readonly string[]; + /** Tools removed from BASE_ALLOWED_TOOLS for this run. */ + disallowedTools?: readonly string[]; + /** Context-mill flow the orchestrator loads. Defaults to `programId`. */ + agentFlow?: string; + /** Task types the program excludes for these flags. The orchestrator adds the CI gates. */ + excludedTaskTypes?: (flags: Record) => readonly string[]; + /** Tasks to queue before the orchestrator's planner runs. */ + seedTasks?: () => SeedTaskEntry[]; + /** Completion hooks, bound by the caller. */ + hooks?: RunHooks; +} + +/** Invocation flags the agent reads. */ +export interface RunFlags { + ci: boolean; + signup: boolean; + debug: boolean; + /** Harness-only: keep the ask bridge in a `ci` run that has an answerer. */ + e2eAsk: boolean; + localMcp: boolean; + captureAio: boolean; + benchmark: boolean; + yaraReport: boolean; +} + +/** + * The invocation snapshot for one agent run. Taken once by the caller; the + * agent never refreshes these from a higher layer. + */ +export interface RunInput { + installDir: string; + /** Resolved credentials, including the host family and its MCP url. */ + credentials: Credentials; + /** Project payload resolved at authentication, for prompt context. */ + project: ApiProject | null; + /** User payload resolved at authentication, for the AI opt-in prompt line. */ + apiUser: ApiUser | null; + /** The skill this run is for: the run's skill id, else its integration label. */ + skillId?: string; + /** Detected framework, when the caller has one. */ + integration?: Integration | null; + /** Docs page for the detected framework, for the orchestrator's preflight message. */ + frameworkDocsUrl?: string; + flags: RunFlags; + /** Where PostHog is, as the CLI was told. */ + host: { + baseUrl?: string; + region?: CloudRegion; + email?: string; + /** `--project-id`, for the auth-error classifier. */ + projectId?: number; + /** `--api-key`, for the auth-error classifier. */ + apiKey?: string; + }; +} + +/** + * The values `prepareRun` resolves from `RunConfig` + `RunInput` and hands to + * both sequences: the gateway mint and the scan-triage classifier built on it. */ export interface BootstrapResult { skillsBaseUrl: string; - /** Auth outputs (incl. the resolved host family and its MCP url), narrowed at the boundary. */ + /** Resolved credentials (incl. the host family and its MCP url). */ credentials: Credentials; /** Program this run is, and the node its gateway spend pins to. */ programId: string; @@ -137,3 +279,83 @@ export interface BootstrapResult { /** Scan-triage classifier on this run's harness. Undefined → skill scans fail closed. */ triageProvider: LLMProvider | undefined; } + +/** + * A decided failure. The same fields `wizardAbort` takes, so the legacy + * adapter passes it through untouched and the exit sequence, codes and + * messages stay exactly what they were. + */ +export interface AgentFailure { + message?: string; + /** Structured error data. Renders via `outroError` instead of `outro`. */ + outroData?: OutroData; + error?: Error; + exitCode?: number; + code?: ErrorCode; + detail?: Record; +} + +export enum RunOutcome { + Success = 'success', + Aborted = 'aborted', + Failed = 'failed', + Crashed = 'crashed', +} + +/** Totals of every `usage` event the run emitted. */ +export interface TokenUsageTotals { + inputTokens: number; + outputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; +} + +/** What the agent reported, accumulated independently of any observer. */ +export interface RunSnapshot { + tasks: import('@lib/agent/progress').TaskSnapshot[]; + statusMessages: string[]; + stage?: string; + usage: TokenUsageTotals; + finalCostUsd?: number; + dashboardUrl?: string; + notebookUrl?: string; +} + +/** A sequence decides an outcome; the dispatcher owns its snapshot. */ +export type SequenceResult = + | { outcome: RunOutcome.Success; outro?: OutroData; failure?: never } + | { + outcome: RunOutcome.Aborted | RunOutcome.Failed; + failure: AgentFailure; + outro?: never; + }; + +/** Every non-success result carries a failure; crashes preserve the original error. */ +export type RunResult = ( + | SequenceResult + | { + outcome: RunOutcome.Crashed; + failure: AgentFailure & { error: Error }; + outro?: never; + } +) & { + /** The skill this run installed or was for. */ + skillId?: string; + snapshot: RunSnapshot; +}; + +export interface RunAgentOptions { + /** Receives every progress event in emission order. Never awaited. */ + onProgress?: (event: import('@lib/agent/progress').AgentProgress) => void; + /** Answers the agent's questions. Absent → no ask bridge, notices declined. */ + interaction?: AgentInteraction; +} + +/** What a sequence receives: the contracts plus the prepared run. */ +export interface SequenceContext { + config: RunConfig; + input: RunInput; + boot: BootstrapResult; + emit: ProgressEmitter; + interaction: AgentInteraction | undefined; +} diff --git a/src/lib/agent/runner/switchboard/sequence.ts b/src/lib/agent/runner/switchboard/sequence.ts index e39e228e8..3d2730405 100644 --- a/src/lib/agent/runner/switchboard/sequence.ts +++ b/src/lib/agent/runner/switchboard/sequence.ts @@ -12,9 +12,7 @@ import { resolveFlagSequence, } from './flags'; import { getHarness, resolveHarness } from './harness'; -import type { WizardSession } from '@lib/wizard-session'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun, BootstrapResult } from '../shared/types'; +import type { SequenceResult, SequenceContext } from '../shared/types'; import { runLinearProgram } from '../sequence/linear'; import { runOrchestrator } from '../sequence/orchestrator/orchestrator-runner'; import { @@ -29,26 +27,18 @@ import { export interface SequenceRunner { readonly name: Sequence; - run( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, - /** Composed sub-run (integration inside self-driving); linear-only. */ - composed: boolean, - ): Promise; + /** Run one program to a decided result. Unexpected errors propagate. */ + run(ctx: SequenceContext): Promise; } export const SEQUENCE_OPTIONS: Partial> = { [Sequence.linear]: { name: Sequence.linear, - run: (session, config, programConfig, boot, composed) => - runLinearProgram(session, config, programConfig, boot, composed), + run: (ctx) => runLinearProgram(ctx), }, [Sequence.orchestrator]: { name: Sequence.orchestrator, - run: (session, config, programConfig, boot, _composed) => - runOrchestrator(session, config, programConfig, boot), + run: (ctx) => runOrchestrator(ctx), }, }; diff --git a/src/lib/detection/__tests__/agentic-progress.test.ts b/src/lib/detection/__tests__/agentic-progress.test.ts new file mode 100644 index 000000000..387bb24fc --- /dev/null +++ b/src/lib/detection/__tests__/agentic-progress.test.ts @@ -0,0 +1,80 @@ +import { detectProjectsWithAgent } from '../agentic'; +import { + initializeAgent, + runAgent as executeAgent, +} from '@lib/agent/agent-interface'; +import { buildSession } from '@lib/wizard-session'; +import { HostResolution } from '@lib/host-resolution'; +import { getUI } from '@ui'; + +vi.mock('@utils/debug'); +vi.mock('@ui', () => ({ getUI: () => ui })); +const ui = vi.hoisted(() => ({ + addTokenUsage: vi.fn(), + setStage: vi.fn(), + pushStatus: vi.fn(), + log: { error: vi.fn() }, +})); +vi.mock('@lib/agent/agent-interface', async (original) => ({ + ...(await original()), + initializeAgent: vi.fn(), + runAgent: vi.fn(), +})); + +it('keeps initialization and execution progress visible during detection', async () => { + const delta = { + inputTokens: 5, + outputTokens: 2, + cacheReadTokens: 0, + cacheCreationTokens: 0, + cacheCreation5m: 0, + cacheCreation1h: 0, + }; + vi.mocked(initializeAgent).mockImplementation((config) => { + config.emit?.({ + kind: 'log', + level: 'error', + message: 'Initialization diagnostic', + }); + return Promise.resolve({ emit: config.emit } as Awaited< + ReturnType + >); + }); + vi.mocked(executeAgent).mockImplementation( + (config, _prompt, _options, _spinner, _messages, middleware) => { + config.emit?.({ kind: 'usage', delta }); + config.emit?.({ kind: 'stage', stage: 'Scanning' }); + config.emit?.({ kind: 'status', message: 'Found a project' }); + config.emit?.({ + kind: 'log', + level: 'error', + message: 'Execution diagnostic', + }); + middleware?.onMessage({ + type: 'result', + result: + '{"projects":[{"path":".","targetId":"node","framework":"Node.js"}]}', + }); + return Promise.resolve({}); + }, + ); + const session = buildSession({ installDir: '/tmp/detection-test' }); + session.credentials = { + accessToken: 'test', + projectApiKey: 'phc_test', + projectId: 1, + host: HostResolution.fromApiHost('https://us.posthog.com'), + }; + const report = await detectProjectsWithAgent(session, { + programId: 'posthog-integration', + targets: [{ id: 'node', name: 'Node.js' }], + }); + expect(report.projects[0].targetId).toBe('node'); + expect(getUI().addTokenUsage).toHaveBeenCalledWith(delta); + expect(ui.setStage).toHaveBeenCalledWith('Scanning'); + expect(ui.pushStatus).toHaveBeenCalledWith('Found a project'); + expect(ui.log.error.mock.calls).toEqual([ + ['Initialization diagnostic'], + ['Execution diagnostic'], + ]); +}); diff --git a/src/lib/detection/agentic.ts b/src/lib/detection/agentic.ts index c95c7c6a3..cce567bbc 100644 --- a/src/lib/detection/agentic.ts +++ b/src/lib/detection/agentic.ts @@ -32,7 +32,8 @@ import { import { analytics } from '@utils/analytics'; import type { WizardSession } from '@lib/wizard-session'; import type { WizardRunOptions } from '@utils/types'; -import type { SpinnerHandle } from '@ui'; +import { getUI, type SpinnerHandle } from '@ui'; +import { createUiReducer } from '@ui/agent-progress'; /** A category the agent classifies each project into (id the agent returns). */ export type DetectTarget = { id: string; name: string }; @@ -382,6 +383,7 @@ export async function detectProjectsWithAgent( : AGENTIC_DETECTION_RETRY_TIMEOUT_MS; const agent = await initializeAgent( { + emit: createUiReducer(getUI()), workingDirectory: cwd, posthogMcpUrl: host.mcpUrl, posthogApiKey: accessToken, diff --git a/src/lib/programs/__tests__/run-agent-legacy.test.ts b/src/lib/programs/__tests__/run-agent-legacy.test.ts new file mode 100644 index 000000000..f96618232 --- /dev/null +++ b/src/lib/programs/__tests__/run-agent-legacy.test.ts @@ -0,0 +1,326 @@ +import { runNonInteractive } from '@lib/runners/run-non-interactive'; +import { runWizard } from '@lib/runners/run-wizard'; +import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { runProgramAgent } from '../run-agent-legacy'; +import { runAgent, RunOutcome, type RunResult } from '@lib/agent/runner'; +import { Harness, Sequence } from '@lib/constants'; +import { buildSession, OutroKind } from '@lib/wizard-session'; +import { HostResolution } from '@lib/host-resolution'; +import { LoggingUI } from '@ui/logging-ui'; +import { InkUI } from '@ui/tui/ink-ui'; +import { startTUI } from '@ui/tui/start-tui'; +import { WizardStore } from '@ui/tui/store'; +import { setUI } from '@ui'; +import { analytics } from '@utils/analytics'; +import { initLogFile, logToFile } from '@utils/debug'; +import { wizardAbort } from '@utils/wizard-abort'; +import type { ProgramConfig } from '../program-step'; + +const streamShutdown = vi.hoisted(() => vi.fn().mockResolvedValue(undefined)); +vi.mock('@env', async (original) => ({ + ...(await original()), + IS_PRODUCTION_BUILD: false, +})); +vi.mock('@lib/local-dev', async (original) => ({ + ...(await original()), + checkLocalServices: vi.fn().mockResolvedValue(null), +})); +vi.mock('@utils/environment', async (original) => ({ + ...(await original()), + readEnvironment: () => ({}), +})); +vi.mock('@lib/gateway-session', async (original) => ({ + ...(await original()), + configureGatewayFromCIEnvironment: vi.fn(), +})); +vi.mock('@lib/task-stream/index', () => ({ + TaskStreamPush: class { + attach = vi.fn(); + shutdown = streamShutdown; + }, + PostHogDestination: class {}, + createFileDestination: () => null, +})); +vi.mock('@ui/tui/start-tui', () => ({ startTUI: vi.fn() })); +vi.mock('@utils/debug'); +vi.mock('@utils/analytics', () => ({ + analytics: { + build: 'test', + runId: 'run-1', + wizardCapture: vi.fn(), + captureException: vi.fn(), + setTag: vi.fn(), + getAllFlagsForWizard: vi.fn().mockResolvedValue({}), + getWizardFlagPayloads: vi.fn().mockReturnValue({}), + shutdown: vi.fn().mockResolvedValue(undefined), + }, + sessionProperties: () => ({}), +})); +vi.mock('@lib/agent/runner', async (original) => ({ + ...(await original()), + runAgent: vi.fn(), +})); +vi.mock('@lib/agent/runner/shared/authenticate', () => ({ + authenticate: vi.fn().mockResolvedValue(undefined), + refreshAccessTokenIfNeeded: vi.fn().mockResolvedValue(undefined), +})); +vi.mock('@lib/agent/claude-settings', () => ({ + checkAllSettingsConflicts: vi.fn().mockReturnValue([]), + restoreClaudeSettings: vi.fn(), +})); +vi.mock('@utils/wizard-abort', async (original) => ({ + ...(await original()), + registerCleanup: vi.fn(), + wizardAbort: vi.fn().mockResolvedValue(undefined), +})); +vi.mock('../posthog-integration/detect', () => ({ + maybeStampAiSdkDetected: vi.fn(), +})); + +const program = (id: ProgramConfig['id'] = 'metrics'): ProgramConfig => ({ + id, + steps: [], + description: 'Test', + run: { + integrationLabel: 'test', + spinnerMessage: 'Working', + successMessage: 'Done', + estimatedDurationMinutes: 1, + reportFile: 'report.md', + docsUrl: 'https://docs.test', + }, +}); +const snapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, +}; +const session = () => ({ + ...buildSession({ ci: true, installDir: '/tmp/adapter-test' }), + credentials: { + accessToken: 'test', + projectApiKey: 'phc_test', + projectId: 1, + host: HostResolution.fromApiHost('https://us.posthog.com'), + }, +}); + +let logSpy: ReturnType; + +/** A run that reports, shows its outro and succeeds. */ +const finishRun: typeof runAgent = (_config, _input, options) => { + options?.onProgress?.({ kind: 'status', message: 'Working' }); + options?.onProgress?.({ + kind: 'completion', + outro: { kind: OutroKind.Success, message: 'Done' }, + }); + options?.onProgress?.({ + kind: 'lifecycle', + phase: 'completed', + message: 'Done', + }); + return Promise.resolve({ outcome: RunOutcome.Success, snapshot }); +}; + +beforeEach(() => { + vi.clearAllMocks(); + vi.mocked(authenticate).mockImplementation((sess) => { + sess.credentials = session().credentials; + return Promise.resolve(); + }); + setUI(new LoggingUI()); + logSpy = vi.spyOn(console, 'log').mockImplementation(() => undefined); + vi.mocked(runAgent).mockImplementation(finishRun); +}); +afterEach(() => logSpy.mockRestore()); + +it.each([ + ['metrics', Harness.pi, Sequence.orchestrator], + ['replay-vision', Harness.anthropic, Sequence.orchestrator], +] as const)( + 'forwards the %s program binding through the real adapter', + async (id, harness, sequence) => { + await runProgramAgent(program(id), session()); + expect(runAgent).toHaveBeenCalledWith( + expect.objectContaining({ + programId: id, + binding: expect.objectContaining({ harness, sequence }), + }), + expect.objectContaining({ flags: expect.objectContaining({ ci: true }) }), + expect.objectContaining({ + onProgress: expect.any(Function), + interaction: expect.any(Object), + }), + ); + expect(logSpy).toHaveBeenCalledWith('◇ Working'); + expect(logSpy).toHaveBeenCalledWith('└ Done'); + expect(initLogFile).toHaveBeenCalledOnce(); + expect(analytics.shutdown).toHaveBeenCalledExactlyOnceWith('success'); + }, +); + +it('sends terminal analytics after the outro and the run, before the host goes on', async () => { + const order: string[] = []; + logSpy.mockImplementation((line) => { + if (line === '└ Done') order.push('outro'); + }); + vi.mocked(runAgent).mockImplementation(async (...args) => { + const result = await finishRun(...args); + order.push('run-returned'); + return result; + }); + vi.mocked(analytics.shutdown).mockImplementation(() => { + order.push('shutdown'); + return Promise.resolve(); + }); + await runProgramAgent(program(), session()); + order.push('host-continues'); + expect(order).toEqual([ + 'outro', + 'run-returned', + 'shutdown', + 'host-continues', + ]); + expect(analytics.shutdown).toHaveBeenCalledExactlyOnceWith('success'); +}); + +it('clamps a composed program to linear and keeps host analytics alive', async () => { + await runProgramAgent(program(), session(), { composed: true }); + expect(runAgent).toHaveBeenCalledWith( + expect.objectContaining({ + composed: true, + binding: expect.objectContaining({ sequence: Sequence.linear }), + }), + expect.anything(), + expect.anything(), + ); + expect(analytics.shutdown).not.toHaveBeenCalled(); + + // The host program's own run, later in the same process, ends it once. + await runProgramAgent(program(), session()); + expect(analytics.shutdown).toHaveBeenCalledExactlyOnceWith('success'); +}); + +it.each([RunOutcome.Aborted, RunOutcome.Failed] as const)( + 'passes a %s result to the existing abort handler', + async (outcome) => { + const failure = { message: 'Failed', exitCode: 2 }; + vi.mocked(runAgent).mockResolvedValue({ outcome, failure, snapshot }); + await runProgramAgent(program(), session()); + expect(wizardAbort).toHaveBeenCalledExactlyOnceWith(failure); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }, +); + +it('rethrows the original crash for the outer runner', async () => { + const error = new Error('mint refused'); + const result: RunResult = { + outcome: RunOutcome.Crashed, + failure: { error }, + snapshot, + }; + vi.mocked(runAgent).mockResolvedValue(result); + await expect(runProgramAgent(program(), session())).rejects.toBe(error); + expect(wizardAbort).not.toHaveBeenCalled(); + expect(analytics.shutdown).not.toHaveBeenCalled(); +}); + +it.each([ + [Harness.pi, Sequence.linear], + [Harness.pi, Sequence.orchestrator], + [Harness.anthropic, Sequence.linear], + [Harness.anthropic, Sequence.orchestrator], +])( + 'runs headless %s/%s through the real CLI adapter', + async (harness, sequence) => { + runNonInteractive( + program(), + { + apiKey: 'phx_test', + projectId: '1', + installDir: '/tmp/adapter-test', + telemetry: false, + harness, + sequence, + }, + 'headless', + ); + await vi.waitFor(() => expect(streamShutdown).toHaveBeenCalledOnce()); + expect(wizardAbort).not.toHaveBeenCalled(); + // One terminal event for the process, sent before the stream settles. + expect(analytics.shutdown).toHaveBeenCalledExactlyOnceWith('success'); + expect( + vi.mocked(analytics.shutdown).mock.invocationCallOrder[0], + ).toBeLessThan(streamShutdown.mock.invocationCallOrder[0]); + expect(runAgent).toHaveBeenCalledWith( + expect.objectContaining({ + binding: expect.objectContaining({ harness, sequence }), + }), + expect.objectContaining({ flags: expect.objectContaining({ ci: true }) }), + expect.anything(), + ); + expect(logSpy).toHaveBeenCalledWith('◇ Working'); + expect(logSpy).toHaveBeenCalledWith('└ Done'); + }, +); + +it('keeps a headless run a success when its terminal analytics flush fails', async () => { + const flushError = new Error('flush timed out'); + vi.mocked(analytics.shutdown).mockRejectedValueOnce(flushError); + runNonInteractive( + program(), + { + apiKey: 'phx_test', + projectId: '1', + installDir: '/tmp/adapter-test', + telemetry: false, + }, + 'headless', + ); + await vi.waitFor(() => expect(streamShutdown).toHaveBeenCalledOnce()); + expect(wizardAbort).not.toHaveBeenCalled(); + expect(analytics.shutdown).toHaveBeenCalledExactlyOnceWith('success'); + expect(logToFile).toHaveBeenCalledWith( + expect.stringContaining('analytics shutdown failed'), + flushError, + ); +}); + +it('keeps a TUI run a success when its terminal analytics flush fails', async () => { + const flushError = new Error('flush timed out'); + vi.mocked(analytics.shutdown).mockRejectedValueOnce(flushError); + const store = new WizardStore('metrics'); + const ui = new InkUI(store); + setUI(ui); + const outroError = vi.spyOn(ui, 'outroError'); + vi.spyOn(store, 'runReadyHooks').mockResolvedValue(undefined); + vi.spyOn(store, 'getGate').mockResolvedValue(undefined); + vi.mocked(startTUI).mockReturnValue({ + store, + unmount: vi.fn(), + waitForSetup: () => Promise.resolve(), + }); + const exit = vi + .spyOn(process, 'exit') + .mockImplementation(() => undefined as never); + + runWizard(program(), { installDir: '/tmp/adapter-test', telemetry: false }); + await vi.waitFor(() => expect(analytics.shutdown).toHaveBeenCalled()); + store.setSkillsComplete(true); + await vi.waitFor(() => expect(exit).toHaveBeenCalled()); + + expect(exit).toHaveBeenCalledExactlyOnceWith(0); + expect(outroError).not.toHaveBeenCalled(); + expect(store.session.outroData?.kind).toBe(OutroKind.Success); + expect(analytics.shutdown).toHaveBeenCalledExactlyOnceWith('success'); + expect(logToFile).toHaveBeenCalledWith( + expect.stringContaining('analytics shutdown failed'), + flushError, + ); + exit.mockRestore(); +}); diff --git a/src/lib/programs/posthog-integration/index.ts b/src/lib/programs/posthog-integration/index.ts index 408d8bdd6..3aab99341 100644 --- a/src/lib/programs/posthog-integration/index.ts +++ b/src/lib/programs/posthog-integration/index.ts @@ -1,5 +1,6 @@ import type { ProgramConfig, ProgramStep } from '@lib/programs/program-step'; -import { runAgent, type ProgramRun } from '@lib/agent/agent-runner'; +import { runProgramAgent } from '@lib/programs/run-agent-legacy'; +import type { ProgramRun } from '@lib/agent/agent-runner'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import type { WizardSession } from '@lib/wizard-session'; import { mayReportScanResults, OutroKind, RunPhase } from '@lib/wizard-session'; @@ -24,7 +25,7 @@ import { requestDeepLink } from '@utils/provisioning'; import { openTrackedLink, withUtm } from '@utils/links'; import type { HostResolution } from '@lib/host-resolution'; import { getDetectedWarehouseSources } from '@lib/programs/warehouse-source/detect'; -import { shouldDisableAsk } from '@lib/agent/runner/shared/bootstrap'; +import { shouldDisableAsk } from '@lib/agent/agent-runner'; import { POSTHOG_INTEGRATION_PROGRAM } from './steps.js'; import { getContentBlocks } from './content/index.js'; import { buildCodingAgentPrompt } from './handoff.js'; @@ -535,7 +536,7 @@ export const integrationRunStep: ProgramStep = { // composed: runs inside the host program (self-driving), so skip the // integration's terminal outro + analytics shutdown of the shared client. run: (session) => - runAgent(posthogIntegrationConfig, session, { composed: true }), + runProgramAgent(posthogIntegrationConfig, session, { composed: true }), isComplete: (session) => session.runPhase === RunPhase.Completed || session.runPhase === RunPhase.Error, diff --git a/src/lib/programs/run-agent-legacy.ts b/src/lib/programs/run-agent-legacy.ts new file mode 100644 index 000000000..4f63c27f4 --- /dev/null +++ b/src/lib/programs/run-agent-legacy.ts @@ -0,0 +1,480 @@ +/** + * The session-driven agent runner every existing caller uses. + * + * `runProgramAgent(programConfig, session)` rebuilds today's behavior on top of the + * functional `runAgent(config, input, options)` in `@lib/agent/runner`: it + * runs the gates the TUI owns (health, settings, AI opt-in, post-auth steps), + * authenticates, resolves the program's binding, builds the agent's inputs + * from the session, maps every progress event back onto `getUI()` one call + * per event, answers the agent's questions through `getUI()`, and applies the + * result — `wizardAbort` for a decided failure, the terminal analytics event + * for a finished top-level run. + * + * This is the only file that knows about `getUI()`, the session and + * `wizardAbort` on the agent's behalf. Programs replace it in Release B. + */ + +import type { WizardSession } from '@lib/wizard-session'; +import { analytics } from '@utils/analytics'; +import { getUI } from '@ui'; +import { createUiReducer, uiInteraction } from '@ui/agent-progress'; +import { + runAgent, + RunOutcome, + resolveBinding, + TASK_OUTCOMES_KEY, + type ProgramRun, + type RunConfig, + type RunInput, + type SwitchboardCtx, +} from '@lib/agent/runner'; +import type { ProgramBinding } from '@lib/agent/runner/switchboard'; +import { buildRunTags } from '@lib/agent/agent-interface'; +import { + backupAndFixClaudeSettings, + checkAllSettingsConflicts, + classifySettingsConflicts, + restoreClaudeSettings, +} from '@lib/agent/claude-settings'; +import { flushScanReport } from '@lib/yara-hooks'; +import { + evaluateWizardReadiness, + WizardReadiness, + SIGNUP_WIZARD_READINESS_CONFIG, + getBlockingServiceKeys, + SERVICE_LABELS, +} from '@lib/health-checks/readiness'; +import { enableDebugLogs, logToFile, initLogFile } from '@utils/debug'; +import { registerCleanup, wizardAbort } from '@utils/wizard-abort'; +import { ErrorCodes } from '@lib/errors'; +import { isNonInteractiveEnvironment } from '@utils/environment'; +import { + getSkillsBaseUrl, + Sequence, + WIZARD_ORCHESTRATOR_FLAG_KEY, + WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY, + type Integration, +} from '@lib/constants'; +import { FRAMEWORK_REGISTRY } from '@lib/registry'; +import { postAuthGateSteps, type ProgramConfig } from './program-step'; +import { + authenticate, + refreshAccessTokenIfNeeded, +} from '@lib/agent/runner/shared/authenticate'; +import { maybeStampAiSdkDetected } from './posthog-integration/detect'; +import { startAuditLedgerWatcher } from './audit/ledger-watcher'; + +/** + * Resolve a ProgramConfig's agent run definition and execute the pipeline. + * Entry point for the runners and for composed run steps. + */ +export async function runProgramAgent( + programConfig: ProgramConfig, + session: WizardSession, + options: { composed?: boolean } = {}, +): Promise { + if (!programConfig.run) { + throw new Error(`Program "${programConfig.id}" has no run configuration.`); + } + + // Before `run()` resolves: an audit seeds the ledger from inside its recipe, + // and a watcher started later would ignore that write as pre-existing. + const ledger = programConfig.auditLedgerFile + ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) + : null; + if (ledger) registerCleanup(() => ledger.stop()); + + try { + const runDef = + typeof programConfig.run === 'function' + ? await programConfig.run(session) + : programConfig.run; + + await runProgram(session, runDef, programConfig, options.composed ?? false); + } finally { + ledger?.stop(); + } +} + +/** + * Gates → authenticate → flags → binding → the functional run → apply result. + * Every step happens in the order it did inside the agent's bootstrap. + */ +async function runProgram( + session: WizardSession, + run: ProgramRun, + programConfig: ProgramConfig, + composed: boolean, +): Promise { + // 1. Init logging + debug + initLogFile(); + session.skillId = run.skillId ?? run.integrationLabel; + logToFile( + `[agent-runner] START ${run.integrationLabel} build=${analytics.build}` + + `${session.ci ? ' (non-interactive)' : ''}`, + ); + if (session.debug) { + enableDebugLogs(); + } + + // 2. Health check (guarded — skip if TUI already ran it). Only + // programs that declare a health-check screen get pre-flight checks; + // for everything else the checks never fire and never block. + await runHealthGate(session, programConfig); + + // 3. Settings conflicts + await runSettingsGate(session); + + analytics.wizardCapture('agent started', { + integration: run.integrationLabel, + program_id: programConfig.id, + skill_id: run.skillId ?? null, + }); + + // 4. Authenticate — idempotent within a run (see authenticate()). A second + // agent run in the same invocation (self-driving's integration phase) reuses + // the first login; it does not launch another OAuth. authenticate() also + // identifies the user and sets analytics groups. + await authenticate(session, programConfig.id); + maybeStampAiSdkDetected(session); + + // 4.5. AI opt-in enforcement. Parks here while AiOptInRequiredScreen is + // up if the org hasn't approved third-party AI — BEFORE the skill + // install and agent start, so no source leaves the machine. The screen + // alone is cosmetic; this await is the actual gate. Resolves + // immediately when the program declared requiresAi: false or in CI. + logToFile('[agent-runner] checking AI opt-in gate'); + await getUI().waitForAiOptIn(); + logToFile('[agent-runner] AI opt-in gate cleared'); + + // Park for any interactive step the user must complete AFTER authenticating + // but BEFORE the agent runs — e.g. the source-maps project picker, which + // needs credentials to scan and writes its choice to frameworkContext that + // the run prompt reads. Generic: await every gated step between auth and run. + for (const step of postAuthGateSteps(programConfig.steps)) { + logToFile(`[agent-runner] awaiting post-auth gate: ${step.id}`); + await getUI().waitForGate(step.id); + logToFile(`[agent-runner] post-auth gate cleared: ${step.id}`); + } + + // Feature flags. Both arms need these, and the routing decision reads them. + const wizardFlags = await analytics.getAllFlagsForWizard(); + const wizardFlagPayloads = analytics.getWizardFlagPayloads(); + + // Gateway trace tags for this run; the binding below stamps its axes on. + const wizardMetadata = buildRunTags({ + programId: programConfig.id, + integration: run.integrationLabel, + runId: analytics.runId, + build: analytics.build, + skillId: run.skillId, + }); + + // The agent can't swap tokens mid-run, so freshness is measured after every + // park above, right before the agent mints. + await refreshAccessTokenIfNeeded(session); + + // Credentials (incl. the resolved host family and its MCP url) live on + // `session.credentials`; narrow once at this boundary — `authenticate` above + // set them — so downstream readers get a non-null type without asserting. + const credentials = session.credentials!; + + // Resolve which sequence and harness will run a program (CLI → PostHog flag → + // per-program binding → default), tag both axes onto analytics, and hand the + // binding to the agent for dispatch. + const switchboard: SwitchboardCtx = { + program: programConfig.id, + composed, + flags: wizardFlags, + flagPayloads: wizardFlagPayloads, + cliHarness: session.harness, + cliSequence: session.sequence, + cliModel: session.model, + }; + const binding = resolveBinding(switchboard); + analytics.setTag('sequence', binding.sequence); + analytics.setTag('harness', binding.harness); + wizardMetadata.SEQUENCE = binding.sequence; + wizardMetadata.HARNESS = binding.harness; + captureSwitchboardDecision(switchboard, binding); + + const ui = getUI(); + + // Cleanup coverage for the abort/cancel path: `wizardAbort` runs the + // registered cleanups, and the agent's own `finally` covers completion. + // flushScanReport is idempotent, so the overlap is a harmless no-op. + registerCleanup(() => flushScanReport({ yaraReport: session.yaraReport })); + + // Linear settings restoration fires on entry to the outro screen, so it + // is registered before the run can reach that screen. Same owner, same + // timing as before; the abort path still restores through the cleanup + // `backupAndFixClaudeSettings` registered. + if (binding.sequence === Sequence.linear) { + ui.onEnterScreen('outro', () => restoreClaudeSettings(session.installDir)); + } + + const framework = session.integration ?? session.skillId ?? undefined; + const config: RunConfig = { + programId: programConfig.id, + run, + composed, + binding, + switchboard, + skillsBaseUrl: getSkillsBaseUrl(), + wizardFlags, + wizardFlagPayloads, + wizardMetadata, + allowedTools: programConfig.allowedTools, + disallowedTools: programConfig.disallowedTools, + agentFlow: programConfig.agentFlow, + excludedTaskTypes: programConfig.excludedTaskTypes, + seedTasks: programConfig.seedTasks + ? () => programConfig.seedTasks!(session) + : undefined, + hooks: { + postRun: run.postRun + ? (creds) => run.postRun!(session, creds) + : undefined, + buildOutroData: run.buildOutroData + ? (creds) => run.buildOutroData!(session, creds) ?? undefined + : undefined, + buildOutroNextSteps: run.buildOutroNextSteps + ? (creds, completed) => + run.buildOutroNextSteps!(session, creds, completed) + : undefined, + recordTaskOutcomes: (outcomes) => { + session.frameworkContext[TASK_OUTCOMES_KEY] = outcomes; + }, + }, + }; + const input: RunInput = { + installDir: session.installDir, + credentials, + project: session.apiProject, + apiUser: session.apiUser, + skillId: session.skillId ?? undefined, + integration: session.integration, + frameworkDocsUrl: framework + ? FRAMEWORK_REGISTRY[framework as Integration]?.metadata.docsUrl + : undefined, + flags: { + ci: session.ci, + signup: session.signup, + debug: session.debug, + e2eAsk: session.e2eAsk, + localMcp: session.localMcp, + captureAio: session.captureAio, + benchmark: session.benchmark, + yaraReport: session.yaraReport, + }, + host: { + baseUrl: session.baseUrl, + region: session.region, + email: session.email, + projectId: session.projectId, + apiKey: session.apiKey, + }, + }; + + const result = await runAgent(config, input, { + onProgress: createUiReducer(ui), + interaction: uiInteraction(ui), + }); + + // The host owns process exits, terminal analytics and rethrowing crashes. + if (result.outcome === RunOutcome.Crashed) { + throw result.failure.error; + } + if (result.outcome !== RunOutcome.Success) { + await wizardAbort(result.failure); + } else if (!composed) { + // A composed sub-run leaves the terminal event to its host program's run. + // The run already succeeded: a failed flush is logged, never the outcome. + try { + await analytics.shutdown('success'); + } catch (error) { + logToFile('[agent-runner] analytics shutdown failed:', error); + } + } +} + +// ── Gates ───────────────────────────────────────────────────────────── + +async function runHealthGate( + session: WizardSession, + programConfig: ProgramConfig, +): Promise { + const hasHealthCheckScreen = programConfig.steps.some( + (s) => s.screenId === 'health-check', + ); + if (session.readinessResult) { + logToFile( + `[agent-runner] readiness pre-computed by TUI: decision=${session.readinessResult.decision}` + + `${ + session.outageDismissed ? ' (outage dismissed by user)' : '' + } — skipping re-check`, + ); + } + if (!hasHealthCheckScreen || session.readinessResult) return; + + logToFile('[agent-runner] evaluating wizard readiness'); + const readinessConfig = session.signup + ? SIGNUP_WIZARD_READINESS_CONFIG + : undefined; + const readiness = await evaluateWizardReadiness(readinessConfig); + logToFile(`[agent-runner] readiness=${readiness.decision}`); + if (readiness.decision === WizardReadiness.No) { + const blockingKeys = getBlockingServiceKeys( + readiness.health, + readinessConfig, + ); + const blockingLabels = blockingKeys.map( + (k) => `${SERVICE_LABELS[k]} (${readiness.health[k].status})`, + ); + logToFile(`[agent-runner] blocked by: ${blockingLabels.join(', ')}`); + + await getUI().showBlockingOutage(readiness); + + // The TUI lets the user continue past an outage; non-interactive runs + // (CI) do the same automatically — the degraded services are reported + // above, but we proceed rather than aborting on a transient upstream blip. + if (!isNonInteractiveEnvironment()) { + await wizardAbort({ + code: ErrorCodes.EnvServiceOutage, + message: + 'Cannot start — external services are down:\n' + + blockingLabels.map((l) => ` - ${l}`).join('\n') + + '\n\nPlease try again later.', + }); + } + } else if (readiness.decision === WizardReadiness.YesWithWarnings) { + getUI().setReadinessWarnings(readiness); + } +} + +async function runSettingsGate(session: WizardSession): Promise { + const settingsConflicts = checkAllSettingsConflicts(session.installDir); + logToFile( + `[agent-runner] settings conflicts: ${ + settingsConflicts.length > 0 + ? settingsConflicts + .map((c) => `${c.source}(${c.keys.join(',')})`) + .join('; ') + : 'none' + }`, + ); + if (settingsConflicts.length === 0) return; + + for (const conflict of settingsConflicts) { + const level = conflict.source === 'managed' ? 'org' : conflict.source; + analytics.wizardCapture('settings conflict detected', { + level, + keys: conflict.keys, + }); + } + + const { autoFix, failClosed, warnOnly } = + classifySettingsConflicts(settingsConflicts); + + // User-global and project-local files are already neutralized — the agent + // runs with settingSources:['project'], so the SDK never reads them. Record + // it and move on; don't make the user act on a setting that can't bite. + for (const conflict of warnOnly) { + logToFile( + `[agent-runner] settings conflict in ${conflict.source} (${conflict.path}) ` + + `neutralized by settingSources:['project'] — not blocking`, + ); + analytics.wizardCapture('settings conflict neutralized', { + level: conflict.source, + keys: conflict.keys, + }); + } + + // Writable project settings.json — the SDK *does* read it, but we can back + // it up and remove it (restored at outro). Neutralize without prompting. + let unfixable = failClosed; + if (autoFix.length > 0) { + const fixed = backupAndFixClaudeSettings(session.installDir); + if (fixed) { + logToFile('[agent-runner] auto-neutralized writable settings conflict'); + analytics.wizardCapture('settings conflict auto-neutralized', { + keys: autoFix.flatMap((c) => c.keys), + }); + } else { + // Couldn't remove it — don't run into the redirect; fail closed instead. + logToFile( + '[agent-runner] could not back up writable settings conflict — failing closed', + ); + unfixable = [...failClosed, ...autoFix]; + } + } + + // What we cannot neutralize (org-managed, always read by the SDK; or a + // writable file we failed to back up) must be fixed by the user. Fail + // closed: the screen names the file + keys and exits. + if (unfixable.length > 0) { + if (isNonInteractiveEnvironment()) { + await wizardAbort({ + code: ErrorCodes.SettingsUnfixableConflict, + message: + 'Cannot start — a Claude settings file redirects the agent away ' + + 'from the PostHog gateway and cannot be neutralized automatically:\n' + + unfixable + .map((c) => ` - ${c.source} (${c.path}): ${c.keys.join(', ')}`) + .join('\n') + + '\n\nRemove the conflicting keys and re-run the wizard.', + }); + } + await getUI().showSettingsOverride(unfixable, () => + backupAndFixClaudeSettings(session.installDir), + ); + logToFile('[agent-runner] settings override resolved'); + } +} + +// ── Switchboard telemetry ───────────────────────────────────────────── + +/** + * One event + one log line per run: what entered the switchboard, which + * precedence rung decided each axis, and the final pick. + */ +function captureSwitchboardDecision( + ctx: SwitchboardCtx, + binding: ProgramBinding, +): void { + const trace = ctx.trace ?? {}; + // Unpinned orchestrator runs choose a model per task from the context-mill agent prompts; the orchestrator logs that map once the prompts load. + const perTaskModel = + binding.sequence === Sequence.orchestrator && trace.model === 'binding'; + const model = perTaskModel ? 'chosen-per-task' : binding.model; + const modelSource = perTaskModel ? 'agent-prompts' : trace.model; + analytics.wizardCapture('switchboard resolved', { + program: ctx.program, + flag_self_driving_use_pi_harness: + ctx.flags[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY], + flag_self_driving_pi_payload: JSON.stringify( + ctx.flagPayloads?.[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY] ?? null, + ), + flag_orchestrator: ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY], + cli_harness: ctx.cliHarness, + cli_sequence: ctx.cliSequence, + cli_model: ctx.cliModel, + harness_source: trace.harness, + model_source: modelSource, + sequence_source: trace.sequence, + harness: binding.harness, + model, + thinking_level: binding.thinkingLevel, + sequence: binding.sequence, + }); + logToFile( + `[switchboard] decision: program=${ctx.program}` + + ` in(orchestrator=${ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY] ?? '-'},` + + ` cli=${ctx.cliHarness ?? '-'}/${ctx.cliSequence ?? '-'}/${ + ctx.cliModel ?? '-' + })` + + ` → harness=${binding.harness} (${trace.harness ?? '?'})` + + ` model=${model} (${modelSource ?? '?'})` + + ` sequence=${binding.sequence} (${trace.sequence ?? '?'})`, + ); +} diff --git a/src/lib/runners/__tests__/mint-recovery.test.ts b/src/lib/runners/__tests__/mint-recovery.test.ts index 24f9b5d83..9a564640e 100644 --- a/src/lib/runners/__tests__/mint-recovery.test.ts +++ b/src/lib/runners/__tests__/mint-recovery.test.ts @@ -1,6 +1,6 @@ import { vi, it, expect, afterEach } from 'vitest'; import { runWizard } from '../run-wizard'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runProgramAgent } from '@lib/programs/run-agent-legacy'; import { startTUI } from '@ui/tui/start-tui'; import { WizardStore } from '@ui/tui/store'; import { InkUI } from '@ui/tui/ink-ui'; @@ -10,7 +10,7 @@ import { ScreenId } from '@ui/tui/router'; import { HostResolution } from '@lib/host-resolution'; import { analytics } from '@utils/analytics'; -vi.mock('@lib/agent/agent-runner', () => ({ runAgent: vi.fn() })); +vi.mock('@lib/programs/run-agent-legacy', () => ({ runProgramAgent: vi.fn() })); vi.mock('@ui/tui/start-tui', () => ({ startTUI: vi.fn() })); vi.mock('@lib/local-dev', async (original) => ({ ...(await original()), @@ -59,7 +59,7 @@ it.each(['continue', 'exit'] as const)( }); // runWizard installs its own session first; auth then sets credentials, // and the agent dies after that. - vi.mocked(runAgent).mockImplementation(() => { + vi.mocked(runProgramAgent).mockImplementation(() => { store.setCredentials({ accessToken: 'tok', projectApiKey: 'pk', diff --git a/src/lib/runners/run-non-interactive.ts b/src/lib/runners/run-non-interactive.ts index 7230c3ec0..10ebfa4ca 100644 --- a/src/lib/runners/run-non-interactive.ts +++ b/src/lib/runners/run-non-interactive.ts @@ -79,7 +79,7 @@ export function validateNonInteractiveOptions( * (`runWizardHeadless`) runs. * * Validates flags, builds a `ci:true` session, runs `config.ciPreRun` (or the - * program's `onReady` hooks by default), executes `runAgent`, and routes any + * program's `onReady` hooks by default), executes `runProgramAgent`, and routes any * failure through `wizardAbort`. `wizardAbort` owns all exits — never add a * raw `process.exit` here. * @@ -335,8 +335,10 @@ export function runNonInteractive( } } - const { runAgent } = await import('@lib/agent/agent-runner'); - await runAgent(config, session); + const { runProgramAgent } = await import( + '@lib/programs/run-agent-legacy' + ); + await runProgramAgent(config, session); await settleStream(RunPhase.Completed); } catch (error) { const errorMessage = diff --git a/src/lib/runners/run-wizard.ts b/src/lib/runners/run-wizard.ts index 1ee2e7120..42b257cd5 100644 --- a/src/lib/runners/run-wizard.ts +++ b/src/lib/runners/run-wizard.ts @@ -1,6 +1,6 @@ import { VERSION } from '@lib/version'; import { logToFile, getLogFilePath } from '@utils/debug'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runProgramAgent } from '@lib/programs/run-agent-legacy'; import { authenticate } from '@lib/agent/runner/shared/authenticate'; import { getProgramConfig } from '@lib/programs/program-registry'; import { getAuditChecks } from '@lib/programs/audit/types'; @@ -60,7 +60,7 @@ async function advanceStep( await step.run(await prepareRunSession(step, store.session)); store.completeRunStep(step.id); } else if (step.screenId === 'run') { - await runAgent(config, await prepareRunSession(step, store.session)); + await runProgramAgent(config, await prepareRunSession(step, store.session)); } else if (step.isComplete) { await store.waitUntil(step.isComplete); } @@ -262,7 +262,7 @@ export function runWizard( }); } else { try { - await runAgent(config, activeTui.store.session); + await runProgramAgent(config, activeTui.store.session); } catch (error) { // The run threw before its own error handling rendered an outro. // Show the handoff screen and let the user's agent take over. diff --git a/src/lib/status-history.ts b/src/lib/status-history.ts new file mode 100644 index 000000000..5931026c6 --- /dev/null +++ b/src/lib/status-history.ts @@ -0,0 +1,9 @@ +/** Shared FIFO window for agent snapshots and the terminal status panel. */ +export const MAX_STATUS_MESSAGES = 10; + +export function appendStatus(messages: string[], message: string): string[] { + if (messages.length > 0 && messages[messages.length - 1] === message) { + return messages; + } + return [...messages.slice(-(MAX_STATUS_MESSAGES - 1)), message]; +} diff --git a/src/lib/wizard-ask-bridge.ts b/src/lib/wizard-ask-bridge.ts index 93b06cefa..1a007a8a8 100644 --- a/src/lib/wizard-ask-bridge.ts +++ b/src/lib/wizard-ask-bridge.ts @@ -53,13 +53,27 @@ export interface AskResponse { export interface WizardAskBridge { /** Open the WizardAsk overlay and resolve with the user's answers. */ request(req: WizardAskRequest): Promise; + /** An unresolved question keeps file mutations paused across concurrent requests. */ + getPendingQuestion: () => PendingQuestion | null; } export interface WizardAskBridgeOptions { /** Returns the active skill id, used as the analytics `source` on the request. */ getSource: () => string; - /** Opens the overlay and resolves once the user submits or cancels. */ - showQuestion: (question: PendingQuestion) => Promise; + /** + * Opens the overlay and resolves once the user submits or cancels. `signal` + * is this question's own: it aborts when the timeout wins the race, and the + * host dismisses this question's overlay. Without that the host keeps its + * pending-question state, and every later `wizard_ask` in the run fails with + * "another request is pending" — one unanswered prompt would block + * credential collection for all remaining sources. The host's abort + * handling must not throw: the bridge cannot catch an abort listener's + * error, and Node rethrows it as an uncaught exception. + */ + showQuestion: ( + question: PendingQuestion, + context: { signal: AbortSignal }, + ) => Promise; /** * Per-question timeout in milliseconds. When the user takes longer than * this to answer, every unanswered field resolves with the @@ -72,14 +86,6 @@ export interface WizardAskBridgeOptions { * Propagated onto every {@link PendingQuestion} this bridge creates. */ richLinks?: boolean; - /** - * Dismiss the host's in-flight question overlay. Called when the timeout - * wins the race: without it the host keeps its pending-question state, and - * every later `wizard_ask` in the run fails with "another request is - * pending" — one unanswered prompt would block credential collection for - * all remaining sources. - */ - cancelQuestion?: () => void; } /** Sentinel returned for unanswered fields on cancellation or timeout. */ @@ -114,8 +120,13 @@ export function createWizardAskBridge( opts: WizardAskBridgeOptions, ): WizardAskBridge { const timeoutMs = opts.timeoutMs ?? DEFAULT_ASK_TIMEOUT_MS; + const pendingQuestions = new Map(); return { + getPendingQuestion: () => { + for (const pending of pendingQuestions.values()) return pending; + return null; + }, async request({ questions, subject }) { const pending: PendingQuestion = { id: randomUUID(), @@ -124,26 +135,29 @@ export function createWizardAskBridge( richLinks: opts.richLinks ?? false, askedAt: new Date().toISOString(), }; + pendingQuestions.set(pending.id, pending); const startedAt = Date.now(); + const controller = new AbortController(); let timer: ReturnType | undefined; let timedOut = false; // Race the user against the timeout. Whichever fires first wins. On - // timeout we also cancel the host's overlay: resolving our side alone - // would leave the host's pending-question state set, and the next - // wizard_ask would be rejected as a duplicate request. + // timeout we also abort this question's signal so the host dismisses its + // overlay: resolving our side alone would leave the host's + // pending-question state set, and the next wizard_ask would be rejected + // as a duplicate request. const timeoutPromise = new Promise((resolve) => { timer = setTimeout(() => { timedOut = true; - opts.cancelQuestion?.(); + controller.abort(); resolve(buildCancelledAnswers(questions)); }, timeoutMs); }); try { const answers = await Promise.race([ - opts.showQuestion(pending), + opts.showQuestion(pending, { signal: controller.signal }), timeoutPromise, ]); const durationMs = Date.now() - startedAt; @@ -168,6 +182,7 @@ export function createWizardAskBridge( return { answers, timedOut }; } finally { if (timer) clearTimeout(timer); + pendingQuestions.delete(pending.id); } }, }; diff --git a/src/ui/__tests__/agent-progress.test.ts b/src/ui/__tests__/agent-progress.test.ts new file mode 100644 index 000000000..bdb1535fd --- /dev/null +++ b/src/ui/__tests__/agent-progress.test.ts @@ -0,0 +1,215 @@ +vi.mock('@ui', () => ({ getUI: vi.fn() })); +vi.mock('@utils/debug'); +vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture: vi.fn() } })); + +import { createUiReducer, uiInteraction } from '../agent-progress'; +import { LoggingUI } from '../logging-ui'; +import type { AgentProgress } from '@lib/agent/progress'; +import { + CANCELLED_SENTINEL, + createWizardAskBridge, +} from '@lib/wizard-ask-bridge'; +import { OutroKind } from '@lib/wizard-session'; +import { logToFile } from '@utils/debug'; + +beforeEach(() => { + vi.mocked(logToFile).mockClear(); +}); + +it('projects every progress event onto the matching UI call, in order', () => { + const ui = new LoggingUI(); + const calls: unknown[][] = []; + const methods = [ + 'startRun', + 'outro', + 'pushStatus', + 'syncTodos', + 'setStage', + 'setDashboardUrl', + 'setNotebookUrl', + 'addTokenUsage', + 'setFinalTokenCostUsd', + 'showAuthError', + 'setOutroData', + ] as const; + for (const method of methods) { + vi.spyOn(ui, method).mockImplementation(((...args: unknown[]) => { + calls.push([method, ...args]); + }) as never); + } + for (const level of ['info', 'warn', 'error', 'success', 'step'] as const) { + vi.spyOn(ui.log, level).mockImplementation((message) => { + calls.push([level, message]); + }); + } + const spinner = vi.spyOn(ui, 'spinner').mockReturnValue({ + start: (message) => { + calls.push(['spinner:start', message]); + }, + message: (message) => { + calls.push(['spinner:message', message]); + }, + stop: (message) => { + calls.push(['spinner:stop', message]); + }, + }); + const tasks = [{ content: 'Install', status: 'completed' }]; + const delta = { + inputTokens: 1, + outputTokens: 2, + cacheReadTokens: 3, + cacheCreationTokens: 4, + cacheCreation5m: 4, + cacheCreation1h: 0, + }; + const detail = { hasSettingsConflict: false, logFilePath: '/tmp/wizard.log' }; + const outro = { kind: OutroKind.Success, message: 'Finished' }; + const events: AgentProgress[] = [ + { kind: 'lifecycle', phase: 'started' }, + { kind: 'spinner', action: 'start', message: 'Starting' }, + { kind: 'spinner', action: 'message', message: 'Working' }, + { kind: 'spinner', action: 'stop' }, + ...(['info', 'warn', 'error', 'success', 'step'] as const).map((level) => ({ + kind: 'log' as const, + level, + message: level, + })), + { kind: 'status', message: 'Configured' }, + { kind: 'tasks', tasks }, + { kind: 'stage', stage: 'Install' }, + { kind: 'url', which: 'dashboard', url: 'https://d/1' }, + { kind: 'url', which: 'notebook', url: 'https://n/1' }, + { kind: 'usage', delta }, + { kind: 'finalCost', usd: 1.25 }, + { kind: 'authError', detail }, + { kind: 'completion', outro }, + { kind: 'lifecycle', phase: 'completed', message: 'Finished' }, + ]; + const reduce = createUiReducer(ui); + expect(spinner).not.toHaveBeenCalled(); + events.forEach(reduce); + expect(spinner).toHaveBeenCalledTimes(1); + expect(calls).toEqual([ + ['startRun'], + ['spinner:start', 'Starting'], + ['spinner:message', 'Working'], + ['spinner:stop', undefined], + ['info', 'info'], + ['warn', 'warn'], + ['error', 'error'], + ['success', 'success'], + ['step', 'step'], + ['pushStatus', 'Configured'], + ['syncTodos', tasks], + ['setStage', 'Install'], + ['setDashboardUrl', 'https://d/1'], + ['setNotebookUrl', 'https://n/1'], + ['addTokenUsage', delta], + ['setFinalTokenCostUsd', 1.25], + ['showAuthError', detail], + ['setOutroData', outro], + ['outro', 'Finished'], + ]); +}); + +const question = { id: 'q', source: 'test', questions: [] }; +const notice = { + title: 'Optional', + body: [], + items: [], + prompt: 'Continue?', + confirmLabel: 'Yes', + cancelLabel: 'No', +}; + +it('forwards answers and notices, leaving the host alone once they settle', async () => { + const ui = new LoggingUI(); + const ask = vi.spyOn(ui, 'requestQuestion').mockResolvedValue({ q: 'yes' }); + const cancelAsk = vi.spyOn(ui, 'cancelPendingQuestion'); + const show = vi.spyOn(ui, 'showTaskNotice').mockResolvedValue(true); + const cancelNotice = vi.spyOn(ui, 'cancelTaskNotice'); + const interaction = uiInteraction(ui); + const asked = new AbortController(); + const noticed = new AbortController(); + await expect( + interaction.ask?.(question, { signal: asked.signal }), + ).resolves.toEqual({ q: 'yes' }); + await expect( + interaction.taskNotice?.(notice, { signal: noticed.signal }), + ).resolves.toBe(true); + // A late abort must not dismiss whatever the host shows next. + asked.abort(); + noticed.abort(); + expect(ask).toHaveBeenCalledWith(question); + expect(show).toHaveBeenCalledWith(notice); + expect(cancelAsk).not.toHaveBeenCalled(); + expect(cancelNotice).not.toHaveBeenCalled(); +}); + +it('dismisses an open question or notice when its signal aborts', () => { + const ui = new LoggingUI(); + vi.spyOn(ui, 'requestQuestion').mockReturnValue(new Promise(() => undefined)); + const cancelAsk = vi.spyOn(ui, 'cancelPendingQuestion'); + vi.spyOn(ui, 'showTaskNotice').mockReturnValue(new Promise(() => undefined)); + const cancelNotice = vi.spyOn(ui, 'cancelTaskNotice'); + const interaction = uiInteraction(ui); + const asked = new AbortController(); + const noticed = new AbortController(); + void interaction.ask?.(question, { signal: asked.signal }); + void interaction.taskNotice?.(notice, { signal: noticed.signal }); + + asked.abort(); + expect(cancelAsk).toHaveBeenCalledOnce(); + expect(cancelNotice).not.toHaveBeenCalled(); + noticed.abort(); + expect(cancelNotice).toHaveBeenCalledOnce(); +}); + +it('settles a timed-out question when the host dismissal throws', async () => { + vi.useFakeTimers(); + try { + const ui = new LoggingUI(); + vi.spyOn(ui, 'requestQuestion').mockReturnValue( + new Promise(() => undefined), + ); + const broken = new Error('overlay broken'); + vi.spyOn(ui, 'cancelPendingQuestion').mockImplementation(() => { + throw broken; + }); + const { ask } = uiInteraction(ui); + if (!ask) throw new Error('uiInteraction answers questions'); + const bridge = createWizardAskBridge({ + getSource: () => 'test', + showQuestion: ask, + timeoutMs: 1000, + }); + const result = bridge.request({ + questions: [{ id: 'goal', prompt: 'Goal?', kind: 'text' }], + }); + vi.advanceTimersByTime(1000); + await expect(result).resolves.toEqual({ + answers: { goal: CANCELLED_SENTINEL }, + timedOut: true, + }); + expect(bridge.getPendingQuestion()).toBeNull(); + // Node rethrows an abort listener's error as an uncaught exception the + // bridge cannot catch, so the answerer logs it instead. + expect(logToFile).toHaveBeenCalledWith(expect.any(String), broken); + } finally { + vi.useRealTimers(); + } +}); + +it('logs a throwing notice dismissal instead of throwing from the abort', () => { + const ui = new LoggingUI(); + vi.spyOn(ui, 'showTaskNotice').mockReturnValue(new Promise(() => undefined)); + const broken = new Error('overlay broken'); + vi.spyOn(ui, 'cancelTaskNotice').mockImplementation(() => { + throw broken; + }); + const noticed = new AbortController(); + void uiInteraction(ui).taskNotice?.(notice, { signal: noticed.signal }); + + noticed.abort(); + expect(logToFile).toHaveBeenCalledWith(expect.any(String), broken); +}); diff --git a/src/ui/agent-progress.ts b/src/ui/agent-progress.ts new file mode 100644 index 000000000..c92f1c108 --- /dev/null +++ b/src/ui/agent-progress.ts @@ -0,0 +1,97 @@ +import type { WizardUI, SpinnerHandle } from './wizard-ui'; +import type { AgentInteraction, AgentProgress } from '@lib/agent/progress'; +import { logToFile } from '@utils/debug'; + +// ── Progress → WizardUI, one call per event ─────────────────────────── + +/** + * The inverse of the agent's former `getUI()` calls: one event, one + * `WizardUI` method, synchronous, in emission order. Because each case maps + * back to exactly the call the agent used to make, the frame and flow goldens + * hold without regeneration. + */ +export function createUiReducer(ui: WizardUI): (event: AgentProgress) => void { + let spinner: SpinnerHandle | undefined; + return (event) => { + switch (event.kind) { + case 'lifecycle': + if (event.phase === 'started') ui.startRun(); + else ui.outro(event.message); + break; + case 'spinner': { + const handle = (spinner ??= ui.spinner()); + handle[event.action](event.message); + break; + } + case 'log': + ui.log[event.level](event.message); + break; + case 'status': + ui.pushStatus(event.message); + break; + case 'tasks': + ui.syncTodos(event.tasks); + break; + case 'stage': + ui.setStage(event.stage); + break; + case 'url': + if (event.which === 'dashboard') ui.setDashboardUrl(event.url); + else ui.setNotebookUrl(event.url); + break; + case 'usage': + ui.addTokenUsage(event.delta); + break; + case 'finalCost': + ui.setFinalTokenCostUsd(event.usd); + break; + case 'authError': + ui.showAuthError(event.detail); + break; + case 'completion': + ui.setOutroData(event.outro); + break; + default: { + const unhandled: never = event; + throw new Error( + `Unhandled agent progress: ${JSON.stringify(unhandled)}`, + ); + } + } + }; +} + +/** The agent's questions, answered wherever `getUI()` answers them today. */ +export function uiInteraction(ui: WizardUI): AgentInteraction { + return { + ask: (question, { signal }) => + dismissOnAbort(ui.requestQuestion(question), signal, () => + ui.cancelPendingQuestion(), + ), + taskNotice: (notice, { signal }) => + dismissOnAbort(ui.showTaskNotice(notice), signal, () => + ui.cancelTaskNotice(), + ), + }; +} + +/** + * Dismiss one open request on abort; a settled one leaves the UI alone. A + * throw inside an abort listener reaches no caller: Node rethrows it as an + * uncaught exception, so a broken overlay is logged here instead. + */ +function dismissOnAbort( + open: Promise, + signal: AbortSignal, + dismiss: () => void, +): Promise { + const onAbort = () => { + try { + dismiss(); + } catch (error) { + logToFile('[agent-progress] dismissing an aborted request failed', error); + } + }; + signal.addEventListener('abort', onAbort, { once: true }); + return open.finally(() => signal.removeEventListener('abort', onAbort)); +} diff --git a/src/ui/tui/constants.ts b/src/ui/tui/constants.ts index 9c055375c..ae420b5c3 100644 --- a/src/ui/tui/constants.ts +++ b/src/ui/tui/constants.ts @@ -6,4 +6,4 @@ */ export const COLLAPSED_COUNT = 2; -export const EXPANDED_COUNT = 10; +export { MAX_STATUS_MESSAGES as EXPANDED_COUNT } from '@lib/status-history'; diff --git a/src/ui/tui/store.ts b/src/ui/tui/store.ts index 4e98e5d0b..f4214510e 100644 --- a/src/ui/tui/store.ts +++ b/src/ui/tui/store.ts @@ -57,7 +57,7 @@ import type { import { getProgramConfig } from '@lib/programs/program-registry'; import { withAiOptInGate } from '@lib/programs/ai-opt-in-gate'; import { reportWarehouseSourcesDetected } from '@lib/programs/posthog-integration/detect'; -import { EXPANDED_COUNT } from '@ui/tui/constants'; +import { appendStatus } from '@lib/status-history'; import { IS_DEV } from '@lib/constants'; import { computeTokenCostUsd } from '@lib/agent/token-pricing'; @@ -120,13 +120,6 @@ interface GateEntry { resolved: boolean; } -/** - * FIFO cap on retained status lines. The status bar is the only consumer and - * renders at most EXPANDED_COUNT lines, so there is no reason to retain more — - * the cap is tied to the window it feeds. - */ -const MAX_STATUS_MESSAGES = EXPANDED_COUNT; - // Capture blocked skill downloads once per readiness result. function captureHealthCheckBlocked(result: WizardReadinessResult): void { try { @@ -1052,15 +1045,8 @@ export class WizardStore { pushStatus(message: string): void { const msgs = this.$statusMessages.get(); - // Skip consecutive duplicate messages (no allocation on the hot path) - if (msgs.length > 0 && msgs[msgs.length - 1] === message) return; - // Nanostore detects change by reference equality, so a new array is - // required. At the cap, allocate exactly once at the final size (dropping - // the oldest entry) rather than push-then-truncate. - const next = - msgs.length >= MAX_STATUS_MESSAGES - ? [...msgs.slice(msgs.length - MAX_STATUS_MESSAGES + 1), message] - : [...msgs, message]; + const next = appendStatus(msgs, message); + if (next === msgs) return; this.$statusMessages.set(next); this.emitChange(); }