diff --git a/.claude/skills/wizard-development/SKILL.md b/.claude/skills/wizard-development/SKILL.md index 3ee4f261d..f56fdc64c 100644 --- a/.claude/skills/wizard-development/SKILL.md +++ b/.claude/skills/wizard-development/SKILL.md @@ -94,8 +94,11 @@ only when it gives a real owner a smaller, reusable boundary. ## Lifecycle and security -[runner/index.ts](../../../src/lib/agent/runner/index.ts) bootstraps, resolves a -binding, and dispatches to a sequence. `agent-runner.ts` is a compatibility +[runner/index.ts](../../../src/lib/agent/runner/index.ts) exports +`runAgent(config, input, options)`: it prepares the run and dispatches to the +sequence the binding names, reporting through `onProgress` and asking through +`interaction`. Gates, authentication and binding resolution live in +`src/lib/programs/run-agent-legacy.ts`. `agent-runner.ts` is a compatibility re-export, not the implementation. Sequences own their lifecycle; harnesses own SDK calls. `ProgramRun.postRun`, `buildOutroData`, `customPrompt`, and `abortCases` are consumed by the linear sequence, not by the orchestrator. Put diff --git a/.claude/skills/wizard-development/references/ARCHITECTURE.md b/.claude/skills/wizard-development/references/ARCHITECTURE.md index bb99255f3..a95bd2b9c 100644 --- a/.claude/skills/wizard-development/references/ARCHITECTURE.md +++ b/.claude/skills/wizard-development/references/ARCHITECTURE.md @@ -16,19 +16,27 @@ readiness hooks, and traverse steps and gates. detection in `onReady`. Noninteractive execution has its own lifecycle in [run-non-interactive.ts](../../../../src/lib/runners/run-non-interactive.ts). -[runner/index.ts](../../../../src/lib/agent/runner/index.ts) resolves a -program's `run` definition, calls shared bootstrap, selects a binding, -dispatches the sequence, and flushes the scanner report on cleanup. The old +[runner/index.ts](../../../../src/lib/agent/runner/index.ts) exports +`runAgent(config, input, options)`: it prepares the run, dispatches the sequence +the binding names, flushes the scanner report and returns a `RunResult`. It +reports through `options.onProgress` and asks through `options.interaction`; it +never calls `getUI()`, reads a session or exits. +[run-agent-legacy.ts](../../../../src/lib/programs/run-agent-legacy.ts) is the +caller today's runners use: it runs the gates, authenticates, resolves the +binding from [bindings.ts](../../../../src/lib/programs/bindings.ts), builds the +agent's inputs from the session, maps progress back onto `getUI()` and hands a +decided failure to `wizardAbort`. [agent-runner.ts](../../../../src/lib/agent/agent-runner.ts) is a compatibility -export. +export of the agent's contracts. -| Layer | Source and responsibility | -| --------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| Bootstrap | [shared/bootstrap.ts](../../../../src/lib/agent/runner/shared/bootstrap.ts): shared health/settings/auth/flag and MCP setup | -| Switchboard | [switchboard/index.ts](../../../../src/lib/agent/runner/switchboard/index.ts): resolve sequence, harness, model and effort override | -| Linear sequence | [sequence/linear.ts](../../../../src/lib/agent/runner/sequence/linear.ts): one conversation, skill/prompt assembly, post-run hooks and outro | -| Orchestrator | [orchestrator-runner.ts](../../../../src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts): seed plan, task queue, focused conversations, handoffs and completion | -| Harness | [harness/types.ts](../../../../src/lib/agent/runner/harness/types.ts): SDK boundary, implemented by Pi and Anthropic | +| Layer | Source and responsibility | +| --------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Contracts | [shared/types.ts](../../../../src/lib/agent/runner/shared/types.ts) and [progress.ts](../../../../src/lib/agent/progress.ts): `RunConfig`, `RunInput`, `RunResult`, `AgentProgress`, `AgentInteraction` | +| Prepare | [shared/bootstrap.ts](../../../../src/lib/agent/runner/shared/bootstrap.ts): logging targets, gateway mint, scan triage | +| Switchboard | [switchboard/index.ts](../../../../src/lib/agent/runner/switchboard/index.ts): resolve sequence, harness, model and effort override | +| Linear sequence | [sequence/linear.ts](../../../../src/lib/agent/runner/sequence/linear.ts): one conversation, skill/prompt assembly, post-run hooks and outro | +| Orchestrator | [orchestrator-runner.ts](../../../../src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts): seed plan, task queue, focused conversations, handoffs and completion | +| Harness | [harness/types.ts](../../../../src/lib/agent/runner/harness/types.ts): SDK boundary, implemented by Pi and Anthropic | The contribution policy is [Pi by default, orchestration preferred](../SKILL.md#execution-policy-and-model-admission). @@ -135,16 +143,20 @@ credentials. ## UI state and agent output -Business logic uses [WizardUI](../../../../src/ui/wizard-ui.ts) through -`getUI()`. [InkUI](../../../../src/ui/tui/ink-ui.ts) updates the TUI store; -[LoggingUI](../../../../src/ui/logging-ui.ts) is available for noninteractive -callers that select it. A missing TTY does not automatically mean an arbitrary -caller uses LoggingUI; snapshot CI drives Ink in a PTY. `requestQuestion` and -task notices are supported interactions, not console prompts to invent in -business logic. +Business logic outside the agent uses +[WizardUI](../../../../src/ui/wizard-ui.ts) through `getUI()`. The agent +(`src/lib/agent`) reports through `AgentProgress` events instead, and the +reducer in +[run-agent-legacy.ts](../../../../src/lib/programs/run-agent-legacy.ts) maps +each event to one `WizardUI` call. [InkUI](../../../../src/ui/tui/ink-ui.ts) +updates the TUI store; [LoggingUI](../../../../src/ui/logging-ui.ts) is +available for noninteractive callers that select it. A missing TTY does not +automatically mean an arbitrary caller uses LoggingUI; snapshot CI drives Ink in +a PTY. `requestQuestion` and task notices are supported interactions, not +console prompts to invent in business logic. Harness adapters translate SDK messages, status markers, task updates and tool -activity into WizardUI calls. Anthropic message processing lives in +activity into progress events. Anthropic message processing lives in [agent-interface.ts](../../../../src/lib/agent/agent-interface.ts); Pi uses its own session event handlers. Orchestrated tasks also have queue and handoff state. Do not assume all harness output passes through `handleSDKMessage`. diff --git a/AGENTS.md b/AGENTS.md index bf6620e5c..a511f8404 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -97,8 +97,8 @@ aliases. | Subcommand | What it audits | | ----------------------------- | ---------------------------------------------------- | -| `wizard audit events` | event capture quality + cost | -| `wizard audit all` | comprehensive audit across every area (**default**) | +| `wizard audit events` | event capture quality + cost | +| `wizard audit all` | comprehensive audit across every area (**default**) | | `wizard audit autocapture` | autocapture setup + cost | | `wizard audit feature-flags` | feature flag usage + cost | | `wizard audit identify` | `$identify` implementation | @@ -170,8 +170,8 @@ nonmutating lint checks, and scope formatting fixes to edited files. Do not add tests for prose, compiler-enforced shapes, or duplicated implementation. Keep new code comments to one line; put longer explanations in linked docs. -Local `--ci`, smoke-test, and full headless runs require two separate secrets: -a PostHog personal API key and an already-issued gateway token supplied through +Local `--ci`, smoke-test, and full headless runs require two separate secrets: a +PostHog personal API key and an already-issued gateway token supplied through `WIZARD_CI_GATEWAY_TOKEN_FILE`, plus the target project ID. Follow the [credential setup](docs/local-dev.md#credentials-for-local-ci-and-headless-runs). @@ -196,7 +196,10 @@ wizard run points. Full catalog: [`docs/local-dev.md`](docs/local-dev.md). - TypeScript everywhere. Use `type` (not `interface`) for framework context types so they satisfy `Record`. - All UI calls go through `getUI()` (returns `WizardUI` interface). Never import - the store directly from business logic. + the store directly from business logic. The agent (`src/lib/agent`) is the + exception: it reports through `AgentProgress` events and asks through + `AgentInteraction`, and never imports `src/ui` (the architecture test enforces + this). - Session mutations go through explicit store setters that call `emitChange()`. Never mutate `session` directly — nanostore holds a shallow copy. - The router resolves the active screen from session state. No imperative diff --git a/scripts/tui-host.no-jest.ts b/scripts/tui-host.no-jest.ts index ded2d55a2..ead31e380 100644 --- a/scripts/tui-host.no-jest.ts +++ b/scripts/tui-host.no-jest.ts @@ -27,10 +27,10 @@ import type { Harness, Sequence } from '@lib/constants'; import { buildSession } from '@lib/wizard-session'; import { initLocalDev } from '@lib/local-dev'; import { configureGatewayFromCIEnvironment } from '@lib/gateway-session'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runAgent } from '@lib/programs/run-agent-legacy'; import { TaskStreamPush, createFileDestination } from '@lib/task-stream/index'; import { getAuditChecks } from '@lib/programs/audit/types'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { authenticate } from '@lib/programs/authenticate'; import { getOrAskForProjectData } from '@utils/setup-utils'; import { logToFile } from '@utils/debug'; import { join } from 'path'; diff --git a/src/__tests__/architecture/import-boundaries.test.ts b/src/__tests__/architecture/import-boundaries.test.ts index 8220182ae..c16a380ef 100644 --- a/src/__tests__/architecture/import-boundaries.test.ts +++ b/src/__tests__/architecture/import-boundaries.test.ts @@ -304,6 +304,11 @@ function analyze(): Analysis { const to = classifySurface(target); if (!allowed.includes(to)) violations.set(key, `matrix:${from}->${to}`); + // The agent reports through `onProgress` and asks through + // `AgentInteraction`; it never reaches for a renderer, not even a type. + if (from === 'agent' && target.startsWith('src/ui/')) { + violations.set(key, 'agent-imports-ui'); + } } } diff --git a/src/__tests__/architecture/known-violations.json b/src/__tests__/architecture/known-violations.json index d523b2a73..f6bead34a 100644 --- a/src/__tests__/architecture/known-violations.json +++ b/src/__tests__/architecture/known-violations.json @@ -2,52 +2,39 @@ "violations": [ "src/commands/factories/family-picker.tsx -> src/commands/command.ts", "src/env.ts -> src/lib/headless-mode.ts", - "src/lib/agent/mcp-prompt-streaming.ts -> src/ui/tui/services/mcp-suggested-prompts-services.ts", "src/lib/detection/agentic.ts -> src/lib/agent/agent-interface.ts", - "src/lib/detection/project-scope.ts -> src/lib/agent/runner/shared/authenticate.ts", "src/lib/errors/agent-map.ts -> src/lib/agent/signals.ts", - "src/lib/programs/agent-skill/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/agent-skill/index.ts -> src/lib/programs/agent-skill/content/index.tsx", "src/lib/programs/ai-observability/index.ts -> src/lib/programs/agent-skill/content/index.tsx", - "src/lib/programs/audit/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/audit/index.ts -> src/lib/agent/agent-runner.ts", + "src/lib/programs/bindings.ts -> src/lib/agent/runner/switchboard/index.ts", "src/lib/programs/dispatch-family.ts -> src/commands/command.ts", "src/lib/programs/dispatch-family.ts -> src/commands/factories/shared.ts", - "src/lib/programs/error-tracking-upload-source-maps/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/error-tracking-upload-source-maps/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/error-tracking-upload-source-maps/index.ts -> src/lib/programs/error-tracking-upload-source-maps/content/index.tsx", "src/lib/programs/error-tracking-upload-source-maps/prompt.ts -> src/lib/agent/agent-interface.ts", - "src/lib/programs/error-tracking/index.ts -> src/lib/agent/runner/shared/types.ts", "src/lib/programs/error-tracking/index.ts -> src/lib/programs/error-tracking/content/index.tsx", "src/lib/programs/error-tracking/index.ts -> src/lib/programs/error-tracking/content/tips.ts", - "src/lib/programs/events-audit/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/mcp-analytics/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/metrics/index.ts -> src/lib/programs/agent-skill/content/index.tsx", - "src/lib/programs/migration/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/migration/index.ts -> src/lib/programs/migration/content/index.tsx", "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/agent-interface.ts", "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/runner/shared/bootstrap.ts", "src/lib/programs/posthog-integration/index.ts -> src/lib/programs/posthog-integration/content/index.tsx", "src/lib/programs/program-registry.ts -> src/lib/programs/agent-skill/content/index.tsx", - "src/lib/programs/program-step.ts -> src/lib/agent/agent-runner.ts", + "src/lib/programs/program-run.ts -> src/lib/agent/runner/shared/types.ts", "src/lib/programs/program-step.ts -> src/ui/tui/components/TipsCard.tsx", "src/lib/programs/program-step.ts -> src/ui/tui/primitives/index.ts", "src/lib/programs/program-step.ts -> src/ui/tui/store.ts", - "src/lib/programs/replay-vision/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/revenue-analytics/detect.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/revenue-analytics/index.ts -> src/lib/programs/revenue-analytics/content/index.tsx", - "src/lib/programs/self-driving/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/self-driving/index.ts -> src/lib/agent/agent-runner.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/agent-interface.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/claude-settings.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/progress.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/index.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/switchboard/index.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/yara-hooks.ts", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/index.tsx", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/pricing.ts", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/tips.ts", "src/lib/programs/self-driving/prompt.ts -> src/lib/agent/agent-interface.ts", - "src/lib/programs/self-driving/prompt.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/warehouse-source/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/warehouse-source/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/warehouse-source/index.ts -> src/lib/programs/warehouse-source/content/index.tsx", - "src/lib/programs/web-analytics-doctor/detect.ts -> src/lib/agent/agent-runner.ts", "src/lib/task-stream/event-plan-watcher.ts -> src/ui/tui/store.ts", "src/lib/task-stream/task-stream-push.ts -> src/ui/tui/store.ts", "src/lib/wizard-session.ts -> src/lib/agent/claude-settings.ts", @@ -69,6 +56,7 @@ "src/ui/tui/store.ts -> src/lib/agent/claude-settings.ts", "src/ui/tui/store.ts -> src/lib/agent/token-pricing.ts", "src/ui/wizard-ui.ts -> src/lib/agent/claude-settings.ts", + "src/ui/wizard-ui.ts -> src/lib/agent/progress.ts", "src/utils/package-manager.ts -> src/telemetry.ts", "src/utils/setup-utils.ts -> src/telemetry.ts", "src/utils/wizard-abort.ts -> src/ui/logging-ui.ts" diff --git a/src/__tests__/cli.test.ts b/src/__tests__/cli.test.ts index 2b8a723f7..e4ebbe951 100644 --- a/src/__tests__/cli.test.ts +++ b/src/__tests__/cli.test.ts @@ -112,7 +112,7 @@ vi.mock('../utils/wizard-abort', async (importOriginal) => ({ ...(await importOriginal()), wizardAbort: vi.fn(), })); -vi.mock('../lib/agent/agent-runner', () => ({ +vi.mock('../lib/programs/run-agent-legacy', () => ({ runAgent: vi.fn().mockResolvedValue(undefined), })); diff --git a/src/__tests__/provision-cli.test.ts b/src/__tests__/provision-cli.test.ts index d1b561ba4..10b6539b2 100644 --- a/src/__tests__/provision-cli.test.ts +++ b/src/__tests__/provision-cli.test.ts @@ -63,7 +63,7 @@ vi.mock('../utils/wizard-abort', async (importOriginal) => ({ ...(await importOriginal()), wizardAbort: vi.fn(), })); -vi.mock('../lib/agent/agent-runner', () => ({ +vi.mock('../lib/programs/run-agent-legacy', () => ({ runAgent: vi.fn().mockResolvedValue(undefined), })); diff --git a/src/lib/__tests__/agent-interface.test.ts b/src/lib/__tests__/agent-interface.test.ts index 3c48975be..7ca36c2ca 100644 --- a/src/lib/__tests__/agent-interface.test.ts +++ b/src/lib/__tests__/agent-interface.test.ts @@ -12,7 +12,6 @@ import { import { AgentOutputSignals } from '@lib/agent/output-signals'; import { RESUME_INSTRUCTION } from '@lib/agent/signals'; import { analytics } from '@utils/analytics'; -import { wizardAbort } from '@utils/wizard-abort'; import { Sequence } from '@lib/constants'; import type { WizardRunOptions } from '@utils/types'; import type { SpinnerHandle } from '@ui'; @@ -24,11 +23,6 @@ import { // Mock dependencies vi.mock('../../utils/analytics'); vi.mock('../../utils/debug'); -// wizardAbort exits the process; the 401 tests below need it to just reject. -vi.mock('@utils/wizard-abort', async (importOriginal) => ({ - ...(await importOriginal()), - wizardAbort: vi.fn(), -})); // Mock the SDK module const mockQuery = vi.fn(); @@ -695,6 +689,10 @@ describe('subprocess gateway credentials', () => { describe('gateway re-mint on 401', () => { const spinner = { start: vi.fn(), stop: vi.fn(), message: vi.fn() }; + // Where the run reports the auth screen; stands where getUI() used to. + const emit = vi.fn(); + const authErrors = () => + emit.mock.calls.filter(([e]) => e.kind === 'authError'); const options: WizardRunOptions = { debug: false, installDir: '/test/dir', @@ -722,6 +720,7 @@ describe('gateway re-mint on 401', () => { triageProvider: () => Promise.resolve('false_positive'), gatewayAuth, refreshGatewayAuth, + emit, }); const run = (cfg: ReturnType) => runAgent(cfg, 'test prompt', options, spinner as unknown as SpinnerHandle, { @@ -778,7 +777,7 @@ describe('gateway re-mint on 401', () => { beforeEach(() => { vi.clearAllMocks(); mockUIInstance.spinner.mockReturnValue(spinner); - vi.mocked(wizardAbort).mockRejectedValue(new Error('wizardAbort: exit')); + emit.mockReset(); }); it('mints once and resumes the session when an aged bearer is rejected', async () => { @@ -794,7 +793,7 @@ describe('gateway re-mint on 401', () => { expect(result).toEqual({}); expect(refresh).toHaveBeenCalledTimes(1); - expect(wizardAbort).not.toHaveBeenCalled(); + expect(authErrors()).toHaveLength(0); expect(mockQuery).toHaveBeenCalledTimes(2); const [first, second] = mockQuery.mock.calls.map((c) => c[0]); expect(first.options.resume).toBeUndefined(); @@ -833,11 +832,12 @@ describe('gateway re-mint on 401', () => { expect(refresh).toHaveBeenCalledTimes(1); expect(mockQuery).toHaveBeenCalledTimes(2); - expect(mockUIInstance.showAuthError).toHaveBeenCalledTimes(1); - expect(wizardAbort).toHaveBeenCalledTimes(1); - // In production wizardAbort exits; the mocked rejection surfaces as the - // run's API error. - expect(result.error).toBe('WIZARD_API_ERROR'); + expect(authErrors()).toHaveLength(1); + // The auth screen is reported, and the decided failure goes back to the + // caller, which owns the exit. + expect(result.error).toBeUndefined(); + expect(result.failure?.message).toBe('Authentication failed (401)'); + expect(result.failure?.code).toBeDefined(); }); it('judges a failed resumed session on its own error, not the old 401', async () => { @@ -874,7 +874,7 @@ describe('gateway re-mint on 401', () => { expect(result.error).toBe('WIZARD_API_ERROR'); expect(result.message).toContain('500'); expect(result.message).not.toContain('401'); - expect(mockUIInstance.showAuthError).not.toHaveBeenCalled(); + expect(authErrors()).toHaveLength(0); }); it('does not re-mint when a fresh bearer is rejected', async () => { @@ -888,9 +888,8 @@ describe('gateway re-mint on 401', () => { // A fresh token the gateway rejects is a bad credential, not age. expect(refresh).not.toHaveBeenCalled(); expect(mockQuery).toHaveBeenCalledTimes(1); - expect(mockUIInstance.showAuthError).toHaveBeenCalledTimes(1); - expect(wizardAbort).toHaveBeenCalledTimes(1); - expect(result.error).toBe('WIZARD_API_ERROR'); + expect(authErrors()).toHaveLength(1); + expect(result.failure?.message).toBe('Authentication failed (401)'); }); }); diff --git a/src/lib/agent/__tests__/agent-prompt.test.ts b/src/lib/agent/__tests__/agent-prompt.test.ts index 2ee6bfdb9..39795283c 100644 --- a/src/lib/agent/__tests__/agent-prompt.test.ts +++ b/src/lib/agent/__tests__/agent-prompt.test.ts @@ -1,5 +1,5 @@ import { assemblePrompt, type PromptContext } from '@lib/agent/agent-prompt'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { HostResolution } from '@lib/host-resolution'; function makeRunDef(overrides: Partial = {}): ProgramRun { diff --git a/src/lib/agent/__tests__/run-agent-standalone.test.ts b/src/lib/agent/__tests__/run-agent-standalone.test.ts new file mode 100644 index 000000000..2f56ffdf3 --- /dev/null +++ b/src/lib/agent/__tests__/run-agent-standalone.test.ts @@ -0,0 +1,389 @@ +/** + * `runAgent(config, input, options)` runs with no UI, no store, no session and + * no program registry: a fake harness stands in for the SDK, and everything + * the run reports arrives through `onProgress` or comes back in the result. + * + * The `@ui` mock below throws on use. It is never reached — that is the + * assertion the whole file rests on. + */ +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { Harness, Sequence } from '@lib/constants'; +import { HostResolution } from '@lib/host-resolution'; +import { OutroKind, type PendingQuestion } from '@lib/wizard-session'; +import { AGENT_ERROR_CODE, ErrorCodes } from '@lib/errors'; +import { AgentErrorType } from '@lib/agent/signals'; +import type { AgentProgress } from '@lib/agent/progress'; +import type { + AgentHarness, + BackendRunInputs, +} from '@lib/agent/runner/harness/types'; + +vi.mock('@ui', () => ({ + getUI: () => { + throw new Error('the agent reached for getUI()'); + }, + setUI: () => { + throw new Error('the agent reached for setUI()'); + }, +})); +vi.mock('@utils/debug'); +vi.mock('@utils/analytics', () => ({ + analytics: { + build: 'test', + runId: 'run-1', + wizardCapture: vi.fn(), + capture: vi.fn(), + captureException: vi.fn(), + setTag: vi.fn(), + shutdown: vi.fn().mockResolvedValue(undefined), + }, +})); +vi.mock('@lib/gateway-session', async (importOriginal) => ({ + ...(await importOriginal()), + gatewayAuth: vi.fn().mockResolvedValue({ + gatewayUrl: 'https://gateway.test', + token: 'phe_run', + teamId: 1, + refreshAtMs: Date.now() + 3_600_000, + }), +})); + +// The fake harness: reports a little of everything, then returns what the +// current test told it to. +const harnessState = vi.hoisted(() => ({ + result: {} as { error?: string; message?: string; failure?: unknown }, + throws: undefined as Error | undefined, + lastInputs: undefined as unknown, + askQuestions: undefined as PendingQuestion['questions'] | undefined, +})); +vi.mock('@lib/agent/runner/switchboard/harness', () => { + const fake: AgentHarness = { + name: Harness.pi, + async run(inputs: BackendRunInputs) { + harnessState.lastInputs = inputs; + const { emit, spinner } = inputs; + emit({ kind: 'log', level: 'step', message: 'Initializing agent' }); + spinner.start('Working'); + emit({ kind: 'status', message: 'Installing the SDK' }); + emit({ + kind: 'tasks', + tasks: [{ content: 'Install', status: 'completed' }], + }); + emit({ kind: 'url', which: 'dashboard', url: 'https://d/1' }); + emit({ + kind: 'usage', + delta: { + inputTokens: 10, + outputTokens: 5, + cacheReadTokens: 0, + cacheCreationTokens: 0, + cacheCreation5m: 0, + cacheCreation1h: 0, + }, + }); + if (harnessState.askQuestions && inputs.askBridge) { + const { answers } = await inputs.askBridge.request({ + questions: harnessState.askQuestions, + }); + emit({ + kind: 'status', + message: `answered:${JSON.stringify(answers)}`, + }); + } + if (harnessState.throws) throw harnessState.throws; + spinner.stop('Done'); + return harnessState.result as never; + }, + }; + return { + HARNESS_OPTIONS: { [Harness.pi]: fake }, + getHarness: () => fake, + resolveHarness: () => ({ harness: Harness.pi, model: 'm' }), + }; +}); + +import { runAgent } from '@lib/agent/runner'; +import type { RunConfig, RunInput } from '@lib/agent/runner'; +import { analytics } from '@utils/analytics'; + +let tmp: string; + +const config = (over: Partial = {}): RunConfig => ({ + programId: 'test-program', + run: { + integrationLabel: 'test-integration', + spinnerMessage: 'Working...', + successMessage: 'Done!', + estimatedDurationMinutes: 1, + reportFile: 'report.md', + docsUrl: 'https://docs.test', + abortCases: [ + { + match: /no stripe/i, + message: 'No Stripe here', + body: 'Stripe is required.', + errorCode: ErrorCodes.AgentAbort, + }, + ], + }, + composed: false, + binding: { sequence: Sequence.linear, harness: Harness.pi, model: 'm' }, + switchboard: { program: 'test-program', flags: {} }, + skillsBaseUrl: 'https://skills.test', + wizardFlags: {}, + wizardFlagPayloads: {}, + wizardMetadata: {}, + ...over, +}); + +const input = (over: Partial = {}): RunInput => ({ + installDir: tmp, + credentials: { + accessToken: 'tok', + projectApiKey: 'phc_test', + host: HostResolution.fromApiHost('https://us.posthog.com'), + projectId: 1, + }, + project: null, + apiUser: null, + skillId: 'test-integration', + flags: { + ci: false, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + host: {}, + ...over, +}); + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'run-agent-standalone-')); + harnessState.result = {}; + harnessState.throws = undefined; + harnessState.lastInputs = undefined; + harnessState.askQuestions = undefined; + vi.mocked(analytics.shutdown).mockClear(); +}); + +afterEach(() => fs.rmSync(tmp, { recursive: true, force: true })); + +describe('runAgent standalone', () => { + it('runs to success with an observer and an answerer', async () => { + const events: AgentProgress[] = []; + const ask = vi.fn(() => Promise.resolve({ q1: 'yes' })); + harnessState.askQuestions = [ + { id: 'q1', prompt: 'Continue?', kind: 'single' as const }, + ]; + + const result = await runAgent(config(), input(), { + onProgress: (e) => events.push(e), + interaction: { ask }, + }); + + expect(result.outcome).toBe('success'); + expect(result.skillId).toBe('test-integration'); + expect(result.outro).toEqual({ + kind: OutroKind.Success, + message: 'Done!', + reportFile: 'report.md', + docsUrl: 'https://docs.test', + continueUrl: undefined, + }); + expect(result.failure).toBeUndefined(); + + // The answerer saw the question the bridge built, and its answer came back. + expect(ask).toHaveBeenCalledTimes(1); + const [question] = ask.mock.calls[0] as unknown as [PendingQuestion]; + expect(question.questions[0].id).toBe('q1'); + expect(question.source).toBe('test-integration'); + expect(events).toContainEqual({ + kind: 'status', + message: 'answered:{"q1":"yes"}', + }); + + // Emission order: started first, completion then completed last. + const kinds = events.map( + (e) => `${e.kind}${'phase' in e ? `:${e.phase}` : ''}`, + ); + expect(kinds[0]).toBe('lifecycle:started'); + expect(kinds.slice(-2)).toEqual(['completion', 'lifecycle:completed']); + + // The agent's own snapshot, independent of the observer. + expect(result.snapshot.dashboardUrl).toBe('https://d/1'); + expect(result.snapshot.tasks).toEqual([ + { content: 'Install', status: 'completed' }, + ]); + expect(result.snapshot.statusMessages[0]).toBe('Installing the SDK'); + expect(result.snapshot.usage).toEqual({ + inputTokens: 10, + outputTokens: 5, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }); + expect(analytics.shutdown).toHaveBeenCalledWith('success'); + }); + + it('runs to a complete result with no options at all', async () => { + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('success'); + expect(result.outro?.kind).toBe(OutroKind.Success); + expect(result.snapshot.dashboardUrl).toBe('https://d/1'); + // No answerer: no bridge is installed, so `wizard_ask` reports unavailable + // rather than hanging. + const inputs = harnessState.lastInputs as BackendRunInputs; + expect(inputs.askBridge).toBeUndefined(); + expect(inputs.getPendingQuestion).toBeUndefined(); + }); + + it('installs no bridge in CI even with an answerer, as before', async () => { + await runAgent( + config(), + input({ + flags: { + ci: true, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + }), + { interaction: { ask: () => Promise.resolve({}) } }, + ); + const inputs = harnessState.lastInputs as BackendRunInputs; + expect(inputs.askBridge).toBeUndefined(); + }); + + it('returns an agent abort as a decided failure with the matched case', async () => { + harnessState.result = { + error: AgentErrorType.ABORT, + message: 'No Stripe found', + }; + const events: AgentProgress[] = []; + + const result = await runAgent(config(), input(), { + onProgress: (e) => events.push(e), + }); + + expect(result.outcome).toBe('aborted'); + expect(result.failure?.code).toBe(ErrorCodes.AgentAbort); + expect(result.failure?.outroData).toMatchObject({ + kind: OutroKind.Error, + message: 'No Stripe here', + body: 'Stripe is required.', + errorDetail: { reason: 'No Stripe found' }, + }); + expect(result.failure?.error?.message).toBe( + 'Agent aborted: No Stripe found', + ); + expect(events.some((e) => e.kind === 'completion')).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('returns a coded failure for a harness error', async () => { + harnessState.result = { error: AgentErrorType.NO_PROGRESS }; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('failed'); + expect(result.failure?.code).toBe( + AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], + ); + expect(result.failure?.message).toContain('without changing your project'); + }); + + it('passes a harness-decided failure through untouched', async () => { + const failure = { code: ErrorCodes.AgentAbort, message: 'decided' }; + harnessState.result = { failure }; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('failed'); + expect(result.failure).toBe(failure); + }); + + it('skips the terminal outro and the shutdown for a composed sub-run', async () => { + const events: AgentProgress[] = []; + + const result = await runAgent(config({ composed: true }), input(), { + onProgress: (e) => events.push(e), + }); + + expect(result.outcome).toBe('success'); + expect(result.outro).toBeUndefined(); + expect(events.some((e) => e.kind === 'completion')).toBe(false); + expect( + events.some((e) => e.kind === 'lifecycle' && e.phase === 'completed'), + ).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('finishes when the observer throws on every event', async () => { + const result = await runAgent(config(), input(), { + onProgress: () => { + throw new Error('projection broke'); + }, + }); + + expect(result.outcome).toBe('success'); + expect(result.snapshot.tasks).toHaveLength(1); + }); + + it('calls the bound hooks with the run credentials', async () => { + const postRun = vi.fn().mockResolvedValue(undefined); + const buildOutroData = vi.fn(() => ({ + kind: OutroKind.Success, + message: 'custom', + })); + + const result = await runAgent( + config({ hooks: { postRun, buildOutroData } }), + input(), + ); + + expect(postRun).toHaveBeenCalledWith( + expect.objectContaining({ projectApiKey: 'phc_test' }), + ); + expect(buildOutroData).toHaveBeenCalledTimes(1); + expect(result.outro).toEqual({ + kind: OutroKind.Success, + message: 'custom', + }); + }); + + it('returns a crash as a result instead of rejecting', async () => { + const boom = new Error('SDK exploded'); + harnessState.throws = boom; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('crashed'); + expect(result.failure?.error).toBe(boom); + expect(result.failure?.message).toBe('SDK exploded'); + expect(result.failure?.code).toBe(ErrorCodes.InternalUnhandled); + // What was reported before the crash survives in the snapshot. + expect(result.snapshot.statusMessages).toContain('Installing the SDK'); + }); + + it('is cancelled by a signal that is already aborted', async () => { + const controller = new AbortController(); + controller.abort(); + + const result = await runAgent(config(), input(), { + signal: controller.signal, + }); + + expect(result.outcome).toBe('cancelled'); + expect(harnessState.lastInputs).toBeUndefined(); + }); +}); diff --git a/src/lib/agent/agent-interface.ts b/src/lib/agent/agent-interface.ts index 815498a25..234a85360 100644 --- a/src/lib/agent/agent-interface.ts +++ b/src/lib/agent/agent-interface.ts @@ -6,8 +6,11 @@ import path from 'path'; import * as os from 'os'; import { createRequire } from 'node:module'; -import { getUI, type SpinnerHandle } from '@ui'; -import type { TokenUsageDelta } from '@ui/wizard-ui'; +import type { + ProgressEmitter, + SpinnerHandle, + TokenUsageDelta, +} from './progress'; import { debug, logToFile, initLogFile, getLogFilePath } from '@utils/debug'; import type { WizardRunOptions } from '@utils/types'; import { analytics } from '@utils/analytics'; @@ -27,7 +30,8 @@ import { type AdditionalFeature, ADDITIONAL_FEATURE_PROMPTS, } from '@lib/wizard-session'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; +import type { AgentFailure } from './runner/shared/types'; import { createCustomHeaders } from '@utils/custom-headers'; import type { HostResolution } from '@lib/host-resolution'; import { @@ -243,6 +247,12 @@ export type AgentConfig = { * `--capture-aio` is off. Constructed once per run by the harness. */ capture?: AioCapture; + /** + * Where the run reports: log lines, tasks, status, URLs, usage, the handoff + * document and the auth-error detail. Absent → the run reports nowhere and + * still completes. + */ + emit?: ProgressEmitter; }; /** @@ -354,8 +364,12 @@ type AgentRunConfig = { program?: string; /** Resolved sequence, for the sequence-axis commandments. */ sequence: Sequence; + /** Where the run reports. A no-op when the caller passed none. */ + emit?: ProgressEmitter; }; +const NO_PROGRESS: ProgressEmitter = () => undefined; + /** * Global identifiers attached to every LLM gateway trace for a run. They ride on * each `$ai_generation` the gateway emits (in the `X-PostHog-Properties` blob @@ -531,6 +545,7 @@ export async function initializeAgent( initLogFile(); logToFile('Agent initialization starting'); logToFile('Install directory:', options.installDir); + const emit = config.emit ?? NO_PROGRESS; try { // Configure model routing (inherited by the SDK subprocess). All model @@ -636,6 +651,7 @@ export async function initializeAgent( askMaxQuestions: config.askMaxQuestions, orchestrator: config.orchestrator, triageProvider, + onHandoffText: (text) => emit({ kind: 'handoff', text }), }); mcpServers['wizard-tools'] = wizardToolsServer; @@ -661,6 +677,7 @@ export async function initializeAgent( program: config.integrationLabel, // A queue context is present only on a task run; that is the sequence. sequence: config.orchestrator ? Sequence.orchestrator : Sequence.linear, + emit, }; logToFile('Agent config:', { @@ -689,9 +706,11 @@ export async function initializeAgent( return agentRunConfig; } catch (error) { - getUI().log.error( - `Failed to initialize agent: ${(error as Error).message}`, - ); + emit({ + kind: 'log', + level: 'error', + message: `Failed to initialize agent: ${(error as Error).message}`, + }); logToFile('Agent initialization error:', error); debug('Agent initialization error:', error); throw error; @@ -740,7 +759,12 @@ export async function runAgent( onMessage(message: any): void; finalize(resultMessage: any, totalDurationMs: number): any; }, -): Promise<{ error?: AgentErrorType; message?: string }> { +): Promise<{ + error?: AgentErrorType; + message?: string; + failure?: AgentFailure; +}> { + const emit = agentConfig.emit ?? NO_PROGRESS; const { spinnerMessage = 'Customizing your PostHog setup...', successMessage = 'PostHog integration complete', @@ -854,7 +878,7 @@ export async function runAgent( // CostTrackerPlugin.onFinalize uses to correct per-turn drift. const totalCostUsd = Number(lastResultMessage?.total_cost_usd ?? 0); if (totalCostUsd > 0) { - getUI().setFinalTokenCostUsd(totalCostUsd); + emit({ kind: 'finalCost', usd: totalCostUsd }); } try { middleware?.finalize(lastResultMessage, durationMs); @@ -879,6 +903,9 @@ export async function runAgent( let sessionId: string | undefined; let reminted = false; let remintRequested = false; + // A 401 on a fresh bearer: the auth screen was reported, and this is the + // failure the caller ends the run with. The query is aborted to unwind. + let authFailure: AgentFailure | undefined; const agentConfigDir = createIsolatedAgentConfigDir(); try { @@ -1162,6 +1189,7 @@ export async function runAgent( options, spinner, signals, + emit, receivedSuccessResult, tasks, agentConfig.suppressTaskRender ?? false, @@ -1240,15 +1268,20 @@ export async function runAgent( ...authError, sessionExpired, }); - getUI().showAuthError({ - hasSettingsConflict: authError.hasSettingsConflict, - conflicts: authError.conflicts, - usingManagedLogin: authError.usingManagedLogin, - credentialPlaces: authError.credentialPlaces, - sessionExpired, - logFilePath: getLogFilePath(), + emit({ + kind: 'authError', + detail: { + hasSettingsConflict: authError.hasSettingsConflict, + conflicts: authError.conflicts, + usingManagedLogin: authError.usingManagedLogin, + credentialPlaces: authError.credentialPlaces, + sessionExpired, + logFilePath: getLogFilePath(), + }, }); - await wizardAbort({ + // The caller ends the run with this; the query is abandoned here + // where the process used to exit. + authFailure = { code: authCode, message: 'Authentication failed (401)', error: new WizardError( @@ -1264,7 +1297,9 @@ export async function runAgent( }, authCode, ), - }); + }; + abortController.abort(); + break; } try { @@ -1290,6 +1325,7 @@ export async function runAgent( } catch (error) { // The abort we asked for; anything else belongs to the outer catch. if (remintRequested) return 'remint'; + if (authFailure) return 'done'; throw error; } return remintRequested ? 'remint' : 'done'; @@ -1321,6 +1357,12 @@ export async function runAgent( await runQuery(sessionId); } + // A fresh bearer was rejected. The auth screen is already up; hand the + // decided failure to the caller, which owns the exit. + if (authFailure) { + return { failure: authFailure }; + } + // A YARA hook detected a terminal violation and aborted the run. if (yaraViolationReason) { logToFile('Agent error: YARA_VIOLATION'); @@ -1421,14 +1463,19 @@ export async function runAgent( // No API error found, re-throw the original exception spinner.stop(errorMessage); - getUI().log.error(`Error: ${(error as Error).message}`); + emit({ + kind: 'log', + level: 'error', + message: `Error: ${(error as Error).message}`, + }); logToFile('Agent run failed:', error); debug('Full error:', error); throw error; } finally { // Always capture run duration, even on abort/error, so we can alert on - // long runs where the user gave up before completion. - if (!receivedSuccessResult) { + // long runs where the user gave up before completion. A 401 never reached + // this block before (the process exited first), so it still does not count. + if (!receivedSuccessResult && !authFailure) { const durationMs = Date.now() - startTime; analytics.wizardCapture('agent aborted', { duration_ms: durationMs, @@ -1737,6 +1784,7 @@ function handleSDKMessage( options: WizardRunOptions, spinner: SpinnerHandle, signals: AgentOutputSignals, + emit: ProgressEmitter, receivedSuccessResult = false, tasks?: Map, // The orchestrator owns the TUI task panel (it renders its queue). Suppress the @@ -1762,7 +1810,7 @@ function handleSDKMessage( const sorted = Array.from(tasks.values()).sort( (a, b) => rank(a.status) - rank(b.status), ); - getUI().syncTodos(sorted); + emit({ kind: 'tasks', tasks: sorted.map((t) => ({ ...t })) }); }; logToFile(`SDK Message: ${message.type}`, JSON.stringify(message, null, 2)); @@ -1777,7 +1825,7 @@ function handleSDKMessage( // dedup for SDK-retried turns — see addTokenUsage's doc comment), so // this stays live-updating for every run, not just `--benchmark`. const tokenUsageDelta = extractTokenUsageDelta(message); - if (tokenUsageDelta) getUI().addTokenUsage(tokenUsageDelta); + if (tokenUsageDelta) emit({ kind: 'usage', delta: tokenUsageDelta }); // Extract text content from assistant messages const content = message.message?.content; @@ -1797,7 +1845,7 @@ function handleSDKMessage( const statusMatch = block.text.match(statusRegex); if (statusMatch) { const statusText = statusMatch[1].trim(); - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } @@ -1811,7 +1859,11 @@ function handleSDKMessage( ); const dashboardMatch = block.text.match(dashboardRegex); if (dashboardMatch) { - getUI().setDashboardUrl(dashboardMatch[1].trim()); + emit({ + kind: 'url', + which: 'dashboard', + url: dashboardMatch[1].trim(), + }); } // Check for [NOTEBOOK_URL] markers @@ -1824,7 +1876,11 @@ function handleSDKMessage( ); const notebookMatch = block.text.match(notebookRegex); if (notebookMatch) { - getUI().setNotebookUrl(notebookMatch[1].trim()); + emit({ + kind: 'url', + which: 'notebook', + url: notebookMatch[1].trim(), + }); } } @@ -1847,7 +1903,7 @@ function handleSDKMessage( // Mirror the active tool into the Visualizer's "stage" indicator. if (block.type === 'tool_use') { const stage = classifyToolToStage((block as ToolUseBlock).name); - if (stage) getUI().setStage(stage); + if (stage) emit({ kind: 'stage', stage }); } } } @@ -1900,7 +1956,7 @@ function handleSDKMessage( // mode race conditions). Full message already logged above via JSON dump. if (message.errors && !receivedSuccessResult) { for (const err of message.errors) { - getUI().log.error(`Error: ${err}`); + emit({ kind: 'log', level: 'error', message: `Error: ${err}` }); logToFile('ERROR:', err); } } @@ -1915,7 +1971,7 @@ function handleSDKMessage( // Full message already logged above via JSON dump. if (message.errors && !receivedSuccessResult) { for (const err of message.errors) { - getUI().log.error(`Error: ${err}`); + emit({ kind: 'log', level: 'error', message: `Error: ${err}` }); logToFile('ERROR:', err); } } diff --git a/src/lib/agent/agent-prompt.ts b/src/lib/agent/agent-prompt.ts index 0427fc4f5..4987adcfb 100644 --- a/src/lib/agent/agent-prompt.ts +++ b/src/lib/agent/agent-prompt.ts @@ -7,7 +7,7 @@ * 3. Skill prompt — "follow SKILL.md" instructions (if a skill was installed) */ -import type { ProgramRun } from './agent-runner.js'; +import type { AgentRunDefinition } from './runner/shared/types'; import type { HostResolution } from '@lib/host-resolution'; /** @@ -62,7 +62,10 @@ Important: You must read a file immediately before attempting to write it, even /** * Assemble the final agent prompt from the program's run config. */ -export function assemblePrompt(runDef: ProgramRun, ctx: PromptContext): string { +export function assemblePrompt( + runDef: AgentRunDefinition, + ctx: PromptContext, +): string { const parts: string[] = []; // Always include the default project prompt diff --git a/src/lib/agent/agent-runner.ts b/src/lib/agent/agent-runner.ts index 8b92c4ebe..5afc4fbe7 100644 --- a/src/lib/agent/agent-runner.ts +++ b/src/lib/agent/agent-runner.ts @@ -1,15 +1,25 @@ /** - * Re-export shim. The runner has been split into agent/runner/. - * Import from there directly; this shim keeps existing importers working. + * Re-export shim for the agent runner. Import from `./runner/index` directly; + * this shim keeps existing importers of the agent's types working. + * + * The session-driven `runAgent(programConfig, session)` every runner and + * program used to import from here now lives in + * `src/lib/programs/run-agent-legacy.ts`. */ export { runAgent, - runProgram, shouldDisableAsk, - type ProgramRun, - type BootstrapResult, type AbortCase, - type PromptContext, + type AgentFailure, + type AgentInteraction, + type AgentProgress, + type AgentRunDefinition, + type BootstrapResult, type Credentials, + type PromptContext, + type RunAgentOptions, + type RunConfig, + type RunInput, + type RunResult, } from './runner/index'; diff --git a/src/lib/agent/mcp-prompt-streaming.ts b/src/lib/agent/mcp-prompt-streaming.ts index 1b9e55182..08b32231a 100644 --- a/src/lib/agent/mcp-prompt-streaming.ts +++ b/src/lib/agent/mcp-prompt-streaming.ts @@ -12,7 +12,22 @@ * `for await (...)` and render as they arrive. */ -import type { AgentChunk } from '@ui/tui/services/mcp-suggested-prompts-services'; +/** + * Discriminated union covering every kind of streamed event a host renders. + * Production yields these from Claude SDK messages; the TUI playground yields + * them from canned scripts. + */ +export type AgentChunk = + | { kind: 'text'; text: string } + /** `command` carries CLI mode's exec command string (`call …`) so the + * screen can recover the inner tool for context-aware follow-ups. */ + | { kind: 'tool-call'; toolName: string; detail: string; command?: string } + | { kind: 'tool-result'; toolName: string; detail: string } + | { kind: 'error'; text: string } + /** Stream completed. `sessionId` is the SDK session ID of the just- + * completed turn; pass it back as `resumeSessionId` on a follow-up + * call to continue the conversation with full history. */ + | { kind: 'done'; sessionId?: string }; import type { Credentials } from '@lib/wizard-session'; import { DEFAULT_AGENT_MODEL, WIZARD_USER_AGENT } from '@lib/constants'; import { logToFile } from '@utils/debug'; diff --git a/src/lib/agent/progress.ts b/src/lib/agent/progress.ts new file mode 100644 index 000000000..cec2ed632 --- /dev/null +++ b/src/lib/agent/progress.ts @@ -0,0 +1,152 @@ +/** + * The agent's public progress and interaction contracts. + * + * `runAgent` reports through one optional callback and asks through one + * optional set of capabilities. Neither reaches into a UI singleton, a store, + * or a session: every payload is copied data, every question is awaited on an + * injected answerer. The legacy adapter in `src/lib/programs/run-agent-legacy.ts` + * maps these back onto `WizardUI` one call per event, so the terminal output of + * every existing runner is unchanged. + */ + +import type { SettingsConflict } from './claude-settings'; +import type { + AskAnswers, + OutroData, + PendingQuestion, + TaskNotice, +} from '@lib/wizard-session'; + +/** + * One assistant turn's token usage, for the hidden Ctrl+T token/cost HUD. + * `model` is the model that produced *this* turn (e.g. the SDK's + * `message.message.model`) — a subagent can run on a different model than + * the main session, and some programs override to Haiku, so pricing must key + * off the per-turn model rather than a single run-wide assumption. Omit only + * when the caller genuinely has no model context (falls back to Sonnet + * pricing — see `pricePerMtokForModel` in `@lib/agent/token-pricing`). + */ +export interface TokenUsageDelta { + inputTokens: number; + outputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; + cacheCreation5m: number; + cacheCreation1h: number; + model?: string; +} + +/** The run spinner a harness drives. In the agent it is a thin emitter. */ +export interface SpinnerHandle { + start(message?: string): void; + stop(message?: string): void; + message(msg?: string): void; +} + +/** + * Context for an auth-error report so the host can pick the right copy. + * + * `hasSettingsConflict` is true when a Claude Code settings file (project, + * project-local, the user's global config, or managed) actually overrides the + * LLM Gateway auth. `conflicts` carries the exact files and keys so the screen + * can name them. When there is no conflict, the 401 has a different cause (bad + * PAT prefix, missing scope, expired key, region mismatch) and we should not + * advise the user to log out of Claude Code. + */ +export interface AuthErrorDetail { + hasSettingsConflict: boolean; + conflicts?: SettingsConflict[]; + /** + * True when the agent SDK authenticated from a stored Claude login + * (`apiKeySource: "/login managed key"`) instead of the wizard's gateway + * token — conflicting Anthropic credentials. Takes priority in the screen. + */ + usingManagedLogin?: boolean; + /** Human-readable places a conflicting Anthropic credential may live. */ + credentialPlaces?: string[]; + /** + * True when a pre-run refresh already failed on a dead grant. The login is + * gone and re-running is the only fix, so this outranks every other branch — + * none of the usual advice (key type, scopes, region) applies. + */ + sessionExpired?: boolean; + logFilePath: string; +} + +/** One task as the host renders it. The same shape `WizardUI.syncTodos` takes. */ +export interface TaskSnapshot { + content: string; + status: string; + activeForm?: string; +} + +export type ProgressLogLevel = 'info' | 'warn' | 'error' | 'success' | 'step'; + +/** + * Everything the agent reports while it runs. One event per former + * `getUI()` call, in the same order, with the same payload, so a reducer that + * maps each case back onto `WizardUI` reproduces today's output exactly. + * + * Payloads are copies. Never a store, a setter, a function or a live + * collection. The callback returns nothing and the agent never branches on it. + */ +export type AgentProgress = + /** The run's main work has started (`WizardUI.startRun`). */ + | { kind: 'lifecycle'; phase: 'started' } + /** The run finished and the host may show its outro (`WizardUI.outro`). */ + | { kind: 'lifecycle'; phase: 'completed'; message: string } + /** The run spinner (`WizardUI.spinner()`), one handle per run. */ + | { + kind: 'spinner'; + action: 'start' | 'stop' | 'message'; + message?: string; + } + /** A log line (`WizardUI.log[level]`). */ + | { kind: 'log'; level: ProgressLogLevel; message: string } + /** A `[STATUS]` line the agent printed (`WizardUI.pushStatus`). */ + | { kind: 'status'; message: string } + /** The full task list, already sorted for display (`WizardUI.syncTodos`). */ + | { kind: 'tasks'; tasks: TaskSnapshot[] } + /** The stage of work derived from the active tool (`WizardUI.setStage`). */ + | { kind: 'stage'; stage: string } + /** A PostHog URL the agent created (`setDashboardUrl` / `setNotebookUrl`). */ + | { kind: 'url'; which: 'dashboard' | 'notebook'; url: string } + /** One assistant turn's token usage (`WizardUI.addTokenUsage`). */ + | { kind: 'usage'; delta: TokenUsageDelta } + /** The SDK's authoritative run cost (`WizardUI.setFinalTokenCostUsd`). */ + | { kind: 'finalCost'; usd: number } + /** The handoff document from `publish_handoff` (`WizardUI.setHandoffText`). */ + | { kind: 'handoff'; text: string } + /** The gateway returned 401; a failure follows (`WizardUI.showAuthError`). */ + | { kind: 'authError'; detail: AuthErrorDetail } + /** The run's final outro payload (`WizardUI.setOutroData`). */ + | { kind: 'completion'; outro: OutroData }; + +export type ProgressEmitter = (event: AgentProgress) => void; + +/** + * The questions the agent may need a person (or a script) to answer. Every + * capability is optional. With none supplied the agent installs no ask bridge, + * so `wizard_ask` returns its existing "not available" error, and an optional + * task notice is declined — the same path a `--ci` run takes today. + */ +export interface AgentInteraction { + /** + * Open a question and resolve with the answers. The bridge that calls this + * owns the timeout, the `__cancelled__` sentinel and the analytics; `signal` + * aborts when the run is cancelled so the host can dismiss its overlay. + */ + ask?: ( + question: PendingQuestion, + context: { signal: AbortSignal }, + ) => Promise; + /** Dismiss the in-flight question as cancelled (timeouts call this). */ + cancelAsk?: () => void; + /** Offer an optional step and resolve with whether to keep it. */ + taskNotice?: ( + notice: TaskNotice, + context: { signal: AbortSignal }, + ) => Promise; + /** Dismiss an in-flight task notice as declined (timeouts call this). */ + cancelTaskNotice?: () => void; +} diff --git a/src/lib/agent/runner/README.md b/src/lib/agent/runner/README.md index 86fee7f1b..2a7a9a6fb 100644 --- a/src/lib/agent/runner/README.md +++ b/src/lib/agent/runner/README.md @@ -39,19 +39,44 @@ for the coordinated change checklist. Five layers, each with its own job. Nothing crosses layers unless it has to. -**The entry point** (`index.ts`) is the front door. It receives a program config -and a session, and orchestrates the run at the highest level. - -**Bootstrap** (`shared/bootstrap.ts`) is the on-ramp. Every run starts with the -same setup work — health checks, settings conflicts, OAuth, PostHog feature flag -fetch, MCP URL, AI opt-in gate. Whether the run turns out to be linear or -orchestrator, anthropic or pi, the setup is the same. +**The entry point** (`index.ts`) is the front door: + +```ts +runAgent(config: RunConfig, input: RunInput, options?: { + onProgress?: (event: AgentProgress) => void; + interaction?: AgentInteraction; + signal?: AbortSignal; +}): Promise +``` -**The switchboard** (`switchboard/`) is the router. Given a program id + the -fetched flags + any CLI overrides, it returns a `ProgramBinding` — which query +`RunConfig` is resolved execution data: the program id, its run definition, the +binding the caller resolved, flags, run tags and tool lists. `RunInput` is the +invocation snapshot: directory, credentials, project, skill. The agent reports +through `onProgress` (one event per former UI call, copied data, never awaited) +and asks through `interaction` (an optional answerer for `wizard_ask` and task +notices). It returns a `RunResult` — outcome, outro, a failure with the same +fields `wizardAbort` takes, and a snapshot of what it reported. It never +renders, never reads a session, never exits the process and never rejects: a +decided failure is `aborted` or `failed`, an error the agent did not decide is +`crashed` with the original error attached. Types live in `shared/types.ts` and +`../progress.ts`. + +Everything the caller decides first — health and settings gates, OAuth, the AI +opt-in gate, post-auth gates, feature flags, token refresh, the binding — lives +in `src/lib/programs/run-agent-legacy.ts`, which also maps progress back onto +`getUI()` for today's runners. + +**Prepare** (`shared/bootstrap.ts`) is the on-ramp inside the agent: logging +targets, the gateway mint and the scan-triage classifier. Whether the run turns +out to be linear or orchestrator, anthropic or pi, the setup is the same. + +**The switchboard** (`switchboard/`) is the router. Given a program id, its +declared binding (`src/lib/programs/bindings.ts`, or `DEFAULT_BINDING`), the +fetched flags and any CLI overrides, it returns a `ProgramBinding` — which query shape (sequence), which agent SDK (harness), which model. Two independent middleware chains, one per axis, apply precedence rules (CLI > flag > program -config > default). This is the only layer that makes routing decisions. +config > default). The caller resolves the run-level binding; the orchestrator +re-resolves the harness per task role from the same inputs. **Sequences** (`sequence/`) are LLM query shapes. Once the switchboard has picked one, that sequence takes over the run and owns _how the LLM's work is @@ -72,8 +97,7 @@ gateway. ## How they connect -- Bootstrap prepares shared services, including triage selected for the resolved - harness. +- Prepare mints the gateway token and builds triage for the resolved harness. - The switchboard knows which sequences and harnesses exist (via its two registries), but not what they do. - A sequence knows how to shape a conversation, but delegates the actual model @@ -84,12 +108,13 @@ Each layer is replaceable. ## Flow -1. Program picked → session built. -2. Bootstrap runs (shared setup, fetches PostHog flags). -3. Switchboard resolves a `ProgramBinding { sequence, harness, model }`. -4. Analytics tags the run with its bindings. -5. Sequence takes over — shapes the LLM's work into one conversation (linear) or - many (orchestrator). -6. Harness drives each conversation through its SDK, using the bound model, on +1. The caller runs its gates, authenticates, fetches PostHog flags and resolves + a `ProgramBinding { sequence, harness, model }`; analytics tags the run. +2. `runAgent(config, input, options)` prepares (mint, triage). +3. Sequence takes over — shapes the LLM's work into one conversation (linear) or + many (orchestrator), reporting through `onProgress`. +4. Harness drives each conversation through its SDK, using the bound model, on the PostHog LLM gateway. -7. Cleanup runs (scan report, settings restore, outro). +5. The scan report flushes; `runAgent` returns a `RunResult`. +6. The caller applies it: an outro is already reported, a decided failure goes + to `wizardAbort`, a crash is rethrown for the runner's own handling. diff --git a/src/lib/agent/runner/__tests__/switchboard.test.ts b/src/lib/agent/runner/__tests__/switchboard.test.ts index 5386fc312..822f7ba2b 100644 --- a/src/lib/agent/runner/__tests__/switchboard.test.ts +++ b/src/lib/agent/runner/__tests__/switchboard.test.ts @@ -23,11 +23,11 @@ import { WIZARD_ORCHESTRATOR_FLAG_KEY, } from '@lib/constants'; import { - PROGRAM_BINDINGS, DEFAULT_BINDING, resolveBinding, type SwitchboardCtx, } from '@lib/agent/runner/switchboard'; +import { PROGRAM_BINDINGS, bindingFor } from '@lib/programs/bindings'; import { modelCapabilities, MINT_ALLOWED_EFFORTS, @@ -69,7 +69,9 @@ describe('switchboard PROGRAM_BINDINGS', () => { if (program === 'metrics') continue; // pinned below if (program === 'replay-vision') continue; // pinned below if (program === 'error-tracking') continue; // pinned below - expect(resolveBinding({ program, flags: {} })).toEqual(DEFAULT_RESOLVED); + expect( + resolveBinding({ program, flags: {}, binding: bindingFor(program) }), + ).toEqual(DEFAULT_RESOLVED); } }); @@ -229,6 +231,7 @@ describe('switchboard composed clamp', () => { for (const program of PROGRAM_IDS) { const ctx: SwitchboardCtx = { program, + binding: bindingFor(program), composed: true, flags: { [WIZARD_ORCHESTRATOR_FLAG_KEY]: 'true' }, trace: {}, diff --git a/src/lib/agent/runner/harness/anthropic/index.ts b/src/lib/agent/runner/harness/anthropic/index.ts index 4f5e449ec..d760e6d60 100644 --- a/src/lib/agent/runner/harness/anthropic/index.ts +++ b/src/lib/agent/runner/harness/anthropic/index.ts @@ -1,6 +1,5 @@ // Supported legacy SDK fallback; both this adapter and Pi implement run and runTask. -import { getUI } from '@ui'; import { Harness } from '@lib/constants'; import { initializeAgent, @@ -9,7 +8,8 @@ import { import { createAioCapture } from '@lib/agent/aio-capture'; import { getLogFilePath, logToFile } from '@utils/debug'; import { detectNodePackageManagers } from '@lib/detection/package-manager'; -import { sessionToOptions } from '@lib/agent/runner/shared/bootstrap'; +import { runOptions } from '@lib/agent/runner/shared/bootstrap'; +import { createEmitLog } from '@lib/agent/runner/shared/progress-collector'; import type { AgentResult, AgentHarness, @@ -22,30 +22,33 @@ export const anthropicBackend: AgentHarness = { async run(inputs: BackendRunInputs): Promise { const { - session, - config, - programConfig, + config: runConfig, + input, boot, + emit, prompt, spinner, askBridge, + getPendingQuestion, middleware, model, } = inputs; + const config = runConfig.run; const { skillsBaseUrl, credentials, wizardFlags, wizardMetadata } = boot; const { accessToken, host, projectApiKey } = credentials; + const log = createEmitLog(emit); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey, apiHost: host.apiHost, runTags: wizardMetadata, }); - getUI().log.step('Initializing Claude agent...'); + log.step('Initializing Claude agent...'); const agent = await initializeAgent( { - workingDirectory: session.installDir, + workingDirectory: input.installDir, posthogMcpUrl: host.mcpUrl, posthogApiKey: accessToken, host, @@ -59,22 +62,23 @@ export const anthropicBackend: AgentHarness = { integrationLabel: config.integrationLabel, askBridge, askMaxQuestions: config.maxQuestions, - allowedTools: programConfig.allowedTools, - disallowedTools: programConfig.disallowedTools, - getPendingQuestion: () => session.pendingQuestion, + allowedTools: runConfig.allowedTools, + disallowedTools: runConfig.disallowedTools, + getPendingQuestion, modelOverride: model, capture, + emit, }, - sessionToOptions(session), + runOptions(input), ); - getUI().log.step(`Verbose logs: ${getLogFilePath()}`); - getUI().log.success("Agent initialized. Let's get cooking!"); + log.step(`Verbose logs: ${getLogFilePath()}`); + log.success("Agent initialized. Let's get cooking!"); logToFile('[agent-runner] agent initialized'); return executeAgent( agent, prompt, - sessionToOptions(session), + runOptions(input), spinner, { estimatedDurationMinutes: config.estimatedDurationMinutes, @@ -94,9 +98,10 @@ export const anthropicBackend: AgentHarness = { async runTask(inputs: TaskRunInputs): Promise { const { - session, - programConfig, + config, + input, boot, + emit, prompt, spinner, model, @@ -111,10 +116,10 @@ export const anthropicBackend: AgentHarness = { requestRemark, analyticsProperties, } = inputs; - const options = sessionToOptions(session); + const options = runOptions(input); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, @@ -125,7 +130,7 @@ export const anthropicBackend: AgentHarness = { // enqueue_task attribute to the right agent when tasks run in parallel. const agent = await initializeAgent( { - workingDirectory: session.installDir, + workingDirectory: input.installDir, posthogMcpUrl: boot.credentials.host.mcpUrl, posthogApiKey: boot.credentials.accessToken, host: boot.credentials.host, @@ -134,12 +139,13 @@ export const anthropicBackend: AgentHarness = { programId: boot.programId, wizardFlags: boot.wizardFlags, wizardMetadata: boot.wizardMetadata, - integrationLabel: programConfig.id, + integrationLabel: config.programId, // Only a task allowed to ask carries a bridge, so the Write/Edit pause // that rides on a pending question stays inside that task's agent. askBridge, orchestrator, capture, + emit, }, options, ); diff --git a/src/lib/agent/runner/harness/pi/index.ts b/src/lib/agent/runner/harness/pi/index.ts index 092e59718..5c5ae3a25 100644 --- a/src/lib/agent/runner/harness/pi/index.ts +++ b/src/lib/agent/runner/harness/pi/index.ts @@ -14,7 +14,6 @@ import fs from 'fs'; import path from 'path'; -import { getUI } from '@ui'; import { getLogFilePath, logToFile } from '@utils/debug'; import { Harness, @@ -41,6 +40,8 @@ import type { TaskRunInputs, } from '../types'; import type { BootstrapResult } from '@lib/agent/runner/shared/types'; +import type { ProgressEmitter } from '@lib/agent/progress'; +import { createEmitLog } from '@lib/agent/runner/shared/progress-collector'; import type { TaskStore } from './tasks'; import { completionFailure, runErrorType } from './completion'; @@ -147,10 +148,19 @@ export function extractText(message: unknown): string { * the MCP creates them) into the outro link, mirroring the anthropic path's * signal parsing (#9). The marker carries the URL the MCP returned. */ -export function applyOutroMarkers(textBlock: string): void { +export function applyOutroMarkers( + textBlock: string, + emit: ProgressEmitter, +): void { const markers: Array<[string, (url: string) => void]> = [ - [AgentSignals.DASHBOARD_URL, (url) => getUI().setDashboardUrl(url)], - [AgentSignals.NOTEBOOK_URL, (url) => getUI().setNotebookUrl(url)], + [ + AgentSignals.DASHBOARD_URL, + (url) => emit({ kind: 'url', which: 'dashboard', url }), + ], + [ + AgentSignals.NOTEBOOK_URL, + (url) => emit({ kind: 'url', which: 'notebook', url }), + ], ]; for (const [marker, apply] of markers) { const idx = textBlock.indexOf(marker); @@ -195,20 +205,22 @@ export const piBackend: AgentHarness = { name: Harness.pi, async run(inputs: BackendRunInputs): Promise { - const { session, boot, prompt, spinner, config, programConfig } = inputs; + const { config: runConfig, input, boot, emit, prompt, spinner } = inputs; + const config = runConfig.run; const modelId = inputs.model; + const log = createEmitLog(emit); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, }); // Init banner (parity #5). - getUI().log.step('Initializing Wizard agent...'); - getUI().log.step(`Verbose logs: ${getLogFilePath()}`); - getUI().log.success("Agent initialized. Let's get cooking!"); + log.step('Initializing Wizard agent...'); + log.step(`Verbose logs: ${getLogFilePath()}`); + log.success("Agent initialized. Let's get cooking!"); spinner.start(config.spinnerMessage ?? 'Customizing your PostHog setup...'); @@ -306,11 +318,11 @@ export const piBackend: AgentHarness = { const { createSecurityExtension } = await import('./security'); const security = createSecurityExtension({ - disallowedTools: programConfig.disallowedTools, + disallowedTools: runConfig.disallowedTools, getWizardAskPending: () => askState.pending, triageProvider: boot.triageProvider, // Where pi's bash runs; the rm allowance is confined to this tree. - workingDirectory: session.installDir, + workingDirectory: input.installDir, }); // Pay warlock's WASM-init + rule-compile cost now, off the tool-call @@ -360,11 +372,11 @@ export const piBackend: AgentHarness = { } const resourceLoader = new DefaultResourceLoader({ - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), systemPrompt: assembleCommandments({ - program: programConfig.id, + program: runConfig.programId, sequence: Sequence.linear, harness: Harness.pi, caps: { bash: true, posthogMcp }, @@ -390,12 +402,14 @@ export const piBackend: AgentHarness = { const { createWizardPiTaskTools } = await import('./tasks'); const { createDispatchAgentTool } = await import('./subagent'); // Created once so the run loop can read the store for the completion guard. - const wizardTaskTools = createWizardPiTaskTools(); + const wizardTaskTools = createWizardPiTaskTools((tasks) => + emit({ kind: 'tasks', tasks }), + ); // The one bash the agent (and its subagents) may use: every subprocess it // spawns gets a scrubbed env, so no secret or ambient variable reaches an // `npm install`. Shared with the subagent so the lockdown is inherited. const scrubbedBash = withMode( - createBashToolDefinition(session.installDir, { + createBashToolDefinition(input.installDir, { spawnHook: (ctx) => ({ ...ctx, env: buildScrubbedEnv() }), }), 'sequential', @@ -406,18 +420,18 @@ export const piBackend: AgentHarness = { // defaults so we can supply the env-scrubbed bash above; read/edit/write // are the stock definitions. Reads run in parallel so a batched turn of // independent reads executes at once; edit/write/bash stay sequential. - withMode(createReadToolDefinition(session.installDir), 'parallel'), - withMode(createEditToolDefinition(session.installDir), 'sequential'), - withMode(createWriteToolDefinition(session.installDir), 'sequential'), + withMode(createReadToolDefinition(input.installDir), 'parallel'), + withMode(createEditToolDefinition(input.installDir), 'sequential'), + withMode(createWriteToolDefinition(input.installDir), 'sequential'), scrubbedBash, // Native ls/find/grep so the agent explores with proper tools instead // of fence-blocked `bash {ls/find}` (the profiled retry-spirals came // from this gap). Parallel — exploration batches cleanly. - withMode(createLsToolDefinition(session.installDir), 'parallel'), - withMode(createFindToolDefinition(session.installDir), 'parallel'), - withMode(createGrepToolDefinition(session.installDir), 'parallel'), + withMode(createLsToolDefinition(input.installDir), 'parallel'), + withMode(createFindToolDefinition(input.installDir), 'parallel'), + withMode(createGrepToolDefinition(input.installDir), 'parallel'), ...createWizardPiTools({ - workingDirectory: session.installDir, + workingDirectory: input.installDir, skillsBaseUrl: boot.skillsBaseUrl, triageProvider: boot.triageProvider, detectPackageManager: config.detectPackageManager, @@ -429,9 +443,10 @@ export const piBackend: AgentHarness = { onAskPendingChange: (pending) => { askState.pending = pending; }, + onHandoffText: (text) => emit({ kind: 'handoff', text }), // Skip wizard_ask when the program disallows it (bare pi tool names // don't match the MCP-prefixed disallow list at the security gate). - disallowedTools: programConfig.disallowedTools, + disallowedTools: runConfig.disallowedTools, }), // Task/todo tools (#526): render the todo list live in the TUI, parity // with the anthropic path. @@ -442,7 +457,7 @@ export const piBackend: AgentHarness = { createDispatchAgentTool({ model, modelRegistry: registry, - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), securityFactory: security.factory as (pi: unknown) => void, bashTool: scrubbedBash, @@ -456,8 +471,8 @@ export const piBackend: AgentHarness = { // Reasoning effort from the switchboard capability matrix (undefined = // pi's default). Sent as `reasoning_effort` for openai-completions. thinkingLevel: caps.thinkingLevel, - cwd: session.installDir, - sessionManager: SessionManager.inMemory(session.installDir), + cwd: input.installDir, + sessionManager: SessionManager.inMemory(input.installDir), resourceLoader, // Disable the default built-in tools; `customTools` re-registers // read/edit/write + an env-scrubbed bash, so no subprocess inherits the @@ -508,12 +523,12 @@ export const piBackend: AgentHarness = { const assistant = extractText(event.message).trim(); if (assistant) { logToFile(`[pi] assistant: ${assistant.slice(0, 1000)}`); - applyOutroMarkers(assistant); + applyOutroMarkers(assistant, emit); // Surface [STATUS] lines into the live spinner + status history, // mirroring the anthropic path — pi otherwise drops them. const statusText = lastStatusLine(assistant); if (statusText) { - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } for (const line of assistant.split('\n')) signals.push(line); @@ -644,7 +659,7 @@ export const piBackend: AgentHarness = { // on completion; pi's `rm` is fence-blocked, so the agent can't — clean it // up host-side rather than leave a stale (often empty) artifact (#15). try { - const planFile = path.join(session.installDir, '.posthog-events.json'); + const planFile = path.join(input.installDir, '.posthog-events.json'); if (fs.existsSync(planFile)) await fs.promises.rm(planFile); } catch (err) { logToFile(`[pi] .posthog-events.json cleanup skipped: ${String(err)}`); @@ -668,7 +683,7 @@ export const piBackend: AgentHarness = { const message = err instanceof Error ? err.message : String(err); logToFile(`[pi] run error: ${message}`); spinner.stop(config.errorMessage ?? `${config.integrationLabel} failed`); - getUI().log.error(`pi backend error: ${message}`); + log.error(`pi backend error: ${message}`); const error = runErrorType(message); captureAborted(error); return { error, message }; diff --git a/src/lib/agent/runner/harness/pi/task.ts b/src/lib/agent/runner/harness/pi/task.ts index 37deec62d..c86e0b1d7 100644 --- a/src/lib/agent/runner/harness/pi/task.ts +++ b/src/lib/agent/runner/harness/pi/task.ts @@ -16,7 +16,6 @@ * Loaded lazily from `index.ts` (typebox/ESM constraint, same as tools.ts). */ -import { getUI } from '@ui'; import { logToFile } from '@utils/debug'; import { analytics } from '@utils/analytics'; import { @@ -166,9 +165,10 @@ function isSettled(ctx: OrchestratorToolsContext): boolean { export async function runPiTask(inputs: TaskRunInputs): Promise { const { - session, - programConfig, + config, + input, boot, + emit, prompt, spinner, model: modelId, @@ -187,7 +187,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { if (spinnerMessage) spinner.start(spinnerMessage); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, @@ -319,10 +319,10 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { const orchestratorTools = allowedOrchestratorTools(disallowedTools); const resourceLoader = new DefaultResourceLoader({ - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), systemPrompt: assembleCommandments({ - program: programConfig.id, + program: config.programId, sequence: Sequence.orchestrator, harness: Harness.pi, caps: { bash: codingTools.has('bash'), posthogMcp }, @@ -339,7 +339,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { // The task's coding tools, gated by its allow list. Reads and searches run // in parallel; mutating tools stay sequential. Bash subprocesses get the // scrubbed env, same as the linear run. - const dir = session.installDir; + const dir = input.installDir; const codingToolFactories = { read: () => withMode(createReadToolDefinition(dir), 'parallel'), edit: () => withMode(createEditToolDefinition(dir), 'sequential'), @@ -375,6 +375,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { onAskPendingChange: (pending) => { askState.pending = pending; }, + onHandoffText: (text) => emit({ kind: 'handoff', text }), }).filter((t) => wizardToolNames.has(t.name)); const { createPiOrchestratorTools } = await import('./orchestrator-tools'); @@ -436,10 +437,10 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { const assistant = extractText(event.message).trim(); if (assistant) { logToFile(`[pi-task] assistant: ${assistant.slice(0, 1000)}`); - applyOutroMarkers(assistant); + applyOutroMarkers(assistant, emit); const statusText = lastStatusLine(assistant); if (statusText) { - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } for (const line of assistant.split('\n')) signals.push(line); diff --git a/src/lib/agent/runner/harness/pi/tasks.ts b/src/lib/agent/runner/harness/pi/tasks.ts index e12f66e1e..0ee0f4462 100644 --- a/src/lib/agent/runner/harness/pi/tasks.ts +++ b/src/lib/agent/runner/harness/pi/tasks.ts @@ -1,15 +1,15 @@ /** * Task/todo parity for pi (#526). The same four Task tools the anthropic path * exposes (TaskCreate/Update/Get/List), as pi `defineTool` tools backed by a - * shared in-memory store. Every mutation pushes the list to the TUI via - * `getUI().syncTodos`, so the todo panel updates live under pi exactly like the - * anthropic path — the thing that was missing before. + * shared in-memory store. Every mutation reports the list through `onSync` + * (a `tasks` progress event), so the todo panel updates live under pi exactly + * like the anthropic path — the thing that was missing before. */ import { Type } from 'typebox'; import { defineTool } from '@earendil-works/pi-coding-agent'; import type { ToolDefinition } from '@earendil-works/pi-coding-agent'; -import { getUI } from '@ui'; +import type { TaskSnapshot } from '@lib/agent/progress'; export type TaskStatus = 'pending' | 'in_progress' | 'completed'; export interface TaskEntry { @@ -26,22 +26,23 @@ function text(s: string): { return { content: [{ type: 'text', text: s }], details: {} }; } -function syncToTui(store: TaskStore): void { - getUI().syncTodos( - Array.from(store.values()).map((t) => ({ - content: t.content, - status: t.status, - activeForm: t.activeForm, - })), - ); +function snapshot(store: TaskStore): TaskSnapshot[] { + return Array.from(store.values()).map((t) => ({ + content: t.content, + status: t.status, + activeForm: t.activeForm, + })); } -/** Build the four Task tools over a fresh store. */ -export function createWizardPiTaskTools(): { +/** Build the four Task tools over a fresh store. `onSync` gets the list after every mutation. */ +export function createWizardPiTaskTools( + onSync: (tasks: TaskSnapshot[]) => void = () => undefined, +): { tools: ToolDefinition[]; store: TaskStore; } { const store: TaskStore = new Map(); + const syncToTui = (): void => onSync(snapshot(store)); const taskCreate = defineTool({ name: 'TaskCreate', @@ -64,7 +65,7 @@ export function createWizardPiTaskTools(): { status: 'pending', activeForm: args.activeForm, }); - syncToTui(store); + syncToTui(); return text(`Created ${id}`); }, }); @@ -97,7 +98,7 @@ export function createWizardPiTaskTools(): { status: (args.status as TaskStatus) ?? existing.status, activeForm: args.activeForm ?? existing.activeForm, }); - syncToTui(store); + syncToTui(); return text(`Updated ${args.taskId}`); }, }); diff --git a/src/lib/agent/runner/harness/pi/tools.ts b/src/lib/agent/runner/harness/pi/tools.ts index b194cc79a..f5d1ab004 100644 --- a/src/lib/agent/runner/harness/pi/tools.ts +++ b/src/lib/agent/runner/harness/pi/tools.ts @@ -93,6 +93,8 @@ export interface PiToolsContext { disallowedTools?: readonly string[]; /** Scan-triage classifier, resolved once in bootstrap. Absent → scans fail closed. */ triageProvider?: LLMProvider; + /** Receives the handoff document `publish_handoff` accepted. */ + onHandoffText?: (text: string) => void; } export function createWizardPiTools(ctx: PiToolsContext): ToolDefinition[] { @@ -553,7 +555,7 @@ export function createWizardPiTools(ctx: PiToolsContext): ToolDefinition[] { }), }), execute(_id, args) { - const result = publishHandoff(args.content); + const result = publishHandoff(args.content, ctx.onHandoffText); logToFile(`[pi] publish_handoff: ${result.message}`); return Promise.resolve(text(result.message)); }, diff --git a/src/lib/agent/runner/harness/types.ts b/src/lib/agent/runner/harness/types.ts index 67c59e5c9..f245889e7 100644 --- a/src/lib/agent/runner/harness/types.ts +++ b/src/lib/agent/runner/harness/types.ts @@ -13,23 +13,27 @@ * per drained task. A harness without orchestrator support omits the method; * `orchestrator-runner.ts` checks for it at the call site and fails loudly * rather than silently downgrading. + * + * A harness reports through `emit` and never reaches for a UI. It returns an + * error classification, or a decided `failure` the sequence must return as + * the run's result. */ -import type { WizardSession } from '@lib/wizard-session'; -import type { AdditionalFeature } from '@lib/wizard-session'; +import type { AdditionalFeature, PendingQuestion } from '@lib/wizard-session'; import type { Harness } from '@lib/constants'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { SpinnerHandle } from '@ui'; import type { WizardAskBridge } from '@lib/wizard-ask-bridge'; import type { AgentErrorType } from '@lib/agent/agent-interface'; +import type { ProgressEmitter, SpinnerHandle } from '@lib/agent/progress'; import type { OrchestratorToolsContext } from '@lib/agent/runner/sequence/orchestrator/queue-tools'; import type { EffortLevel, ThinkingLevel, } from '@lib/agent/runner/switchboard/models'; import type { - ProgramRun, + AgentFailure, BootstrapResult, + RunConfig, + RunInput, } from '@lib/agent/runner/shared/types'; /** The benchmark/telemetry hook threaded through a run, if enabled. */ @@ -40,14 +44,14 @@ export interface RunMiddleware { /** * Everything a runner needs to run one program. Assembled by `linear.ts` from - * the bootstrap result and the program config; the runner consumes it and never + * the prepared run and the run config; the runner consumes it and never * re-derives run context. */ export interface BackendRunInputs { - session: WizardSession; - config: ProgramRun; - programConfig: ProgramConfig; + config: RunConfig; + input: RunInput; boot: BootstrapResult; + emit: ProgressEmitter; /** The fully assembled prompt. */ prompt: string; /** Installed framework-skill path, when the program installs one. */ @@ -56,7 +60,9 @@ export interface BackendRunInputs { spinner: SpinnerHandle; /** Interactive question bridge; undefined in CI/headless (ask disabled). */ askBridge?: WizardAskBridge; - /** Benchmark middleware, when `session.benchmark` is set. */ + /** The question on screen, for the Write/Edit guard. Undefined without a bridge. */ + getPendingQuestion?: () => PendingQuestion | null; + /** Benchmark middleware, when `--benchmark` is set. */ middleware?: RunMiddleware; /** Gateway model id resolved from the (runner, model) pair. */ model: string; @@ -64,8 +70,16 @@ export interface BackendRunInputs { thinkingLevel?: EffortLevel; } -/** What a runner reports back: an error classification, or nothing on success. */ -export type AgentResult = { error?: AgentErrorType; message?: string }; +/** + * What a runner reports back: an error classification, or nothing on success. + * `failure` is a fully decided abort the harness already reported to the host + * (the 401 auth screen); the sequence returns it as the run's result. + */ +export type AgentResult = { + error?: AgentErrorType; + message?: string; + failure?: AgentFailure; +}; /** * One orchestrator-mode unit of work — the seed plan, or one drained task. @@ -75,9 +89,10 @@ export type AgentResult = { error?: AgentErrorType; message?: string }; * them from the program-level config the linear pipeline assembles once. */ export interface TaskRunInputs { - session: WizardSession; - programConfig: ProgramConfig; + config: RunConfig; + input: RunInput; boot: BootstrapResult; + emit: ProgressEmitter; /** The fully assembled per-task or seed prompt. */ prompt: string; spinner: SpinnerHandle; diff --git a/src/lib/agent/runner/index.ts b/src/lib/agent/runner/index.ts index f11bf73c8..993cdb89b 100644 --- a/src/lib/agent/runner/index.ts +++ b/src/lib/agent/runner/index.ts @@ -1,214 +1,129 @@ /** * Unified program runner — dispatcher. * - * Single configurable pipeline for all programs. Each program - * provides a ProgramRun (via the `run` field on ProgramConfig) - * that controls: - * - Whether a skill is pre-installed or discovered at runtime - * - How the agent prompt is built - * - What MCP servers and package manager detector to use - * - What happens after the agent completes + * One callable, `runAgent(config, input, options)`, runs a program's agent + * pipeline to a decided result. `config` is resolved execution data (which + * program, how it is routed, which flags apply); `input` is the invocation + * snapshot (directory, credentials, project); `options` carries an optional + * progress observer, an optional answerer and an optional cancel signal. * - * The pipeline runs a shared bootstrap (logging, health check, settings, OAuth, - * flags, MCP url), then forks. The `orchestrator` variant routes to the - * experimental task-queue runner. Every other variant runs the fixed linear - * pipeline: + * The pipeline prepares the run (logging targets, gateway mint, scan triage), + * then forks. The `orchestrator` variant routes to the task-queue runner. + * Every other variant runs the fixed linear pipeline: * [skill install] → agent init → prompt → run → errors → [postRun] → outro + * + * The agent reports and asks, it never renders, never reads a session, never + * exits the process and never rejects. A decided failure comes back in + * `RunResult.failure` with the same fields `wizardAbort` takes; an error the + * agent did not decide (a refused mint, an SDK crash) comes back as + * `outcome: 'crashed'` with the original error attached, so a caller can keep + * handling it the way it always did. The legacy adapter in + * `src/lib/programs/run-agent-legacy.ts` rebuilds today's session-driven + * behavior on top of this call for every existing caller. */ -import type { WizardSession } from '../../wizard-session'; -import { analytics } from '@utils/analytics'; -import { - Sequence, - WIZARD_ORCHESTRATOR_FLAG_KEY, - WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY, -} from '@lib/constants'; +import { Sequence } from '@lib/constants'; +import { classifyRunFailure } from '@lib/errors'; import { logToFile } from '@utils/debug'; -import { getUI } from '../../../ui'; -import type { ProgramConfig } from '../../programs/program-step'; -import type { ProgramRun, BootstrapResult } from './shared/types'; -import { bootstrapProgram } from './shared/bootstrap'; -import { - getSequence, - resolveBinding, - type ProgramBinding, - type SwitchboardCtx, -} from './switchboard'; +import type { + RunAgentOptions, + RunConfig, + RunInput, + RunResult, +} from './shared/types'; +import { prepareRun } from './shared/bootstrap'; +import { createProgressCollector } from './shared/progress-collector'; +import { getSequence } from './switchboard'; import { flushScanReport } from '../../yara-hooks'; -import { startAuditLedgerWatcher } from '../../programs/audit/ledger-watcher'; -import { registerCleanup } from '../../../utils/wizard-abort'; export type { - ProgramRun, - BootstrapResult, AbortCase, - PromptContext, + AgentFailure, + AgentRunDefinition, + BootstrapResult, Credentials, + PromptContext, + ResolvedBinding, + RunAgentOptions, + RunConfig, + RunFlags, + RunHooks, + RunInput, + RunOutcome, + RunResult, + RunSnapshot, + SeedTaskEntry, } from './shared/types'; +export type { + AgentInteraction, + AgentProgress, + ProgressEmitter, +} from '@lib/agent/progress'; export { shouldDisableAsk } from './shared/bootstrap'; - -/** - * Resolve a ProgramConfig's agent run definition and execute the pipeline. - * Entry point for bin.ts — handles buildRunConfig, bootstrap, and (future) run field. - */ -export async function runAgent( - programConfig: ProgramConfig, - session: WizardSession, - options: { composed?: boolean } = {}, -): Promise { - if (!programConfig.run) { - throw new Error(`Program "${programConfig.id}" has no run configuration.`); - } - - // Before `run()` resolves: an audit seeds the ledger from inside its recipe, - // and a watcher started later would ignore that write as pre-existing. - const ledger = programConfig.auditLedgerFile - ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) - : null; - if (ledger) registerCleanup(() => ledger.stop()); - - try { - const runDef = - typeof programConfig.run === 'function' - ? await programConfig.run(session) - : programConfig.run; - - await runProgram(session, runDef, programConfig, options); - } finally { - ledger?.stop(); - } -} +export { DEFAULT_BINDING, resolveBinding } from './switchboard'; +export type { ProgramBinding, SwitchboardCtx } from './switchboard'; /** * Run a program's agent pipeline. * - * Bootstrap → bind the program via the switchboard (resolve which sequence - * and harness will run it, tag both axes) → dispatch to the resolved - * sequence's runner. + * Prepare → dispatch to the sequence the binding names → return its result + * with the agent's own snapshot of what it reported. Missing observers change + * nothing; a throwing observer is logged and the run continues. */ -export async function runProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - options: { composed?: boolean } = {}, -): Promise { - const boot = await bootstrapProgram(session, config, programConfig); +export async function runAgent( + config: RunConfig, + input: RunInput, + options: RunAgentOptions = {}, +): Promise { + const collector = createProgressCollector(options.onProgress); + const { emit } = collector; + const signal = options.signal ?? new AbortController().signal; + const log = (message: string) => + emit({ kind: 'log', level: 'info', message }); // Flush the warlock scan report once, at this single seam, on every - // termination path and for every harness (linear, orchestrator, or future): - // - registerCleanup covers the abort/cancel path (wizardAbort runs the - // registered cleanups; the success path never calls them) - // - the finally covers normal completion and any direct throw that unwinds - // through here - // flushScanReport is idempotent (it zeroes scan state), so the overlap is a - // harmless no-op. No harness has to know reporting exists. - registerCleanup(() => flushScanReport(session)); + // termination path and for every harness (linear, orchestrator, or future). + // flushScanReport is idempotent (it zeroes scan state), so a caller that also + // flushes from its own cleanup path sees a harmless no-op. No harness has to + // know reporting exists. try { - const binding = resolveProgramRunner( - session, - programConfig, - boot, - options.composed ?? false, - ); - if (binding.sequence === Sequence.orchestrator) { - getUI().log.info('Task-queue orchestrator enabled.'); + const boot = await prepareRun(config, input); + if (config.binding.sequence === Sequence.orchestrator) { + log('Task-queue orchestrator enabled.'); } - return await getSequence(binding.sequence).run( - session, + logToFile( + `[agent-runner] run program=${config.programId} sequence=${config.binding.sequence}` + + ` harness=${config.binding.harness} composed=${config.composed}`, + ); + const result = await getSequence(config.binding.sequence).run({ config, - programConfig, + input, boot, - options.composed ?? false, - ); + emit, + interaction: options.interaction, + signal, + }); + return { + ...result, + skillId: input.skillId, + snapshot: collector.snapshot(), + }; + } catch (error) { + // Not a decision the agent made. Hand it back whole rather than throw, so + // every ending of a run is a result the caller reads the same way. + const failure = classifyRunFailure(error); + logToFile('[agent-runner] run crashed:', error); + return { + outcome: 'crashed', + skillId: input.skillId, + failure: { + code: failure.code, + message: failure.message, + error: error instanceof Error ? error : new Error(String(error)), + }, + snapshot: collector.snapshot(), + }; } finally { - flushScanReport(session); + flushScanReport({ yaraReport: input.flags.yaraReport }, log); } } - -/** - * Resolve which sequence and harness will run a program (CLI → PostHog flag → - * per-program binding → default), tag both axes onto analytics, and return the - * binding for downstream dispatch. - * - * The one place `runner/index.ts` reaches into the switchboard — every other - * concern (bootstrap, cleanup, dispatch, per-task per-role harness picks) is - * either upstream or downstream of this call. - */ -function resolveProgramRunner( - session: WizardSession, - programConfig: ProgramConfig, - boot: BootstrapResult, - composed: boolean, -): ProgramBinding { - const ctx = { - program: programConfig.id, - composed, - flags: boot.wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliSequence: session.sequence, - cliModel: session.model, - }; - const binding = resolveBinding(ctx); - tagBinding(boot, binding); - captureSwitchboardDecision(ctx, binding); - return binding; -} - -/** - * One event + one log line per run: what entered the switchboard, which - * precedence rung decided each axis, and the final pick. - */ -function captureSwitchboardDecision( - ctx: SwitchboardCtx, - binding: ProgramBinding, -): void { - const trace = ctx.trace ?? {}; - // Unpinned orchestrator runs choose a model per task from the context-mill agent prompts; the orchestrator logs that map once the prompts load. - const perTaskModel = - binding.sequence === Sequence.orchestrator && trace.model === 'binding'; - const model = perTaskModel ? 'chosen-per-task' : binding.model; - const modelSource = perTaskModel ? 'agent-prompts' : trace.model; - analytics.wizardCapture('switchboard resolved', { - program: ctx.program, - flag_self_driving_use_pi_harness: - ctx.flags[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY], - flag_self_driving_pi_payload: JSON.stringify( - ctx.flagPayloads?.[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY] ?? null, - ), - flag_orchestrator: ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY], - cli_harness: ctx.cliHarness, - cli_sequence: ctx.cliSequence, - cli_model: ctx.cliModel, - harness_source: trace.harness, - model_source: modelSource, - sequence_source: trace.sequence, - harness: binding.harness, - model, - thinking_level: binding.thinkingLevel, - sequence: binding.sequence, - }); - logToFile( - `[switchboard] decision: program=${ctx.program}` + - ` in(orchestrator=${ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY] ?? '-'},` + - ` cli=${ctx.cliHarness ?? '-'}/${ctx.cliSequence ?? '-'}/${ - ctx.cliModel ?? '-' - })` + - ` → harness=${binding.harness} (${trace.harness ?? '?'})` + - ` model=${model} (${modelSource ?? '?'})` + - ` sequence=${binding.sequence} (${trace.sequence ?? '?'})`, - ); -} - -/** - * Tag the run with its two routing axes. Sequence is stable for the whole - * run; harness reflects the run-level (default-role) resolution — orchestrator - * per-task calls emit their own `harness` property in their events so per-task - * aggregations attribute correctly. - */ -function tagBinding(boot: BootstrapResult, binding: ProgramBinding): void { - analytics.setTag('sequence', binding.sequence); - analytics.setTag('harness', binding.harness); - boot.wizardMetadata.SEQUENCE = binding.sequence; - boot.wizardMetadata.HARNESS = binding.harness; -} diff --git a/src/lib/agent/runner/sequence/linear.ts b/src/lib/agent/runner/sequence/linear.ts index 94ab86841..fa3794233 100644 --- a/src/lib/agent/runner/sequence/linear.ts +++ b/src/lib/agent/runner/sequence/linear.ts @@ -1,112 +1,127 @@ /** * The linear pipeline. Single execution path for all non-orchestrator programs, * both skill-based (revenue analytics) and framework-based (core integration). - * The `ProgramRun` controls what varies between them; `programConfig` carries the - * program-level static metadata (tool allow/disallow lists, etc.). + * The `AgentRunDefinition` controls what varies between them; `RunConfig` + * carries the program-level static metadata (tool allow/disallow lists, etc.). + * + * Reports through `emit`, asks through `interaction`, and returns a decided + * `RunResult`. Every former `getUI()` call is one progress event in the same + * place; every former `wizardAbort` is a returned failure with the same + * arguments, so the caller's exit sequence is unchanged. */ -import type { WizardSession } from '../../../wizard-session'; -import { OutroKind } from '../../../wizard-session'; -import { getUI } from '../../../../ui'; +import { OutroKind, type OutroData } from '@lib/wizard-session'; import { AgentErrorType, AgentSignals } from '../../agent-interface'; -import { restoreClaudeSettings } from '../../claude-settings'; import { logToFile } from '../../../../utils/debug'; import { createBenchmarkPipeline } from '../../../middleware/benchmark'; -import { - wizardAbort, - WizardError, - registerCleanup, -} from '../../../../utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; import { ErrorCodes, AGENT_ERROR_CODE } from '@lib/errors'; import { analytics } from '../../../../utils/analytics'; -import { - formatScanReport, - formatYaraAbortMessage, - writeScanReport, -} from '../../../yara-hooks'; +import { formatYaraAbortMessage } from '../../../yara-hooks'; import { installSkillById } from '../../../wizard-tools'; -import { createWizardAskBridge } from '../../../wizard-ask-bridge'; -import type { ProgramConfig } from '../../../programs/program-step'; import { assemblePrompt } from '../../agent-prompt'; -import type { ProgramRun, BootstrapResult } from '../shared/types'; -import { abortOnInstallFailure } from '../shared/errors'; -import { shouldDisableAsk, sessionToOptions } from '../shared/bootstrap'; -import { resolveHarness, getHarness } from '../switchboard'; +import type { + AgentFailure, + RunResult, + RunSnapshot, + SequenceContext, +} from '../shared/types'; +import { installFailure } from '../shared/errors'; +import { shouldDisableAsk, runOptions } from '../shared/bootstrap'; +import { createEmitSpinner } from '../shared/progress-collector'; +import { createAskBridge } from '../shared/ask'; +import { getHarness } from '../switchboard'; -export async function runLinearProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, - composed = false, -): Promise { - const { skillsBaseUrl, credentials, wizardFlags, project } = boot; +/** The snapshot the sequence returns; `runAgent` fills it from its collector. */ +const PENDING_SNAPSHOT: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, +}; + +const failed = (failure: AgentFailure): RunResult => ({ + outcome: 'failed', + failure, + snapshot: PENDING_SNAPSHOT, +}); + +const cancelled = (): RunResult => ({ + outcome: 'cancelled', + failure: { message: 'Wizard setup cancelled.' }, + snapshot: PENDING_SNAPSHOT, +}); + +export async function runLinearProgram({ + config, + input, + boot, + emit, + interaction, + signal, +}: SequenceContext): Promise { + const { run, composed } = config; + const { skillsBaseUrl, credentials, project } = boot; const { projectApiKey, host, projectId } = credentials; + if (signal.aborted) return cancelled(); + // 5. Skill install (if skillId provided) let skillPath: string | undefined; - if (config.skillId) { - logToFile(`[agent-runner] installing skill ${config.skillId}`); + if (run.skillId) { + logToFile(`[agent-runner] installing skill ${run.skillId}`); const installResult = await installSkillById( - config.skillId, - session.installDir, + run.skillId, + input.installDir, skillsBaseUrl, { triage: boot.triageProvider }, ); if (installResult.kind !== 'ok') { - await abortOnInstallFailure(config.integrationLabel, installResult); - return; + return failed(installFailure(run.integrationLabel, installResult)); } skillPath = installResult.path; logToFile(`[agent-runner] skill installed at ${skillPath}`); } - // 6. Initialize agent - const spinner = getUI().spinner(); - - const restoreSettings = () => restoreClaudeSettings(session.installDir); - getUI().onEnterScreen('outro', restoreSettings); + if (signal.aborted) return cancelled(); - if (session.yaraReport) { - registerCleanup(() => { - const reportPath = writeScanReport(); - if (reportPath) { - const summary = formatScanReport(); - getUI().log.info(`YARA scan report: ${reportPath}${summary ?? ''}`); - } - }); - } + // 6. Initialize agent + const spinner = createEmitSpinner(emit); - getUI().startRun(); + emit({ kind: 'lifecycle', phase: 'started' }); // wizard_ask needs an answerer. A human answers at the keyboard; the e2e // snapshot/MCP host answers via its driver and sets WIZARD_ASK_AUTODRIVE. // CI/signup with neither has no answerer, so we omit the bridge and the tool // returns an actionable error rather than hanging on a never-resolving prompt. const askDisabled = - shouldDisableAsk(session) && process.env.WIZARD_ASK_AUTODRIVE !== '1'; - const askBridge = askDisabled + shouldDisableAsk(input.flags) && process.env.WIZARD_ASK_AUTODRIVE !== '1'; + const ask = askDisabled ? undefined - : createWizardAskBridge({ - getSource: () => session.skillId ?? config.integrationLabel, - showQuestion: (q) => getUI().requestQuestion(q), - cancelQuestion: () => getUI().cancelPendingQuestion(), - richLinks: config.richLinks ?? false, - timeoutMs: config.askTimeoutMs, + : createAskBridge(interaction, signal, { + getSource: () => input.skillId ?? run.integrationLabel, + richLinks: run.richLinks ?? false, + timeoutMs: run.askTimeoutMs, }); - const middleware = session.benchmark - ? createBenchmarkPipeline(spinner, sessionToOptions(session)) + const middleware = input.flags.benchmark + ? createBenchmarkPipeline(spinner, runOptions(input), undefined, { + log: (message) => emit({ kind: 'log', level: 'info', message }), + }) : undefined; // 7. Build prompt - const prompt = assemblePrompt(config, { + const prompt = assemblePrompt(run, { projectId, projectApiKey, host, skillPath, orgAiDataProcessingApproved: - session.apiUser?.organization?.is_ai_data_processing_approved ?? null, + input.apiUser?.organization?.is_ai_data_processing_approved ?? null, teamProductOptIns: project ? { sessionReplay: project.session_recording_opt_in ?? null, @@ -117,37 +132,35 @@ export async function runLinearProgram( }); logToFile(`[agent-runner] prompt assembled (${prompt.length} chars)`); - // 8. Resolve the (runner, model) pair from the central plan and run the agent - // through the selected runner. The runner owns the agent loop + model - // transport; everything around it (skill install, prompt, ask bridge, error - // routing, outro) stays here so every runner shares it. - const pick = resolveHarness({ - program: programConfig.id, - flags: wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliModel: session.model, - }); - const agentResult = await getHarness(pick.harness).run({ - session, + // 8. Run the agent through the run-level harness. The harness owns the agent + // loop + model transport; everything around it (skill install, prompt, ask + // bridge, error routing, outro) stays here so every harness shares it. + const { harness, model, thinkingLevel } = config.binding; + const agentResult = await getHarness(harness).run({ config, - programConfig, + input, boot, + emit, prompt, skillPath, spinner, - askBridge, + askBridge: ask?.bridge, + getPendingQuestion: ask?.getPendingQuestion, middleware, - model: pick.model, - thinkingLevel: pick.thinkingLevel, + model, + thinkingLevel, }); - // 9. Error handling (full set from both runners) + // 9. Error handling (full set from both harnesses) + if (agentResult.failure) { + return failed(agentResult.failure); + } + if (agentResult.error === AgentErrorType.ABORT) { const reason = agentResult.message ?? ''; - const matched = config.abortCases?.find((c) => c.match.test(reason)); + const matched = run.abortCases?.find((c) => c.match.test(reason)); const abortCode = matched?.errorCode ?? ErrorCodes.AgentAbort; - const outroData: WizardSession['outroData'] = matched + const outroData: OutroData = matched ? { kind: OutroKind.Error, message: matched.message, @@ -158,44 +171,48 @@ export async function runLinearProgram( } : { kind: OutroKind.Error, - message: `${config.integrationLabel} aborted`, + message: `${run.integrationLabel} aborted`, body: reason || 'The agent aborted the program.', - docsUrl: config.docsUrl, + docsUrl: run.docsUrl, errorCode: abortCode, errorDetail: { reason }, }; analytics.wizardCapture('agent aborted', { - integration: config.integrationLabel, + integration: run.integrationLabel, reason, matched: matched?.message ?? null, }); - await wizardAbort({ - outroData, - code: abortCode, - error: new WizardError( - `Agent aborted: ${reason}`, - { - integration: config.integrationLabel, - error_type: AgentErrorType.ABORT, - reason, - }, - abortCode, - ), - }); + return { + outcome: 'aborted', + failure: { + outroData, + code: abortCode, + error: new WizardError( + `Agent aborted: ${reason}`, + { + integration: run.integrationLabel, + error_type: AgentErrorType.ABORT, + reason, + }, + abortCode, + ), + }, + snapshot: PENDING_SNAPSHOT, + }; } if (agentResult.error === AgentErrorType.MCP_MISSING) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.MCP_MISSING], message: 'Could not access the PostHog MCP server\n\n' + 'The wizard was unable to connect to the PostHog MCP server.\n' + 'This could be due to a network issue or a configuration problem.\n\n' + - `Please try again, or check the documentation:\n${config.docsUrl}`, + `Please try again, or check the documentation:\n${run.docsUrl}`, error: new WizardError( 'Agent could not access PostHog MCP server', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.MCP_MISSING, signal: AgentSignals.ERROR_MCP_MISSING, }, @@ -205,16 +222,16 @@ export async function runLinearProgram( } if (agentResult.error === AgentErrorType.RESOURCE_MISSING) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.RESOURCE_MISSING], message: 'Could not access the setup resource\n\n' + 'This may indicate a version mismatch or a temporary service issue.\n\n' + - `Please try again, or check the documentation:\n${config.docsUrl}`, + `Please try again, or check the documentation:\n${run.docsUrl}`, error: new WizardError( 'Agent could not access setup resource', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.RESOURCE_MISSING, signal: AgentSignals.ERROR_RESOURCE_MISSING, }, @@ -224,13 +241,13 @@ export async function runLinearProgram( } if (agentResult.error === AgentErrorType.YARA_VIOLATION) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.YARA_VIOLATION], message: formatYaraAbortMessage(), error: new WizardError( 'YARA scanner terminated session', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.YARA_VIOLATION, }, AGENT_ERROR_CODE[AgentErrorType.YARA_VIOLATION], @@ -240,10 +257,10 @@ export async function runLinearProgram( if (agentResult.error === AgentErrorType.NO_PROGRESS) { analytics.wizardCapture('agent no progress', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.NO_PROGRESS, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], message: 'The Wizard exited without changing your project. Please contact the ' + @@ -251,7 +268,7 @@ export async function runLinearProgram( error: new WizardError( 'Agent made no progress', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.NO_PROGRESS, }, AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], @@ -261,10 +278,10 @@ export async function runLinearProgram( if (agentResult.error === AgentErrorType.INCOMPLETE_TASKS) { analytics.wizardCapture('agent incomplete tasks', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.INCOMPLETE_TASKS, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.INCOMPLETE_TASKS], message: 'The Wizard exited without completing its planned tasks. Please contact ' + @@ -272,7 +289,7 @@ export async function runLinearProgram( error: new WizardError( 'Agent left planned tasks incomplete', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.INCOMPLETE_TASKS, }, AGENT_ERROR_CODE[AgentErrorType.INCOMPLETE_TASKS], @@ -285,12 +302,12 @@ export async function runLinearProgram( agentResult.error === AgentErrorType.API_ERROR ) { analytics.wizardCapture('agent api error', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: agentResult.error, error_message: agentResult.message, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[agentResult.error], message: `API Error\n\n${ agentResult.message || 'Unknown error' @@ -298,7 +315,7 @@ export async function runLinearProgram( error: new WizardError( `API error: ${agentResult.message}`, { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: agentResult.error, }, AGENT_ERROR_CODE[agentResult.error], @@ -307,39 +324,36 @@ export async function runLinearProgram( } // 10. Post-run hooks - if (config.postRun) { - await config.postRun(session, credentials); + if (config.hooks?.postRun) { + await config.hooks.postRun(credentials); } // A composed sub-run (integration inside self-driving) skips the terminal // outro + analytics shutdown so the shared client survives the host's run. - if (composed) return; + if (composed) { + return { outcome: 'success', snapshot: PENDING_SNAPSHOT }; + } // 11. Outro - // Push outro data through the UI (not via direct `session.outroData = ...` - // mutation) so the live store gets the value. agent-runner's `session` - // parameter is captured at runAgent() invocation time, and any `setKey` - // call between then and here (e.g. setDashboardUrl, setNotebookUrl) forks - // the session reference — direct mutation then lands on a stale snapshot - // that the screen never reads. UI.setOutroData() goes through the store - // and also merges in any post-snapshot URLs from the live session. - const outroData = config.buildOutroData - ? config.buildOutroData(session, credentials) + const outroData: OutroData | undefined = config.hooks?.buildOutroData + ? config.hooks.buildOutroData(credentials) : { kind: OutroKind.Success, - message: config.successMessage, - reportFile: config.reportFile, - docsUrl: config.docsUrl, - continueUrl: session.signup + message: run.successMessage, + reportFile: run.reportFile, + docsUrl: run.docsUrl, + continueUrl: input.flags.signup ? `${host.appHost}/products?source=wizard` : undefined, }; if (outroData) { - getUI().setOutroData(outroData); + emit({ kind: 'completion', outro: outroData }); } - getUI().outro(config.successMessage); + emit({ kind: 'lifecycle', phase: 'completed', message: run.successMessage }); // 12. Analytics shutdown await analytics.shutdown('success'); + + return { outcome: 'success', outro: outroData, snapshot: PENDING_SNAPSHOT }; } diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts index d2379b32c..9ee3fca2a 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts @@ -10,9 +10,6 @@ import * as fs from 'fs'; import * as os from 'os'; import * as path from 'path'; -vi.mock('@ui', () => ({ - getUI: () => ({ showTaskNotice: vi.fn(), cancelTaskNotice: vi.fn() }), -})); vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture: vi.fn(), diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts index c6acf1014..40c6642f8 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts @@ -19,9 +19,6 @@ const { showTaskNotice, cancelTaskNotice, wizardCapture, captureException } = captureException: vi.fn(), })); -vi.mock('@ui', () => ({ - getUI: () => ({ showTaskNotice, cancelTaskNotice }), -})); vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture, @@ -38,6 +35,12 @@ import { TASK_NOTICE_TIMEOUT_MS, } from '@lib/agent/runner/sequence/orchestrator/orchestrator-runner'; +/** The answerer under test, standing where `getUI()` used to. */ +const interaction = { + taskNotice: (notice: TaskNotice) => showTaskNotice(notice), + cancelTaskNotice: () => cancelTaskNotice(), +}; + const NOTICE: TaskNotice = { title: 'Connect your data sources', body: ['We found some sources.'], @@ -67,7 +70,7 @@ describe('task notice timeout', () => { // Nobody presses anything. showTaskNotice.mockReturnValue(new Promise(() => undefined)); - const promise = offerSeededTask(NOTICE, 1000); + const promise = offerSeededTask(NOTICE, 1000, interaction); vi.advanceTimersByTime(1000); await expect(promise).resolves.toEqual({ keep: false, timedOut: true }); @@ -83,7 +86,7 @@ describe('task notice timeout', () => { try { showTaskNotice.mockResolvedValue(true); - const result = await offerSeededTask(NOTICE, 1000); + const result = await offerSeededTask(NOTICE, 1000, interaction); expect(result).toEqual({ keep: true, timedOut: false }); vi.advanceTimersByTime(5000); @@ -102,10 +105,12 @@ describe('task notice timeout', () => { // Both decline, but only one of them means "the user was not there" — // the run reports them differently, and now skips them differently too. - await expect(offerSeededTask(NOTICE, 1000)).resolves.toEqual({ - keep: false, - timedOut: false, - }); + await expect(offerSeededTask(NOTICE, 1000, interaction)).resolves.toEqual( + { + keep: false, + timedOut: false, + }, + ); expect(cancelTaskNotice).not.toHaveBeenCalled(); } finally { vi.useRealTimers(); @@ -164,7 +169,9 @@ describe('askSeededConsent', () => { it('records an acceptance', async () => { showTaskNotice.mockResolvedValue(true); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, 1000, interaction), + ).resolves.toEqual({ keep: true, timedOut: false, errored: false, @@ -174,7 +181,9 @@ describe('askSeededConsent', () => { it('records an explicit decline', async () => { showTaskNotice.mockResolvedValue(false); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, 1000, interaction), + ).resolves.toEqual({ keep: false, timedOut: false, errored: false, @@ -186,7 +195,7 @@ describe('askSeededConsent', () => { try { showTaskNotice.mockReturnValue(new Promise(() => undefined)); - const promise = askSeededConsent('warehouse', NOTICE, 1000); + const promise = askSeededConsent('warehouse', NOTICE, 1000, interaction); vi.advanceTimersByTime(1000); await expect(promise).resolves.toEqual({ @@ -202,7 +211,7 @@ describe('askSeededConsent', () => { it('reports the answer on one event per offer', async () => { showTaskNotice.mockResolvedValue(true); - await askSeededConsent('warehouse', NOTICE, 1000); + await askSeededConsent('warehouse', NOTICE, 1000, interaction); expect(wizardCapture).toHaveBeenCalledTimes(1); expect(wizardCapture).toHaveBeenCalledWith( @@ -216,7 +225,9 @@ describe('askSeededConsent', () => { // that could not be put to the user must never be read as a yes. showTaskNotice.mockRejectedValue(new Error('UI blew up')); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, 1000, interaction), + ).resolves.toEqual({ keep: false, timedOut: false, errored: true, @@ -247,7 +258,9 @@ describe('offering notices from the seed loop', () => { const answers: { keep: boolean; timedOut: boolean; errored: boolean }[] = []; for (const entry of entries) { - answers.push(await askSeededConsent(entry.type, NOTICE, 60_000)); + answers.push( + await askSeededConsent(entry.type, NOTICE, 60_000, interaction), + ); } return answers; }; diff --git a/src/lib/agent/runner/sequence/orchestrator/executor.ts b/src/lib/agent/runner/sequence/orchestrator/executor.ts index 616d29442..e17e8c269 100644 --- a/src/lib/agent/runner/sequence/orchestrator/executor.ts +++ b/src/lib/agent/runner/sequence/orchestrator/executor.ts @@ -10,6 +10,7 @@ * injected: the real one spins up a fresh agent, the tests use a fake. */ import { analytics } from '@utils/analytics'; +import type { AgentFailure } from '../../shared/types'; import { logToFile } from '@utils/debug'; import { TaskStatus, type QueueStore, type QueuedTask } from './queue'; @@ -33,6 +34,19 @@ export type TaskResolver = ( * (via the task agent calling complete_task). */ export type RunTask = (task: QueuedTask) => Promise; +/** + * Thrown by a `RunTask` when its harness returned a decided failure that ends + * the whole run (a 401 the auth screen already reported), not just the task. + * `drainQueue` lets it through so the sequence can return the failure where + * the harness used to exit the process. + */ +export class RunTaskFatal extends Error { + constructor(public readonly failure: AgentFailure) { + super(failure.message ?? 'agent run failed'); + this.name = 'RunTaskFatal'; + } +} + export interface DrainOptions { /** Backstop against a pathological always-one-more-pending loop. */ maxStarts: number; @@ -51,6 +65,7 @@ async function runOne( try { await runTask(task); } catch (error) { + if (error instanceof RunTaskFatal) throw error; // The task threw rather than reporting. The outcome check below handles // the queue; the exception itself should never be silent. logToFile(`[executor] runTask threw for ${task.type}:`, error); diff --git a/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts b/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts index 8778431ad..502b3a02c 100644 --- a/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts +++ b/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts @@ -20,31 +20,28 @@ import { writeFileSync, } from 'fs'; import * as path from 'path'; -import { - OutroKind, - type TaskNotice, - type WizardSession, -} from '@lib/wizard-session'; -import { - POSTHOG_DOCS_URL, - WIZARD_CONTACT_EMAIL, - type Integration, -} from '@lib/constants'; -import { FRAMEWORK_REGISTRY } from '@lib/registry'; +import { OutroKind, type TaskNotice } from '@lib/wizard-session'; +import { POSTHOG_DOCS_URL, WIZARD_CONTACT_EMAIL } from '@lib/constants'; import { installSkillById, fetchSkillMenu, type SkillEntry, } from '@lib/wizard-tools'; -import { getUI } from '@ui'; import { analytics } from '@utils/analytics'; import { ciExcludedTaskTypes } from '@utils/ci-flag-overrides'; import { logToFile } from '@utils/debug'; import { ringTerminalBell } from '@utils/terminal-bell'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; import { ErrorCodes } from '@lib/errors'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { BootstrapResult, ProgramRun } from '../../shared/types'; +import type { AgentInteraction } from '@lib/agent/progress'; +import type { + AgentFailure, + RunResult, + RunSnapshot, + SequenceContext, +} from '../../shared/types'; +import { createEmitSpinner } from '../../shared/progress-collector'; +import { createAskBridge } from '../../shared/ask'; import { areSeededTasksEnabled, getHarness, @@ -61,14 +58,11 @@ import { TaskStatus, type QueuedTask, } from './queue'; -import { drainQueue, type RunTask } from './executor'; +import { drainQueue, RunTaskFatal, type RunTask } from './executor'; import { RunMetrics } from './run-metrics'; import { dependencyClosure, uncoveredBySink } from './queue-tools'; import { deferSeededTasks } from './seeded-deps'; -import { - createWizardAskBridge, - LONGER_ASK_TIMEOUT_MS, -} from '@lib/wizard-ask-bridge'; +import { LONGER_ASK_TIMEOUT_MS } from '@lib/wizard-ask-bridge'; import { shouldDisableAsk } from '../../shared/bootstrap'; import { agentRunTools, @@ -183,7 +177,7 @@ export function resolveSkillVariantId( } /** - * The framework reference is the full `integration` skill. `session.skillId` is + * The framework reference is the full `integration` skill. `input.skillId` is * the bare framework (e.g. `django`), but the skill menu ids it as * `integration-`. */ @@ -222,7 +216,14 @@ export const TASK_NOTICE_TIMEOUT_MS = 5 * 60 * 1000; export async function offerSeededTask( notice: TaskNotice, timeoutMs: number = TASK_NOTICE_TIMEOUT_MS, + interaction?: AgentInteraction, + signal: AbortSignal = new AbortController().signal, ): Promise<{ keep: boolean; timedOut: boolean }> { + // No one to show the notice to: a step nobody can answer for must not run. + // The same answer a non-interactive host gives today. + if (!interaction?.taskNotice) return { keep: false, timedOut: false }; + const { taskNotice, cancelTaskNotice } = interaction; + let timer: ReturnType | undefined; let timedOut = false; const timeout = new Promise((resolve) => { @@ -230,12 +231,12 @@ export async function offerSeededTask( timedOut = true; // Dismisses the overlay and settles the showTaskNotice promise too, so // the losing side of the race cannot leave a modal on screen. - getUI().cancelTaskNotice(); + cancelTaskNotice?.(); resolve(false); }, timeoutMs); }); try { - const keep = await Promise.race([getUI().showTaskNotice(notice), timeout]); + const keep = await Promise.race([taskNotice(notice, { signal }), timeout]); return { keep, timedOut }; } finally { if (timer) clearTimeout(timer); @@ -337,8 +338,15 @@ export async function askSeededConsent( type: string, notice: TaskNotice, timeoutMs?: number, + interaction?: AgentInteraction, + signal?: AbortSignal, ): Promise { - const consent = await offerSeededTask(notice, timeoutMs).then( + const consent = await offerSeededTask( + notice, + timeoutMs, + interaction, + signal, + ).then( (answer): SeededConsent => ({ ...answer, errored: false }), (err: unknown): SeededConsent => { logToFile( @@ -427,33 +435,58 @@ export function displayOrder( .map((entry) => entry.task); } -export async function runOrchestrator( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, -): Promise { +/** The snapshot the sequence returns; `runAgent` fills it from its collector. */ +const PENDING_SNAPSHOT: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, +}; + +const failed = (failure: AgentFailure): RunResult => ({ + outcome: 'failed', + failure, + snapshot: PENDING_SNAPSHOT, +}); + +export async function runOrchestrator({ + config, + input, + boot, + emit, + interaction, + signal, +}: SequenceContext): Promise { const runId = randomUUID(); + const { run } = config; + const programId = config.programId; + + if (signal.aborted) { + return { + outcome: 'cancelled', + failure: { message: 'Wizard setup cancelled.' }, + snapshot: PENDING_SNAPSHOT, + }; + } // Switchboard context — reused for every per-role harness resolution below. - const switchboardCtx = { - program: programConfig.id, - flags: boot.wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliSequence: session.sequence, - cliModel: session.model, - }; + // The caller resolved the run-level binding from it; per-task roles overlay + // `binding.contextMillOverride[role]` on the same inputs. + const switchboardCtx = { ...config.switchboard, trace: undefined }; // The WHAT (agent prompts) is served from context-mill. Fetch the registry // once up front: its types drive enqueue validation, and resolving a task to // its run config is then synchronous, with no mid-drain network latency. - const flow = programConfig.agentFlow ?? programConfig.id; + const flow = config.agentFlow ?? programId; const registry = await loadAgentRegistry(boot.skillsBaseUrl, flow, { exclude: ciExcludedTaskTypes(), // Baked into the prompts at load, so enqueue, dispatch, and telemetry all read one effective spec. overrides: resolveStageOverrides( - programConfig.id, + programId, boot.wizardFlags, boot.wizardFlagPayloads, ), @@ -492,7 +525,7 @@ export async function runOrchestrator( ? Date.parse(t.finishedAt) - Date.parse(t.startedAt) : undefined; - const store = new QueueStore(session.installDir, runId, { + const store = new QueueStore(input.installDir, runId, { onTransition: (event, task) => { const pick = resolveHarness(switchboardCtx, task.type); // Mirror dispatch's allow-list fallback so attribution names the model that runs. @@ -565,20 +598,20 @@ export async function runOrchestrator( let commandmentsPath: string | undefined; let referenceInstallPath: string | undefined; const menuSkillEntries = await fetchSkillMenuEntries(boot.skillsBaseUrl); - // The framework key for reference + variant resolution. `session.integration` - // is the detected framework and always wins; `session.skillId` is the - // fallback for the basic-integration path, where bootstrap sets it to the + // The framework key for reference + variant resolution. `input.integration` + // is the detected framework and always wins; `input.skillId` is the + // fallback for the basic-integration path, where the caller sets it to the // framework label. Programs whose run config carries their own skill id // (agent-skill commands like replay-vision) would otherwise leak that id in - // here as a bogus framework after bootstrap overwrites the detect result. - const framework = session.integration ?? session.skillId ?? undefined; + // here as a bogus framework after the caller overwrites the detect result. + const framework = input.integration ?? input.skillId ?? undefined; const referenceSkillId = framework ? resolveReferenceSkillId(menuSkillEntries, framework) : undefined; if (referenceSkillId) { const ref = await installSkillById( referenceSkillId, - session.installDir, + input.installDir, boot.skillsBaseUrl, { skillsRoot: path.join(QUEUE_DIR_NAME, 'reference'), @@ -588,11 +621,11 @@ export async function runOrchestrator( if (ref.kind === 'ok') { referenceInstallPath = ref.path; const example = path.join(ref.path, 'references', 'EXAMPLE.md'); - if (existsSync(path.join(session.installDir, example))) { + if (existsSync(path.join(input.installDir, example))) { examplePath = example; } const commandments = path.join(ref.path, 'references', 'COMMANDMENTS.md'); - if (existsSync(path.join(session.installDir, commandments))) { + if (existsSync(path.join(input.installDir, commandments))) { commandmentsPath = commandments; } } else { @@ -627,11 +660,9 @@ export async function runOrchestrator( } } if (missingVariants.length > 0) { - // The framework's own docs page from its config; generic docs when detection found none. - const docsUrl = framework - ? FRAMEWORK_REGISTRY[framework as Integration]?.metadata.docsUrl - : undefined; - await wizardAbort({ + // The framework's own docs page, resolved by the caller; generic docs when detection found none. + const docsUrl = framework ? input.frameworkDocsUrl : undefined; + return failed({ code: ErrorCodes.AgentOrchestratorSkillVariantMissing, message: 'Setup instructions for this project failed to download.\n' + @@ -662,27 +693,28 @@ export async function runOrchestrator( }; logToFile( - `[orchestrator] START program=${programConfig.id} dir=${session.installDir} run=${runId}`, + `[orchestrator] START program=${programId} dir=${input.installDir} run=${runId}`, ); analytics.wizardCapture('orchestrator started', { - program_id: programConfig.id, + program_id: programId, }); - getUI().startRun(); + emit({ kind: 'lifecycle', phase: 'started' }); // Label precedence: what the orchestrator set at enqueue, then the agent // prompt's default, then the bare type. const labelFor = (t: { type: string; label?: string }) => t.label ?? registry.get(t.type)?.label ?? t.type; const renderQueue = () => - getUI().syncTodos( - displayOrder(store.list(), (t) => + emit({ + kind: 'tasks', + tasks: displayOrder(store.list(), (t) => registry.runnerSeededTypes.includes(t.type), ).map((t) => ({ content: labelFor(t), status: toTodoStatus(t.status), activeForm: labelFor(t), })), - ); + }); // Each task's run binds the wizard-tools MCP server to a per-task // orchestrator context so complete_task / enqueue_task attribute correctly @@ -709,7 +741,7 @@ export async function runOrchestrator( // Kill switch: off (or unset), the wizard queues nothing itself and the run // is byte-identical to a project with no detected sources. const seedEntries = areSeededTasksEnabled(boot.wizardFlags) - ? programConfig.seedTasks?.(session) ?? [] + ? config.seedTasks?.() ?? [] : []; const seededTypes: string[] = []; // Kept so their dependencies can be resolved once the planner has run — they @@ -758,7 +790,13 @@ export async function runOrchestrator( // not, which is why the offer lives here and not there. seededConsent.set( task.id, - await askSeededConsent(seeded.type, seeded.notice), + await askSeededConsent( + seeded.type, + seeded.notice, + undefined, + interaction, + signal, + ), ); } logToFile(`[orchestrator] runner-seeded task ${seeded.type}`); @@ -790,11 +828,11 @@ export async function runOrchestrator( // One bridge for the run, handed only to a task whose prompt allows asking. // Absent in CI and signup, where nobody can answer. - const askBridge = shouldDisableAsk(session) + const askBridge = shouldDisableAsk(input.flags) ? undefined - : createWizardAskBridge({ - getSource: () => session.skillId ?? programConfig.id, - showQuestion: (q) => { + : createAskBridge(interaction, signal, { + getSource: () => input.skillId ?? programId, + beforeShow: () => { // How late the first ask lands is the measure of this run shape: it // should follow the autonomous work, not interrupt it. metrics.recordAsk(Date.now()); @@ -804,16 +842,14 @@ export async function runOrchestrator( // unanswered ask still times out into the deep-link fallback — it just // gives a person who stepped away a chance to come back first. ringTerminalBell(); - return getUI().requestQuestion(q); }, - cancelQuestion: () => getUI().cancelPendingQuestion(), - richLinks: config.richLinks ?? false, + richLinks: run.richLinks ?? false, // A task ask waits on a person, and the drain waits it out — the // executor holds the task's promise — so this is the only real limit. timeoutMs: LONGER_ASK_TIMEOUT_MS, - }); + })?.bridge; - const spinner = getUI().spinner(); + const spinner = createEmitSpinner(emit); // 1. Seed the queue with the orchestrator agent. It is itself an agent prompt // (the WHAT), so its model and tools come from its frontmatter. The seed @@ -826,9 +862,10 @@ export async function runOrchestrator( const seedHarness = requireTaskHarness(seedPick); const seedModel = promptModelFor(seedPrompt, seedPick.harness); const seedResult = await seedHarness.runTask({ - session, - programConfig, + config, + input, boot, + emit, prompt: assembleSeedPrompt(promptContext, seedPrompt.body, store.list()), spinner, model: requireKnownModel(seedModel.model, seedPick.model), @@ -841,6 +878,9 @@ export async function runOrchestrator( requestRemark: false, analyticsProperties: { task_type: 'seed', harness: seedPick.harness }, }); + // A decided failure (a 401 the harness already reported) ends the run here, + // before anything is queued, exactly where the harness used to exit. + if (seedResult.failure) return failed(seedResult.failure); if (seedResult.error) { logToFile( `[orchestrator] seed error: ${seedResult.error} ${ @@ -941,7 +981,7 @@ export async function runOrchestrator( analytics.wizardCapture('orchestrator sink invariant violated', { uncovered_types: unwaited.map((t) => t.type), }); - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorSinkInvariant, message: `The wizard could not plan this setup: the final step would have skipped ${unwaited .map((t) => t.type) @@ -965,7 +1005,7 @@ export async function runOrchestrator( // Task agents can install durable skills mid-run (load_skill), and only the // framework reference docs earn a place — snapshot what was already there so // the sweeps remove exactly what this run added. - const claudeSkillsDir = path.join(session.installDir, '.claude', 'skills'); + const claudeSkillsDir = path.join(input.installDir, '.claude', 'skills'); const preexistingSkills = new Set( existsSync(claudeSkillsDir) ? readdirSync(claudeSkillsDir) : [], ); @@ -998,7 +1038,7 @@ export async function runOrchestrator( } const result = await installSkillById( variantId, - session.installDir, + input.installDir, boot.skillsBaseUrl, { skillsRoot: taskSkillsRoot, triage: boot.triageProvider }, ); @@ -1029,10 +1069,11 @@ export async function runOrchestrator( const taskPick = resolveHarness(switchboardCtx, task.type); const taskHarness = requireTaskHarness(taskPick); const taskModel = taskModelSpec(registry, task, taskPick.harness); - await taskHarness.runTask({ - session, - programConfig, + const taskResult = await taskHarness.runTask({ + config, + input, boot, + emit, prompt: assembleTaskPrompt(promptContext, resolved.prompt, skillPaths), spinner, model: requireKnownModel(taskModel.model, taskPick.model), @@ -1051,6 +1092,10 @@ export async function runOrchestrator( harness: taskPick.harness, }, }); + // A decided failure (a 401 the harness already reported) is the run's, + // not the task's: stop the drain and report it, where the harness used + // to exit the process. + if (taskResult.failure) throw new RunTaskFatal(taskResult.failure); } finally { // Durable skills a task installed are irrelevant to later tasks — and // the sdk harness auto-loads .claude/skills into every agent — so sweep @@ -1074,13 +1119,17 @@ export async function runOrchestrator( renderQueue(); } + let fatal: AgentFailure | undefined; try { await drainQueue(store, runTask); + } catch (error) { + if (!(error instanceof RunTaskFatal)) throw error; + fatal = error.failure; } finally { try { if (referenceSkillId && referenceInstallPath) { promoteReferenceSkill( - path.join(session.installDir, referenceInstallPath), + path.join(input.installDir, referenceInstallPath), claudeSkillsDir, referenceSkillId, ); @@ -1095,7 +1144,7 @@ export async function runOrchestrator( // cache folder (queue, handoffs, reference example, installed task // instructions). The .DELETE-ME.md inside is the fallback if we don't. try { - rmSync(path.join(session.installDir, QUEUE_DIR_NAME), { + rmSync(path.join(input.installDir, QUEUE_DIR_NAME), { recursive: true, force: true, }); @@ -1119,6 +1168,8 @@ export async function runOrchestrator( } } + if (fatal) return failed(fatal); + renderQueue(); const summary = store.summary(); @@ -1172,7 +1223,7 @@ export async function runOrchestrator( ', ', )}.\n\nPlease try again, approving all permissions on the PostHog authorization screen. If it still fails, report it to: ${WIZARD_CONTACT_EMAIL}` : `The wizard was unable to set up PostHog: ${whatFailed}.\n\nPlease report this to: ${WIZARD_CONTACT_EMAIL}`; - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorTasksFailed, message, error: new WizardError( @@ -1193,7 +1244,7 @@ export async function runOrchestrator( // usable model response (e.g. the gateway returned empty completions). // "0/0 completed" is a dead run, not a success with an empty denominator. if (summary.total === 0) { - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorHollowRun, message: `The wizard was unable to set up PostHog: the planning step produced no work, so nothing ran.\n\nPlease try again — and if it happens again, report it to: ${WIZARD_CONTACT_EMAIL}`, error: new WizardError( @@ -1219,19 +1270,20 @@ export async function runOrchestrator( } steps completed${ stepNotes.length > 0 ? ` (${stepNotes.join(', ')})` : '' }.`; - getUI().setOutroData({ + const outro = { kind: OutroKind.Success, message, body: conflict ? `⚠ Build conflict: ${conflict}\nFull details are in the setup report.` : undefined, docsUrl: 'https://posthog.com/docs/ai-engineering/ai-wizard', - nextSteps: config.buildOutroNextSteps?.( - session, + nextSteps: config.hooks?.buildOutroNextSteps?.( boot.credentials, completedSeededTypes(store, seededTasks), ), - }); - getUI().outro(message); + }; + emit({ kind: 'completion', outro }); + emit({ kind: 'lifecycle', phase: 'completed', message }); await analytics.shutdown('success'); + return { outcome: 'success', outro, snapshot: PENDING_SNAPSHOT }; } diff --git a/src/lib/agent/runner/shared/ask.ts b/src/lib/agent/runner/shared/ask.ts new file mode 100644 index 000000000..640c0a781 --- /dev/null +++ b/src/lib/agent/runner/shared/ask.ts @@ -0,0 +1,61 @@ +/** + * The ask bridge over an injected answerer. + * + * `createWizardAskBridge` already owns request ids, the timeout race, the + * `__cancelled__` sentinel and the analytics; it only ever needed a + * `showQuestion` and a `cancelQuestion`. Here those come from + * `AgentInteraction` instead of `getUI()`. With no answerer there is no bridge, + * so `wizard_ask` reports its existing "not available" error rather than + * hanging on a question nobody can see. + */ + +import { + createWizardAskBridge, + type WizardAskBridge, +} from '@lib/wizard-ask-bridge'; +import type { PendingQuestion } from '@lib/wizard-session'; +import type { AgentInteraction } from '@lib/agent/progress'; + +export interface AskBridgeHandle { + bridge: WizardAskBridge; + /** + * The question currently on screen, or null. The anthropic harness reads it + * to block Write/Edit while an overlay is open, the same guard pi keeps in + * its security extension. + */ + getPendingQuestion: () => PendingQuestion | null; +} + +export function createAskBridge( + interaction: AgentInteraction | undefined, + signal: AbortSignal, + options: { + getSource: () => string; + richLinks: boolean; + timeoutMs?: number; + /** Runs before each question is shown (the orchestrator's bell and metric). */ + beforeShow?: () => void; + }, +): AskBridgeHandle | undefined { + const ask = interaction?.ask; + if (!ask) return undefined; + + let pending: PendingQuestion | null = null; + const bridge = createWizardAskBridge({ + getSource: options.getSource, + showQuestion: async (question) => { + options.beforeShow?.(); + pending = question; + try { + return await ask(question, { signal }); + } finally { + pending = null; + } + }, + cancelQuestion: interaction?.cancelAsk, + richLinks: options.richLinks, + timeoutMs: options.timeoutMs, + }); + + return { bridge, getPendingQuestion: () => pending }; +} diff --git a/src/lib/agent/runner/shared/bootstrap.ts b/src/lib/agent/runner/shared/bootstrap.ts index dc12a2b60..1b776a87c 100644 --- a/src/lib/agent/runner/shared/bootstrap.ts +++ b/src/lib/agent/runner/shared/bootstrap.ts @@ -1,46 +1,22 @@ /** - * Shared bootstrap for the runner pipeline. + * Shared preparation for the runner pipeline. * - * Runs before the fork into the linear or orchestrator arm: logging, health - * check, settings conflicts, OAuth and credentials, feature flags, variant - * metadata, and MCP url. Sets `session.credentials`, role, and user as side - * effects. Returns the values the arms still need. + * Runs before the fork into the linear or orchestrator arm: logging targets, + * the gateway mint and the scan-triage classifier built on it. Everything the + * caller must decide first — health gates, settings conflicts, authentication, + * the AI opt-in gate, post-auth gates, feature flags, run tags, token refresh — + * arrives already resolved in `RunConfig` and `RunInput`. */ -import type { WizardSession } from '@lib/wizard-session'; import { analytics } from '@utils/analytics'; -import { getUI } from '@ui'; -import { authenticate, refreshAccessTokenIfNeeded } from './authenticate'; -import { maybeStampAiSdkDetected } from '@lib/programs/posthog-integration/detect'; import { createTriageLLMProvider } from '@lib/agent/triage-provider'; import { gatewayAuth } from '@lib/gateway-session'; -import { resolveHarness } from '../switchboard'; -import { buildRunTags } from '@lib/agent/agent-interface'; -import { - checkAllSettingsConflicts, - backupAndFixClaudeSettings, - classifySettingsConflicts, -} from '@lib/agent/claude-settings'; -import { - evaluateWizardReadiness, - WizardReadiness, - SIGNUP_WIZARD_READINESS_CONFIG, - getBlockingServiceKeys, - SERVICE_LABELS, -} from '@lib/health-checks/readiness'; import { enableDebugLogs, logToFile, initLogFile } from '@utils/debug'; -import { wizardAbort } from '@utils/wizard-abort'; -import { ErrorCodes } from '@lib/errors'; -import { isNonInteractiveEnvironment } from '@utils/environment'; -import { CallType, getSkillsBaseUrl, IS_DEV } from '@lib/constants'; +import { CallType, IS_DEV } from '@lib/constants'; import { VERSION } from '@lib/version'; import { mcpUrlFor } from '@lib/host-resolution'; import type { WizardRunOptions } from '@utils/types'; -import { - postAuthGateSteps, - type ProgramConfig, -} from '@lib/programs/program-step'; -import type { ProgramRun, BootstrapResult } from './types'; +import type { BootstrapResult, RunConfig, RunFlags, RunInput } from './types'; // ── Helpers ────────────────────────────────────────────────────────── @@ -51,7 +27,7 @@ import type { ProgramRun, BootstrapResult } from './types'; * the program's `disallowedTools` so the SDK rejects calls outright. * Extracted so the policy can be unit-tested directly. * - * `session.e2eAsk` is the one escape hatch. The e2e harness runs a `ci` + * `e2eAsk` is the one escape hatch. The e2e harness runs a `ci` * session, but it does have an answerer — the driver loop answers each * `wizard_ask` batch from the program's e2e profile. Without the flag the * agent-in-the-loop layer (the ask bridge in both sequence arms, and the @@ -61,254 +37,68 @@ import type { ProgramRun, BootstrapResult } from './types'; * populates it, so plain `--ci` and `--signup` runs behave exactly as before. */ export function shouldDisableAsk( - session: Pick, + flags: Pick, ): boolean { - return (session.ci || session.signup) && !session.e2eAsk; + return (flags.ci || flags.signup) && !flags.e2eAsk; } -export function sessionToOptions(session: WizardSession): WizardRunOptions { +/** The option bag the agent interface and the middleware read. */ +export function runOptions(input: RunInput): WizardRunOptions { return { - installDir: session.installDir, - debug: session.debug, - signup: session.signup, - ci: session.ci, - benchmark: session.benchmark, - projectId: session.projectId, - apiKey: session.apiKey, - yaraReport: session.yaraReport, + installDir: input.installDir, + debug: input.flags.debug, + signup: input.flags.signup, + ci: input.flags.ci, + benchmark: input.flags.benchmark, + projectId: input.host.projectId, + apiKey: input.host.apiKey, + yaraReport: input.flags.yaraReport, }; } -// ── Bootstrap ───────────────────────────────────────────────────────── +// ── Prepare ─────────────────────────────────────────────────────────── /** - * Shared setup for both arms: logging, health check, settings conflicts, OAuth - * and credentials, then the feature flags, variant metadata, and MCP url. Sets - * `session.credentials`, role, and user as a side effect. Returns the values the - * arms still need. + * Shared setup for both arms: logging targets, then the gateway mint and the + * triage classifier. Throws when the mint is refused, so the run fails before + * any agent starts — the caller maps that the way it maps any unexpected error. */ -export async function bootstrapProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, +export async function prepareRun( + config: RunConfig, + input: RunInput, ): Promise { + const { run } = config; + // 1. Init logging + debug initLogFile(); - session.skillId = config.skillId ?? config.integrationLabel; logToFile( - `[agent-runner] START ${config.integrationLabel} build=${analytics.build}` + - `${session.ci ? ' (non-interactive)' : ''}`, + `[agent-runner] START ${run.integrationLabel} build=${analytics.build}` + + `${input.flags.ci ? ' (non-interactive)' : ''}`, ); - if (session.debug) { + if (input.flags.debug) { enableDebugLogs(); } - const skillsBaseUrl = getSkillsBaseUrl(); + const { skillsBaseUrl } = config; // Where this run actually points. The three services switch independently, // so otherwise "why did it use prod skills?" means reading three call sites. logToFile( `[agent-runner] targets build=${VERSION}${IS_DEV ? '/dev' : ''} ` + `skills=${skillsBaseUrl} ` + - `mcp=${mcpUrlFor(session.localMcp)} ` + - `posthog=${session.baseUrl ?? 'region-resolved'}`, - ); - - // 2. Health check (guarded — skip if TUI already ran it). Only - // programs that declare a health-check screen get pre-flight checks; - // for everything else the checks never fire and never block. - const hasHealthCheckScreen = programConfig.steps.some( - (s) => s.screenId === 'health-check', - ); - if (session.readinessResult) { - logToFile( - `[agent-runner] readiness pre-computed by TUI: decision=${session.readinessResult.decision}` + - `${ - session.outageDismissed ? ' (outage dismissed by user)' : '' - } — skipping re-check`, - ); - } - if (hasHealthCheckScreen && !session.readinessResult) { - logToFile('[agent-runner] evaluating wizard readiness'); - const readinessConfig = session.signup - ? SIGNUP_WIZARD_READINESS_CONFIG - : undefined; - const readiness = await evaluateWizardReadiness(readinessConfig); - logToFile(`[agent-runner] readiness=${readiness.decision}`); - if (readiness.decision === WizardReadiness.No) { - const blockingKeys = getBlockingServiceKeys( - readiness.health, - readinessConfig, - ); - const blockingLabels = blockingKeys.map( - (k) => `${SERVICE_LABELS[k]} (${readiness.health[k].status})`, - ); - logToFile(`[agent-runner] blocked by: ${blockingLabels.join(', ')}`); - - await getUI().showBlockingOutage(readiness); - - // The TUI lets the user continue past an outage; non-interactive runs - // (CI) do the same automatically — the degraded services are reported - // above, but we proceed rather than aborting on a transient upstream blip. - if (!isNonInteractiveEnvironment()) { - await wizardAbort({ - code: ErrorCodes.EnvServiceOutage, - message: - 'Cannot start — external services are down:\n' + - blockingLabels.map((l) => ` - ${l}`).join('\n') + - '\n\nPlease try again later.', - }); - } - } else if (readiness.decision === WizardReadiness.YesWithWarnings) { - getUI().setReadinessWarnings(readiness); - } - } - - // 3. Settings conflicts - const settingsConflicts = checkAllSettingsConflicts(session.installDir); - logToFile( - `[agent-runner] settings conflicts: ${ - settingsConflicts.length > 0 - ? settingsConflicts - .map((c) => `${c.source}(${c.keys.join(',')})`) - .join('; ') - : 'none' - }`, + `mcp=${mcpUrlFor(input.flags.localMcp)} ` + + `posthog=${input.host.baseUrl ?? 'region-resolved'}`, ); - if (settingsConflicts.length > 0) { - for (const conflict of settingsConflicts) { - const level = conflict.source === 'managed' ? 'org' : conflict.source; - analytics.wizardCapture('settings conflict detected', { - level, - keys: conflict.keys, - }); - } - - const { autoFix, failClosed, warnOnly } = - classifySettingsConflicts(settingsConflicts); - - // User-global and project-local files are already neutralized — the agent - // runs with settingSources:['project'], so the SDK never reads them. Record - // it and move on; don't make the user act on a setting that can't bite. - for (const conflict of warnOnly) { - logToFile( - `[agent-runner] settings conflict in ${conflict.source} (${conflict.path}) ` + - `neutralized by settingSources:['project'] — not blocking`, - ); - analytics.wizardCapture('settings conflict neutralized', { - level: conflict.source, - keys: conflict.keys, - }); - } - - // Writable project settings.json — the SDK *does* read it, but we can back - // it up and remove it (restored at outro). Neutralize without prompting. - let unfixable = failClosed; - if (autoFix.length > 0) { - const fixed = backupAndFixClaudeSettings(session.installDir); - if (fixed) { - logToFile('[agent-runner] auto-neutralized writable settings conflict'); - analytics.wizardCapture('settings conflict auto-neutralized', { - keys: autoFix.flatMap((c) => c.keys), - }); - } else { - // Couldn't remove it — don't run into the redirect; fail closed instead. - logToFile( - '[agent-runner] could not back up writable settings conflict — failing closed', - ); - unfixable = [...failClosed, ...autoFix]; - } - } - - // What we cannot neutralize (org-managed, always read by the SDK; or a - // writable file we failed to back up) must be fixed by the user. Fail - // closed: the screen names the file + keys and exits. - if (unfixable.length > 0) { - if (isNonInteractiveEnvironment()) { - await wizardAbort({ - code: ErrorCodes.SettingsUnfixableConflict, - message: - 'Cannot start — a Claude settings file redirects the agent away ' + - 'from the PostHog gateway and cannot be neutralized automatically:\n' + - unfixable - .map((c) => ` - ${c.source} (${c.path}): ${c.keys.join(', ')}`) - .join('\n') + - '\n\nRemove the conflicting keys and re-run the wizard.', - }); - } - await getUI().showSettingsOverride(unfixable, () => - backupAndFixClaudeSettings(session.installDir), - ); - logToFile('[agent-runner] settings override resolved'); - } - } - - analytics.wizardCapture('agent started', { - integration: config.integrationLabel, - program_id: programConfig.id, - skill_id: config.skillId ?? null, - }); - - // 4. Authenticate — idempotent within a run (see authenticate()). A second - // agent run in the same invocation (self-driving's integration phase) reuses - // the first login; it does not launch another OAuth. authenticate() also - // identifies the user and sets analytics groups. - await authenticate(session, programConfig.id); - maybeStampAiSdkDetected(session); - const project = session.apiProject; - - // 4.5. AI opt-in enforcement. Parks here while AiOptInRequiredScreen is - // up if the org hasn't approved third-party AI — BEFORE the skill - // install and agent start, so no source leaves the machine. The screen - // alone is cosmetic; this await is the actual gate. Resolves - // immediately when the program declared requiresAi: false or in CI. - // In bootstrapProgram so both the linear and orchestrator arms gate. - logToFile('[agent-runner] checking AI opt-in gate'); - await getUI().waitForAiOptIn(); - logToFile('[agent-runner] AI opt-in gate cleared'); - - // Park for any interactive step the user must complete AFTER authenticating - // but BEFORE the agent runs — e.g. the source-maps project picker, which - // needs credentials to scan and writes its choice to frameworkContext that - // the run prompt reads. Generic: await every gated step between auth and run. - for (const step of postAuthGateSteps(programConfig.steps)) { - logToFile(`[agent-runner] awaiting post-auth gate: ${step.id}`); - await getUI().waitForGate(step.id); - logToFile(`[agent-runner] post-auth gate cleared: ${step.id}`); - } - - // Feature flags. Both arms need these, and the fork decision reads the flags. - // This map is PostHog-side only — CLI `--harness` / `--sequence` precedence - // lives at the resolution sites (`runner/index.ts` for sequence, - // `resolveHarness` for harness), not here. - const wizardFlags = await analytics.getAllFlagsForWizard(); - const wizardFlagPayloads = analytics.getWizardFlagPayloads(); - - // Gateway trace tags for this run. The runner stamps its variant onto this - // after the fork (see runProgram), so the value reflects which arm ran. - const wizardMetadata = buildRunTags({ - programId: programConfig.id, - integration: config.integrationLabel, - runId: analytics.runId, - build: analytics.build, - skillId: config.skillId, - }); - - // The agent can't swap tokens mid-run, so freshness is measured after every park above, right before the mint. - await refreshAccessTokenIfNeeded(session); - - // Credentials (incl. the resolved host family and its MCP url) live on - // `session.credentials`; narrow once at this boundary — `authenticate` above - // set them — so downstream readers get a non-null type without asserting. - const credentials = session.credentials!; + const { credentials } = input; + const { wizardFlags, wizardFlagPayloads, wizardMetadata, programId } = config; // Mint now so a refusal fails the boot before any agent starts. Later // readers re-resolve through the cache, which re-mints past the refresh // point. const currentGatewayAuth = () => - gatewayAuth(credentials.host, credentials.accessToken, programConfig.id); + gatewayAuth(credentials.host, credentials.accessToken, programId); await currentGatewayAuth(); return { @@ -316,33 +106,24 @@ export async function bootstrapProgram( credentials, // Carried so per-task sessions re-resolve against the same program the boot // minted for, rather than digging it back out of the metadata bag. - programId: programConfig.id, + programId, wizardFlags, wizardFlagPayloads, wizardMetadata, - project, - // Resolved once, here: the only place holding both the switchboard inputs + project: input.project, + // Resolved once, here: the only place holding both the run-level harness // and the gateway auth. Every skill install downstream reads it off boot. - triageProvider: createTriageLLMProvider( - async () => { - const auth = await currentGatewayAuth(); - return { - baseURL: auth.gatewayUrl, - authToken: auth.token, - teamId: auth.teamId, - // `call_type` splits scan spend out of the program's agent cost, - // the same tag the in-run triage provider carries. - wizardMetadata: { ...wizardMetadata, call_type: CallType.yaraTriage }, - wizardFlags, - }; - }, - resolveHarness({ - program: programConfig.id, - flags: wizardFlags, - flagPayloads: wizardFlagPayloads, - cliHarness: session.harness, - cliModel: session.model, - }).harness, - ), + triageProvider: createTriageLLMProvider(async () => { + const auth = await currentGatewayAuth(); + return { + baseURL: auth.gatewayUrl, + authToken: auth.token, + teamId: auth.teamId, + // `call_type` splits scan spend out of the program's agent cost, + // the same tag the in-run triage provider carries. + wizardMetadata: { ...wizardMetadata, call_type: CallType.yaraTriage }, + wizardFlags, + }; + }, config.binding.harness), }; } diff --git a/src/lib/agent/runner/shared/errors.ts b/src/lib/agent/runner/shared/errors.ts index 376cb45d4..b3778a9b1 100644 --- a/src/lib/agent/runner/shared/errors.ts +++ b/src/lib/agent/runner/shared/errors.ts @@ -4,14 +4,14 @@ import type { InstallSkillResult } from '@lib/wizard-tools'; import { skillErrorCode } from '@lib/errors'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; +import type { AgentFailure } from './types'; -export async function abortOnInstallFailure( +/** The failure a skill install error decides. The caller reports and exits. */ +export function installFailure( integrationLabel: string, - result: InstallSkillResult, -): Promise { - if (result.kind === 'ok') return; - + result: Exclude, +): AgentFailure { const code = skillErrorCode(result) ?? undefined; const message = (() => { @@ -25,7 +25,7 @@ export async function abortOnInstallFailure( } })(); - await wizardAbort({ + return { message, code, error: new WizardError( @@ -40,5 +40,5 @@ export async function abortOnInstallFailure( }, code, ), - }); + }; } diff --git a/src/lib/agent/runner/shared/progress-collector.ts b/src/lib/agent/runner/shared/progress-collector.ts new file mode 100644 index 000000000..bdbde70f2 --- /dev/null +++ b/src/lib/agent/runner/shared/progress-collector.ts @@ -0,0 +1,120 @@ +/** + * The agent's own record of what it reported. + * + * `runAgent` accumulates its final `RunSnapshot` here, independently of any + * observer, so a missing or throwing `onProgress` never makes the result + * incomplete. The emitter wraps the caller's callback: it applies the event + * here first, then hands a copy to the observer, and logs rather than + * propagates anything the observer throws. + */ + +import { logToFile } from '@utils/debug'; +import type { + AgentProgress, + ProgressEmitter, + SpinnerHandle, +} from '@lib/agent/progress'; +import type { RunSnapshot } from './types'; + +export interface ProgressCollector { + emit: ProgressEmitter; + snapshot(): RunSnapshot; +} + +export function createProgressCollector( + onProgress?: (event: AgentProgress) => void, +): ProgressCollector { + const snapshot: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, + }; + + const apply = (event: AgentProgress): void => { + switch (event.kind) { + case 'tasks': + snapshot.tasks = event.tasks.map((t) => ({ ...t })); + break; + case 'status': + snapshot.statusMessages.push(event.message); + break; + case 'stage': + snapshot.stage = event.stage; + break; + case 'url': + if (event.which === 'dashboard') snapshot.dashboardUrl = event.url; + else snapshot.notebookUrl = event.url; + break; + case 'usage': + snapshot.usage.inputTokens += event.delta.inputTokens; + snapshot.usage.outputTokens += event.delta.outputTokens; + snapshot.usage.cacheReadTokens += event.delta.cacheReadTokens; + snapshot.usage.cacheCreationTokens += event.delta.cacheCreationTokens; + break; + case 'finalCost': + snapshot.finalCostUsd = event.usd; + break; + case 'handoff': + snapshot.handoffText = event.text; + break; + default: + break; + } + }; + + const emit: ProgressEmitter = (event) => { + apply(event); + if (!onProgress) return; + try { + onProgress(event); + } catch (error) { + // A broken projection is the host's problem, not the run's. Say so in + // the log and carry on; the snapshot above is the source of truth. + logToFile( + `[agent] progress observer threw on ${event.kind}:`, + error instanceof Error ? error.message : error, + ); + } + }; + + return { + emit, + snapshot: () => ({ + ...snapshot, + tasks: snapshot.tasks.map((t) => ({ ...t })), + statusMessages: [...snapshot.statusMessages], + usage: { ...snapshot.usage }, + }), + }; +} + +/** The run spinner as a progress emitter. One per run, like `getUI().spinner()`. */ +export function createEmitSpinner(emit: ProgressEmitter): SpinnerHandle { + return { + start: (message) => emit({ kind: 'spinner', action: 'start', message }), + stop: (message) => emit({ kind: 'spinner', action: 'stop', message }), + message: (message) => emit({ kind: 'spinner', action: 'message', message }), + }; +} + +/** `WizardUI.log` as a progress emitter. */ +export function createEmitLog(emit: ProgressEmitter): { + info(message: string): void; + warn(message: string): void; + error(message: string): void; + success(message: string): void; + step(message: string): void; +} { + return { + info: (message) => emit({ kind: 'log', level: 'info', message }), + warn: (message) => emit({ kind: 'log', level: 'warn', message }), + error: (message) => emit({ kind: 'log', level: 'error', message }), + success: (message) => emit({ kind: 'log', level: 'success', message }), + step: (message) => emit({ kind: 'log', level: 'step', message }), + }; +} diff --git a/src/lib/agent/runner/shared/types.ts b/src/lib/agent/runner/shared/types.ts index 4f4292f83..ae0123ce0 100644 --- a/src/lib/agent/runner/shared/types.ts +++ b/src/lib/agent/runner/shared/types.ts @@ -1,16 +1,30 @@ /** - * Shared types for the runner pipeline. + * The agent's run contracts. + * + * `runAgent(config, input, options)` takes resolved execution data and an + * invocation snapshot, reports through `options.onProgress`, asks through + * `options.interaction`, and returns a `RunResult`. Nothing here names a UI, + * a store, a session or a program registry: the caller resolves those and + * hands over plain data. `src/lib/programs/run-agent-legacy.ts` is the caller + * that rebuilds today's session-driven behavior on top of this contract. */ import type { - Credentials, AdditionalFeature, - WizardSession, + CloudRegion, + Credentials, + OutroData, + TaskNotice, } from '@lib/wizard-session'; import type { PromptContext } from '@lib/agent/agent-prompt'; import type { PackageManagerDetector } from '@lib/detection/package-manager'; -import type { ApiProject } from '@lib/api'; +import type { ApiProject, ApiUser } from '@lib/api'; +import type { Harness, Integration, Sequence } from '@lib/constants'; +import type { ErrorCode } from '@lib/errors'; import type { LLMProvider } from '@posthog/warlock'; +import type { AgentInteraction, ProgressEmitter } from '@lib/agent/progress'; +import type { EffortLevel } from '../switchboard/models'; +import type { SwitchboardCtx } from '../switchboard'; export type { PromptContext, Credentials }; @@ -23,17 +37,18 @@ export interface AbortCase { message: string; body: string; docsUrl?: string; - errorCode?: import('@lib/errors').ErrorCode; + errorCode?: ErrorCode; } /** - * Unified agent run configuration. + * What the agent reads of a program's run definition. * - * Every program provides one of these — either as a static object - * or via a function that builds one from the session. The runner - * assembles the final prompt from `prompt` + `skillId`. + * A program's `ProgramRun` (`src/lib/programs/program-run.ts`) extends this + * with hooks that take the wizard session; the caller binds those and hands + * the agent `RunConfig.hooks` instead. Everything below is plain data or a + * pure function of agent-side context. */ -export interface ProgramRun { +export interface AgentRunDefinition { /** Analytics label (e.g. 'revenue-analytics-setup', 'nextjs') */ integrationLabel: string; /** Skill ID to pre-install. Omit for agent-driven skill discovery. */ @@ -53,33 +68,6 @@ export interface ProgramRun { additionalFeatureQueue?: readonly AdditionalFeature[]; /** Known `[ABORT] ` cases this program can render. */ abortCases?: AbortCase[]; - /** Runs after agent completes, before outro (e.g. env var upload). */ - postRun?: (session: WizardSession, credentials: Credentials) => Promise; - /** Custom outro data. Omit for default built from successMessage/reportFile/docsUrl. */ - buildOutroData?: ( - session: WizardSession, - credentials: Credentials, - ) => WizardSession['outroData']; - /** - * Outro bullets for a sequence that composes its own outro data. - * - * `buildOutroData` is the linear sequence's seam: it hands the program the - * whole outro. The orchestrated sequence cannot, because its message is the - * drain's result — how many steps ran, what was skipped, which conflict the - * review step left. So a program with next steps to offer had nowhere to put - * them there, and the integration's data-source links were built and then - * dropped on every orchestrated run. This hook keeps the message with the - * sequence and the bullets with the program. - * - * `completedSeededTypes` names the runner-seeded task types that finished - * successfully, so a program can leave out a step its own seeded task - * already did — the sequence stays ignorant of what any type means. - */ - buildOutroNextSteps?: ( - session: WizardSession, - credentials: Credentials, - completedSeededTypes: readonly string[], - ) => { heading: string; items: string[] } | undefined; /** * Per-run cap on `wizard_ask` invocations. Defaults to 10. The 4th call * always returns a "batch your questions" error regardless of the cap. @@ -116,15 +104,132 @@ export interface ProgramRun { resolveStepKey?: (stepName: string | undefined) => string | undefined; } +/** A task the caller queues itself before the orchestrator's planner runs. */ +export interface SeedTaskEntry { + type: string; + label?: string; + inputs?: Record; + /** Shown before the run starts, letting the user decline the task. */ + notice?: TaskNotice; +} + +/** + * Completion hooks the caller binds for the agent. Each receives the run's + * resolved credentials, exactly what the linear and orchestrator sequences + * passed alongside the session before. + */ +export interface RunHooks { + /** Runs after the agent completes, before the outro (linear only). */ + postRun?: (credentials: Credentials) => Promise; + /** Custom outro data (linear only). Omit for the default outro. */ + buildOutroData?: (credentials: Credentials) => OutroData | undefined; + /** Outro bullets for the orchestrated sequence. */ + buildOutroNextSteps?: ( + credentials: Credentials, + completedSeededTypes: readonly string[], + ) => { heading: string; items: string[] } | undefined; +} + +/** The run-level routing decision the caller made. */ +export interface ResolvedBinding { + sequence: Sequence; + harness: Harness; + /** Gateway model id. */ + model: string; + /** Reasoning-effort override. Absent → the model's table default. */ + thinkingLevel?: EffortLevel; +} + +/** + * Resolved execution data for one agent run. The caller has already decided + * which program this is, how it is routed and which flags apply; the agent + * treats every label as opaque. + */ +export interface RunConfig { + /** Program id: gateway spend pin, analytics label, commandments axis. */ + programId: string; + /** The program's run definition, minus the session-bound hooks. */ + run: AgentRunDefinition; + /** A composed sub-run skips the terminal outro and the analytics shutdown. */ + composed: boolean; + /** Run-level sequence, harness and model. */ + binding: ResolvedBinding; + /** + * The inputs the run-level binding was resolved from. The orchestrator + * re-resolves the harness per task role from these; nothing else reads them. + */ + switchboard: SwitchboardCtx; + /** Primary skills origin (context-mill dev or GitHub Releases). */ + skillsBaseUrl: string; + /** Feature flag key → variant, evaluated before the run. */ + wizardFlags: Record; + /** Flag payloads from the same snapshot. */ + wizardFlagPayloads: Record; + /** Gateway trace tags for this run, already stamped with sequence and harness. */ + wizardMetadata: Record; + /** Extra tools added on top of BASE_ALLOWED_TOOLS for this run. */ + allowedTools?: readonly string[]; + /** Tools removed from BASE_ALLOWED_TOOLS for this run. */ + disallowedTools?: readonly string[]; + /** Context-mill flow the orchestrator loads. Defaults to `programId`. */ + agentFlow?: string; + /** Tasks to queue before the orchestrator's planner runs. */ + seedTasks?: () => SeedTaskEntry[]; + /** Completion hooks, bound by the caller. */ + hooks?: RunHooks; +} + +/** Invocation flags the agent reads. */ +export interface RunFlags { + ci: boolean; + signup: boolean; + debug: boolean; + /** Harness-only: keep the ask bridge in a `ci` run that has an answerer. */ + e2eAsk: boolean; + localMcp: boolean; + captureAio: boolean; + benchmark: boolean; + yaraReport: boolean; +} + +/** + * The invocation snapshot for one agent run. Taken once by the caller; the + * agent never refreshes these from a higher layer. + */ +export interface RunInput { + installDir: string; + /** Resolved credentials, including the host family and its MCP url. */ + credentials: Credentials; + /** Project payload resolved at authentication, for prompt context. */ + project: ApiProject | null; + /** User payload resolved at authentication, for the AI opt-in prompt line. */ + apiUser: ApiUser | null; + /** The skill this run is for: the run's skill id, else its integration label. */ + skillId?: string; + /** Detected framework, when the caller has one. */ + integration?: Integration | null; + /** Docs page for the detected framework, for the orchestrator's preflight message. */ + frameworkDocsUrl?: string; + flags: RunFlags; + /** Where PostHog is, as the CLI was told. */ + host: { + baseUrl?: string; + region?: CloudRegion; + email?: string; + /** `--project-id`, for the auth-error classifier. */ + projectId?: number; + /** `--api-key`, for the auth-error classifier. */ + apiKey?: string; + }; +} + /** - * Result of the shared bootstrap, consumed by both the linear and the - * orchestrator arm. `bootstrapProgram` runs `authenticate` before returning, so - * `credentials` is guaranteed non-null here — the single narrowing point owns - * the invariant, and downstream readers get a properly non-null type for free. + * The values `prepareRun` resolves from `RunConfig` + `RunInput` and hands to + * both sequences: the gateway mint and the scan-triage classifier built on it. */ export interface BootstrapResult { skillsBaseUrl: string; - /** Auth outputs (incl. the resolved host family and its MCP url), narrowed at the boundary. */ + /** Resolved credentials (incl. the host family and its MCP url). */ credentials: Credentials; /** Program this run is, and the node its gateway spend pins to. */ programId: string; @@ -137,3 +242,88 @@ export interface BootstrapResult { /** Scan-triage classifier on this run's harness. Undefined → skill scans fail closed. */ triageProvider: LLMProvider | undefined; } + +/** + * A decided failure. The same fields `wizardAbort` takes, so the legacy + * adapter passes it through untouched and the exit sequence, codes and + * messages stay exactly what they were. + */ +export interface AgentFailure { + message?: string; + /** Structured error data. Renders via `outroError` instead of `outro`. */ + outroData?: OutroData; + error?: Error; + exitCode?: number; + code?: ErrorCode; + detail?: Record; +} + +export type RunOutcome = + /** The run finished; `outro` holds what to show. */ + | 'success' + /** The agent chose to stop (`[ABORT]`); `failure` holds the rendered case. */ + | 'aborted' + /** A coded failure the agent decided; `failure` holds it. */ + | 'failed' + /** `options.signal` aborted the run. */ + | 'cancelled' + /** + * Something threw that the agent did not decide (a refused gateway mint, an + * SDK crash). `failure.error` is the original error and `failure.code` its + * classification, so a caller can report it or rethrow it as it did before. + */ + | 'crashed'; + +/** Totals of every `usage` event the run emitted. */ +export interface TokenUsageTotals { + inputTokens: number; + outputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; +} + +/** What the agent reported, accumulated independently of any observer. */ +export interface RunSnapshot { + tasks: import('@lib/agent/progress').TaskSnapshot[]; + statusMessages: string[]; + stage?: string; + handoffText?: string; + usage: TokenUsageTotals; + finalCostUsd?: number; + dashboardUrl?: string; + notebookUrl?: string; +} + +/** + * `runAgent` never rejects: every ending is one of these, and every ending + * that is not `success` carries a `failure` the caller can act on. + */ +export interface RunResult { + outcome: RunOutcome; + /** The skill this run installed or was for. */ + skillId?: string; + /** Success outro. Absent for a composed sub-run, which has no terminal outro. */ + outro?: OutroData; + /** Present whenever `outcome` is not `success`. */ + failure?: AgentFailure; + snapshot: RunSnapshot; +} + +export interface RunAgentOptions { + /** Receives every progress event in emission order. Never awaited. */ + onProgress?: (event: import('@lib/agent/progress').AgentProgress) => void; + /** Answers the agent's questions. Absent → no ask bridge, notices declined. */ + interaction?: AgentInteraction; + /** Cancels the run at its next phase boundary and any pending question. */ + signal?: AbortSignal; +} + +/** What a sequence receives: the contracts plus the prepared run. */ +export interface SequenceContext { + config: RunConfig; + input: RunInput; + boot: BootstrapResult; + emit: ProgressEmitter; + interaction: AgentInteraction | undefined; + signal: AbortSignal; +} diff --git a/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts b/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts index ebddc32e2..501a29006 100644 --- a/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts +++ b/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts @@ -11,6 +11,7 @@ import { type SwitchboardTrace, } from '@lib/agent/runner/switchboard'; import type { EffortLevel } from '@lib/agent/runner/switchboard/models'; +import { bindingFor } from '@lib/programs/bindings'; /** The complete resolved binding — every axis stated, nothing implicit. */ export interface ExpectedBinding { @@ -38,7 +39,12 @@ export function runBindingCases( it(c.name, () => { if (c.surface) setSurface?.(c.surface); try { - const ctx: SwitchboardCtx = { ...c.ctx }; + // The program's declared binding rides along, as the legacy adapter + // supplies it; a case may pin its own. + const ctx: SwitchboardCtx = { + binding: bindingFor(c.ctx.program), + ...c.ctx, + }; expect(resolveBinding(ctx)).toEqual(c.binding); if (c.trace) expect(ctx.trace).toEqual(c.trace); } finally { diff --git a/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts b/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts index f0540ab5b..cf509b9ed 100644 --- a/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts +++ b/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts @@ -27,6 +27,7 @@ import { } from '@lib/agent/runner/switchboard/flags/orchestrator'; import { SELF_DRIVING_EXPERIMENT } from '@lib/agent/runner/switchboard/flags/self-driving'; import { runBindingCases } from './binding-cases'; +import { bindingFor } from '@lib/programs/bindings'; const envState = vi.hoisted(() => ({ runSurface: 'local' as 'cloud' | 'local', @@ -261,7 +262,12 @@ describe('isolation — everything on at once', () => { it('only the two covered programs move; each lands exactly on its own row', () => { for (const program of PROGRAM_IDS) { - const ctx: SwitchboardCtx = { program, flags, flagPayloads }; + const ctx: SwitchboardCtx = { + program, + flags, + flagPayloads, + binding: bindingFor(program), + }; const resolved = resolveBinding(ctx); if (program === 'posthog-integration') { expect(resolved).toEqual(ORCHESTRATOR_PI_DEFAULT); @@ -325,6 +331,7 @@ describe('isolation — everything on at once', () => { it('regression (2026-07-17): self-driving never rides the global orchestrator flag into the orchestrator', () => { const binding = resolveBinding({ program: 'self-driving', + binding: bindingFor('self-driving'), flags: { [ORCH]: 'true', [SD]: 'true' }, flagPayloads: { [SD]: { model: 'gpt-5-6-terra' } }, }); diff --git a/src/lib/agent/runner/switchboard/flags/index.ts b/src/lib/agent/runner/switchboard/flags/index.ts index b9d3d03a2..4df468611 100644 --- a/src/lib/agent/runner/switchboard/flags/index.ts +++ b/src/lib/agent/runner/switchboard/flags/index.ts @@ -7,7 +7,6 @@ import { RUN_SURFACE } from '@env'; import { logToFile } from '@utils/debug'; import type { Sequence } from '@lib/constants'; -import type { ProgramId } from '@lib/programs/program-registry'; import { ORCHESTRATOR_HARNESS_ROUTE, ORCHESTRATOR_SEQUENCE_ROUTE, @@ -34,7 +33,7 @@ export const SEQUENCE_EXPERIMENTS: readonly SequenceExperiment[] = [ /** The flag-driven route for a program, or undefined when no experiment covers it or its flags don't validly route. */ export function resolveFlagRoute( - program: ProgramId, + program: string, flags: Record, flagPayloads?: Record, ): FlagRoute | undefined { @@ -46,7 +45,7 @@ export function resolveFlagRoute( /** The flag-driven sequence for a program, or undefined when no sequence experiment covers it with its flag on. Surface/build scoping is the flag's own job (see `flagPersonProperties`). */ export function resolveFlagSequence( - program: ProgramId, + program: string, flags: Record, ): Sequence | undefined { return SEQUENCE_EXPERIMENTS.find( @@ -56,7 +55,7 @@ export function resolveFlagSequence( /** The per-stage overrides for a program's run, or undefined (prompt frontmatter stays). Applied once, where the agent prompts are loaded. */ export function resolveStageOverrides( - program: ProgramId, + program: string, flags: Record, flagPayloads?: Record, ): Record | undefined { diff --git a/src/lib/agent/runner/switchboard/flags/schemes.ts b/src/lib/agent/runner/switchboard/flags/schemes.ts index f55f30141..ee074d0ec 100644 --- a/src/lib/agent/runner/switchboard/flags/schemes.ts +++ b/src/lib/agent/runner/switchboard/flags/schemes.ts @@ -13,7 +13,6 @@ import { Sequence, SONNET_5_MODEL, } from '@lib/constants'; -import type { ProgramId } from '@lib/programs/program-registry'; import { logToFile } from '@utils/debug'; import type { EffortLevel } from '../models'; @@ -68,13 +67,13 @@ export type ConfigFlag = HarnessConfigFlag | PayloadConfigFlag; * program. */ export interface HarnessExperiment { - program: ProgramId; + program: string; flags: ConfigFlag; } /** A sequence-axis experiment: one boolean flag, inert outside its listed programs. */ export interface SequenceExperiment { - programs: readonly ProgramId[]; + programs: readonly string[]; flag: string; /** Sequence the flag routes covered programs to. */ sequence: Sequence; diff --git a/src/lib/agent/runner/switchboard/harness.ts b/src/lib/agent/runner/switchboard/harness.ts index 8710a166f..837a4cbe6 100644 --- a/src/lib/agent/runner/switchboard/harness.ts +++ b/src/lib/agent/runner/switchboard/harness.ts @@ -10,8 +10,7 @@ import { piBackend } from '../harness/pi'; import type { AgentHarness } from '../harness/types'; import { resolveFlagRoute } from './flags'; import { - DEFAULT_BINDING, - PROGRAM_BINDINGS, + bindingOf, runChain, type HarnessPick, type Middleware, @@ -86,7 +85,7 @@ export function resolveHarness( const pick = runChain(HARNESS_MIDDLEWARE, ctx, () => { if (ctx.trace) Object.assign(ctx.trace, { harness: 'binding', model: 'binding' }); - const binding = PROGRAM_BINDINGS[ctx.program] ?? DEFAULT_BINDING; + const binding = bindingOf(ctx); return { harness: binding.harness, model: binding.model, diff --git a/src/lib/agent/runner/switchboard/index.ts b/src/lib/agent/runner/switchboard/index.ts index f6d641a66..31d645bd1 100644 --- a/src/lib/agent/runner/switchboard/index.ts +++ b/src/lib/agent/runner/switchboard/index.ts @@ -1,13 +1,6 @@ // Resolves routing; model additions also require mint allowlists and gateway prompt/transport support. -import { - DEFAULT_AGENT_MODEL, - GPT5_6_SOL_MODEL, - GPT5_6_TERRA_MODEL, - Harness, - Sequence, -} from '@lib/constants'; -import type { ProgramId } from '@lib/programs/program-registry'; +import { GPT5_6_SOL_MODEL, Harness, Sequence } from '@lib/constants'; import { resolveHarness } from './harness'; import type { EffortLevel } from './models'; import { resolveSequence } from './sequence'; @@ -29,7 +22,14 @@ export interface SwitchboardTrace { /** Everything a resolver middleware may branch on. Built once per run. */ export interface SwitchboardCtx { - program: ProgramId; + /** Program id. An opaque label here: experiments match on it, nothing else reads it. */ + program: string; + /** + * The program's declared binding, resolved by the caller (see + * `src/lib/programs/bindings.ts`). Absent → `DEFAULT_BINDING`, for + * standalone and skill-only runs that belong to no registered program. + */ + binding?: ProgramBinding; /** Composed sub-run (a dependency inside a parent program). Structurally linear — no override can orchestrate it. */ composed?: boolean; flags: Record; @@ -110,60 +110,10 @@ export const DEFAULT_BINDING: ProgramBinding = { thinkingLevel: 'medium', }; -/** - * Per-program routing. Kept in lockstep with `PROGRAM_REGISTRY` by the - * switchboard test. Anything absent falls back to `DEFAULT_BINDING`. - */ -export const PROGRAM_BINDINGS: Partial> = { - 'posthog-integration': DEFAULT_BINDING, - 'revenue-analytics-setup': DEFAULT_BINDING, - 'warehouse-source': DEFAULT_BINDING, - 'error-tracking-upload-source-maps': { - sequence: Sequence.linear, - harness: Harness.pi, - model: GPT5_6_SOL_MODEL, - thinkingLevel: 'medium', - }, - audit: DEFAULT_BINDING, - 'events-audit': DEFAULT_BINDING, - 'posthog-doctor': DEFAULT_BINDING, - 'web-analytics-doctor': DEFAULT_BINDING, - migration: DEFAULT_BINDING, - 'self-driving': DEFAULT_BINDING, - 'agent-skill': DEFAULT_BINDING, - 'mcp-add': DEFAULT_BINDING, - 'mcp-remove': DEFAULT_BINDING, - 'mcp-tutorial': DEFAULT_BINDING, - 'mcp-analytics': DEFAULT_BINDING, - // Orchestrator on pi. The binding routes only; every stage's model and - // effort are pinned context-mill side in the flow's frontmatter - // (`model_pi`/`effort_pi`: terra seed, sol tasks, luna report). - metrics: { - sequence: Sequence.orchestrator, - harness: Harness.pi, - model: DEFAULT_AGENT_MODEL, - }, - 'replay-vision': { - sequence: Sequence.orchestrator, - harness: Harness.anthropic, - model: DEFAULT_AGENT_MODEL, - }, - // Orchestrator on pi, like metrics. The binding routes only; every stage's - // model and effort are pinned context-mill side in the flow's frontmatter - // (`model_pi`/`effort_pi`: terra seed, install and init, sol tasks, luna report). - 'error-tracking': { - sequence: Sequence.orchestrator, - harness: Harness.pi, - model: DEFAULT_AGENT_MODEL, - }, - 'ai-observability': { - sequence: Sequence.linear, - harness: Harness.pi, - model: GPT5_6_TERRA_MODEL, - thinkingLevel: 'high', - }, - slack: DEFAULT_BINDING, -}; +/** The binding a context resolves from: the caller's, else the agent default. */ +export function bindingOf(ctx: SwitchboardCtx): ProgramBinding { + return ctx.binding ?? DEFAULT_BINDING; +} // ── Unified resolver ──────────────────────────────────────────────────── diff --git a/src/lib/agent/runner/switchboard/sequence.ts b/src/lib/agent/runner/switchboard/sequence.ts index e39e228e8..8e76c8599 100644 --- a/src/lib/agent/runner/switchboard/sequence.ts +++ b/src/lib/agent/runner/switchboard/sequence.ts @@ -12,43 +12,27 @@ import { resolveFlagSequence, } from './flags'; import { getHarness, resolveHarness } from './harness'; -import type { WizardSession } from '@lib/wizard-session'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun, BootstrapResult } from '../shared/types'; +import type { RunResult, SequenceContext } from '../shared/types'; import { runLinearProgram } from '../sequence/linear'; import { runOrchestrator } from '../sequence/orchestrator/orchestrator-runner'; -import { - DEFAULT_BINDING, - PROGRAM_BINDINGS, - runChain, - type Middleware, - type SwitchboardCtx, -} from '.'; +import { bindingOf, runChain, type Middleware, type SwitchboardCtx } from '.'; // ── Registry ──────────────────────────────────────────────────────────── export interface SequenceRunner { readonly name: Sequence; - run( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, - /** Composed sub-run (integration inside self-driving); linear-only. */ - composed: boolean, - ): Promise; + /** Run one program to a decided result. Unexpected errors propagate. */ + run(ctx: SequenceContext): Promise; } export const SEQUENCE_OPTIONS: Partial> = { [Sequence.linear]: { name: Sequence.linear, - run: (session, config, programConfig, boot, composed) => - runLinearProgram(session, config, programConfig, boot, composed), + run: (ctx) => runLinearProgram(ctx), }, [Sequence.orchestrator]: { name: Sequence.orchestrator, - run: (session, config, programConfig, boot, _composed) => - runOrchestrator(session, config, programConfig, boot), + run: (ctx) => runOrchestrator(ctx), }, }; @@ -129,8 +113,7 @@ const SEQUENCE_MIDDLEWARE: Middleware[] = [ export function resolveSequence(ctx: SwitchboardCtx): Sequence { const sequence = runChain(SEQUENCE_MIDDLEWARE, ctx, () => { if (ctx.trace) ctx.trace.sequence = 'binding'; - const binding = PROGRAM_BINDINGS[ctx.program] ?? DEFAULT_BINDING; - return binding.sequence; + return bindingOf(ctx).sequence; }); logToFile( `[switchboard] resolved: program=${ctx.program} sequence=${sequence}` + diff --git a/src/lib/detection/__tests__/project-scope.test.ts b/src/lib/detection/__tests__/project-scope.test.ts index 46aedaf04..18f9173dc 100644 --- a/src/lib/detection/__tests__/project-scope.test.ts +++ b/src/lib/detection/__tests__/project-scope.test.ts @@ -10,12 +10,12 @@ import { AGENTIC_DETECTION_TIMEOUT_MS, WIZARD_BASIC_INTEGRATION_AGENTIC_DETECTION_FLAG_KEY, } from '@lib/constants'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { authenticate } from '@lib/programs/authenticate'; import { buildSession } from '@lib/wizard-session'; import { analytics } from '@utils/analytics'; // Mock only the two network edges of scopeInstallDirToProject; everything else runs real. -vi.mock('@lib/agent/runner/shared/authenticate', () => ({ +vi.mock('@lib/programs/authenticate', () => ({ authenticate: vi.fn().mockResolvedValue(undefined), })); vi.mock('@lib/detection/agentic', async (importOriginal) => ({ diff --git a/src/lib/detection/project-scope.ts b/src/lib/detection/project-scope.ts index 73279b075..da4ffaf0a 100644 --- a/src/lib/detection/project-scope.ts +++ b/src/lib/detection/project-scope.ts @@ -8,7 +8,7 @@ import { type DetectEvent, type DetectTarget, } from './agentic.js'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { authenticate } from '@lib/programs/authenticate'; import { FRAMEWORK_REGISTRY } from '@lib/registry'; import { AGENTIC_DETECTION_TIMEOUT_MS, diff --git a/src/lib/errors/index.ts b/src/lib/errors/index.ts index 5558c7942..30e98f4ec 100644 --- a/src/lib/errors/index.ts +++ b/src/lib/errors/index.ts @@ -18,3 +18,4 @@ export { } from './emit'; export { sanitizeErrorDetail } from './sanitize'; export { classifyRunFailure, type RunFailure } from './run-failure'; +export { WizardError } from './wizard-error'; diff --git a/src/lib/errors/wizard-error.ts b/src/lib/errors/wizard-error.ts new file mode 100644 index 000000000..fb4690749 --- /dev/null +++ b/src/lib/errors/wizard-error.ts @@ -0,0 +1,20 @@ +import type { ErrorCode } from './codes'; + +/** + * A data carrier for a decided failure: message, analytics context and the + * catalog code. Passed to `wizardAbort()` by the caller that owns the exit; + * never thrown by the agent. + */ +export class WizardError extends Error { + readonly code?: ErrorCode; + + constructor( + message: string, + public readonly context?: Record, + code?: ErrorCode, + ) { + super(message); + this.name = 'WizardError'; + this.code = code; + } +} diff --git a/src/lib/middleware/benchmark.ts b/src/lib/middleware/benchmark.ts index 20017ad75..1e6c0937b 100644 --- a/src/lib/middleware/benchmark.ts +++ b/src/lib/middleware/benchmark.ts @@ -7,7 +7,7 @@ * pipeline.finalize(resultMessage, durationMs); */ -import { getUI, type SpinnerHandle } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; import { logToFile, getLogFilePath, configureLogFile } from '@utils/debug'; import { MiddlewarePipeline } from './pipeline'; import { PhaseDetector } from './phase-detector'; @@ -68,8 +68,10 @@ export function createBenchmarkPipeline( spinner: SpinnerHandle, options: WizardRunOptions, configOverride?: BenchmarkConfig, + reporting: { log?: (message: string) => void } = {}, ): MiddlewarePipeline { const config = configOverride ?? loadBenchmarkConfig(options.installDir); + const log = reporting.log ?? (() => undefined); configureLogFile({ path: config.output.logPath, @@ -80,13 +82,12 @@ export function createBenchmarkPipeline( spinner, phased: false, outputPath: config.output.benchmarkPath, + log, }); if (!config.output.suppressWizardLogs) { - getUI().log.info( - `${AgentSignals.BENCHMARK} Verbose logs: ${getLogFilePath()}`, - ); - getUI().log.info( + log(`${AgentSignals.BENCHMARK} Verbose logs: ${getLogFilePath()}`); + log( `${AgentSignals.BENCHMARK} Benchmark data will be written to: ${config.output.benchmarkPath}`, ); } diff --git a/src/lib/middleware/benchmarks/index.ts b/src/lib/middleware/benchmarks/index.ts index 470dbd82b..61f845e3e 100644 --- a/src/lib/middleware/benchmarks/index.ts +++ b/src/lib/middleware/benchmarks/index.ts @@ -30,8 +30,8 @@ const PLUGIN_REGISTRY: Record = { contextSize: () => new ContextSizeTrackerPlugin(), cost: () => new CostTrackerPlugin(), duration: () => new DurationTrackerPlugin(), - summary: (opts) => new SummaryPlugin(opts.spinner!), - jsonWriter: (opts) => new JsonWriterPlugin(opts.outputPath!), + summary: (opts) => new SummaryPlugin(opts.spinner!, opts.log), + jsonWriter: (opts) => new JsonWriterPlugin(opts.outputPath!, opts.log), }; /** diff --git a/src/lib/middleware/benchmarks/json-writer.ts b/src/lib/middleware/benchmarks/json-writer.ts index 253225335..43d94dbf9 100644 --- a/src/lib/middleware/benchmarks/json-writer.ts +++ b/src/lib/middleware/benchmarks/json-writer.ts @@ -6,7 +6,6 @@ */ import fs from 'fs'; -import { getUI } from '@ui'; import { logToFile } from '@utils/debug'; import { AgentSignals } from '@lib/agent/agent-interface'; import type { @@ -57,7 +56,13 @@ export class JsonWriterPlugin implements Middleware { private outputPath: string; - constructor(outputPath: string) { + private readonly log: (message: string) => void; + + constructor( + outputPath: string, + log: (message: string) => void = () => undefined, + ) { + this.log = log; this.outputPath = outputPath; } @@ -170,7 +175,7 @@ export class JsonWriterPlugin implements Middleware { try { fs.writeFileSync(this.outputPath, JSON.stringify(data, null, 2)); logToFile(`Benchmark data written to ${this.outputPath}`); - getUI().log.info( + this.log( `● ${AgentSignals.BENCHMARK} Results written to ${this.outputPath}`, ); } catch (error) { diff --git a/src/lib/middleware/benchmarks/summary.ts b/src/lib/middleware/benchmarks/summary.ts index 45c3b0a8e..161b515a0 100644 --- a/src/lib/middleware/benchmarks/summary.ts +++ b/src/lib/middleware/benchmarks/summary.ts @@ -1,4 +1,4 @@ -import { getUI, type SpinnerHandle } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; import { AgentSignals } from '@lib/agent/agent-interface'; import type { Middleware, @@ -97,8 +97,14 @@ export class SummaryPlugin implements Middleware { private spinner: SpinnerHandle; - constructor(spinner: SpinnerHandle) { + private readonly log: (message: string) => void; + + constructor( + spinner: SpinnerHandle, + log: (message: string) => void = () => undefined, + ) { this.spinner = spinner; + this.log = log; } onPhaseTransition( @@ -117,7 +123,7 @@ export class SummaryPlugin implements Middleware { this.spinner.stop(`${AgentSignals.BENCHMARK} ${fromPhase}`); } - getUI().log.info(`${AgentSignals.BENCHMARK} Starting phase: ${toPhase}`); + this.log(`${AgentSignals.BENCHMARK} Starting phase: ${toPhase}`); this.spinner.start(`Integrating PostHog (${toPhase})...`); } @@ -135,31 +141,31 @@ export class SummaryPlugin implements Middleware { const phaseCount = duration?.phaseSnapshots.length ?? 0; const totalCost = cost?.totalCost ?? 0; - getUI().log.info(''); - getUI().log.info( + this.log(''); + this.log( `◇ ${AgentSignals.BENCHMARK} ${phaseCount} phases in ${fmtDuration( totalDurationMs, )}, cost: ${fmtCost(totalCost)}`, ); - getUI().log.info( + this.log( ` total in: ${fmtTok(tokens?.totalInput ?? 0)}, out: ${fmtTok( tokens?.totalOutput ?? 0, )}, cache_read: ${fmtTok(cache?.totalRead ?? 0)}, cache_5m: ${fmtTok( cache?.totalCreation5m ?? 0, )}, cache_1h: ${fmtTok(cache?.totalCreation1h ?? 0)}`, ); - getUI().log.info(''); - getUI().log.info(`● ${AgentSignals.BENCHMARK} Summary by phase:`); + this.log(''); + this.log(`● ${AgentSignals.BENCHMARK} Summary by phase:`); if (duration?.phaseSnapshots) { for (let i = 0; i < duration.phaseSnapshots.length; i++) { const stats = getPhaseStats(i, ctx); if (stats) { - getUI().log.info(printPhase(stats)); + this.log(printPhase(stats)); } } } - getUI().log.info(''); + this.log(''); } } diff --git a/src/lib/middleware/types.ts b/src/lib/middleware/types.ts index ea1715700..3b8ab8c46 100644 --- a/src/lib/middleware/types.ts +++ b/src/lib/middleware/types.ts @@ -5,7 +5,7 @@ * and can publish data to a shared store for downstream middleware to read. */ -import type { SpinnerHandle } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; export type SDKMessage = any; @@ -57,4 +57,6 @@ export interface MiddlewareFactoryOptions { spinner?: SpinnerHandle; outputPath?: string; phased?: boolean; + /** Where the summary and writer plugins print their lines. */ + log?: (message: string) => void; } diff --git a/src/lib/programs/__tests__/agent-skill.test.ts b/src/lib/programs/__tests__/agent-skill.test.ts index 132586e1d..b836f0486 100644 --- a/src/lib/programs/__tests__/agent-skill.test.ts +++ b/src/lib/programs/__tests__/agent-skill.test.ts @@ -3,7 +3,7 @@ import { AGENT_SKILL_STEPS, type SkillProgramOptions, } from '@lib/programs/agent-skill/index'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { buildSession, RunPhase } from '@lib/wizard-session'; import { HostResolution } from '@lib/host-resolution'; diff --git a/src/lib/programs/__tests__/error-tracking.test.ts b/src/lib/programs/__tests__/error-tracking.test.ts index 7cf0561d8..5ecef5279 100644 --- a/src/lib/programs/__tests__/error-tracking.test.ts +++ b/src/lib/programs/__tests__/error-tracking.test.ts @@ -1,6 +1,6 @@ import { beforeEach, describe, expect, test, vi } from 'vitest'; -import type { ProgramRun } from '@lib/agent/runner/shared/types'; +import type { ProgramRun } from '@lib/programs/program-run'; import { Integration } from '@lib/constants'; import type { AgenticDetectionReport } from '@lib/detection/agentic'; import { detectFramework } from '@lib/detection/index'; diff --git a/src/lib/programs/__tests__/metrics-program.test.ts b/src/lib/programs/__tests__/metrics-program.test.ts index ae5936408..46f381471 100644 --- a/src/lib/programs/__tests__/metrics-program.test.ts +++ b/src/lib/programs/__tests__/metrics-program.test.ts @@ -1,7 +1,7 @@ import { AGENT_SKILL_STEPS } from '@lib/programs/agent-skill/index'; import { getProgramConfig, Program } from '@lib/programs/program-registry'; import { metricsConfig } from '@lib/programs/metrics/index'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { metricsCommand } from '../../../commands/metrics'; diff --git a/src/lib/agent/runner/shared/__tests__/refresh-access-token-if-needed.test.ts b/src/lib/programs/__tests__/refresh-access-token-if-needed.test.ts similarity index 100% rename from src/lib/agent/runner/shared/__tests__/refresh-access-token-if-needed.test.ts rename to src/lib/programs/__tests__/refresh-access-token-if-needed.test.ts diff --git a/src/lib/programs/agent-skill/index.ts b/src/lib/programs/agent-skill/index.ts index dbd207ae0..b2f6198a7 100644 --- a/src/lib/programs/agent-skill/index.ts +++ b/src/lib/programs/agent-skill/index.ts @@ -20,7 +20,7 @@ */ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun, AbortCase } from '@lib/agent/agent-runner'; +import type { ProgramRun, AbortCase } from '@lib/programs/program-run'; import { AGENT_SKILL_STEPS } from './steps.js'; import { getContentBlocks } from './content/index.js'; diff --git a/src/lib/programs/audit/detect.ts b/src/lib/programs/audit/detect.ts index 2cfa299a5..6ffbb69c8 100644 --- a/src/lib/programs/audit/detect.ts +++ b/src/lib/programs/audit/detect.ts @@ -1,4 +1,4 @@ -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; /** `[ABORT] ` cases the audit skill can emit. Reason strings are diff --git a/src/lib/programs/audit/index.ts b/src/lib/programs/audit/index.ts index fc5f60297..261fbaefc 100644 --- a/src/lib/programs/audit/index.ts +++ b/src/lib/programs/audit/index.ts @@ -3,7 +3,7 @@ import { createSkillProgram, } from '@lib/programs/agent-skill/index'; import type { ProgramStep, ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { OutroKind } from '@lib/wizard-session'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; diff --git a/src/lib/agent/runner/shared/authenticate.ts b/src/lib/programs/authenticate.ts similarity index 98% rename from src/lib/agent/runner/shared/authenticate.ts rename to src/lib/programs/authenticate.ts index 8644a328c..1c55453dd 100644 --- a/src/lib/agent/runner/shared/authenticate.ts +++ b/src/lib/programs/authenticate.ts @@ -11,7 +11,7 @@ */ import type { Credentials, WizardSession } from '@lib/wizard-session'; -import type { ProgramId } from '@lib/programs/program-registry'; +import type { ProgramId } from './program-registry'; import { getOrAskForProjectData } from '@utils/setup-utils'; import { refreshAccessToken } from '@utils/oauth'; import { OAuthError } from '@utils/oauth-errors'; diff --git a/src/lib/programs/bindings.ts b/src/lib/programs/bindings.ts new file mode 100644 index 000000000..1d445e08f --- /dev/null +++ b/src/lib/programs/bindings.ts @@ -0,0 +1,77 @@ +/** + * Per-program routing: which sequence, harness and model run each program. + * + * Kept in lockstep with `PROGRAM_REGISTRY` by the switchboard test. Anything + * absent falls back to the agent's `DEFAULT_BINDING`. The agent never reads + * this table: `run-agent-legacy.ts` looks a program up here and hands the + * binding to the switchboard as `SwitchboardCtx.binding`. + */ + +import { + DEFAULT_AGENT_MODEL, + GPT5_6_SOL_MODEL, + GPT5_6_TERRA_MODEL, + Harness, + Sequence, +} from '@lib/constants'; +import { + DEFAULT_BINDING, + type ProgramBinding, +} from '@lib/agent/runner/switchboard'; +import type { ProgramId } from './program-registry'; + +export const PROGRAM_BINDINGS: Partial> = { + 'posthog-integration': DEFAULT_BINDING, + 'revenue-analytics-setup': DEFAULT_BINDING, + 'warehouse-source': DEFAULT_BINDING, + 'error-tracking-upload-source-maps': { + sequence: Sequence.linear, + harness: Harness.pi, + model: GPT5_6_SOL_MODEL, + thinkingLevel: 'medium', + }, + audit: DEFAULT_BINDING, + 'events-audit': DEFAULT_BINDING, + 'posthog-doctor': DEFAULT_BINDING, + 'web-analytics-doctor': DEFAULT_BINDING, + migration: DEFAULT_BINDING, + 'self-driving': DEFAULT_BINDING, + 'agent-skill': DEFAULT_BINDING, + 'mcp-add': DEFAULT_BINDING, + 'mcp-remove': DEFAULT_BINDING, + 'mcp-tutorial': DEFAULT_BINDING, + 'mcp-analytics': DEFAULT_BINDING, + // Orchestrator on pi. The binding routes only; every stage's model and + // effort are pinned context-mill side in the flow's frontmatter + // (`model_pi`/`effort_pi`: terra seed, sol tasks, luna report). + metrics: { + sequence: Sequence.orchestrator, + harness: Harness.pi, + model: DEFAULT_AGENT_MODEL, + }, + 'replay-vision': { + sequence: Sequence.orchestrator, + harness: Harness.anthropic, + model: DEFAULT_AGENT_MODEL, + }, + // Orchestrator on pi, like metrics. The binding routes only; every stage's + // model and effort are pinned context-mill side in the flow's frontmatter + // (`model_pi`/`effort_pi`: terra seed, install and init, sol tasks, luna report). + 'error-tracking': { + sequence: Sequence.orchestrator, + harness: Harness.pi, + model: DEFAULT_AGENT_MODEL, + }, + 'ai-observability': { + sequence: Sequence.linear, + harness: Harness.pi, + model: GPT5_6_TERRA_MODEL, + thinkingLevel: 'high', + }, + slack: DEFAULT_BINDING, +}; + +/** The binding for a program, or the agent default when none is declared. */ +export function bindingFor(programId: string): ProgramBinding { + return PROGRAM_BINDINGS[programId] ?? DEFAULT_BINDING; +} diff --git a/src/lib/programs/error-tracking-upload-source-maps/detect.ts b/src/lib/programs/error-tracking-upload-source-maps/detect.ts index 2d81b0061..f6ddbfb09 100644 --- a/src/lib/programs/error-tracking-upload-source-maps/detect.ts +++ b/src/lib/programs/error-tracking-upload-source-maps/detect.ts @@ -17,7 +17,7 @@ import { safeReadFile, } from '@utils/bounded-fs'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; /** diff --git a/src/lib/programs/error-tracking-upload-source-maps/index.ts b/src/lib/programs/error-tracking-upload-source-maps/index.ts index c79f245e2..afdc17007 100644 --- a/src/lib/programs/error-tracking-upload-source-maps/index.ts +++ b/src/lib/programs/error-tracking-upload-source-maps/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { OutroKind } from '@lib/wizard-session'; import { ERROR_TRACKING_UPLOAD_SOURCE_MAPS_PROGRAM } from './steps.js'; diff --git a/src/lib/programs/error-tracking/index.ts b/src/lib/programs/error-tracking/index.ts index 2fd6a7cb1..68e7986b5 100644 --- a/src/lib/programs/error-tracking/index.ts +++ b/src/lib/programs/error-tracking/index.ts @@ -2,7 +2,7 @@ import { Integration } from '@lib/constants'; import { detectFramework } from '@lib/detection/index'; import { scopeInstallDirToProject } from '@lib/detection/project-scope'; import { FRAMEWORK_REGISTRY } from '@lib/registry'; -import type { ProgramRun } from '@lib/agent/runner/shared/types'; +import type { ProgramRun } from '@lib/programs/program-run'; import { AGENT_SKILL_STEPS } from '@lib/programs/agent-skill/steps'; import { getContentBlocks } from '@lib/programs/error-tracking/content/index'; import { getTips } from '@lib/programs/error-tracking/content/tips'; diff --git a/src/lib/programs/events-audit/index.ts b/src/lib/programs/events-audit/index.ts index 080e59d69..4cfa397f0 100644 --- a/src/lib/programs/events-audit/index.ts +++ b/src/lib/programs/events-audit/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { OutroKind } from '@lib/wizard-session'; import { SPINNER_MESSAGE } from '@lib/framework-config'; diff --git a/src/lib/programs/mcp-analytics/index.ts b/src/lib/programs/mcp-analytics/index.ts index f9199d76e..5231a1d0a 100644 --- a/src/lib/programs/mcp-analytics/index.ts +++ b/src/lib/programs/mcp-analytics/index.ts @@ -1,4 +1,4 @@ -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; import { createSkillProgram } from '@lib/programs/agent-skill/index'; diff --git a/src/lib/programs/migration/index.ts b/src/lib/programs/migration/index.ts index 27cf0b3d0..9c0ddbad2 100644 --- a/src/lib/programs/migration/index.ts +++ b/src/lib/programs/migration/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import { MIGRATION_PROGRAM } from './steps.js'; import { getContentBlocks } from './content/index.js'; diff --git a/src/lib/programs/posthog-integration/index.ts b/src/lib/programs/posthog-integration/index.ts index c3ac0b913..433db9e7a 100644 --- a/src/lib/programs/posthog-integration/index.ts +++ b/src/lib/programs/posthog-integration/index.ts @@ -1,5 +1,6 @@ import type { ProgramConfig, ProgramStep } from '@lib/programs/program-step'; -import { runAgent, type ProgramRun } from '@lib/agent/agent-runner'; +import { runAgent } from '@lib/programs/run-agent-legacy'; +import type { ProgramRun } from '@lib/programs/program-run'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import type { WizardSession } from '@lib/wizard-session'; import { mayReportScanResults, OutroKind, RunPhase } from '@lib/wizard-session'; @@ -21,7 +22,7 @@ import { requestDeepLink } from '@utils/provisioning'; import { openTrackedLink, withUtm } from '@utils/links'; import type { HostResolution } from '@lib/host-resolution'; import { getDetectedWarehouseSources } from '@lib/programs/warehouse-source/detect'; -import { shouldDisableAsk } from '@lib/agent/runner/shared/bootstrap'; +import { shouldDisableAsk } from '@lib/agent/agent-runner'; import { POSTHOG_INTEGRATION_PROGRAM } from './steps.js'; import { getContentBlocks } from './content/index.js'; import { buildCodingAgentPrompt } from './handoff.js'; diff --git a/src/lib/programs/program-run.ts b/src/lib/programs/program-run.ts new file mode 100644 index 000000000..a2cb5ff3a --- /dev/null +++ b/src/lib/programs/program-run.ts @@ -0,0 +1,48 @@ +/** + * A program's agent run definition. + * + * Every program provides one of these — either as a static object or via a + * function that builds one from the session. The agent reads the + * `AgentRunDefinition` part; the three hooks below take the wizard session and + * are bound by `run-agent-legacy.ts` before the agent sees them. + */ + +import type { Credentials, WizardSession } from '@lib/wizard-session'; +import type { AgentRunDefinition } from '@lib/agent/runner/shared/types'; + +export type { + AbortCase, + AgentRunDefinition, + PromptContext, +} from '@lib/agent/runner/shared/types'; +export type { Credentials }; + +export interface ProgramRun extends AgentRunDefinition { + /** Runs after agent completes, before outro (e.g. env var upload). */ + postRun?: (session: WizardSession, credentials: Credentials) => Promise; + /** Custom outro data. Omit for default built from successMessage/reportFile/docsUrl. */ + buildOutroData?: ( + session: WizardSession, + credentials: Credentials, + ) => WizardSession['outroData']; + /** + * Outro bullets for a sequence that composes its own outro data. + * + * `buildOutroData` is the linear sequence's seam: it hands the program the + * whole outro. The orchestrated sequence cannot, because its message is the + * drain's result — how many steps ran, what was skipped, which conflict the + * review step left. So a program with next steps to offer had nowhere to put + * them there, and the integration's data-source links were built and then + * dropped on every orchestrated run. This hook keeps the message with the + * sequence and the bullets with the program. + * + * `completedSeededTypes` names the runner-seeded task types that finished + * successfully, so a program can leave out a step its own seeded task + * already did — the sequence stays ignorant of what any type means. + */ + buildOutroNextSteps?: ( + session: WizardSession, + credentials: Credentials, + completedSeededTypes: readonly string[], + ) => { heading: string; items: string[] } | undefined; +} diff --git a/src/lib/programs/program-step.ts b/src/lib/programs/program-step.ts index 81e55a655..5c24decf4 100644 --- a/src/lib/programs/program-step.ts +++ b/src/lib/programs/program-step.ts @@ -4,7 +4,7 @@ import type { TaskNotice, } from '@lib/wizard-session'; import type { WizardReadinessResult } from '@lib/health-checks/readiness'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from './program-run.js'; import type { Integration } from '@lib/constants'; import type { FrameworkConfig } from '@lib/framework-config'; import type { ContentBlock } from '@ui/tui/primitives/index'; diff --git a/src/lib/programs/replay-vision/index.ts b/src/lib/programs/replay-vision/index.ts index 046f76e25..540436b0a 100644 --- a/src/lib/programs/replay-vision/index.ts +++ b/src/lib/programs/replay-vision/index.ts @@ -1,4 +1,4 @@ -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { Integration } from '@lib/constants'; import { detectFramework, gatherFrameworkContext } from '@lib/detection/index'; import { scopeInstallDirToProject } from '@lib/detection/project-scope'; diff --git a/src/lib/programs/revenue-analytics/detect.ts b/src/lib/programs/revenue-analytics/detect.ts index de75eae0b..8ff8528dd 100644 --- a/src/lib/programs/revenue-analytics/detect.ts +++ b/src/lib/programs/revenue-analytics/detect.ts @@ -7,7 +7,7 @@ import { existsSync, statSync } from 'fs'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { findPackageJsons } from '@lib/programs/shared/package-scanning'; export { diff --git a/src/lib/programs/run-agent-legacy.ts b/src/lib/programs/run-agent-legacy.ts new file mode 100644 index 000000000..92f264570 --- /dev/null +++ b/src/lib/programs/run-agent-legacy.ts @@ -0,0 +1,539 @@ +/** + * The session-driven agent runner every existing caller uses. + * + * `runAgent(programConfig, session)` rebuilds today's behavior on top of the + * functional `runAgent(config, input, options)` in `@lib/agent/runner`: it + * runs the gates the TUI owns (health, settings, AI opt-in, post-auth steps), + * authenticates, resolves the program's binding, builds the agent's inputs + * from the session, maps every progress event back onto `getUI()` one call + * per event, answers the agent's questions through `getUI()`, and applies the + * result — `wizardAbort` for a decided failure, nothing more for success. + * + * This is the only file that knows about `getUI()`, the session and + * `wizardAbort` on the agent's behalf. Programs replace it in Release B. + */ + +import type { WizardSession } from '@lib/wizard-session'; +import { analytics } from '@utils/analytics'; +import { getUI, type WizardUI } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; +import type { AgentInteraction, AgentProgress } from '@lib/agent/progress'; +import { + runAgent as runAgentFunctional, + resolveBinding, + type RunConfig, + type RunInput, + type SwitchboardCtx, +} from '@lib/agent/runner'; +import type { ProgramBinding } from '@lib/agent/runner/switchboard'; +import { buildRunTags } from '@lib/agent/agent-interface'; +import { + backupAndFixClaudeSettings, + checkAllSettingsConflicts, + classifySettingsConflicts, + restoreClaudeSettings, +} from '@lib/agent/claude-settings'; +import { flushScanReport } from '@lib/yara-hooks'; +import { + evaluateWizardReadiness, + WizardReadiness, + SIGNUP_WIZARD_READINESS_CONFIG, + getBlockingServiceKeys, + SERVICE_LABELS, +} from '@lib/health-checks/readiness'; +import { enableDebugLogs, logToFile, initLogFile } from '@utils/debug'; +import { registerCleanup, wizardAbort } from '@utils/wizard-abort'; +import { ErrorCodes } from '@lib/errors'; +import { isNonInteractiveEnvironment } from '@utils/environment'; +import { + getSkillsBaseUrl, + Sequence, + WIZARD_ORCHESTRATOR_FLAG_KEY, + WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY, + type Integration, +} from '@lib/constants'; +import { FRAMEWORK_REGISTRY } from '@lib/registry'; +import { postAuthGateSteps, type ProgramConfig } from './program-step'; +import type { ProgramRun } from './program-run'; +import { authenticate, refreshAccessTokenIfNeeded } from './authenticate'; +import { maybeStampAiSdkDetected } from './posthog-integration/detect'; +import { startAuditLedgerWatcher } from './audit/ledger-watcher'; +import { bindingFor } from './bindings'; + +/** + * Resolve a ProgramConfig's agent run definition and execute the pipeline. + * Entry point for the runners and for composed run steps. + */ +export async function runAgent( + programConfig: ProgramConfig, + session: WizardSession, + options: { composed?: boolean } = {}, +): Promise { + if (!programConfig.run) { + throw new Error(`Program "${programConfig.id}" has no run configuration.`); + } + + // Before `run()` resolves: an audit seeds the ledger from inside its recipe, + // and a watcher started later would ignore that write as pre-existing. + const ledger = programConfig.auditLedgerFile + ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) + : null; + if (ledger) registerCleanup(() => ledger.stop()); + + try { + const runDef = + typeof programConfig.run === 'function' + ? await programConfig.run(session) + : programConfig.run; + + await runProgram(session, runDef, programConfig, options.composed ?? false); + } finally { + ledger?.stop(); + } +} + +/** + * Gates → authenticate → flags → binding → the functional run → apply result. + * Every step happens in the order it did inside the agent's bootstrap. + */ +async function runProgram( + session: WizardSession, + run: ProgramRun, + programConfig: ProgramConfig, + composed: boolean, +): Promise { + // 1. Init logging + debug + initLogFile(); + session.skillId = run.skillId ?? run.integrationLabel; + logToFile( + `[agent-runner] START ${run.integrationLabel} build=${analytics.build}` + + `${session.ci ? ' (non-interactive)' : ''}`, + ); + if (session.debug) { + enableDebugLogs(); + } + + // 2. Health check (guarded — skip if TUI already ran it). Only + // programs that declare a health-check screen get pre-flight checks; + // for everything else the checks never fire and never block. + await runHealthGate(session, programConfig); + + // 3. Settings conflicts + await runSettingsGate(session); + + analytics.wizardCapture('agent started', { + integration: run.integrationLabel, + program_id: programConfig.id, + skill_id: run.skillId ?? null, + }); + + // 4. Authenticate — idempotent within a run (see authenticate()). A second + // agent run in the same invocation (self-driving's integration phase) reuses + // the first login; it does not launch another OAuth. authenticate() also + // identifies the user and sets analytics groups. + await authenticate(session, programConfig.id); + maybeStampAiSdkDetected(session); + + // 4.5. AI opt-in enforcement. Parks here while AiOptInRequiredScreen is + // up if the org hasn't approved third-party AI — BEFORE the skill + // install and agent start, so no source leaves the machine. The screen + // alone is cosmetic; this await is the actual gate. Resolves + // immediately when the program declared requiresAi: false or in CI. + logToFile('[agent-runner] checking AI opt-in gate'); + await getUI().waitForAiOptIn(); + logToFile('[agent-runner] AI opt-in gate cleared'); + + // Park for any interactive step the user must complete AFTER authenticating + // but BEFORE the agent runs — e.g. the source-maps project picker, which + // needs credentials to scan and writes its choice to frameworkContext that + // the run prompt reads. Generic: await every gated step between auth and run. + for (const step of postAuthGateSteps(programConfig.steps)) { + logToFile(`[agent-runner] awaiting post-auth gate: ${step.id}`); + await getUI().waitForGate(step.id); + logToFile(`[agent-runner] post-auth gate cleared: ${step.id}`); + } + + // Feature flags. Both arms need these, and the routing decision reads them. + const wizardFlags = await analytics.getAllFlagsForWizard(); + const wizardFlagPayloads = analytics.getWizardFlagPayloads(); + + // Gateway trace tags for this run; the binding below stamps its axes on. + const wizardMetadata = buildRunTags({ + programId: programConfig.id, + integration: run.integrationLabel, + runId: analytics.runId, + build: analytics.build, + skillId: run.skillId, + }); + + // The agent can't swap tokens mid-run, so freshness is measured after every + // park above, right before the agent mints. + await refreshAccessTokenIfNeeded(session); + + // Credentials (incl. the resolved host family and its MCP url) live on + // `session.credentials`; narrow once at this boundary — `authenticate` above + // set them — so downstream readers get a non-null type without asserting. + const credentials = session.credentials!; + + // Resolve which sequence and harness will run a program (CLI → PostHog flag → + // per-program binding → default), tag both axes onto analytics, and hand the + // binding to the agent for dispatch. + const switchboard: SwitchboardCtx = { + program: programConfig.id, + composed, + flags: wizardFlags, + flagPayloads: wizardFlagPayloads, + cliHarness: session.harness, + cliSequence: session.sequence, + cliModel: session.model, + binding: bindingFor(programConfig.id), + }; + const binding = resolveBinding(switchboard); + analytics.setTag('sequence', binding.sequence); + analytics.setTag('harness', binding.harness); + wizardMetadata.SEQUENCE = binding.sequence; + wizardMetadata.HARNESS = binding.harness; + captureSwitchboardDecision(switchboard, binding); + + const ui = getUI(); + + // Cleanup coverage for the abort/cancel path: `wizardAbort` runs the + // registered cleanups, and the agent's own `finally` covers completion. + // flushScanReport is idempotent, so the overlap is a harmless no-op. + registerCleanup(() => + flushScanReport({ yaraReport: session.yaraReport }, (message) => + ui.log.info(message), + ), + ); + + // Linear settings restoration fires on entry to the outro screen, so it + // is registered before the run can reach that screen. Same owner, same + // timing as before; the abort path still restores through the cleanup + // `backupAndFixClaudeSettings` registered. + if (binding.sequence === Sequence.linear) { + ui.onEnterScreen('outro', () => restoreClaudeSettings(session.installDir)); + } + + const framework = session.integration ?? session.skillId ?? undefined; + const config: RunConfig = { + programId: programConfig.id, + run, + composed, + binding, + switchboard, + skillsBaseUrl: getSkillsBaseUrl(), + wizardFlags, + wizardFlagPayloads, + wizardMetadata, + allowedTools: programConfig.allowedTools, + disallowedTools: programConfig.disallowedTools, + agentFlow: programConfig.agentFlow, + seedTasks: programConfig.seedTasks + ? () => programConfig.seedTasks!(session) + : undefined, + hooks: { + postRun: run.postRun + ? (creds) => run.postRun!(session, creds) + : undefined, + buildOutroData: run.buildOutroData + ? (creds) => run.buildOutroData!(session, creds) ?? undefined + : undefined, + buildOutroNextSteps: run.buildOutroNextSteps + ? (creds, completed) => + run.buildOutroNextSteps!(session, creds, completed) + : undefined, + }, + }; + const input: RunInput = { + installDir: session.installDir, + credentials, + project: session.apiProject, + apiUser: session.apiUser, + skillId: session.skillId ?? undefined, + integration: session.integration, + frameworkDocsUrl: framework + ? FRAMEWORK_REGISTRY[framework as Integration]?.metadata.docsUrl + : undefined, + flags: { + ci: session.ci, + signup: session.signup, + debug: session.debug, + e2eAsk: session.e2eAsk, + localMcp: session.localMcp, + captureAio: session.captureAio, + benchmark: session.benchmark, + yaraReport: session.yaraReport, + }, + host: { + baseUrl: session.baseUrl, + region: session.region, + email: session.email, + projectId: session.projectId, + apiKey: session.apiKey, + }, + }; + + const result = await runAgentFunctional(config, input, { + onProgress: createUiReducer(ui), + interaction: uiInteraction(ui), + }); + + // Success already rendered through the reducer (outro data, outro line) and + // shut analytics down. A decided failure exits exactly as it always has. An + // error the agent did not decide used to propagate out of this call, and the + // runners handle it (mint-failure screen, classifyRunFailure): keep throwing + // the original error for them. + if (result.outcome === 'crashed') { + throw result.failure?.error ?? new Error(result.failure?.message); + } + if (result.outcome !== 'success') { + await wizardAbort(result.failure); + } +} + +// ── Gates ───────────────────────────────────────────────────────────── + +async function runHealthGate( + session: WizardSession, + programConfig: ProgramConfig, +): Promise { + const hasHealthCheckScreen = programConfig.steps.some( + (s) => s.screenId === 'health-check', + ); + if (session.readinessResult) { + logToFile( + `[agent-runner] readiness pre-computed by TUI: decision=${session.readinessResult.decision}` + + `${ + session.outageDismissed ? ' (outage dismissed by user)' : '' + } — skipping re-check`, + ); + } + if (!hasHealthCheckScreen || session.readinessResult) return; + + logToFile('[agent-runner] evaluating wizard readiness'); + const readinessConfig = session.signup + ? SIGNUP_WIZARD_READINESS_CONFIG + : undefined; + const readiness = await evaluateWizardReadiness(readinessConfig); + logToFile(`[agent-runner] readiness=${readiness.decision}`); + if (readiness.decision === WizardReadiness.No) { + const blockingKeys = getBlockingServiceKeys( + readiness.health, + readinessConfig, + ); + const blockingLabels = blockingKeys.map( + (k) => `${SERVICE_LABELS[k]} (${readiness.health[k].status})`, + ); + logToFile(`[agent-runner] blocked by: ${blockingLabels.join(', ')}`); + + await getUI().showBlockingOutage(readiness); + + // The TUI lets the user continue past an outage; non-interactive runs + // (CI) do the same automatically — the degraded services are reported + // above, but we proceed rather than aborting on a transient upstream blip. + if (!isNonInteractiveEnvironment()) { + await wizardAbort({ + code: ErrorCodes.EnvServiceOutage, + message: + 'Cannot start — external services are down:\n' + + blockingLabels.map((l) => ` - ${l}`).join('\n') + + '\n\nPlease try again later.', + }); + } + } else if (readiness.decision === WizardReadiness.YesWithWarnings) { + getUI().setReadinessWarnings(readiness); + } +} + +async function runSettingsGate(session: WizardSession): Promise { + const settingsConflicts = checkAllSettingsConflicts(session.installDir); + logToFile( + `[agent-runner] settings conflicts: ${ + settingsConflicts.length > 0 + ? settingsConflicts + .map((c) => `${c.source}(${c.keys.join(',')})`) + .join('; ') + : 'none' + }`, + ); + if (settingsConflicts.length === 0) return; + + for (const conflict of settingsConflicts) { + const level = conflict.source === 'managed' ? 'org' : conflict.source; + analytics.wizardCapture('settings conflict detected', { + level, + keys: conflict.keys, + }); + } + + const { autoFix, failClosed, warnOnly } = + classifySettingsConflicts(settingsConflicts); + + // User-global and project-local files are already neutralized — the agent + // runs with settingSources:['project'], so the SDK never reads them. Record + // it and move on; don't make the user act on a setting that can't bite. + for (const conflict of warnOnly) { + logToFile( + `[agent-runner] settings conflict in ${conflict.source} (${conflict.path}) ` + + `neutralized by settingSources:['project'] — not blocking`, + ); + analytics.wizardCapture('settings conflict neutralized', { + level: conflict.source, + keys: conflict.keys, + }); + } + + // Writable project settings.json — the SDK *does* read it, but we can back + // it up and remove it (restored at outro). Neutralize without prompting. + let unfixable = failClosed; + if (autoFix.length > 0) { + const fixed = backupAndFixClaudeSettings(session.installDir); + if (fixed) { + logToFile('[agent-runner] auto-neutralized writable settings conflict'); + analytics.wizardCapture('settings conflict auto-neutralized', { + keys: autoFix.flatMap((c) => c.keys), + }); + } else { + // Couldn't remove it — don't run into the redirect; fail closed instead. + logToFile( + '[agent-runner] could not back up writable settings conflict — failing closed', + ); + unfixable = [...failClosed, ...autoFix]; + } + } + + // What we cannot neutralize (org-managed, always read by the SDK; or a + // writable file we failed to back up) must be fixed by the user. Fail + // closed: the screen names the file + keys and exits. + if (unfixable.length > 0) { + if (isNonInteractiveEnvironment()) { + await wizardAbort({ + code: ErrorCodes.SettingsUnfixableConflict, + message: + 'Cannot start — a Claude settings file redirects the agent away ' + + 'from the PostHog gateway and cannot be neutralized automatically:\n' + + unfixable + .map((c) => ` - ${c.source} (${c.path}): ${c.keys.join(', ')}`) + .join('\n') + + '\n\nRemove the conflicting keys and re-run the wizard.', + }); + } + await getUI().showSettingsOverride(unfixable, () => + backupAndFixClaudeSettings(session.installDir), + ); + logToFile('[agent-runner] settings override resolved'); + } +} + +// ── Switchboard telemetry ───────────────────────────────────────────── + +/** + * One event + one log line per run: what entered the switchboard, which + * precedence rung decided each axis, and the final pick. + */ +function captureSwitchboardDecision( + ctx: SwitchboardCtx, + binding: ProgramBinding, +): void { + const trace = ctx.trace ?? {}; + // Unpinned orchestrator runs choose a model per task from the context-mill agent prompts; the orchestrator logs that map once the prompts load. + const perTaskModel = + binding.sequence === Sequence.orchestrator && trace.model === 'binding'; + const model = perTaskModel ? 'chosen-per-task' : binding.model; + const modelSource = perTaskModel ? 'agent-prompts' : trace.model; + analytics.wizardCapture('switchboard resolved', { + program: ctx.program, + flag_self_driving_use_pi_harness: + ctx.flags[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY], + flag_self_driving_pi_payload: JSON.stringify( + ctx.flagPayloads?.[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY] ?? null, + ), + flag_orchestrator: ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY], + cli_harness: ctx.cliHarness, + cli_sequence: ctx.cliSequence, + cli_model: ctx.cliModel, + harness_source: trace.harness, + model_source: modelSource, + sequence_source: trace.sequence, + harness: binding.harness, + model, + thinking_level: binding.thinkingLevel, + sequence: binding.sequence, + }); + logToFile( + `[switchboard] decision: program=${ctx.program}` + + ` in(orchestrator=${ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY] ?? '-'},` + + ` cli=${ctx.cliHarness ?? '-'}/${ctx.cliSequence ?? '-'}/${ + ctx.cliModel ?? '-' + })` + + ` → harness=${binding.harness} (${trace.harness ?? '?'})` + + ` model=${model} (${modelSource ?? '?'})` + + ` sequence=${binding.sequence} (${trace.sequence ?? '?'})`, + ); +} + +// ── Progress → WizardUI, one call per event ─────────────────────────── + +/** + * The inverse of the agent's former `getUI()` calls: one event, one + * `WizardUI` method, synchronous, in emission order. Because each case maps + * back to exactly the call the agent used to make, the frame and flow goldens + * hold without regeneration. + */ +export function createUiReducer(ui: WizardUI): (event: AgentProgress) => void { + let spinner: SpinnerHandle | undefined; + return (event) => { + switch (event.kind) { + case 'lifecycle': + if (event.phase === 'started') ui.startRun(); + else ui.outro(event.message); + break; + case 'spinner': { + const handle = (spinner ??= ui.spinner()); + handle[event.action](event.message); + break; + } + case 'log': + ui.log[event.level](event.message); + break; + case 'status': + ui.pushStatus(event.message); + break; + case 'tasks': + ui.syncTodos(event.tasks); + break; + case 'stage': + ui.setStage(event.stage); + break; + case 'url': + if (event.which === 'dashboard') ui.setDashboardUrl(event.url); + else ui.setNotebookUrl(event.url); + break; + case 'usage': + ui.addTokenUsage(event.delta); + break; + case 'finalCost': + ui.setFinalTokenCostUsd(event.usd); + break; + case 'handoff': + ui.setHandoffText(event.text); + break; + case 'authError': + ui.showAuthError(event.detail); + break; + case 'completion': + ui.setOutroData(event.outro); + break; + } + }; +} + +/** The agent's questions, answered wherever `getUI()` answers them today. */ +export function uiInteraction(ui: WizardUI): AgentInteraction { + return { + ask: (question) => ui.requestQuestion(question), + cancelAsk: () => ui.cancelPendingQuestion(), + taskNotice: (notice) => ui.showTaskNotice(notice), + cancelTaskNotice: () => ui.cancelTaskNotice(), + }; +} diff --git a/src/lib/programs/self-driving/detect.ts b/src/lib/programs/self-driving/detect.ts index d9c74f055..804fcd245 100644 --- a/src/lib/programs/self-driving/detect.ts +++ b/src/lib/programs/self-driving/detect.ts @@ -29,7 +29,7 @@ import { import { join } from 'path'; import { analytics } from '@utils/analytics'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; import { detectWarehouseSources } from '@lib/warehouse-sources/detect'; import type { DetectedSource } from '@lib/warehouse-sources/types'; diff --git a/src/lib/programs/self-driving/index.ts b/src/lib/programs/self-driving/index.ts index b507cc18e..38a0ad930 100644 --- a/src/lib/programs/self-driving/index.ts +++ b/src/lib/programs/self-driving/index.ts @@ -1,7 +1,7 @@ import { join } from 'path'; import { access, rm } from 'node:fs/promises'; import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { OutroKind, type WizardSession } from '@lib/wizard-session'; import { createSkillProgram } from '../agent-skill/index.js'; import { SELF_DRIVING_PROGRAM } from './steps.js'; diff --git a/src/lib/programs/self-driving/prompt.ts b/src/lib/programs/self-driving/prompt.ts index 5bb0a6af2..5bfe59839 100644 --- a/src/lib/programs/self-driving/prompt.ts +++ b/src/lib/programs/self-driving/prompt.ts @@ -1,5 +1,5 @@ import { AgentSignals } from '@lib/agent/agent-interface'; -import type { PromptContext } from '@lib/agent/agent-runner'; +import type { PromptContext } from '@lib/programs/program-run'; import type { DetectedSource } from '@lib/warehouse-sources/types'; /** diff --git a/src/lib/programs/warehouse-source/detect.ts b/src/lib/programs/warehouse-source/detect.ts index 9e97b7d0c..0386593e1 100644 --- a/src/lib/programs/warehouse-source/detect.ts +++ b/src/lib/programs/warehouse-source/detect.ts @@ -9,7 +9,7 @@ import { existsSync, statSync } from 'fs'; import { analytics } from '@utils/analytics'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { detectWarehouseSources } from '@lib/warehouse-sources/detect'; import type { DetectedSource } from '@lib/warehouse-sources/types'; diff --git a/src/lib/programs/warehouse-source/index.ts b/src/lib/programs/warehouse-source/index.ts index 059e240a0..4b31f3851 100644 --- a/src/lib/programs/warehouse-source/index.ts +++ b/src/lib/programs/warehouse-source/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { LONGER_ASK_TIMEOUT_MS } from '@lib/wizard-ask-bridge'; import { WAREHOUSE_SOURCE_PROGRAM } from './steps.js'; diff --git a/src/lib/programs/web-analytics-doctor/detect.ts b/src/lib/programs/web-analytics-doctor/detect.ts index 749f4e75d..ec224425c 100644 --- a/src/lib/programs/web-analytics-doctor/detect.ts +++ b/src/lib/programs/web-analytics-doctor/detect.ts @@ -1,6 +1,6 @@ import { existsSync, statSync } from 'fs'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; import { findPackageJsons } from '@lib/programs/shared/package-scanning'; diff --git a/src/lib/runners/__tests__/mint-recovery.test.ts b/src/lib/runners/__tests__/mint-recovery.test.ts index 24f9b5d83..076bf9a20 100644 --- a/src/lib/runners/__tests__/mint-recovery.test.ts +++ b/src/lib/runners/__tests__/mint-recovery.test.ts @@ -1,6 +1,6 @@ import { vi, it, expect, afterEach } from 'vitest'; import { runWizard } from '../run-wizard'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runAgent } from '@lib/programs/run-agent-legacy'; import { startTUI } from '@ui/tui/start-tui'; import { WizardStore } from '@ui/tui/store'; import { InkUI } from '@ui/tui/ink-ui'; @@ -10,7 +10,7 @@ import { ScreenId } from '@ui/tui/router'; import { HostResolution } from '@lib/host-resolution'; import { analytics } from '@utils/analytics'; -vi.mock('@lib/agent/agent-runner', () => ({ runAgent: vi.fn() })); +vi.mock('@lib/programs/run-agent-legacy', () => ({ runAgent: vi.fn() })); vi.mock('@ui/tui/start-tui', () => ({ startTUI: vi.fn() })); vi.mock('@lib/local-dev', async (original) => ({ ...(await original()), diff --git a/src/lib/runners/run-non-interactive.ts b/src/lib/runners/run-non-interactive.ts index 7230c3ec0..a4fd1f037 100644 --- a/src/lib/runners/run-non-interactive.ts +++ b/src/lib/runners/run-non-interactive.ts @@ -335,7 +335,7 @@ export function runNonInteractive( } } - const { runAgent } = await import('@lib/agent/agent-runner'); + const { runAgent } = await import('@lib/programs/run-agent-legacy'); await runAgent(config, session); await settleStream(RunPhase.Completed); } catch (error) { diff --git a/src/lib/runners/run-wizard.ts b/src/lib/runners/run-wizard.ts index 1ee2e7120..a0e0879ff 100644 --- a/src/lib/runners/run-wizard.ts +++ b/src/lib/runners/run-wizard.ts @@ -1,7 +1,7 @@ import { VERSION } from '@lib/version'; import { logToFile, getLogFilePath } from '@utils/debug'; -import { runAgent } from '@lib/agent/agent-runner'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { runAgent } from '@lib/programs/run-agent-legacy'; +import { authenticate } from '@lib/programs/authenticate'; import { getProgramConfig } from '@lib/programs/program-registry'; import { getAuditChecks } from '@lib/programs/audit/types'; import { maybeStampAiSdkDetected } from '@lib/programs/posthog-integration/detect'; diff --git a/src/lib/wizard-tools/__tests__/handoff.test.ts b/src/lib/wizard-tools/__tests__/handoff.test.ts index bbe2aae14..ea773abe8 100644 --- a/src/lib/wizard-tools/__tests__/handoff.test.ts +++ b/src/lib/wizard-tools/__tests__/handoff.test.ts @@ -11,34 +11,31 @@ import { } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; -import { getUI, setUI } from '@ui'; -import type { WizardUI } from '@ui/wizard-ui'; -import { MAX_HANDOFF_TEXT_CHARS, publishHandoff } from '../handoff'; +import { + MAX_HANDOFF_TEXT_CHARS, + publishHandoff as publishHandoffTo, +} from '../handoff'; describe('publishHandoff', () => { const captured: string[] = []; - let previousUI: WizardUI; let temporaryDirectory: string; let ambientOutputPath: string | undefined; + // The reporting seam: what the agent's `handoff` progress event carries. + const publishHandoff = (content: string) => + publishHandoffTo(content, (text) => { + captured.push(text); + }); beforeEach(() => { captured.length = 0; - previousUI = getUI(); temporaryDirectory = mkdtempSync(join(tmpdir(), 'wizard-handoff-')); // Save the ambient value so a developer running these tests with the var // exported doesn't have their environment silently clobbered. ambientOutputPath = process.env.POSTHOG_HANDOFF_OUTPUT_PATH; delete process.env.POSTHOG_HANDOFF_OUTPUT_PATH; - setUI({ - ...previousUI, - setHandoffText: (text: string) => { - captured.push(text); - }, - } as WizardUI); }); afterEach(() => { - setUI(previousUI); rmSync(temporaryDirectory, { recursive: true, force: true }); if (ambientOutputPath === undefined) { delete process.env.POSTHOG_HANDOFF_OUTPUT_PATH; @@ -47,7 +44,7 @@ describe('publishHandoff', () => { } }); - it('publishes the content through the UI seam', () => { + it('publishes the content through the reporting seam', () => { const result = publishHandoff('# Setup report\n\nAll done.'); expect(result.ok).toBe(true); expect(captured).toEqual(['# Setup report\n\nAll done.']); diff --git a/src/lib/wizard-tools/handoff.ts b/src/lib/wizard-tools/handoff.ts index 4dedf0428..4f3b3217d 100644 --- a/src/lib/wizard-tools/handoff.ts +++ b/src/lib/wizard-tools/handoff.ts @@ -3,7 +3,6 @@ * markdown) in one explicit call, replacing the report file + watcher path. */ -import { getUI } from '@ui'; import { analytics } from '@utils/analytics'; import { logToFile } from '@utils/debug'; import { runtimeEnv } from '@env'; @@ -134,7 +133,10 @@ function writeHandoffFileAtomically( } } -export function publishHandoff(content: string): PublishHandoffResult { +export function publishHandoff( + content: string, + onHandoffText?: (text: string) => void, +): PublishHandoffResult { if (content.trim() === '') { analytics.wizardCapture('handoff published', { handoff_ok: false, @@ -148,7 +150,7 @@ export function publishHandoff(content: string): PublishHandoffResult { } const truncated = content.length > MAX_HANDOFF_TEXT_CHARS; const text = truncated ? content.slice(0, MAX_HANDOFF_TEXT_CHARS) : content; - getUI().setHandoffText(text); + onHandoffText?.(text); const handoffOutputPath = runtimeEnv('POSTHOG_HANDOFF_OUTPUT_PATH'); let handoffOutputWritten: boolean | undefined; diff --git a/src/lib/wizard-tools/mcp.ts b/src/lib/wizard-tools/mcp.ts index 4ad43e235..493724991 100644 --- a/src/lib/wizard-tools/mcp.ts +++ b/src/lib/wizard-tools/mcp.ts @@ -147,6 +147,9 @@ export interface WizardToolsOptions { /** Scan-triage classifier for install_skill's scan, resolved by the caller. */ triageProvider: LLMProvider; + + /** Receives the handoff document `publish_handoff` accepted. */ + onHandoffText?: (text: string) => void; } /** Default per-run cap on wizard_ask calls when no override is provided. */ @@ -168,6 +171,7 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { secretVault = createSecretVault(), orchestrator, triageProvider, + onHandoffText, } = options; const sdk = await getSDKModule(); const { tool, createSdkMcpServer } = sdk; @@ -755,7 +759,7 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { content: z.string().describe(PUBLISH_HANDOFF_CONTENT_DESCRIPTION), }, (args: { content: string }) => { - const result = publishHandoff(args.content); + const result = publishHandoff(args.content, onHandoffText); logToFile(`publish_handoff: ${result.message}`); return { content: [{ type: 'text' as const, text: result.message }], diff --git a/src/lib/yara-hooks.ts b/src/lib/yara-hooks.ts index 3ec2cb266..63dadafe3 100644 --- a/src/lib/yara-hooks.ts +++ b/src/lib/yara-hooks.ts @@ -35,8 +35,6 @@ import type { import { logToFile } from '@utils/debug'; import { readFileHead } from '@utils/bounded-fs'; import { analytics } from '@utils/analytics'; -import { getUI } from '@ui'; -import type { WizardSession } from '@lib/wizard-session'; import { isSkillInstallCommand } from './skill-install'; import { highestSeverityMatch, @@ -431,13 +429,14 @@ export function captureScanReport(): void { * scanCount === 0 and every step no-ops. */ export function flushScanReport( - session: Pick, + options: { yaraReport: boolean }, + log: (message: string) => void, ): void { - if (session.yaraReport) { + if (options.yaraReport) { const reportPath = writeScanReport(); if (reportPath) { const summary = formatScanReport(); - getUI().log.info(`YARA scan report: ${reportPath}${summary ?? ''}`); + log(`YARA scan report: ${reportPath}${summary ?? ''}`); } } captureScanReport(); diff --git a/src/ui/tui/services/mcp-suggested-prompts-services.ts b/src/ui/tui/services/mcp-suggested-prompts-services.ts index 8b41505ca..f5ba6cd66 100644 --- a/src/ui/tui/services/mcp-suggested-prompts-services.ts +++ b/src/ui/tui/services/mcp-suggested-prompts-services.ts @@ -21,22 +21,8 @@ import { } from '@lib/mcp-project-profile'; import { seedDemoEvents as runSeed } from '@lib/mcp-seed-events'; -/** - * Discriminated union covering every kind of streamed event the screen - * needs to render. Production yields these from Claude SDK messages; - * the playground yields them from canned scripts. - */ -export type AgentChunk = - | { kind: 'text'; text: string } - /** `command` carries CLI mode's exec command string (`call …`) so the - * screen can recover the inner tool for context-aware follow-ups. */ - | { kind: 'tool-call'; toolName: string; detail: string; command?: string } - | { kind: 'tool-result'; toolName: string; detail: string } - | { kind: 'error'; text: string } - /** Stream completed. `sessionId` is the SDK session ID of the just- - * completed turn; pass it back as `resumeSessionId` on a follow-up - * call to continue the conversation with full history. */ - | { kind: 'done'; sessionId?: string }; +export type { AgentChunk } from '@lib/agent/mcp-prompt-streaming'; +import type { AgentChunk } from '@lib/agent/mcp-prompt-streaming'; export interface McpSuggestedPromptsServices { /** diff --git a/src/ui/wizard-ui.ts b/src/ui/wizard-ui.ts index 38ad29235..7a96fea1b 100644 --- a/src/ui/wizard-ui.ts +++ b/src/ui/wizard-ui.ts @@ -9,6 +9,11 @@ */ import type { SettingsConflict } from '@lib/agent/claude-settings'; +import type { + AuthErrorDetail, + SpinnerHandle, + TokenUsageDelta, +} from '@lib/agent/progress'; import type { WizardReadinessResult } from '@lib/health-checks/readiness'; import type { ApiUser } from '@lib/api'; import type { Credentials, TaskNotice } from '@lib/wizard-session'; @@ -29,60 +34,11 @@ export function isTaskStatus(value: string): value is TaskStatus { return (Object.values(TaskStatus) as string[]).includes(value); } -/** - * One assistant turn's token usage, for the hidden Ctrl+T token/cost HUD. - * `model` is the model that produced *this* turn (e.g. the SDK's - * `message.message.model`) — a subagent can run on a different model than - * the main session, and some programs override to Haiku, so pricing must key - * off the per-turn model rather than a single run-wide assumption. Omit only - * when the caller genuinely has no model context (falls back to Sonnet - * pricing — see `pricePerMtokForModel` in `@lib/agent/token-pricing`). - */ -export interface TokenUsageDelta { - inputTokens: number; - outputTokens: number; - cacheReadTokens: number; - cacheCreationTokens: number; - cacheCreation5m: number; - cacheCreation1h: number; - model?: string; -} - -export interface SpinnerHandle { - start(message?: string): void; - stop(message?: string): void; - message(msg?: string): void; -} - -/** - * Context passed to `showAuthError` so the screen can pick the right copy. - * - * `hasSettingsConflict` is true when a Claude Code settings file (project, - * project-local, the user's global config, or managed) actually overrides the - * LLM Gateway auth. `conflicts` carries the exact files and keys so the screen - * can name them. When there is no conflict, the 401 has a different cause (bad - * PAT prefix, missing scope, expired key, region mismatch) and we should not - * advise the user to log out of Claude Code. - */ -export interface AuthErrorDetail { - hasSettingsConflict: boolean; - conflicts?: SettingsConflict[]; - /** - * True when the agent SDK authenticated from a stored Claude login - * (`apiKeySource: "/login managed key"`) instead of the wizard's gateway - * token — conflicting Anthropic credentials. Takes priority in the screen. - */ - usingManagedLogin?: boolean; - /** Human-readable places a conflicting Anthropic credential may live. */ - credentialPlaces?: string[]; - /** - * True when a pre-run refresh already failed on a dead grant. The login is - * gone and re-running is the only fix, so this outranks every other branch — - * none of the usual advice (key type, scopes, region) applies. - */ - sessionExpired?: boolean; - logFilePath: string; -} +export type { + AuthErrorDetail, + SpinnerHandle, + TokenUsageDelta, +} from '@lib/agent/progress'; export interface WizardUI { // ── Lifecycle messages ──────────────────────────────────────────── diff --git a/src/utils/wizard-abort.ts b/src/utils/wizard-abort.ts index 7805edcc4..bb3731a8e 100644 --- a/src/utils/wizard-abort.ts +++ b/src/utils/wizard-abort.ts @@ -13,20 +13,9 @@ import { LoggingUI } from '@ui/logging-ui'; import { OutroKind, type OutroData } from '@lib/wizard-session'; import type { ErrorCode } from '@lib/errors'; import { emitWizardError, sanitizeErrorDetail } from '@lib/errors'; +import { WizardError } from '@lib/errors/wizard-error'; -export class WizardError extends Error { - readonly code?: ErrorCode; - - constructor( - message: string, - public readonly context?: Record, - code?: ErrorCode, - ) { - super(message); - this.name = 'WizardError'; - this.code = code; - } -} +export { WizardError }; interface WizardAbortOptions { message?: string;