From a18d15fbd74bda5a070880e576597a61a7cc8fcc Mon Sep 17 00:00:00 2001 From: "Vincent (Wen Yu) Ge" Date: Mon, 21 Sep 2026 17:41:12 -0400 Subject: [PATCH] refactor(agent): make runAgent a function with progress and interaction contracts runAgent(config, input, {onProgress?, interaction?, signal?}) returns a RunResult and never rejects. RunConfig and RunInput replace the session and program config reads, AgentProgress replaces the 58 getUI() calls, an optional AgentInteraction replaces the getUI() answerer, and every former wizardAbort returns as a failure with the same fields. Unexpected throws return as outcome 'crashed' with the original error. Gates, authenticate, token refresh, flag fetch, binding resolution and the exit move to src/lib/programs/run-agent-legacy.ts, which maps each progress event to one WizardUI call so every existing caller keeps its output. PROGRAM_BINDINGS, ProgramRun and authenticate leave the agent for programs. The architecture test forbids agent -> src/ui imports; known violations go from 72 to 60. A standalone test runs the agent with @ui mocked to throw. Generated-By: PostHog Desktop Task-Id: d14e92bb-6ee1-49b5-8502-39cb80079589 --- .claude/skills/wizard-development/SKILL.md | 7 +- .../references/ARCHITECTURE.md | 50 +- AGENTS.md | 13 +- scripts/tui-host.no-jest.ts | 4 +- .../architecture/import-boundaries.test.ts | 5 + .../architecture/known-violations.json | 30 +- src/__tests__/cli.test.ts | 2 +- src/__tests__/provision-cli.test.ts | 2 +- src/lib/__tests__/agent-interface.test.ts | 33 +- src/lib/agent/__tests__/agent-prompt.test.ts | 2 +- .../__tests__/run-agent-standalone.test.ts | 389 +++++++++++++ src/lib/agent/agent-interface.ts | 112 +++- src/lib/agent/agent-prompt.ts | 7 +- src/lib/agent/agent-runner.ts | 22 +- src/lib/agent/mcp-prompt-streaming.ts | 17 +- src/lib/agent/progress.ts | 152 +++++ src/lib/agent/runner/README.md | 65 ++- .../runner/__tests__/switchboard.test.ts | 7 +- .../agent/runner/harness/anthropic/index.ts | 48 +- src/lib/agent/runner/harness/pi/index.ts | 75 ++- src/lib/agent/runner/harness/pi/task.ts | 19 +- src/lib/agent/runner/harness/pi/tasks.ts | 33 +- src/lib/agent/runner/harness/pi/tools.ts | 4 +- src/lib/agent/runner/harness/types.ts | 43 +- src/lib/agent/runner/index.ts | 285 ++++----- src/lib/agent/runner/sequence/linear.ts | 282 ++++----- .../__tests__/seeded-decline-skip.test.ts | 3 - .../__tests__/task-notice-timeout.test.ts | 43 +- .../runner/sequence/orchestrator/executor.ts | 15 + .../orchestrator/orchestrator-runner.ts | 228 +++++--- src/lib/agent/runner/shared/ask.ts | 61 ++ src/lib/agent/runner/shared/bootstrap.ts | 327 ++--------- src/lib/agent/runner/shared/errors.ts | 16 +- .../agent/runner/shared/progress-collector.ts | 120 ++++ src/lib/agent/runner/shared/types.ts | 274 +++++++-- .../flags/__tests__/binding-cases.ts | 8 +- .../switchboard/flags/__tests__/flags.test.ts | 9 +- .../agent/runner/switchboard/flags/index.ts | 7 +- .../agent/runner/switchboard/flags/schemes.ts | 5 +- src/lib/agent/runner/switchboard/harness.ts | 5 +- src/lib/agent/runner/switchboard/index.ts | 76 +-- src/lib/agent/runner/switchboard/sequence.ts | 31 +- .../detection/__tests__/project-scope.test.ts | 4 +- src/lib/detection/project-scope.ts | 2 +- src/lib/errors/index.ts | 1 + src/lib/errors/wizard-error.ts | 20 + src/lib/middleware/benchmark.ts | 11 +- src/lib/middleware/benchmarks/index.ts | 4 +- src/lib/middleware/benchmarks/json-writer.ts | 11 +- src/lib/middleware/benchmarks/summary.ts | 26 +- src/lib/middleware/types.ts | 4 +- .../programs/__tests__/agent-skill.test.ts | 2 +- .../programs/__tests__/error-tracking.test.ts | 2 +- .../__tests__/metrics-program.test.ts | 2 +- .../refresh-access-token-if-needed.test.ts | 0 src/lib/programs/agent-skill/index.ts | 2 +- src/lib/programs/audit/detect.ts | 2 +- src/lib/programs/audit/index.ts | 2 +- .../shared => programs}/authenticate.ts | 2 +- src/lib/programs/bindings.ts | 77 +++ .../detect.ts | 2 +- .../index.ts | 2 +- src/lib/programs/error-tracking/index.ts | 2 +- src/lib/programs/events-audit/index.ts | 2 +- src/lib/programs/mcp-analytics/index.ts | 2 +- src/lib/programs/migration/index.ts | 2 +- src/lib/programs/posthog-integration/index.ts | 5 +- src/lib/programs/program-run.ts | 48 ++ src/lib/programs/program-step.ts | 2 +- src/lib/programs/replay-vision/index.ts | 2 +- src/lib/programs/revenue-analytics/detect.ts | 2 +- src/lib/programs/run-agent-legacy.ts | 539 ++++++++++++++++++ src/lib/programs/self-driving/detect.ts | 2 +- src/lib/programs/self-driving/index.ts | 2 +- src/lib/programs/self-driving/prompt.ts | 2 +- src/lib/programs/warehouse-source/detect.ts | 2 +- src/lib/programs/warehouse-source/index.ts | 2 +- .../programs/web-analytics-doctor/detect.ts | 2 +- .../runners/__tests__/mint-recovery.test.ts | 4 +- src/lib/runners/run-non-interactive.ts | 2 +- src/lib/runners/run-wizard.ts | 4 +- .../wizard-tools/__tests__/handoff.test.ts | 23 +- src/lib/wizard-tools/handoff.ts | 8 +- src/lib/wizard-tools/mcp.ts | 6 +- src/lib/yara-hooks.ts | 9 +- .../mcp-suggested-prompts-services.ts | 18 +- src/ui/wizard-ui.ts | 64 +-- src/utils/wizard-abort.ts | 15 +- 88 files changed, 2661 insertions(+), 1224 deletions(-) create mode 100644 src/lib/agent/__tests__/run-agent-standalone.test.ts create mode 100644 src/lib/agent/progress.ts create mode 100644 src/lib/agent/runner/shared/ask.ts create mode 100644 src/lib/agent/runner/shared/progress-collector.ts create mode 100644 src/lib/errors/wizard-error.ts rename src/lib/{agent/runner/shared => programs}/__tests__/refresh-access-token-if-needed.test.ts (100%) rename src/lib/{agent/runner/shared => programs}/authenticate.ts (98%) create mode 100644 src/lib/programs/bindings.ts create mode 100644 src/lib/programs/program-run.ts create mode 100644 src/lib/programs/run-agent-legacy.ts diff --git a/.claude/skills/wizard-development/SKILL.md b/.claude/skills/wizard-development/SKILL.md index 3ee4f261d..f56fdc64c 100644 --- a/.claude/skills/wizard-development/SKILL.md +++ b/.claude/skills/wizard-development/SKILL.md @@ -94,8 +94,11 @@ only when it gives a real owner a smaller, reusable boundary. ## Lifecycle and security -[runner/index.ts](../../../src/lib/agent/runner/index.ts) bootstraps, resolves a -binding, and dispatches to a sequence. `agent-runner.ts` is a compatibility +[runner/index.ts](../../../src/lib/agent/runner/index.ts) exports +`runAgent(config, input, options)`: it prepares the run and dispatches to the +sequence the binding names, reporting through `onProgress` and asking through +`interaction`. Gates, authentication and binding resolution live in +`src/lib/programs/run-agent-legacy.ts`. `agent-runner.ts` is a compatibility re-export, not the implementation. Sequences own their lifecycle; harnesses own SDK calls. `ProgramRun.postRun`, `buildOutroData`, `customPrompt`, and `abortCases` are consumed by the linear sequence, not by the orchestrator. Put diff --git a/.claude/skills/wizard-development/references/ARCHITECTURE.md b/.claude/skills/wizard-development/references/ARCHITECTURE.md index bb99255f3..a95bd2b9c 100644 --- a/.claude/skills/wizard-development/references/ARCHITECTURE.md +++ b/.claude/skills/wizard-development/references/ARCHITECTURE.md @@ -16,19 +16,27 @@ readiness hooks, and traverse steps and gates. detection in `onReady`. Noninteractive execution has its own lifecycle in [run-non-interactive.ts](../../../../src/lib/runners/run-non-interactive.ts). -[runner/index.ts](../../../../src/lib/agent/runner/index.ts) resolves a -program's `run` definition, calls shared bootstrap, selects a binding, -dispatches the sequence, and flushes the scanner report on cleanup. The old +[runner/index.ts](../../../../src/lib/agent/runner/index.ts) exports +`runAgent(config, input, options)`: it prepares the run, dispatches the sequence +the binding names, flushes the scanner report and returns a `RunResult`. It +reports through `options.onProgress` and asks through `options.interaction`; it +never calls `getUI()`, reads a session or exits. +[run-agent-legacy.ts](../../../../src/lib/programs/run-agent-legacy.ts) is the +caller today's runners use: it runs the gates, authenticates, resolves the +binding from [bindings.ts](../../../../src/lib/programs/bindings.ts), builds the +agent's inputs from the session, maps progress back onto `getUI()` and hands a +decided failure to `wizardAbort`. [agent-runner.ts](../../../../src/lib/agent/agent-runner.ts) is a compatibility -export. +export of the agent's contracts. -| Layer | Source and responsibility | -| --------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| Bootstrap | [shared/bootstrap.ts](../../../../src/lib/agent/runner/shared/bootstrap.ts): shared health/settings/auth/flag and MCP setup | -| Switchboard | [switchboard/index.ts](../../../../src/lib/agent/runner/switchboard/index.ts): resolve sequence, harness, model and effort override | -| Linear sequence | [sequence/linear.ts](../../../../src/lib/agent/runner/sequence/linear.ts): one conversation, skill/prompt assembly, post-run hooks and outro | -| Orchestrator | [orchestrator-runner.ts](../../../../src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts): seed plan, task queue, focused conversations, handoffs and completion | -| Harness | [harness/types.ts](../../../../src/lib/agent/runner/harness/types.ts): SDK boundary, implemented by Pi and Anthropic | +| Layer | Source and responsibility | +| --------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Contracts | [shared/types.ts](../../../../src/lib/agent/runner/shared/types.ts) and [progress.ts](../../../../src/lib/agent/progress.ts): `RunConfig`, `RunInput`, `RunResult`, `AgentProgress`, `AgentInteraction` | +| Prepare | [shared/bootstrap.ts](../../../../src/lib/agent/runner/shared/bootstrap.ts): logging targets, gateway mint, scan triage | +| Switchboard | [switchboard/index.ts](../../../../src/lib/agent/runner/switchboard/index.ts): resolve sequence, harness, model and effort override | +| Linear sequence | [sequence/linear.ts](../../../../src/lib/agent/runner/sequence/linear.ts): one conversation, skill/prompt assembly, post-run hooks and outro | +| Orchestrator | [orchestrator-runner.ts](../../../../src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts): seed plan, task queue, focused conversations, handoffs and completion | +| Harness | [harness/types.ts](../../../../src/lib/agent/runner/harness/types.ts): SDK boundary, implemented by Pi and Anthropic | The contribution policy is [Pi by default, orchestration preferred](../SKILL.md#execution-policy-and-model-admission). @@ -135,16 +143,20 @@ credentials. ## UI state and agent output -Business logic uses [WizardUI](../../../../src/ui/wizard-ui.ts) through -`getUI()`. [InkUI](../../../../src/ui/tui/ink-ui.ts) updates the TUI store; -[LoggingUI](../../../../src/ui/logging-ui.ts) is available for noninteractive -callers that select it. A missing TTY does not automatically mean an arbitrary -caller uses LoggingUI; snapshot CI drives Ink in a PTY. `requestQuestion` and -task notices are supported interactions, not console prompts to invent in -business logic. +Business logic outside the agent uses +[WizardUI](../../../../src/ui/wizard-ui.ts) through `getUI()`. The agent +(`src/lib/agent`) reports through `AgentProgress` events instead, and the +reducer in +[run-agent-legacy.ts](../../../../src/lib/programs/run-agent-legacy.ts) maps +each event to one `WizardUI` call. [InkUI](../../../../src/ui/tui/ink-ui.ts) +updates the TUI store; [LoggingUI](../../../../src/ui/logging-ui.ts) is +available for noninteractive callers that select it. A missing TTY does not +automatically mean an arbitrary caller uses LoggingUI; snapshot CI drives Ink in +a PTY. `requestQuestion` and task notices are supported interactions, not +console prompts to invent in business logic. Harness adapters translate SDK messages, status markers, task updates and tool -activity into WizardUI calls. Anthropic message processing lives in +activity into progress events. Anthropic message processing lives in [agent-interface.ts](../../../../src/lib/agent/agent-interface.ts); Pi uses its own session event handlers. Orchestrated tasks also have queue and handoff state. Do not assume all harness output passes through `handleSDKMessage`. diff --git a/AGENTS.md b/AGENTS.md index bf6620e5c..a511f8404 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -97,8 +97,8 @@ aliases. | Subcommand | What it audits | | ----------------------------- | ---------------------------------------------------- | -| `wizard audit events` | event capture quality + cost | -| `wizard audit all` | comprehensive audit across every area (**default**) | +| `wizard audit events` | event capture quality + cost | +| `wizard audit all` | comprehensive audit across every area (**default**) | | `wizard audit autocapture` | autocapture setup + cost | | `wizard audit feature-flags` | feature flag usage + cost | | `wizard audit identify` | `$identify` implementation | @@ -170,8 +170,8 @@ nonmutating lint checks, and scope formatting fixes to edited files. Do not add tests for prose, compiler-enforced shapes, or duplicated implementation. Keep new code comments to one line; put longer explanations in linked docs. -Local `--ci`, smoke-test, and full headless runs require two separate secrets: -a PostHog personal API key and an already-issued gateway token supplied through +Local `--ci`, smoke-test, and full headless runs require two separate secrets: a +PostHog personal API key and an already-issued gateway token supplied through `WIZARD_CI_GATEWAY_TOKEN_FILE`, plus the target project ID. Follow the [credential setup](docs/local-dev.md#credentials-for-local-ci-and-headless-runs). @@ -196,7 +196,10 @@ wizard run points. Full catalog: [`docs/local-dev.md`](docs/local-dev.md). - TypeScript everywhere. Use `type` (not `interface`) for framework context types so they satisfy `Record`. - All UI calls go through `getUI()` (returns `WizardUI` interface). Never import - the store directly from business logic. + the store directly from business logic. The agent (`src/lib/agent`) is the + exception: it reports through `AgentProgress` events and asks through + `AgentInteraction`, and never imports `src/ui` (the architecture test enforces + this). - Session mutations go through explicit store setters that call `emitChange()`. Never mutate `session` directly — nanostore holds a shallow copy. - The router resolves the active screen from session state. No imperative diff --git a/scripts/tui-host.no-jest.ts b/scripts/tui-host.no-jest.ts index ded2d55a2..ead31e380 100644 --- a/scripts/tui-host.no-jest.ts +++ b/scripts/tui-host.no-jest.ts @@ -27,10 +27,10 @@ import type { Harness, Sequence } from '@lib/constants'; import { buildSession } from '@lib/wizard-session'; import { initLocalDev } from '@lib/local-dev'; import { configureGatewayFromCIEnvironment } from '@lib/gateway-session'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runAgent } from '@lib/programs/run-agent-legacy'; import { TaskStreamPush, createFileDestination } from '@lib/task-stream/index'; import { getAuditChecks } from '@lib/programs/audit/types'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { authenticate } from '@lib/programs/authenticate'; import { getOrAskForProjectData } from '@utils/setup-utils'; import { logToFile } from '@utils/debug'; import { join } from 'path'; diff --git a/src/__tests__/architecture/import-boundaries.test.ts b/src/__tests__/architecture/import-boundaries.test.ts index 8220182ae..c16a380ef 100644 --- a/src/__tests__/architecture/import-boundaries.test.ts +++ b/src/__tests__/architecture/import-boundaries.test.ts @@ -304,6 +304,11 @@ function analyze(): Analysis { const to = classifySurface(target); if (!allowed.includes(to)) violations.set(key, `matrix:${from}->${to}`); + // The agent reports through `onProgress` and asks through + // `AgentInteraction`; it never reaches for a renderer, not even a type. + if (from === 'agent' && target.startsWith('src/ui/')) { + violations.set(key, 'agent-imports-ui'); + } } } diff --git a/src/__tests__/architecture/known-violations.json b/src/__tests__/architecture/known-violations.json index d523b2a73..f6bead34a 100644 --- a/src/__tests__/architecture/known-violations.json +++ b/src/__tests__/architecture/known-violations.json @@ -2,52 +2,39 @@ "violations": [ "src/commands/factories/family-picker.tsx -> src/commands/command.ts", "src/env.ts -> src/lib/headless-mode.ts", - "src/lib/agent/mcp-prompt-streaming.ts -> src/ui/tui/services/mcp-suggested-prompts-services.ts", "src/lib/detection/agentic.ts -> src/lib/agent/agent-interface.ts", - "src/lib/detection/project-scope.ts -> src/lib/agent/runner/shared/authenticate.ts", "src/lib/errors/agent-map.ts -> src/lib/agent/signals.ts", - "src/lib/programs/agent-skill/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/agent-skill/index.ts -> src/lib/programs/agent-skill/content/index.tsx", "src/lib/programs/ai-observability/index.ts -> src/lib/programs/agent-skill/content/index.tsx", - "src/lib/programs/audit/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/audit/index.ts -> src/lib/agent/agent-runner.ts", + "src/lib/programs/bindings.ts -> src/lib/agent/runner/switchboard/index.ts", "src/lib/programs/dispatch-family.ts -> src/commands/command.ts", "src/lib/programs/dispatch-family.ts -> src/commands/factories/shared.ts", - "src/lib/programs/error-tracking-upload-source-maps/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/error-tracking-upload-source-maps/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/error-tracking-upload-source-maps/index.ts -> src/lib/programs/error-tracking-upload-source-maps/content/index.tsx", "src/lib/programs/error-tracking-upload-source-maps/prompt.ts -> src/lib/agent/agent-interface.ts", - "src/lib/programs/error-tracking/index.ts -> src/lib/agent/runner/shared/types.ts", "src/lib/programs/error-tracking/index.ts -> src/lib/programs/error-tracking/content/index.tsx", "src/lib/programs/error-tracking/index.ts -> src/lib/programs/error-tracking/content/tips.ts", - "src/lib/programs/events-audit/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/mcp-analytics/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/metrics/index.ts -> src/lib/programs/agent-skill/content/index.tsx", - "src/lib/programs/migration/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/migration/index.ts -> src/lib/programs/migration/content/index.tsx", "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/agent-interface.ts", "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/posthog-integration/index.ts -> src/lib/agent/runner/shared/bootstrap.ts", "src/lib/programs/posthog-integration/index.ts -> src/lib/programs/posthog-integration/content/index.tsx", "src/lib/programs/program-registry.ts -> src/lib/programs/agent-skill/content/index.tsx", - "src/lib/programs/program-step.ts -> src/lib/agent/agent-runner.ts", + "src/lib/programs/program-run.ts -> src/lib/agent/runner/shared/types.ts", "src/lib/programs/program-step.ts -> src/ui/tui/components/TipsCard.tsx", "src/lib/programs/program-step.ts -> src/ui/tui/primitives/index.ts", "src/lib/programs/program-step.ts -> src/ui/tui/store.ts", - "src/lib/programs/replay-vision/index.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/revenue-analytics/detect.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/revenue-analytics/index.ts -> src/lib/programs/revenue-analytics/content/index.tsx", - "src/lib/programs/self-driving/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/self-driving/index.ts -> src/lib/agent/agent-runner.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/agent-interface.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/claude-settings.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/progress.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/index.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/agent/runner/switchboard/index.ts", + "src/lib/programs/run-agent-legacy.ts -> src/lib/yara-hooks.ts", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/index.tsx", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/pricing.ts", "src/lib/programs/self-driving/index.ts -> src/lib/programs/self-driving/content/tips.ts", "src/lib/programs/self-driving/prompt.ts -> src/lib/agent/agent-interface.ts", - "src/lib/programs/self-driving/prompt.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/warehouse-source/detect.ts -> src/lib/agent/agent-runner.ts", - "src/lib/programs/warehouse-source/index.ts -> src/lib/agent/agent-runner.ts", "src/lib/programs/warehouse-source/index.ts -> src/lib/programs/warehouse-source/content/index.tsx", - "src/lib/programs/web-analytics-doctor/detect.ts -> src/lib/agent/agent-runner.ts", "src/lib/task-stream/event-plan-watcher.ts -> src/ui/tui/store.ts", "src/lib/task-stream/task-stream-push.ts -> src/ui/tui/store.ts", "src/lib/wizard-session.ts -> src/lib/agent/claude-settings.ts", @@ -69,6 +56,7 @@ "src/ui/tui/store.ts -> src/lib/agent/claude-settings.ts", "src/ui/tui/store.ts -> src/lib/agent/token-pricing.ts", "src/ui/wizard-ui.ts -> src/lib/agent/claude-settings.ts", + "src/ui/wizard-ui.ts -> src/lib/agent/progress.ts", "src/utils/package-manager.ts -> src/telemetry.ts", "src/utils/setup-utils.ts -> src/telemetry.ts", "src/utils/wizard-abort.ts -> src/ui/logging-ui.ts" diff --git a/src/__tests__/cli.test.ts b/src/__tests__/cli.test.ts index 2b8a723f7..e4ebbe951 100644 --- a/src/__tests__/cli.test.ts +++ b/src/__tests__/cli.test.ts @@ -112,7 +112,7 @@ vi.mock('../utils/wizard-abort', async (importOriginal) => ({ ...(await importOriginal()), wizardAbort: vi.fn(), })); -vi.mock('../lib/agent/agent-runner', () => ({ +vi.mock('../lib/programs/run-agent-legacy', () => ({ runAgent: vi.fn().mockResolvedValue(undefined), })); diff --git a/src/__tests__/provision-cli.test.ts b/src/__tests__/provision-cli.test.ts index d1b561ba4..10b6539b2 100644 --- a/src/__tests__/provision-cli.test.ts +++ b/src/__tests__/provision-cli.test.ts @@ -63,7 +63,7 @@ vi.mock('../utils/wizard-abort', async (importOriginal) => ({ ...(await importOriginal()), wizardAbort: vi.fn(), })); -vi.mock('../lib/agent/agent-runner', () => ({ +vi.mock('../lib/programs/run-agent-legacy', () => ({ runAgent: vi.fn().mockResolvedValue(undefined), })); diff --git a/src/lib/__tests__/agent-interface.test.ts b/src/lib/__tests__/agent-interface.test.ts index 3c48975be..7ca36c2ca 100644 --- a/src/lib/__tests__/agent-interface.test.ts +++ b/src/lib/__tests__/agent-interface.test.ts @@ -12,7 +12,6 @@ import { import { AgentOutputSignals } from '@lib/agent/output-signals'; import { RESUME_INSTRUCTION } from '@lib/agent/signals'; import { analytics } from '@utils/analytics'; -import { wizardAbort } from '@utils/wizard-abort'; import { Sequence } from '@lib/constants'; import type { WizardRunOptions } from '@utils/types'; import type { SpinnerHandle } from '@ui'; @@ -24,11 +23,6 @@ import { // Mock dependencies vi.mock('../../utils/analytics'); vi.mock('../../utils/debug'); -// wizardAbort exits the process; the 401 tests below need it to just reject. -vi.mock('@utils/wizard-abort', async (importOriginal) => ({ - ...(await importOriginal()), - wizardAbort: vi.fn(), -})); // Mock the SDK module const mockQuery = vi.fn(); @@ -695,6 +689,10 @@ describe('subprocess gateway credentials', () => { describe('gateway re-mint on 401', () => { const spinner = { start: vi.fn(), stop: vi.fn(), message: vi.fn() }; + // Where the run reports the auth screen; stands where getUI() used to. + const emit = vi.fn(); + const authErrors = () => + emit.mock.calls.filter(([e]) => e.kind === 'authError'); const options: WizardRunOptions = { debug: false, installDir: '/test/dir', @@ -722,6 +720,7 @@ describe('gateway re-mint on 401', () => { triageProvider: () => Promise.resolve('false_positive'), gatewayAuth, refreshGatewayAuth, + emit, }); const run = (cfg: ReturnType) => runAgent(cfg, 'test prompt', options, spinner as unknown as SpinnerHandle, { @@ -778,7 +777,7 @@ describe('gateway re-mint on 401', () => { beforeEach(() => { vi.clearAllMocks(); mockUIInstance.spinner.mockReturnValue(spinner); - vi.mocked(wizardAbort).mockRejectedValue(new Error('wizardAbort: exit')); + emit.mockReset(); }); it('mints once and resumes the session when an aged bearer is rejected', async () => { @@ -794,7 +793,7 @@ describe('gateway re-mint on 401', () => { expect(result).toEqual({}); expect(refresh).toHaveBeenCalledTimes(1); - expect(wizardAbort).not.toHaveBeenCalled(); + expect(authErrors()).toHaveLength(0); expect(mockQuery).toHaveBeenCalledTimes(2); const [first, second] = mockQuery.mock.calls.map((c) => c[0]); expect(first.options.resume).toBeUndefined(); @@ -833,11 +832,12 @@ describe('gateway re-mint on 401', () => { expect(refresh).toHaveBeenCalledTimes(1); expect(mockQuery).toHaveBeenCalledTimes(2); - expect(mockUIInstance.showAuthError).toHaveBeenCalledTimes(1); - expect(wizardAbort).toHaveBeenCalledTimes(1); - // In production wizardAbort exits; the mocked rejection surfaces as the - // run's API error. - expect(result.error).toBe('WIZARD_API_ERROR'); + expect(authErrors()).toHaveLength(1); + // The auth screen is reported, and the decided failure goes back to the + // caller, which owns the exit. + expect(result.error).toBeUndefined(); + expect(result.failure?.message).toBe('Authentication failed (401)'); + expect(result.failure?.code).toBeDefined(); }); it('judges a failed resumed session on its own error, not the old 401', async () => { @@ -874,7 +874,7 @@ describe('gateway re-mint on 401', () => { expect(result.error).toBe('WIZARD_API_ERROR'); expect(result.message).toContain('500'); expect(result.message).not.toContain('401'); - expect(mockUIInstance.showAuthError).not.toHaveBeenCalled(); + expect(authErrors()).toHaveLength(0); }); it('does not re-mint when a fresh bearer is rejected', async () => { @@ -888,9 +888,8 @@ describe('gateway re-mint on 401', () => { // A fresh token the gateway rejects is a bad credential, not age. expect(refresh).not.toHaveBeenCalled(); expect(mockQuery).toHaveBeenCalledTimes(1); - expect(mockUIInstance.showAuthError).toHaveBeenCalledTimes(1); - expect(wizardAbort).toHaveBeenCalledTimes(1); - expect(result.error).toBe('WIZARD_API_ERROR'); + expect(authErrors()).toHaveLength(1); + expect(result.failure?.message).toBe('Authentication failed (401)'); }); }); diff --git a/src/lib/agent/__tests__/agent-prompt.test.ts b/src/lib/agent/__tests__/agent-prompt.test.ts index 2ee6bfdb9..39795283c 100644 --- a/src/lib/agent/__tests__/agent-prompt.test.ts +++ b/src/lib/agent/__tests__/agent-prompt.test.ts @@ -1,5 +1,5 @@ import { assemblePrompt, type PromptContext } from '@lib/agent/agent-prompt'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { HostResolution } from '@lib/host-resolution'; function makeRunDef(overrides: Partial = {}): ProgramRun { diff --git a/src/lib/agent/__tests__/run-agent-standalone.test.ts b/src/lib/agent/__tests__/run-agent-standalone.test.ts new file mode 100644 index 000000000..2f56ffdf3 --- /dev/null +++ b/src/lib/agent/__tests__/run-agent-standalone.test.ts @@ -0,0 +1,389 @@ +/** + * `runAgent(config, input, options)` runs with no UI, no store, no session and + * no program registry: a fake harness stands in for the SDK, and everything + * the run reports arrives through `onProgress` or comes back in the result. + * + * The `@ui` mock below throws on use. It is never reached — that is the + * assertion the whole file rests on. + */ +import * as fs from 'fs'; +import * as os from 'os'; +import * as path from 'path'; +import { Harness, Sequence } from '@lib/constants'; +import { HostResolution } from '@lib/host-resolution'; +import { OutroKind, type PendingQuestion } from '@lib/wizard-session'; +import { AGENT_ERROR_CODE, ErrorCodes } from '@lib/errors'; +import { AgentErrorType } from '@lib/agent/signals'; +import type { AgentProgress } from '@lib/agent/progress'; +import type { + AgentHarness, + BackendRunInputs, +} from '@lib/agent/runner/harness/types'; + +vi.mock('@ui', () => ({ + getUI: () => { + throw new Error('the agent reached for getUI()'); + }, + setUI: () => { + throw new Error('the agent reached for setUI()'); + }, +})); +vi.mock('@utils/debug'); +vi.mock('@utils/analytics', () => ({ + analytics: { + build: 'test', + runId: 'run-1', + wizardCapture: vi.fn(), + capture: vi.fn(), + captureException: vi.fn(), + setTag: vi.fn(), + shutdown: vi.fn().mockResolvedValue(undefined), + }, +})); +vi.mock('@lib/gateway-session', async (importOriginal) => ({ + ...(await importOriginal()), + gatewayAuth: vi.fn().mockResolvedValue({ + gatewayUrl: 'https://gateway.test', + token: 'phe_run', + teamId: 1, + refreshAtMs: Date.now() + 3_600_000, + }), +})); + +// The fake harness: reports a little of everything, then returns what the +// current test told it to. +const harnessState = vi.hoisted(() => ({ + result: {} as { error?: string; message?: string; failure?: unknown }, + throws: undefined as Error | undefined, + lastInputs: undefined as unknown, + askQuestions: undefined as PendingQuestion['questions'] | undefined, +})); +vi.mock('@lib/agent/runner/switchboard/harness', () => { + const fake: AgentHarness = { + name: Harness.pi, + async run(inputs: BackendRunInputs) { + harnessState.lastInputs = inputs; + const { emit, spinner } = inputs; + emit({ kind: 'log', level: 'step', message: 'Initializing agent' }); + spinner.start('Working'); + emit({ kind: 'status', message: 'Installing the SDK' }); + emit({ + kind: 'tasks', + tasks: [{ content: 'Install', status: 'completed' }], + }); + emit({ kind: 'url', which: 'dashboard', url: 'https://d/1' }); + emit({ + kind: 'usage', + delta: { + inputTokens: 10, + outputTokens: 5, + cacheReadTokens: 0, + cacheCreationTokens: 0, + cacheCreation5m: 0, + cacheCreation1h: 0, + }, + }); + if (harnessState.askQuestions && inputs.askBridge) { + const { answers } = await inputs.askBridge.request({ + questions: harnessState.askQuestions, + }); + emit({ + kind: 'status', + message: `answered:${JSON.stringify(answers)}`, + }); + } + if (harnessState.throws) throw harnessState.throws; + spinner.stop('Done'); + return harnessState.result as never; + }, + }; + return { + HARNESS_OPTIONS: { [Harness.pi]: fake }, + getHarness: () => fake, + resolveHarness: () => ({ harness: Harness.pi, model: 'm' }), + }; +}); + +import { runAgent } from '@lib/agent/runner'; +import type { RunConfig, RunInput } from '@lib/agent/runner'; +import { analytics } from '@utils/analytics'; + +let tmp: string; + +const config = (over: Partial = {}): RunConfig => ({ + programId: 'test-program', + run: { + integrationLabel: 'test-integration', + spinnerMessage: 'Working...', + successMessage: 'Done!', + estimatedDurationMinutes: 1, + reportFile: 'report.md', + docsUrl: 'https://docs.test', + abortCases: [ + { + match: /no stripe/i, + message: 'No Stripe here', + body: 'Stripe is required.', + errorCode: ErrorCodes.AgentAbort, + }, + ], + }, + composed: false, + binding: { sequence: Sequence.linear, harness: Harness.pi, model: 'm' }, + switchboard: { program: 'test-program', flags: {} }, + skillsBaseUrl: 'https://skills.test', + wizardFlags: {}, + wizardFlagPayloads: {}, + wizardMetadata: {}, + ...over, +}); + +const input = (over: Partial = {}): RunInput => ({ + installDir: tmp, + credentials: { + accessToken: 'tok', + projectApiKey: 'phc_test', + host: HostResolution.fromApiHost('https://us.posthog.com'), + projectId: 1, + }, + project: null, + apiUser: null, + skillId: 'test-integration', + flags: { + ci: false, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + host: {}, + ...over, +}); + +beforeEach(() => { + tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'run-agent-standalone-')); + harnessState.result = {}; + harnessState.throws = undefined; + harnessState.lastInputs = undefined; + harnessState.askQuestions = undefined; + vi.mocked(analytics.shutdown).mockClear(); +}); + +afterEach(() => fs.rmSync(tmp, { recursive: true, force: true })); + +describe('runAgent standalone', () => { + it('runs to success with an observer and an answerer', async () => { + const events: AgentProgress[] = []; + const ask = vi.fn(() => Promise.resolve({ q1: 'yes' })); + harnessState.askQuestions = [ + { id: 'q1', prompt: 'Continue?', kind: 'single' as const }, + ]; + + const result = await runAgent(config(), input(), { + onProgress: (e) => events.push(e), + interaction: { ask }, + }); + + expect(result.outcome).toBe('success'); + expect(result.skillId).toBe('test-integration'); + expect(result.outro).toEqual({ + kind: OutroKind.Success, + message: 'Done!', + reportFile: 'report.md', + docsUrl: 'https://docs.test', + continueUrl: undefined, + }); + expect(result.failure).toBeUndefined(); + + // The answerer saw the question the bridge built, and its answer came back. + expect(ask).toHaveBeenCalledTimes(1); + const [question] = ask.mock.calls[0] as unknown as [PendingQuestion]; + expect(question.questions[0].id).toBe('q1'); + expect(question.source).toBe('test-integration'); + expect(events).toContainEqual({ + kind: 'status', + message: 'answered:{"q1":"yes"}', + }); + + // Emission order: started first, completion then completed last. + const kinds = events.map( + (e) => `${e.kind}${'phase' in e ? `:${e.phase}` : ''}`, + ); + expect(kinds[0]).toBe('lifecycle:started'); + expect(kinds.slice(-2)).toEqual(['completion', 'lifecycle:completed']); + + // The agent's own snapshot, independent of the observer. + expect(result.snapshot.dashboardUrl).toBe('https://d/1'); + expect(result.snapshot.tasks).toEqual([ + { content: 'Install', status: 'completed' }, + ]); + expect(result.snapshot.statusMessages[0]).toBe('Installing the SDK'); + expect(result.snapshot.usage).toEqual({ + inputTokens: 10, + outputTokens: 5, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }); + expect(analytics.shutdown).toHaveBeenCalledWith('success'); + }); + + it('runs to a complete result with no options at all', async () => { + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('success'); + expect(result.outro?.kind).toBe(OutroKind.Success); + expect(result.snapshot.dashboardUrl).toBe('https://d/1'); + // No answerer: no bridge is installed, so `wizard_ask` reports unavailable + // rather than hanging. + const inputs = harnessState.lastInputs as BackendRunInputs; + expect(inputs.askBridge).toBeUndefined(); + expect(inputs.getPendingQuestion).toBeUndefined(); + }); + + it('installs no bridge in CI even with an answerer, as before', async () => { + await runAgent( + config(), + input({ + flags: { + ci: true, + signup: false, + debug: false, + e2eAsk: false, + localMcp: false, + captureAio: false, + benchmark: false, + yaraReport: false, + }, + }), + { interaction: { ask: () => Promise.resolve({}) } }, + ); + const inputs = harnessState.lastInputs as BackendRunInputs; + expect(inputs.askBridge).toBeUndefined(); + }); + + it('returns an agent abort as a decided failure with the matched case', async () => { + harnessState.result = { + error: AgentErrorType.ABORT, + message: 'No Stripe found', + }; + const events: AgentProgress[] = []; + + const result = await runAgent(config(), input(), { + onProgress: (e) => events.push(e), + }); + + expect(result.outcome).toBe('aborted'); + expect(result.failure?.code).toBe(ErrorCodes.AgentAbort); + expect(result.failure?.outroData).toMatchObject({ + kind: OutroKind.Error, + message: 'No Stripe here', + body: 'Stripe is required.', + errorDetail: { reason: 'No Stripe found' }, + }); + expect(result.failure?.error?.message).toBe( + 'Agent aborted: No Stripe found', + ); + expect(events.some((e) => e.kind === 'completion')).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('returns a coded failure for a harness error', async () => { + harnessState.result = { error: AgentErrorType.NO_PROGRESS }; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('failed'); + expect(result.failure?.code).toBe( + AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], + ); + expect(result.failure?.message).toContain('without changing your project'); + }); + + it('passes a harness-decided failure through untouched', async () => { + const failure = { code: ErrorCodes.AgentAbort, message: 'decided' }; + harnessState.result = { failure }; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('failed'); + expect(result.failure).toBe(failure); + }); + + it('skips the terminal outro and the shutdown for a composed sub-run', async () => { + const events: AgentProgress[] = []; + + const result = await runAgent(config({ composed: true }), input(), { + onProgress: (e) => events.push(e), + }); + + expect(result.outcome).toBe('success'); + expect(result.outro).toBeUndefined(); + expect(events.some((e) => e.kind === 'completion')).toBe(false); + expect( + events.some((e) => e.kind === 'lifecycle' && e.phase === 'completed'), + ).toBe(false); + expect(analytics.shutdown).not.toHaveBeenCalled(); + }); + + it('finishes when the observer throws on every event', async () => { + const result = await runAgent(config(), input(), { + onProgress: () => { + throw new Error('projection broke'); + }, + }); + + expect(result.outcome).toBe('success'); + expect(result.snapshot.tasks).toHaveLength(1); + }); + + it('calls the bound hooks with the run credentials', async () => { + const postRun = vi.fn().mockResolvedValue(undefined); + const buildOutroData = vi.fn(() => ({ + kind: OutroKind.Success, + message: 'custom', + })); + + const result = await runAgent( + config({ hooks: { postRun, buildOutroData } }), + input(), + ); + + expect(postRun).toHaveBeenCalledWith( + expect.objectContaining({ projectApiKey: 'phc_test' }), + ); + expect(buildOutroData).toHaveBeenCalledTimes(1); + expect(result.outro).toEqual({ + kind: OutroKind.Success, + message: 'custom', + }); + }); + + it('returns a crash as a result instead of rejecting', async () => { + const boom = new Error('SDK exploded'); + harnessState.throws = boom; + + const result = await runAgent(config(), input()); + + expect(result.outcome).toBe('crashed'); + expect(result.failure?.error).toBe(boom); + expect(result.failure?.message).toBe('SDK exploded'); + expect(result.failure?.code).toBe(ErrorCodes.InternalUnhandled); + // What was reported before the crash survives in the snapshot. + expect(result.snapshot.statusMessages).toContain('Installing the SDK'); + }); + + it('is cancelled by a signal that is already aborted', async () => { + const controller = new AbortController(); + controller.abort(); + + const result = await runAgent(config(), input(), { + signal: controller.signal, + }); + + expect(result.outcome).toBe('cancelled'); + expect(harnessState.lastInputs).toBeUndefined(); + }); +}); diff --git a/src/lib/agent/agent-interface.ts b/src/lib/agent/agent-interface.ts index 815498a25..234a85360 100644 --- a/src/lib/agent/agent-interface.ts +++ b/src/lib/agent/agent-interface.ts @@ -6,8 +6,11 @@ import path from 'path'; import * as os from 'os'; import { createRequire } from 'node:module'; -import { getUI, type SpinnerHandle } from '@ui'; -import type { TokenUsageDelta } from '@ui/wizard-ui'; +import type { + ProgressEmitter, + SpinnerHandle, + TokenUsageDelta, +} from './progress'; import { debug, logToFile, initLogFile, getLogFilePath } from '@utils/debug'; import type { WizardRunOptions } from '@utils/types'; import { analytics } from '@utils/analytics'; @@ -27,7 +30,8 @@ import { type AdditionalFeature, ADDITIONAL_FEATURE_PROMPTS, } from '@lib/wizard-session'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; +import type { AgentFailure } from './runner/shared/types'; import { createCustomHeaders } from '@utils/custom-headers'; import type { HostResolution } from '@lib/host-resolution'; import { @@ -243,6 +247,12 @@ export type AgentConfig = { * `--capture-aio` is off. Constructed once per run by the harness. */ capture?: AioCapture; + /** + * Where the run reports: log lines, tasks, status, URLs, usage, the handoff + * document and the auth-error detail. Absent → the run reports nowhere and + * still completes. + */ + emit?: ProgressEmitter; }; /** @@ -354,8 +364,12 @@ type AgentRunConfig = { program?: string; /** Resolved sequence, for the sequence-axis commandments. */ sequence: Sequence; + /** Where the run reports. A no-op when the caller passed none. */ + emit?: ProgressEmitter; }; +const NO_PROGRESS: ProgressEmitter = () => undefined; + /** * Global identifiers attached to every LLM gateway trace for a run. They ride on * each `$ai_generation` the gateway emits (in the `X-PostHog-Properties` blob @@ -531,6 +545,7 @@ export async function initializeAgent( initLogFile(); logToFile('Agent initialization starting'); logToFile('Install directory:', options.installDir); + const emit = config.emit ?? NO_PROGRESS; try { // Configure model routing (inherited by the SDK subprocess). All model @@ -636,6 +651,7 @@ export async function initializeAgent( askMaxQuestions: config.askMaxQuestions, orchestrator: config.orchestrator, triageProvider, + onHandoffText: (text) => emit({ kind: 'handoff', text }), }); mcpServers['wizard-tools'] = wizardToolsServer; @@ -661,6 +677,7 @@ export async function initializeAgent( program: config.integrationLabel, // A queue context is present only on a task run; that is the sequence. sequence: config.orchestrator ? Sequence.orchestrator : Sequence.linear, + emit, }; logToFile('Agent config:', { @@ -689,9 +706,11 @@ export async function initializeAgent( return agentRunConfig; } catch (error) { - getUI().log.error( - `Failed to initialize agent: ${(error as Error).message}`, - ); + emit({ + kind: 'log', + level: 'error', + message: `Failed to initialize agent: ${(error as Error).message}`, + }); logToFile('Agent initialization error:', error); debug('Agent initialization error:', error); throw error; @@ -740,7 +759,12 @@ export async function runAgent( onMessage(message: any): void; finalize(resultMessage: any, totalDurationMs: number): any; }, -): Promise<{ error?: AgentErrorType; message?: string }> { +): Promise<{ + error?: AgentErrorType; + message?: string; + failure?: AgentFailure; +}> { + const emit = agentConfig.emit ?? NO_PROGRESS; const { spinnerMessage = 'Customizing your PostHog setup...', successMessage = 'PostHog integration complete', @@ -854,7 +878,7 @@ export async function runAgent( // CostTrackerPlugin.onFinalize uses to correct per-turn drift. const totalCostUsd = Number(lastResultMessage?.total_cost_usd ?? 0); if (totalCostUsd > 0) { - getUI().setFinalTokenCostUsd(totalCostUsd); + emit({ kind: 'finalCost', usd: totalCostUsd }); } try { middleware?.finalize(lastResultMessage, durationMs); @@ -879,6 +903,9 @@ export async function runAgent( let sessionId: string | undefined; let reminted = false; let remintRequested = false; + // A 401 on a fresh bearer: the auth screen was reported, and this is the + // failure the caller ends the run with. The query is aborted to unwind. + let authFailure: AgentFailure | undefined; const agentConfigDir = createIsolatedAgentConfigDir(); try { @@ -1162,6 +1189,7 @@ export async function runAgent( options, spinner, signals, + emit, receivedSuccessResult, tasks, agentConfig.suppressTaskRender ?? false, @@ -1240,15 +1268,20 @@ export async function runAgent( ...authError, sessionExpired, }); - getUI().showAuthError({ - hasSettingsConflict: authError.hasSettingsConflict, - conflicts: authError.conflicts, - usingManagedLogin: authError.usingManagedLogin, - credentialPlaces: authError.credentialPlaces, - sessionExpired, - logFilePath: getLogFilePath(), + emit({ + kind: 'authError', + detail: { + hasSettingsConflict: authError.hasSettingsConflict, + conflicts: authError.conflicts, + usingManagedLogin: authError.usingManagedLogin, + credentialPlaces: authError.credentialPlaces, + sessionExpired, + logFilePath: getLogFilePath(), + }, }); - await wizardAbort({ + // The caller ends the run with this; the query is abandoned here + // where the process used to exit. + authFailure = { code: authCode, message: 'Authentication failed (401)', error: new WizardError( @@ -1264,7 +1297,9 @@ export async function runAgent( }, authCode, ), - }); + }; + abortController.abort(); + break; } try { @@ -1290,6 +1325,7 @@ export async function runAgent( } catch (error) { // The abort we asked for; anything else belongs to the outer catch. if (remintRequested) return 'remint'; + if (authFailure) return 'done'; throw error; } return remintRequested ? 'remint' : 'done'; @@ -1321,6 +1357,12 @@ export async function runAgent( await runQuery(sessionId); } + // A fresh bearer was rejected. The auth screen is already up; hand the + // decided failure to the caller, which owns the exit. + if (authFailure) { + return { failure: authFailure }; + } + // A YARA hook detected a terminal violation and aborted the run. if (yaraViolationReason) { logToFile('Agent error: YARA_VIOLATION'); @@ -1421,14 +1463,19 @@ export async function runAgent( // No API error found, re-throw the original exception spinner.stop(errorMessage); - getUI().log.error(`Error: ${(error as Error).message}`); + emit({ + kind: 'log', + level: 'error', + message: `Error: ${(error as Error).message}`, + }); logToFile('Agent run failed:', error); debug('Full error:', error); throw error; } finally { // Always capture run duration, even on abort/error, so we can alert on - // long runs where the user gave up before completion. - if (!receivedSuccessResult) { + // long runs where the user gave up before completion. A 401 never reached + // this block before (the process exited first), so it still does not count. + if (!receivedSuccessResult && !authFailure) { const durationMs = Date.now() - startTime; analytics.wizardCapture('agent aborted', { duration_ms: durationMs, @@ -1737,6 +1784,7 @@ function handleSDKMessage( options: WizardRunOptions, spinner: SpinnerHandle, signals: AgentOutputSignals, + emit: ProgressEmitter, receivedSuccessResult = false, tasks?: Map, // The orchestrator owns the TUI task panel (it renders its queue). Suppress the @@ -1762,7 +1810,7 @@ function handleSDKMessage( const sorted = Array.from(tasks.values()).sort( (a, b) => rank(a.status) - rank(b.status), ); - getUI().syncTodos(sorted); + emit({ kind: 'tasks', tasks: sorted.map((t) => ({ ...t })) }); }; logToFile(`SDK Message: ${message.type}`, JSON.stringify(message, null, 2)); @@ -1777,7 +1825,7 @@ function handleSDKMessage( // dedup for SDK-retried turns — see addTokenUsage's doc comment), so // this stays live-updating for every run, not just `--benchmark`. const tokenUsageDelta = extractTokenUsageDelta(message); - if (tokenUsageDelta) getUI().addTokenUsage(tokenUsageDelta); + if (tokenUsageDelta) emit({ kind: 'usage', delta: tokenUsageDelta }); // Extract text content from assistant messages const content = message.message?.content; @@ -1797,7 +1845,7 @@ function handleSDKMessage( const statusMatch = block.text.match(statusRegex); if (statusMatch) { const statusText = statusMatch[1].trim(); - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } @@ -1811,7 +1859,11 @@ function handleSDKMessage( ); const dashboardMatch = block.text.match(dashboardRegex); if (dashboardMatch) { - getUI().setDashboardUrl(dashboardMatch[1].trim()); + emit({ + kind: 'url', + which: 'dashboard', + url: dashboardMatch[1].trim(), + }); } // Check for [NOTEBOOK_URL] markers @@ -1824,7 +1876,11 @@ function handleSDKMessage( ); const notebookMatch = block.text.match(notebookRegex); if (notebookMatch) { - getUI().setNotebookUrl(notebookMatch[1].trim()); + emit({ + kind: 'url', + which: 'notebook', + url: notebookMatch[1].trim(), + }); } } @@ -1847,7 +1903,7 @@ function handleSDKMessage( // Mirror the active tool into the Visualizer's "stage" indicator. if (block.type === 'tool_use') { const stage = classifyToolToStage((block as ToolUseBlock).name); - if (stage) getUI().setStage(stage); + if (stage) emit({ kind: 'stage', stage }); } } } @@ -1900,7 +1956,7 @@ function handleSDKMessage( // mode race conditions). Full message already logged above via JSON dump. if (message.errors && !receivedSuccessResult) { for (const err of message.errors) { - getUI().log.error(`Error: ${err}`); + emit({ kind: 'log', level: 'error', message: `Error: ${err}` }); logToFile('ERROR:', err); } } @@ -1915,7 +1971,7 @@ function handleSDKMessage( // Full message already logged above via JSON dump. if (message.errors && !receivedSuccessResult) { for (const err of message.errors) { - getUI().log.error(`Error: ${err}`); + emit({ kind: 'log', level: 'error', message: `Error: ${err}` }); logToFile('ERROR:', err); } } diff --git a/src/lib/agent/agent-prompt.ts b/src/lib/agent/agent-prompt.ts index 0427fc4f5..4987adcfb 100644 --- a/src/lib/agent/agent-prompt.ts +++ b/src/lib/agent/agent-prompt.ts @@ -7,7 +7,7 @@ * 3. Skill prompt — "follow SKILL.md" instructions (if a skill was installed) */ -import type { ProgramRun } from './agent-runner.js'; +import type { AgentRunDefinition } from './runner/shared/types'; import type { HostResolution } from '@lib/host-resolution'; /** @@ -62,7 +62,10 @@ Important: You must read a file immediately before attempting to write it, even /** * Assemble the final agent prompt from the program's run config. */ -export function assemblePrompt(runDef: ProgramRun, ctx: PromptContext): string { +export function assemblePrompt( + runDef: AgentRunDefinition, + ctx: PromptContext, +): string { const parts: string[] = []; // Always include the default project prompt diff --git a/src/lib/agent/agent-runner.ts b/src/lib/agent/agent-runner.ts index 8b92c4ebe..5afc4fbe7 100644 --- a/src/lib/agent/agent-runner.ts +++ b/src/lib/agent/agent-runner.ts @@ -1,15 +1,25 @@ /** - * Re-export shim. The runner has been split into agent/runner/. - * Import from there directly; this shim keeps existing importers working. + * Re-export shim for the agent runner. Import from `./runner/index` directly; + * this shim keeps existing importers of the agent's types working. + * + * The session-driven `runAgent(programConfig, session)` every runner and + * program used to import from here now lives in + * `src/lib/programs/run-agent-legacy.ts`. */ export { runAgent, - runProgram, shouldDisableAsk, - type ProgramRun, - type BootstrapResult, type AbortCase, - type PromptContext, + type AgentFailure, + type AgentInteraction, + type AgentProgress, + type AgentRunDefinition, + type BootstrapResult, type Credentials, + type PromptContext, + type RunAgentOptions, + type RunConfig, + type RunInput, + type RunResult, } from './runner/index'; diff --git a/src/lib/agent/mcp-prompt-streaming.ts b/src/lib/agent/mcp-prompt-streaming.ts index 1b9e55182..08b32231a 100644 --- a/src/lib/agent/mcp-prompt-streaming.ts +++ b/src/lib/agent/mcp-prompt-streaming.ts @@ -12,7 +12,22 @@ * `for await (...)` and render as they arrive. */ -import type { AgentChunk } from '@ui/tui/services/mcp-suggested-prompts-services'; +/** + * Discriminated union covering every kind of streamed event a host renders. + * Production yields these from Claude SDK messages; the TUI playground yields + * them from canned scripts. + */ +export type AgentChunk = + | { kind: 'text'; text: string } + /** `command` carries CLI mode's exec command string (`call …`) so the + * screen can recover the inner tool for context-aware follow-ups. */ + | { kind: 'tool-call'; toolName: string; detail: string; command?: string } + | { kind: 'tool-result'; toolName: string; detail: string } + | { kind: 'error'; text: string } + /** Stream completed. `sessionId` is the SDK session ID of the just- + * completed turn; pass it back as `resumeSessionId` on a follow-up + * call to continue the conversation with full history. */ + | { kind: 'done'; sessionId?: string }; import type { Credentials } from '@lib/wizard-session'; import { DEFAULT_AGENT_MODEL, WIZARD_USER_AGENT } from '@lib/constants'; import { logToFile } from '@utils/debug'; diff --git a/src/lib/agent/progress.ts b/src/lib/agent/progress.ts new file mode 100644 index 000000000..cec2ed632 --- /dev/null +++ b/src/lib/agent/progress.ts @@ -0,0 +1,152 @@ +/** + * The agent's public progress and interaction contracts. + * + * `runAgent` reports through one optional callback and asks through one + * optional set of capabilities. Neither reaches into a UI singleton, a store, + * or a session: every payload is copied data, every question is awaited on an + * injected answerer. The legacy adapter in `src/lib/programs/run-agent-legacy.ts` + * maps these back onto `WizardUI` one call per event, so the terminal output of + * every existing runner is unchanged. + */ + +import type { SettingsConflict } from './claude-settings'; +import type { + AskAnswers, + OutroData, + PendingQuestion, + TaskNotice, +} from '@lib/wizard-session'; + +/** + * One assistant turn's token usage, for the hidden Ctrl+T token/cost HUD. + * `model` is the model that produced *this* turn (e.g. the SDK's + * `message.message.model`) — a subagent can run on a different model than + * the main session, and some programs override to Haiku, so pricing must key + * off the per-turn model rather than a single run-wide assumption. Omit only + * when the caller genuinely has no model context (falls back to Sonnet + * pricing — see `pricePerMtokForModel` in `@lib/agent/token-pricing`). + */ +export interface TokenUsageDelta { + inputTokens: number; + outputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; + cacheCreation5m: number; + cacheCreation1h: number; + model?: string; +} + +/** The run spinner a harness drives. In the agent it is a thin emitter. */ +export interface SpinnerHandle { + start(message?: string): void; + stop(message?: string): void; + message(msg?: string): void; +} + +/** + * Context for an auth-error report so the host can pick the right copy. + * + * `hasSettingsConflict` is true when a Claude Code settings file (project, + * project-local, the user's global config, or managed) actually overrides the + * LLM Gateway auth. `conflicts` carries the exact files and keys so the screen + * can name them. When there is no conflict, the 401 has a different cause (bad + * PAT prefix, missing scope, expired key, region mismatch) and we should not + * advise the user to log out of Claude Code. + */ +export interface AuthErrorDetail { + hasSettingsConflict: boolean; + conflicts?: SettingsConflict[]; + /** + * True when the agent SDK authenticated from a stored Claude login + * (`apiKeySource: "/login managed key"`) instead of the wizard's gateway + * token — conflicting Anthropic credentials. Takes priority in the screen. + */ + usingManagedLogin?: boolean; + /** Human-readable places a conflicting Anthropic credential may live. */ + credentialPlaces?: string[]; + /** + * True when a pre-run refresh already failed on a dead grant. The login is + * gone and re-running is the only fix, so this outranks every other branch — + * none of the usual advice (key type, scopes, region) applies. + */ + sessionExpired?: boolean; + logFilePath: string; +} + +/** One task as the host renders it. The same shape `WizardUI.syncTodos` takes. */ +export interface TaskSnapshot { + content: string; + status: string; + activeForm?: string; +} + +export type ProgressLogLevel = 'info' | 'warn' | 'error' | 'success' | 'step'; + +/** + * Everything the agent reports while it runs. One event per former + * `getUI()` call, in the same order, with the same payload, so a reducer that + * maps each case back onto `WizardUI` reproduces today's output exactly. + * + * Payloads are copies. Never a store, a setter, a function or a live + * collection. The callback returns nothing and the agent never branches on it. + */ +export type AgentProgress = + /** The run's main work has started (`WizardUI.startRun`). */ + | { kind: 'lifecycle'; phase: 'started' } + /** The run finished and the host may show its outro (`WizardUI.outro`). */ + | { kind: 'lifecycle'; phase: 'completed'; message: string } + /** The run spinner (`WizardUI.spinner()`), one handle per run. */ + | { + kind: 'spinner'; + action: 'start' | 'stop' | 'message'; + message?: string; + } + /** A log line (`WizardUI.log[level]`). */ + | { kind: 'log'; level: ProgressLogLevel; message: string } + /** A `[STATUS]` line the agent printed (`WizardUI.pushStatus`). */ + | { kind: 'status'; message: string } + /** The full task list, already sorted for display (`WizardUI.syncTodos`). */ + | { kind: 'tasks'; tasks: TaskSnapshot[] } + /** The stage of work derived from the active tool (`WizardUI.setStage`). */ + | { kind: 'stage'; stage: string } + /** A PostHog URL the agent created (`setDashboardUrl` / `setNotebookUrl`). */ + | { kind: 'url'; which: 'dashboard' | 'notebook'; url: string } + /** One assistant turn's token usage (`WizardUI.addTokenUsage`). */ + | { kind: 'usage'; delta: TokenUsageDelta } + /** The SDK's authoritative run cost (`WizardUI.setFinalTokenCostUsd`). */ + | { kind: 'finalCost'; usd: number } + /** The handoff document from `publish_handoff` (`WizardUI.setHandoffText`). */ + | { kind: 'handoff'; text: string } + /** The gateway returned 401; a failure follows (`WizardUI.showAuthError`). */ + | { kind: 'authError'; detail: AuthErrorDetail } + /** The run's final outro payload (`WizardUI.setOutroData`). */ + | { kind: 'completion'; outro: OutroData }; + +export type ProgressEmitter = (event: AgentProgress) => void; + +/** + * The questions the agent may need a person (or a script) to answer. Every + * capability is optional. With none supplied the agent installs no ask bridge, + * so `wizard_ask` returns its existing "not available" error, and an optional + * task notice is declined — the same path a `--ci` run takes today. + */ +export interface AgentInteraction { + /** + * Open a question and resolve with the answers. The bridge that calls this + * owns the timeout, the `__cancelled__` sentinel and the analytics; `signal` + * aborts when the run is cancelled so the host can dismiss its overlay. + */ + ask?: ( + question: PendingQuestion, + context: { signal: AbortSignal }, + ) => Promise; + /** Dismiss the in-flight question as cancelled (timeouts call this). */ + cancelAsk?: () => void; + /** Offer an optional step and resolve with whether to keep it. */ + taskNotice?: ( + notice: TaskNotice, + context: { signal: AbortSignal }, + ) => Promise; + /** Dismiss an in-flight task notice as declined (timeouts call this). */ + cancelTaskNotice?: () => void; +} diff --git a/src/lib/agent/runner/README.md b/src/lib/agent/runner/README.md index 86fee7f1b..2a7a9a6fb 100644 --- a/src/lib/agent/runner/README.md +++ b/src/lib/agent/runner/README.md @@ -39,19 +39,44 @@ for the coordinated change checklist. Five layers, each with its own job. Nothing crosses layers unless it has to. -**The entry point** (`index.ts`) is the front door. It receives a program config -and a session, and orchestrates the run at the highest level. - -**Bootstrap** (`shared/bootstrap.ts`) is the on-ramp. Every run starts with the -same setup work — health checks, settings conflicts, OAuth, PostHog feature flag -fetch, MCP URL, AI opt-in gate. Whether the run turns out to be linear or -orchestrator, anthropic or pi, the setup is the same. +**The entry point** (`index.ts`) is the front door: + +```ts +runAgent(config: RunConfig, input: RunInput, options?: { + onProgress?: (event: AgentProgress) => void; + interaction?: AgentInteraction; + signal?: AbortSignal; +}): Promise +``` -**The switchboard** (`switchboard/`) is the router. Given a program id + the -fetched flags + any CLI overrides, it returns a `ProgramBinding` — which query +`RunConfig` is resolved execution data: the program id, its run definition, the +binding the caller resolved, flags, run tags and tool lists. `RunInput` is the +invocation snapshot: directory, credentials, project, skill. The agent reports +through `onProgress` (one event per former UI call, copied data, never awaited) +and asks through `interaction` (an optional answerer for `wizard_ask` and task +notices). It returns a `RunResult` — outcome, outro, a failure with the same +fields `wizardAbort` takes, and a snapshot of what it reported. It never +renders, never reads a session, never exits the process and never rejects: a +decided failure is `aborted` or `failed`, an error the agent did not decide is +`crashed` with the original error attached. Types live in `shared/types.ts` and +`../progress.ts`. + +Everything the caller decides first — health and settings gates, OAuth, the AI +opt-in gate, post-auth gates, feature flags, token refresh, the binding — lives +in `src/lib/programs/run-agent-legacy.ts`, which also maps progress back onto +`getUI()` for today's runners. + +**Prepare** (`shared/bootstrap.ts`) is the on-ramp inside the agent: logging +targets, the gateway mint and the scan-triage classifier. Whether the run turns +out to be linear or orchestrator, anthropic or pi, the setup is the same. + +**The switchboard** (`switchboard/`) is the router. Given a program id, its +declared binding (`src/lib/programs/bindings.ts`, or `DEFAULT_BINDING`), the +fetched flags and any CLI overrides, it returns a `ProgramBinding` — which query shape (sequence), which agent SDK (harness), which model. Two independent middleware chains, one per axis, apply precedence rules (CLI > flag > program -config > default). This is the only layer that makes routing decisions. +config > default). The caller resolves the run-level binding; the orchestrator +re-resolves the harness per task role from the same inputs. **Sequences** (`sequence/`) are LLM query shapes. Once the switchboard has picked one, that sequence takes over the run and owns _how the LLM's work is @@ -72,8 +97,7 @@ gateway. ## How they connect -- Bootstrap prepares shared services, including triage selected for the resolved - harness. +- Prepare mints the gateway token and builds triage for the resolved harness. - The switchboard knows which sequences and harnesses exist (via its two registries), but not what they do. - A sequence knows how to shape a conversation, but delegates the actual model @@ -84,12 +108,13 @@ Each layer is replaceable. ## Flow -1. Program picked → session built. -2. Bootstrap runs (shared setup, fetches PostHog flags). -3. Switchboard resolves a `ProgramBinding { sequence, harness, model }`. -4. Analytics tags the run with its bindings. -5. Sequence takes over — shapes the LLM's work into one conversation (linear) or - many (orchestrator). -6. Harness drives each conversation through its SDK, using the bound model, on +1. The caller runs its gates, authenticates, fetches PostHog flags and resolves + a `ProgramBinding { sequence, harness, model }`; analytics tags the run. +2. `runAgent(config, input, options)` prepares (mint, triage). +3. Sequence takes over — shapes the LLM's work into one conversation (linear) or + many (orchestrator), reporting through `onProgress`. +4. Harness drives each conversation through its SDK, using the bound model, on the PostHog LLM gateway. -7. Cleanup runs (scan report, settings restore, outro). +5. The scan report flushes; `runAgent` returns a `RunResult`. +6. The caller applies it: an outro is already reported, a decided failure goes + to `wizardAbort`, a crash is rethrown for the runner's own handling. diff --git a/src/lib/agent/runner/__tests__/switchboard.test.ts b/src/lib/agent/runner/__tests__/switchboard.test.ts index 5386fc312..822f7ba2b 100644 --- a/src/lib/agent/runner/__tests__/switchboard.test.ts +++ b/src/lib/agent/runner/__tests__/switchboard.test.ts @@ -23,11 +23,11 @@ import { WIZARD_ORCHESTRATOR_FLAG_KEY, } from '@lib/constants'; import { - PROGRAM_BINDINGS, DEFAULT_BINDING, resolveBinding, type SwitchboardCtx, } from '@lib/agent/runner/switchboard'; +import { PROGRAM_BINDINGS, bindingFor } from '@lib/programs/bindings'; import { modelCapabilities, MINT_ALLOWED_EFFORTS, @@ -69,7 +69,9 @@ describe('switchboard PROGRAM_BINDINGS', () => { if (program === 'metrics') continue; // pinned below if (program === 'replay-vision') continue; // pinned below if (program === 'error-tracking') continue; // pinned below - expect(resolveBinding({ program, flags: {} })).toEqual(DEFAULT_RESOLVED); + expect( + resolveBinding({ program, flags: {}, binding: bindingFor(program) }), + ).toEqual(DEFAULT_RESOLVED); } }); @@ -229,6 +231,7 @@ describe('switchboard composed clamp', () => { for (const program of PROGRAM_IDS) { const ctx: SwitchboardCtx = { program, + binding: bindingFor(program), composed: true, flags: { [WIZARD_ORCHESTRATOR_FLAG_KEY]: 'true' }, trace: {}, diff --git a/src/lib/agent/runner/harness/anthropic/index.ts b/src/lib/agent/runner/harness/anthropic/index.ts index 4f5e449ec..d760e6d60 100644 --- a/src/lib/agent/runner/harness/anthropic/index.ts +++ b/src/lib/agent/runner/harness/anthropic/index.ts @@ -1,6 +1,5 @@ // Supported legacy SDK fallback; both this adapter and Pi implement run and runTask. -import { getUI } from '@ui'; import { Harness } from '@lib/constants'; import { initializeAgent, @@ -9,7 +8,8 @@ import { import { createAioCapture } from '@lib/agent/aio-capture'; import { getLogFilePath, logToFile } from '@utils/debug'; import { detectNodePackageManagers } from '@lib/detection/package-manager'; -import { sessionToOptions } from '@lib/agent/runner/shared/bootstrap'; +import { runOptions } from '@lib/agent/runner/shared/bootstrap'; +import { createEmitLog } from '@lib/agent/runner/shared/progress-collector'; import type { AgentResult, AgentHarness, @@ -22,30 +22,33 @@ export const anthropicBackend: AgentHarness = { async run(inputs: BackendRunInputs): Promise { const { - session, - config, - programConfig, + config: runConfig, + input, boot, + emit, prompt, spinner, askBridge, + getPendingQuestion, middleware, model, } = inputs; + const config = runConfig.run; const { skillsBaseUrl, credentials, wizardFlags, wizardMetadata } = boot; const { accessToken, host, projectApiKey } = credentials; + const log = createEmitLog(emit); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey, apiHost: host.apiHost, runTags: wizardMetadata, }); - getUI().log.step('Initializing Claude agent...'); + log.step('Initializing Claude agent...'); const agent = await initializeAgent( { - workingDirectory: session.installDir, + workingDirectory: input.installDir, posthogMcpUrl: host.mcpUrl, posthogApiKey: accessToken, host, @@ -59,22 +62,23 @@ export const anthropicBackend: AgentHarness = { integrationLabel: config.integrationLabel, askBridge, askMaxQuestions: config.maxQuestions, - allowedTools: programConfig.allowedTools, - disallowedTools: programConfig.disallowedTools, - getPendingQuestion: () => session.pendingQuestion, + allowedTools: runConfig.allowedTools, + disallowedTools: runConfig.disallowedTools, + getPendingQuestion, modelOverride: model, capture, + emit, }, - sessionToOptions(session), + runOptions(input), ); - getUI().log.step(`Verbose logs: ${getLogFilePath()}`); - getUI().log.success("Agent initialized. Let's get cooking!"); + log.step(`Verbose logs: ${getLogFilePath()}`); + log.success("Agent initialized. Let's get cooking!"); logToFile('[agent-runner] agent initialized'); return executeAgent( agent, prompt, - sessionToOptions(session), + runOptions(input), spinner, { estimatedDurationMinutes: config.estimatedDurationMinutes, @@ -94,9 +98,10 @@ export const anthropicBackend: AgentHarness = { async runTask(inputs: TaskRunInputs): Promise { const { - session, - programConfig, + config, + input, boot, + emit, prompt, spinner, model, @@ -111,10 +116,10 @@ export const anthropicBackend: AgentHarness = { requestRemark, analyticsProperties, } = inputs; - const options = sessionToOptions(session); + const options = runOptions(input); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, @@ -125,7 +130,7 @@ export const anthropicBackend: AgentHarness = { // enqueue_task attribute to the right agent when tasks run in parallel. const agent = await initializeAgent( { - workingDirectory: session.installDir, + workingDirectory: input.installDir, posthogMcpUrl: boot.credentials.host.mcpUrl, posthogApiKey: boot.credentials.accessToken, host: boot.credentials.host, @@ -134,12 +139,13 @@ export const anthropicBackend: AgentHarness = { programId: boot.programId, wizardFlags: boot.wizardFlags, wizardMetadata: boot.wizardMetadata, - integrationLabel: programConfig.id, + integrationLabel: config.programId, // Only a task allowed to ask carries a bridge, so the Write/Edit pause // that rides on a pending question stays inside that task's agent. askBridge, orchestrator, capture, + emit, }, options, ); diff --git a/src/lib/agent/runner/harness/pi/index.ts b/src/lib/agent/runner/harness/pi/index.ts index 092e59718..5c5ae3a25 100644 --- a/src/lib/agent/runner/harness/pi/index.ts +++ b/src/lib/agent/runner/harness/pi/index.ts @@ -14,7 +14,6 @@ import fs from 'fs'; import path from 'path'; -import { getUI } from '@ui'; import { getLogFilePath, logToFile } from '@utils/debug'; import { Harness, @@ -41,6 +40,8 @@ import type { TaskRunInputs, } from '../types'; import type { BootstrapResult } from '@lib/agent/runner/shared/types'; +import type { ProgressEmitter } from '@lib/agent/progress'; +import { createEmitLog } from '@lib/agent/runner/shared/progress-collector'; import type { TaskStore } from './tasks'; import { completionFailure, runErrorType } from './completion'; @@ -147,10 +148,19 @@ export function extractText(message: unknown): string { * the MCP creates them) into the outro link, mirroring the anthropic path's * signal parsing (#9). The marker carries the URL the MCP returned. */ -export function applyOutroMarkers(textBlock: string): void { +export function applyOutroMarkers( + textBlock: string, + emit: ProgressEmitter, +): void { const markers: Array<[string, (url: string) => void]> = [ - [AgentSignals.DASHBOARD_URL, (url) => getUI().setDashboardUrl(url)], - [AgentSignals.NOTEBOOK_URL, (url) => getUI().setNotebookUrl(url)], + [ + AgentSignals.DASHBOARD_URL, + (url) => emit({ kind: 'url', which: 'dashboard', url }), + ], + [ + AgentSignals.NOTEBOOK_URL, + (url) => emit({ kind: 'url', which: 'notebook', url }), + ], ]; for (const [marker, apply] of markers) { const idx = textBlock.indexOf(marker); @@ -195,20 +205,22 @@ export const piBackend: AgentHarness = { name: Harness.pi, async run(inputs: BackendRunInputs): Promise { - const { session, boot, prompt, spinner, config, programConfig } = inputs; + const { config: runConfig, input, boot, emit, prompt, spinner } = inputs; + const config = runConfig.run; const modelId = inputs.model; + const log = createEmitLog(emit); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, }); // Init banner (parity #5). - getUI().log.step('Initializing Wizard agent...'); - getUI().log.step(`Verbose logs: ${getLogFilePath()}`); - getUI().log.success("Agent initialized. Let's get cooking!"); + log.step('Initializing Wizard agent...'); + log.step(`Verbose logs: ${getLogFilePath()}`); + log.success("Agent initialized. Let's get cooking!"); spinner.start(config.spinnerMessage ?? 'Customizing your PostHog setup...'); @@ -306,11 +318,11 @@ export const piBackend: AgentHarness = { const { createSecurityExtension } = await import('./security'); const security = createSecurityExtension({ - disallowedTools: programConfig.disallowedTools, + disallowedTools: runConfig.disallowedTools, getWizardAskPending: () => askState.pending, triageProvider: boot.triageProvider, // Where pi's bash runs; the rm allowance is confined to this tree. - workingDirectory: session.installDir, + workingDirectory: input.installDir, }); // Pay warlock's WASM-init + rule-compile cost now, off the tool-call @@ -360,11 +372,11 @@ export const piBackend: AgentHarness = { } const resourceLoader = new DefaultResourceLoader({ - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), systemPrompt: assembleCommandments({ - program: programConfig.id, + program: runConfig.programId, sequence: Sequence.linear, harness: Harness.pi, caps: { bash: true, posthogMcp }, @@ -390,12 +402,14 @@ export const piBackend: AgentHarness = { const { createWizardPiTaskTools } = await import('./tasks'); const { createDispatchAgentTool } = await import('./subagent'); // Created once so the run loop can read the store for the completion guard. - const wizardTaskTools = createWizardPiTaskTools(); + const wizardTaskTools = createWizardPiTaskTools((tasks) => + emit({ kind: 'tasks', tasks }), + ); // The one bash the agent (and its subagents) may use: every subprocess it // spawns gets a scrubbed env, so no secret or ambient variable reaches an // `npm install`. Shared with the subagent so the lockdown is inherited. const scrubbedBash = withMode( - createBashToolDefinition(session.installDir, { + createBashToolDefinition(input.installDir, { spawnHook: (ctx) => ({ ...ctx, env: buildScrubbedEnv() }), }), 'sequential', @@ -406,18 +420,18 @@ export const piBackend: AgentHarness = { // defaults so we can supply the env-scrubbed bash above; read/edit/write // are the stock definitions. Reads run in parallel so a batched turn of // independent reads executes at once; edit/write/bash stay sequential. - withMode(createReadToolDefinition(session.installDir), 'parallel'), - withMode(createEditToolDefinition(session.installDir), 'sequential'), - withMode(createWriteToolDefinition(session.installDir), 'sequential'), + withMode(createReadToolDefinition(input.installDir), 'parallel'), + withMode(createEditToolDefinition(input.installDir), 'sequential'), + withMode(createWriteToolDefinition(input.installDir), 'sequential'), scrubbedBash, // Native ls/find/grep so the agent explores with proper tools instead // of fence-blocked `bash {ls/find}` (the profiled retry-spirals came // from this gap). Parallel — exploration batches cleanly. - withMode(createLsToolDefinition(session.installDir), 'parallel'), - withMode(createFindToolDefinition(session.installDir), 'parallel'), - withMode(createGrepToolDefinition(session.installDir), 'parallel'), + withMode(createLsToolDefinition(input.installDir), 'parallel'), + withMode(createFindToolDefinition(input.installDir), 'parallel'), + withMode(createGrepToolDefinition(input.installDir), 'parallel'), ...createWizardPiTools({ - workingDirectory: session.installDir, + workingDirectory: input.installDir, skillsBaseUrl: boot.skillsBaseUrl, triageProvider: boot.triageProvider, detectPackageManager: config.detectPackageManager, @@ -429,9 +443,10 @@ export const piBackend: AgentHarness = { onAskPendingChange: (pending) => { askState.pending = pending; }, + onHandoffText: (text) => emit({ kind: 'handoff', text }), // Skip wizard_ask when the program disallows it (bare pi tool names // don't match the MCP-prefixed disallow list at the security gate). - disallowedTools: programConfig.disallowedTools, + disallowedTools: runConfig.disallowedTools, }), // Task/todo tools (#526): render the todo list live in the TUI, parity // with the anthropic path. @@ -442,7 +457,7 @@ export const piBackend: AgentHarness = { createDispatchAgentTool({ model, modelRegistry: registry, - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), securityFactory: security.factory as (pi: unknown) => void, bashTool: scrubbedBash, @@ -456,8 +471,8 @@ export const piBackend: AgentHarness = { // Reasoning effort from the switchboard capability matrix (undefined = // pi's default). Sent as `reasoning_effort` for openai-completions. thinkingLevel: caps.thinkingLevel, - cwd: session.installDir, - sessionManager: SessionManager.inMemory(session.installDir), + cwd: input.installDir, + sessionManager: SessionManager.inMemory(input.installDir), resourceLoader, // Disable the default built-in tools; `customTools` re-registers // read/edit/write + an env-scrubbed bash, so no subprocess inherits the @@ -508,12 +523,12 @@ export const piBackend: AgentHarness = { const assistant = extractText(event.message).trim(); if (assistant) { logToFile(`[pi] assistant: ${assistant.slice(0, 1000)}`); - applyOutroMarkers(assistant); + applyOutroMarkers(assistant, emit); // Surface [STATUS] lines into the live spinner + status history, // mirroring the anthropic path — pi otherwise drops them. const statusText = lastStatusLine(assistant); if (statusText) { - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } for (const line of assistant.split('\n')) signals.push(line); @@ -644,7 +659,7 @@ export const piBackend: AgentHarness = { // on completion; pi's `rm` is fence-blocked, so the agent can't — clean it // up host-side rather than leave a stale (often empty) artifact (#15). try { - const planFile = path.join(session.installDir, '.posthog-events.json'); + const planFile = path.join(input.installDir, '.posthog-events.json'); if (fs.existsSync(planFile)) await fs.promises.rm(planFile); } catch (err) { logToFile(`[pi] .posthog-events.json cleanup skipped: ${String(err)}`); @@ -668,7 +683,7 @@ export const piBackend: AgentHarness = { const message = err instanceof Error ? err.message : String(err); logToFile(`[pi] run error: ${message}`); spinner.stop(config.errorMessage ?? `${config.integrationLabel} failed`); - getUI().log.error(`pi backend error: ${message}`); + log.error(`pi backend error: ${message}`); const error = runErrorType(message); captureAborted(error); return { error, message }; diff --git a/src/lib/agent/runner/harness/pi/task.ts b/src/lib/agent/runner/harness/pi/task.ts index 37deec62d..c86e0b1d7 100644 --- a/src/lib/agent/runner/harness/pi/task.ts +++ b/src/lib/agent/runner/harness/pi/task.ts @@ -16,7 +16,6 @@ * Loaded lazily from `index.ts` (typebox/ESM constraint, same as tools.ts). */ -import { getUI } from '@ui'; import { logToFile } from '@utils/debug'; import { analytics } from '@utils/analytics'; import { @@ -166,9 +165,10 @@ function isSettled(ctx: OrchestratorToolsContext): boolean { export async function runPiTask(inputs: TaskRunInputs): Promise { const { - session, - programConfig, + config, + input, boot, + emit, prompt, spinner, model: modelId, @@ -187,7 +187,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { if (spinnerMessage) spinner.start(spinnerMessage); const capture = createAioCapture({ - enabled: session.captureAio, + enabled: input.flags.captureAio, projectApiKey: boot.credentials.projectApiKey, apiHost: boot.credentials.host.apiHost, runTags: boot.wizardMetadata, @@ -319,10 +319,10 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { const orchestratorTools = allowedOrchestratorTools(disallowedTools); const resourceLoader = new DefaultResourceLoader({ - cwd: session.installDir, + cwd: input.installDir, agentDir: getAgentDir(), systemPrompt: assembleCommandments({ - program: programConfig.id, + program: config.programId, sequence: Sequence.orchestrator, harness: Harness.pi, caps: { bash: codingTools.has('bash'), posthogMcp }, @@ -339,7 +339,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { // The task's coding tools, gated by its allow list. Reads and searches run // in parallel; mutating tools stay sequential. Bash subprocesses get the // scrubbed env, same as the linear run. - const dir = session.installDir; + const dir = input.installDir; const codingToolFactories = { read: () => withMode(createReadToolDefinition(dir), 'parallel'), edit: () => withMode(createEditToolDefinition(dir), 'sequential'), @@ -375,6 +375,7 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { onAskPendingChange: (pending) => { askState.pending = pending; }, + onHandoffText: (text) => emit({ kind: 'handoff', text }), }).filter((t) => wizardToolNames.has(t.name)); const { createPiOrchestratorTools } = await import('./orchestrator-tools'); @@ -436,10 +437,10 @@ export async function runPiTask(inputs: TaskRunInputs): Promise { const assistant = extractText(event.message).trim(); if (assistant) { logToFile(`[pi-task] assistant: ${assistant.slice(0, 1000)}`); - applyOutroMarkers(assistant); + applyOutroMarkers(assistant, emit); const statusText = lastStatusLine(assistant); if (statusText) { - getUI().pushStatus(statusText); + emit({ kind: 'status', message: statusText }); spinner.message(statusText); } for (const line of assistant.split('\n')) signals.push(line); diff --git a/src/lib/agent/runner/harness/pi/tasks.ts b/src/lib/agent/runner/harness/pi/tasks.ts index e12f66e1e..0ee0f4462 100644 --- a/src/lib/agent/runner/harness/pi/tasks.ts +++ b/src/lib/agent/runner/harness/pi/tasks.ts @@ -1,15 +1,15 @@ /** * Task/todo parity for pi (#526). The same four Task tools the anthropic path * exposes (TaskCreate/Update/Get/List), as pi `defineTool` tools backed by a - * shared in-memory store. Every mutation pushes the list to the TUI via - * `getUI().syncTodos`, so the todo panel updates live under pi exactly like the - * anthropic path — the thing that was missing before. + * shared in-memory store. Every mutation reports the list through `onSync` + * (a `tasks` progress event), so the todo panel updates live under pi exactly + * like the anthropic path — the thing that was missing before. */ import { Type } from 'typebox'; import { defineTool } from '@earendil-works/pi-coding-agent'; import type { ToolDefinition } from '@earendil-works/pi-coding-agent'; -import { getUI } from '@ui'; +import type { TaskSnapshot } from '@lib/agent/progress'; export type TaskStatus = 'pending' | 'in_progress' | 'completed'; export interface TaskEntry { @@ -26,22 +26,23 @@ function text(s: string): { return { content: [{ type: 'text', text: s }], details: {} }; } -function syncToTui(store: TaskStore): void { - getUI().syncTodos( - Array.from(store.values()).map((t) => ({ - content: t.content, - status: t.status, - activeForm: t.activeForm, - })), - ); +function snapshot(store: TaskStore): TaskSnapshot[] { + return Array.from(store.values()).map((t) => ({ + content: t.content, + status: t.status, + activeForm: t.activeForm, + })); } -/** Build the four Task tools over a fresh store. */ -export function createWizardPiTaskTools(): { +/** Build the four Task tools over a fresh store. `onSync` gets the list after every mutation. */ +export function createWizardPiTaskTools( + onSync: (tasks: TaskSnapshot[]) => void = () => undefined, +): { tools: ToolDefinition[]; store: TaskStore; } { const store: TaskStore = new Map(); + const syncToTui = (): void => onSync(snapshot(store)); const taskCreate = defineTool({ name: 'TaskCreate', @@ -64,7 +65,7 @@ export function createWizardPiTaskTools(): { status: 'pending', activeForm: args.activeForm, }); - syncToTui(store); + syncToTui(); return text(`Created ${id}`); }, }); @@ -97,7 +98,7 @@ export function createWizardPiTaskTools(): { status: (args.status as TaskStatus) ?? existing.status, activeForm: args.activeForm ?? existing.activeForm, }); - syncToTui(store); + syncToTui(); return text(`Updated ${args.taskId}`); }, }); diff --git a/src/lib/agent/runner/harness/pi/tools.ts b/src/lib/agent/runner/harness/pi/tools.ts index b194cc79a..f5d1ab004 100644 --- a/src/lib/agent/runner/harness/pi/tools.ts +++ b/src/lib/agent/runner/harness/pi/tools.ts @@ -93,6 +93,8 @@ export interface PiToolsContext { disallowedTools?: readonly string[]; /** Scan-triage classifier, resolved once in bootstrap. Absent → scans fail closed. */ triageProvider?: LLMProvider; + /** Receives the handoff document `publish_handoff` accepted. */ + onHandoffText?: (text: string) => void; } export function createWizardPiTools(ctx: PiToolsContext): ToolDefinition[] { @@ -553,7 +555,7 @@ export function createWizardPiTools(ctx: PiToolsContext): ToolDefinition[] { }), }), execute(_id, args) { - const result = publishHandoff(args.content); + const result = publishHandoff(args.content, ctx.onHandoffText); logToFile(`[pi] publish_handoff: ${result.message}`); return Promise.resolve(text(result.message)); }, diff --git a/src/lib/agent/runner/harness/types.ts b/src/lib/agent/runner/harness/types.ts index 67c59e5c9..f245889e7 100644 --- a/src/lib/agent/runner/harness/types.ts +++ b/src/lib/agent/runner/harness/types.ts @@ -13,23 +13,27 @@ * per drained task. A harness without orchestrator support omits the method; * `orchestrator-runner.ts` checks for it at the call site and fails loudly * rather than silently downgrading. + * + * A harness reports through `emit` and never reaches for a UI. It returns an + * error classification, or a decided `failure` the sequence must return as + * the run's result. */ -import type { WizardSession } from '@lib/wizard-session'; -import type { AdditionalFeature } from '@lib/wizard-session'; +import type { AdditionalFeature, PendingQuestion } from '@lib/wizard-session'; import type { Harness } from '@lib/constants'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { SpinnerHandle } from '@ui'; import type { WizardAskBridge } from '@lib/wizard-ask-bridge'; import type { AgentErrorType } from '@lib/agent/agent-interface'; +import type { ProgressEmitter, SpinnerHandle } from '@lib/agent/progress'; import type { OrchestratorToolsContext } from '@lib/agent/runner/sequence/orchestrator/queue-tools'; import type { EffortLevel, ThinkingLevel, } from '@lib/agent/runner/switchboard/models'; import type { - ProgramRun, + AgentFailure, BootstrapResult, + RunConfig, + RunInput, } from '@lib/agent/runner/shared/types'; /** The benchmark/telemetry hook threaded through a run, if enabled. */ @@ -40,14 +44,14 @@ export interface RunMiddleware { /** * Everything a runner needs to run one program. Assembled by `linear.ts` from - * the bootstrap result and the program config; the runner consumes it and never + * the prepared run and the run config; the runner consumes it and never * re-derives run context. */ export interface BackendRunInputs { - session: WizardSession; - config: ProgramRun; - programConfig: ProgramConfig; + config: RunConfig; + input: RunInput; boot: BootstrapResult; + emit: ProgressEmitter; /** The fully assembled prompt. */ prompt: string; /** Installed framework-skill path, when the program installs one. */ @@ -56,7 +60,9 @@ export interface BackendRunInputs { spinner: SpinnerHandle; /** Interactive question bridge; undefined in CI/headless (ask disabled). */ askBridge?: WizardAskBridge; - /** Benchmark middleware, when `session.benchmark` is set. */ + /** The question on screen, for the Write/Edit guard. Undefined without a bridge. */ + getPendingQuestion?: () => PendingQuestion | null; + /** Benchmark middleware, when `--benchmark` is set. */ middleware?: RunMiddleware; /** Gateway model id resolved from the (runner, model) pair. */ model: string; @@ -64,8 +70,16 @@ export interface BackendRunInputs { thinkingLevel?: EffortLevel; } -/** What a runner reports back: an error classification, or nothing on success. */ -export type AgentResult = { error?: AgentErrorType; message?: string }; +/** + * What a runner reports back: an error classification, or nothing on success. + * `failure` is a fully decided abort the harness already reported to the host + * (the 401 auth screen); the sequence returns it as the run's result. + */ +export type AgentResult = { + error?: AgentErrorType; + message?: string; + failure?: AgentFailure; +}; /** * One orchestrator-mode unit of work — the seed plan, or one drained task. @@ -75,9 +89,10 @@ export type AgentResult = { error?: AgentErrorType; message?: string }; * them from the program-level config the linear pipeline assembles once. */ export interface TaskRunInputs { - session: WizardSession; - programConfig: ProgramConfig; + config: RunConfig; + input: RunInput; boot: BootstrapResult; + emit: ProgressEmitter; /** The fully assembled per-task or seed prompt. */ prompt: string; spinner: SpinnerHandle; diff --git a/src/lib/agent/runner/index.ts b/src/lib/agent/runner/index.ts index f11bf73c8..993cdb89b 100644 --- a/src/lib/agent/runner/index.ts +++ b/src/lib/agent/runner/index.ts @@ -1,214 +1,129 @@ /** * Unified program runner — dispatcher. * - * Single configurable pipeline for all programs. Each program - * provides a ProgramRun (via the `run` field on ProgramConfig) - * that controls: - * - Whether a skill is pre-installed or discovered at runtime - * - How the agent prompt is built - * - What MCP servers and package manager detector to use - * - What happens after the agent completes + * One callable, `runAgent(config, input, options)`, runs a program's agent + * pipeline to a decided result. `config` is resolved execution data (which + * program, how it is routed, which flags apply); `input` is the invocation + * snapshot (directory, credentials, project); `options` carries an optional + * progress observer, an optional answerer and an optional cancel signal. * - * The pipeline runs a shared bootstrap (logging, health check, settings, OAuth, - * flags, MCP url), then forks. The `orchestrator` variant routes to the - * experimental task-queue runner. Every other variant runs the fixed linear - * pipeline: + * The pipeline prepares the run (logging targets, gateway mint, scan triage), + * then forks. The `orchestrator` variant routes to the task-queue runner. + * Every other variant runs the fixed linear pipeline: * [skill install] → agent init → prompt → run → errors → [postRun] → outro + * + * The agent reports and asks, it never renders, never reads a session, never + * exits the process and never rejects. A decided failure comes back in + * `RunResult.failure` with the same fields `wizardAbort` takes; an error the + * agent did not decide (a refused mint, an SDK crash) comes back as + * `outcome: 'crashed'` with the original error attached, so a caller can keep + * handling it the way it always did. The legacy adapter in + * `src/lib/programs/run-agent-legacy.ts` rebuilds today's session-driven + * behavior on top of this call for every existing caller. */ -import type { WizardSession } from '../../wizard-session'; -import { analytics } from '@utils/analytics'; -import { - Sequence, - WIZARD_ORCHESTRATOR_FLAG_KEY, - WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY, -} from '@lib/constants'; +import { Sequence } from '@lib/constants'; +import { classifyRunFailure } from '@lib/errors'; import { logToFile } from '@utils/debug'; -import { getUI } from '../../../ui'; -import type { ProgramConfig } from '../../programs/program-step'; -import type { ProgramRun, BootstrapResult } from './shared/types'; -import { bootstrapProgram } from './shared/bootstrap'; -import { - getSequence, - resolveBinding, - type ProgramBinding, - type SwitchboardCtx, -} from './switchboard'; +import type { + RunAgentOptions, + RunConfig, + RunInput, + RunResult, +} from './shared/types'; +import { prepareRun } from './shared/bootstrap'; +import { createProgressCollector } from './shared/progress-collector'; +import { getSequence } from './switchboard'; import { flushScanReport } from '../../yara-hooks'; -import { startAuditLedgerWatcher } from '../../programs/audit/ledger-watcher'; -import { registerCleanup } from '../../../utils/wizard-abort'; export type { - ProgramRun, - BootstrapResult, AbortCase, - PromptContext, + AgentFailure, + AgentRunDefinition, + BootstrapResult, Credentials, + PromptContext, + ResolvedBinding, + RunAgentOptions, + RunConfig, + RunFlags, + RunHooks, + RunInput, + RunOutcome, + RunResult, + RunSnapshot, + SeedTaskEntry, } from './shared/types'; +export type { + AgentInteraction, + AgentProgress, + ProgressEmitter, +} from '@lib/agent/progress'; export { shouldDisableAsk } from './shared/bootstrap'; - -/** - * Resolve a ProgramConfig's agent run definition and execute the pipeline. - * Entry point for bin.ts — handles buildRunConfig, bootstrap, and (future) run field. - */ -export async function runAgent( - programConfig: ProgramConfig, - session: WizardSession, - options: { composed?: boolean } = {}, -): Promise { - if (!programConfig.run) { - throw new Error(`Program "${programConfig.id}" has no run configuration.`); - } - - // Before `run()` resolves: an audit seeds the ledger from inside its recipe, - // and a watcher started later would ignore that write as pre-existing. - const ledger = programConfig.auditLedgerFile - ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) - : null; - if (ledger) registerCleanup(() => ledger.stop()); - - try { - const runDef = - typeof programConfig.run === 'function' - ? await programConfig.run(session) - : programConfig.run; - - await runProgram(session, runDef, programConfig, options); - } finally { - ledger?.stop(); - } -} +export { DEFAULT_BINDING, resolveBinding } from './switchboard'; +export type { ProgramBinding, SwitchboardCtx } from './switchboard'; /** * Run a program's agent pipeline. * - * Bootstrap → bind the program via the switchboard (resolve which sequence - * and harness will run it, tag both axes) → dispatch to the resolved - * sequence's runner. + * Prepare → dispatch to the sequence the binding names → return its result + * with the agent's own snapshot of what it reported. Missing observers change + * nothing; a throwing observer is logged and the run continues. */ -export async function runProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - options: { composed?: boolean } = {}, -): Promise { - const boot = await bootstrapProgram(session, config, programConfig); +export async function runAgent( + config: RunConfig, + input: RunInput, + options: RunAgentOptions = {}, +): Promise { + const collector = createProgressCollector(options.onProgress); + const { emit } = collector; + const signal = options.signal ?? new AbortController().signal; + const log = (message: string) => + emit({ kind: 'log', level: 'info', message }); // Flush the warlock scan report once, at this single seam, on every - // termination path and for every harness (linear, orchestrator, or future): - // - registerCleanup covers the abort/cancel path (wizardAbort runs the - // registered cleanups; the success path never calls them) - // - the finally covers normal completion and any direct throw that unwinds - // through here - // flushScanReport is idempotent (it zeroes scan state), so the overlap is a - // harmless no-op. No harness has to know reporting exists. - registerCleanup(() => flushScanReport(session)); + // termination path and for every harness (linear, orchestrator, or future). + // flushScanReport is idempotent (it zeroes scan state), so a caller that also + // flushes from its own cleanup path sees a harmless no-op. No harness has to + // know reporting exists. try { - const binding = resolveProgramRunner( - session, - programConfig, - boot, - options.composed ?? false, - ); - if (binding.sequence === Sequence.orchestrator) { - getUI().log.info('Task-queue orchestrator enabled.'); + const boot = await prepareRun(config, input); + if (config.binding.sequence === Sequence.orchestrator) { + log('Task-queue orchestrator enabled.'); } - return await getSequence(binding.sequence).run( - session, + logToFile( + `[agent-runner] run program=${config.programId} sequence=${config.binding.sequence}` + + ` harness=${config.binding.harness} composed=${config.composed}`, + ); + const result = await getSequence(config.binding.sequence).run({ config, - programConfig, + input, boot, - options.composed ?? false, - ); + emit, + interaction: options.interaction, + signal, + }); + return { + ...result, + skillId: input.skillId, + snapshot: collector.snapshot(), + }; + } catch (error) { + // Not a decision the agent made. Hand it back whole rather than throw, so + // every ending of a run is a result the caller reads the same way. + const failure = classifyRunFailure(error); + logToFile('[agent-runner] run crashed:', error); + return { + outcome: 'crashed', + skillId: input.skillId, + failure: { + code: failure.code, + message: failure.message, + error: error instanceof Error ? error : new Error(String(error)), + }, + snapshot: collector.snapshot(), + }; } finally { - flushScanReport(session); + flushScanReport({ yaraReport: input.flags.yaraReport }, log); } } - -/** - * Resolve which sequence and harness will run a program (CLI → PostHog flag → - * per-program binding → default), tag both axes onto analytics, and return the - * binding for downstream dispatch. - * - * The one place `runner/index.ts` reaches into the switchboard — every other - * concern (bootstrap, cleanup, dispatch, per-task per-role harness picks) is - * either upstream or downstream of this call. - */ -function resolveProgramRunner( - session: WizardSession, - programConfig: ProgramConfig, - boot: BootstrapResult, - composed: boolean, -): ProgramBinding { - const ctx = { - program: programConfig.id, - composed, - flags: boot.wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliSequence: session.sequence, - cliModel: session.model, - }; - const binding = resolveBinding(ctx); - tagBinding(boot, binding); - captureSwitchboardDecision(ctx, binding); - return binding; -} - -/** - * One event + one log line per run: what entered the switchboard, which - * precedence rung decided each axis, and the final pick. - */ -function captureSwitchboardDecision( - ctx: SwitchboardCtx, - binding: ProgramBinding, -): void { - const trace = ctx.trace ?? {}; - // Unpinned orchestrator runs choose a model per task from the context-mill agent prompts; the orchestrator logs that map once the prompts load. - const perTaskModel = - binding.sequence === Sequence.orchestrator && trace.model === 'binding'; - const model = perTaskModel ? 'chosen-per-task' : binding.model; - const modelSource = perTaskModel ? 'agent-prompts' : trace.model; - analytics.wizardCapture('switchboard resolved', { - program: ctx.program, - flag_self_driving_use_pi_harness: - ctx.flags[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY], - flag_self_driving_pi_payload: JSON.stringify( - ctx.flagPayloads?.[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY] ?? null, - ), - flag_orchestrator: ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY], - cli_harness: ctx.cliHarness, - cli_sequence: ctx.cliSequence, - cli_model: ctx.cliModel, - harness_source: trace.harness, - model_source: modelSource, - sequence_source: trace.sequence, - harness: binding.harness, - model, - thinking_level: binding.thinkingLevel, - sequence: binding.sequence, - }); - logToFile( - `[switchboard] decision: program=${ctx.program}` + - ` in(orchestrator=${ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY] ?? '-'},` + - ` cli=${ctx.cliHarness ?? '-'}/${ctx.cliSequence ?? '-'}/${ - ctx.cliModel ?? '-' - })` + - ` → harness=${binding.harness} (${trace.harness ?? '?'})` + - ` model=${model} (${modelSource ?? '?'})` + - ` sequence=${binding.sequence} (${trace.sequence ?? '?'})`, - ); -} - -/** - * Tag the run with its two routing axes. Sequence is stable for the whole - * run; harness reflects the run-level (default-role) resolution — orchestrator - * per-task calls emit their own `harness` property in their events so per-task - * aggregations attribute correctly. - */ -function tagBinding(boot: BootstrapResult, binding: ProgramBinding): void { - analytics.setTag('sequence', binding.sequence); - analytics.setTag('harness', binding.harness); - boot.wizardMetadata.SEQUENCE = binding.sequence; - boot.wizardMetadata.HARNESS = binding.harness; -} diff --git a/src/lib/agent/runner/sequence/linear.ts b/src/lib/agent/runner/sequence/linear.ts index 94ab86841..fa3794233 100644 --- a/src/lib/agent/runner/sequence/linear.ts +++ b/src/lib/agent/runner/sequence/linear.ts @@ -1,112 +1,127 @@ /** * The linear pipeline. Single execution path for all non-orchestrator programs, * both skill-based (revenue analytics) and framework-based (core integration). - * The `ProgramRun` controls what varies between them; `programConfig` carries the - * program-level static metadata (tool allow/disallow lists, etc.). + * The `AgentRunDefinition` controls what varies between them; `RunConfig` + * carries the program-level static metadata (tool allow/disallow lists, etc.). + * + * Reports through `emit`, asks through `interaction`, and returns a decided + * `RunResult`. Every former `getUI()` call is one progress event in the same + * place; every former `wizardAbort` is a returned failure with the same + * arguments, so the caller's exit sequence is unchanged. */ -import type { WizardSession } from '../../../wizard-session'; -import { OutroKind } from '../../../wizard-session'; -import { getUI } from '../../../../ui'; +import { OutroKind, type OutroData } from '@lib/wizard-session'; import { AgentErrorType, AgentSignals } from '../../agent-interface'; -import { restoreClaudeSettings } from '../../claude-settings'; import { logToFile } from '../../../../utils/debug'; import { createBenchmarkPipeline } from '../../../middleware/benchmark'; -import { - wizardAbort, - WizardError, - registerCleanup, -} from '../../../../utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; import { ErrorCodes, AGENT_ERROR_CODE } from '@lib/errors'; import { analytics } from '../../../../utils/analytics'; -import { - formatScanReport, - formatYaraAbortMessage, - writeScanReport, -} from '../../../yara-hooks'; +import { formatYaraAbortMessage } from '../../../yara-hooks'; import { installSkillById } from '../../../wizard-tools'; -import { createWizardAskBridge } from '../../../wizard-ask-bridge'; -import type { ProgramConfig } from '../../../programs/program-step'; import { assemblePrompt } from '../../agent-prompt'; -import type { ProgramRun, BootstrapResult } from '../shared/types'; -import { abortOnInstallFailure } from '../shared/errors'; -import { shouldDisableAsk, sessionToOptions } from '../shared/bootstrap'; -import { resolveHarness, getHarness } from '../switchboard'; +import type { + AgentFailure, + RunResult, + RunSnapshot, + SequenceContext, +} from '../shared/types'; +import { installFailure } from '../shared/errors'; +import { shouldDisableAsk, runOptions } from '../shared/bootstrap'; +import { createEmitSpinner } from '../shared/progress-collector'; +import { createAskBridge } from '../shared/ask'; +import { getHarness } from '../switchboard'; -export async function runLinearProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, - composed = false, -): Promise { - const { skillsBaseUrl, credentials, wizardFlags, project } = boot; +/** The snapshot the sequence returns; `runAgent` fills it from its collector. */ +const PENDING_SNAPSHOT: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, +}; + +const failed = (failure: AgentFailure): RunResult => ({ + outcome: 'failed', + failure, + snapshot: PENDING_SNAPSHOT, +}); + +const cancelled = (): RunResult => ({ + outcome: 'cancelled', + failure: { message: 'Wizard setup cancelled.' }, + snapshot: PENDING_SNAPSHOT, +}); + +export async function runLinearProgram({ + config, + input, + boot, + emit, + interaction, + signal, +}: SequenceContext): Promise { + const { run, composed } = config; + const { skillsBaseUrl, credentials, project } = boot; const { projectApiKey, host, projectId } = credentials; + if (signal.aborted) return cancelled(); + // 5. Skill install (if skillId provided) let skillPath: string | undefined; - if (config.skillId) { - logToFile(`[agent-runner] installing skill ${config.skillId}`); + if (run.skillId) { + logToFile(`[agent-runner] installing skill ${run.skillId}`); const installResult = await installSkillById( - config.skillId, - session.installDir, + run.skillId, + input.installDir, skillsBaseUrl, { triage: boot.triageProvider }, ); if (installResult.kind !== 'ok') { - await abortOnInstallFailure(config.integrationLabel, installResult); - return; + return failed(installFailure(run.integrationLabel, installResult)); } skillPath = installResult.path; logToFile(`[agent-runner] skill installed at ${skillPath}`); } - // 6. Initialize agent - const spinner = getUI().spinner(); - - const restoreSettings = () => restoreClaudeSettings(session.installDir); - getUI().onEnterScreen('outro', restoreSettings); + if (signal.aborted) return cancelled(); - if (session.yaraReport) { - registerCleanup(() => { - const reportPath = writeScanReport(); - if (reportPath) { - const summary = formatScanReport(); - getUI().log.info(`YARA scan report: ${reportPath}${summary ?? ''}`); - } - }); - } + // 6. Initialize agent + const spinner = createEmitSpinner(emit); - getUI().startRun(); + emit({ kind: 'lifecycle', phase: 'started' }); // wizard_ask needs an answerer. A human answers at the keyboard; the e2e // snapshot/MCP host answers via its driver and sets WIZARD_ASK_AUTODRIVE. // CI/signup with neither has no answerer, so we omit the bridge and the tool // returns an actionable error rather than hanging on a never-resolving prompt. const askDisabled = - shouldDisableAsk(session) && process.env.WIZARD_ASK_AUTODRIVE !== '1'; - const askBridge = askDisabled + shouldDisableAsk(input.flags) && process.env.WIZARD_ASK_AUTODRIVE !== '1'; + const ask = askDisabled ? undefined - : createWizardAskBridge({ - getSource: () => session.skillId ?? config.integrationLabel, - showQuestion: (q) => getUI().requestQuestion(q), - cancelQuestion: () => getUI().cancelPendingQuestion(), - richLinks: config.richLinks ?? false, - timeoutMs: config.askTimeoutMs, + : createAskBridge(interaction, signal, { + getSource: () => input.skillId ?? run.integrationLabel, + richLinks: run.richLinks ?? false, + timeoutMs: run.askTimeoutMs, }); - const middleware = session.benchmark - ? createBenchmarkPipeline(spinner, sessionToOptions(session)) + const middleware = input.flags.benchmark + ? createBenchmarkPipeline(spinner, runOptions(input), undefined, { + log: (message) => emit({ kind: 'log', level: 'info', message }), + }) : undefined; // 7. Build prompt - const prompt = assemblePrompt(config, { + const prompt = assemblePrompt(run, { projectId, projectApiKey, host, skillPath, orgAiDataProcessingApproved: - session.apiUser?.organization?.is_ai_data_processing_approved ?? null, + input.apiUser?.organization?.is_ai_data_processing_approved ?? null, teamProductOptIns: project ? { sessionReplay: project.session_recording_opt_in ?? null, @@ -117,37 +132,35 @@ export async function runLinearProgram( }); logToFile(`[agent-runner] prompt assembled (${prompt.length} chars)`); - // 8. Resolve the (runner, model) pair from the central plan and run the agent - // through the selected runner. The runner owns the agent loop + model - // transport; everything around it (skill install, prompt, ask bridge, error - // routing, outro) stays here so every runner shares it. - const pick = resolveHarness({ - program: programConfig.id, - flags: wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliModel: session.model, - }); - const agentResult = await getHarness(pick.harness).run({ - session, + // 8. Run the agent through the run-level harness. The harness owns the agent + // loop + model transport; everything around it (skill install, prompt, ask + // bridge, error routing, outro) stays here so every harness shares it. + const { harness, model, thinkingLevel } = config.binding; + const agentResult = await getHarness(harness).run({ config, - programConfig, + input, boot, + emit, prompt, skillPath, spinner, - askBridge, + askBridge: ask?.bridge, + getPendingQuestion: ask?.getPendingQuestion, middleware, - model: pick.model, - thinkingLevel: pick.thinkingLevel, + model, + thinkingLevel, }); - // 9. Error handling (full set from both runners) + // 9. Error handling (full set from both harnesses) + if (agentResult.failure) { + return failed(agentResult.failure); + } + if (agentResult.error === AgentErrorType.ABORT) { const reason = agentResult.message ?? ''; - const matched = config.abortCases?.find((c) => c.match.test(reason)); + const matched = run.abortCases?.find((c) => c.match.test(reason)); const abortCode = matched?.errorCode ?? ErrorCodes.AgentAbort; - const outroData: WizardSession['outroData'] = matched + const outroData: OutroData = matched ? { kind: OutroKind.Error, message: matched.message, @@ -158,44 +171,48 @@ export async function runLinearProgram( } : { kind: OutroKind.Error, - message: `${config.integrationLabel} aborted`, + message: `${run.integrationLabel} aborted`, body: reason || 'The agent aborted the program.', - docsUrl: config.docsUrl, + docsUrl: run.docsUrl, errorCode: abortCode, errorDetail: { reason }, }; analytics.wizardCapture('agent aborted', { - integration: config.integrationLabel, + integration: run.integrationLabel, reason, matched: matched?.message ?? null, }); - await wizardAbort({ - outroData, - code: abortCode, - error: new WizardError( - `Agent aborted: ${reason}`, - { - integration: config.integrationLabel, - error_type: AgentErrorType.ABORT, - reason, - }, - abortCode, - ), - }); + return { + outcome: 'aborted', + failure: { + outroData, + code: abortCode, + error: new WizardError( + `Agent aborted: ${reason}`, + { + integration: run.integrationLabel, + error_type: AgentErrorType.ABORT, + reason, + }, + abortCode, + ), + }, + snapshot: PENDING_SNAPSHOT, + }; } if (agentResult.error === AgentErrorType.MCP_MISSING) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.MCP_MISSING], message: 'Could not access the PostHog MCP server\n\n' + 'The wizard was unable to connect to the PostHog MCP server.\n' + 'This could be due to a network issue or a configuration problem.\n\n' + - `Please try again, or check the documentation:\n${config.docsUrl}`, + `Please try again, or check the documentation:\n${run.docsUrl}`, error: new WizardError( 'Agent could not access PostHog MCP server', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.MCP_MISSING, signal: AgentSignals.ERROR_MCP_MISSING, }, @@ -205,16 +222,16 @@ export async function runLinearProgram( } if (agentResult.error === AgentErrorType.RESOURCE_MISSING) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.RESOURCE_MISSING], message: 'Could not access the setup resource\n\n' + 'This may indicate a version mismatch or a temporary service issue.\n\n' + - `Please try again, or check the documentation:\n${config.docsUrl}`, + `Please try again, or check the documentation:\n${run.docsUrl}`, error: new WizardError( 'Agent could not access setup resource', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.RESOURCE_MISSING, signal: AgentSignals.ERROR_RESOURCE_MISSING, }, @@ -224,13 +241,13 @@ export async function runLinearProgram( } if (agentResult.error === AgentErrorType.YARA_VIOLATION) { - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.YARA_VIOLATION], message: formatYaraAbortMessage(), error: new WizardError( 'YARA scanner terminated session', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.YARA_VIOLATION, }, AGENT_ERROR_CODE[AgentErrorType.YARA_VIOLATION], @@ -240,10 +257,10 @@ export async function runLinearProgram( if (agentResult.error === AgentErrorType.NO_PROGRESS) { analytics.wizardCapture('agent no progress', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.NO_PROGRESS, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], message: 'The Wizard exited without changing your project. Please contact the ' + @@ -251,7 +268,7 @@ export async function runLinearProgram( error: new WizardError( 'Agent made no progress', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.NO_PROGRESS, }, AGENT_ERROR_CODE[AgentErrorType.NO_PROGRESS], @@ -261,10 +278,10 @@ export async function runLinearProgram( if (agentResult.error === AgentErrorType.INCOMPLETE_TASKS) { analytics.wizardCapture('agent incomplete tasks', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.INCOMPLETE_TASKS, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[AgentErrorType.INCOMPLETE_TASKS], message: 'The Wizard exited without completing its planned tasks. Please contact ' + @@ -272,7 +289,7 @@ export async function runLinearProgram( error: new WizardError( 'Agent left planned tasks incomplete', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: AgentErrorType.INCOMPLETE_TASKS, }, AGENT_ERROR_CODE[AgentErrorType.INCOMPLETE_TASKS], @@ -285,12 +302,12 @@ export async function runLinearProgram( agentResult.error === AgentErrorType.API_ERROR ) { analytics.wizardCapture('agent api error', { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: agentResult.error, error_message: agentResult.message, }); - await wizardAbort({ + return failed({ code: AGENT_ERROR_CODE[agentResult.error], message: `API Error\n\n${ agentResult.message || 'Unknown error' @@ -298,7 +315,7 @@ export async function runLinearProgram( error: new WizardError( `API error: ${agentResult.message}`, { - integration: config.integrationLabel, + integration: run.integrationLabel, error_type: agentResult.error, }, AGENT_ERROR_CODE[agentResult.error], @@ -307,39 +324,36 @@ export async function runLinearProgram( } // 10. Post-run hooks - if (config.postRun) { - await config.postRun(session, credentials); + if (config.hooks?.postRun) { + await config.hooks.postRun(credentials); } // A composed sub-run (integration inside self-driving) skips the terminal // outro + analytics shutdown so the shared client survives the host's run. - if (composed) return; + if (composed) { + return { outcome: 'success', snapshot: PENDING_SNAPSHOT }; + } // 11. Outro - // Push outro data through the UI (not via direct `session.outroData = ...` - // mutation) so the live store gets the value. agent-runner's `session` - // parameter is captured at runAgent() invocation time, and any `setKey` - // call between then and here (e.g. setDashboardUrl, setNotebookUrl) forks - // the session reference — direct mutation then lands on a stale snapshot - // that the screen never reads. UI.setOutroData() goes through the store - // and also merges in any post-snapshot URLs from the live session. - const outroData = config.buildOutroData - ? config.buildOutroData(session, credentials) + const outroData: OutroData | undefined = config.hooks?.buildOutroData + ? config.hooks.buildOutroData(credentials) : { kind: OutroKind.Success, - message: config.successMessage, - reportFile: config.reportFile, - docsUrl: config.docsUrl, - continueUrl: session.signup + message: run.successMessage, + reportFile: run.reportFile, + docsUrl: run.docsUrl, + continueUrl: input.flags.signup ? `${host.appHost}/products?source=wizard` : undefined, }; if (outroData) { - getUI().setOutroData(outroData); + emit({ kind: 'completion', outro: outroData }); } - getUI().outro(config.successMessage); + emit({ kind: 'lifecycle', phase: 'completed', message: run.successMessage }); // 12. Analytics shutdown await analytics.shutdown('success'); + + return { outcome: 'success', outro: outroData, snapshot: PENDING_SNAPSHOT }; } diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts index d2379b32c..9ee3fca2a 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/seeded-decline-skip.test.ts @@ -10,9 +10,6 @@ import * as fs from 'fs'; import * as os from 'os'; import * as path from 'path'; -vi.mock('@ui', () => ({ - getUI: () => ({ showTaskNotice: vi.fn(), cancelTaskNotice: vi.fn() }), -})); vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture: vi.fn(), diff --git a/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts b/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts index c6acf1014..40c6642f8 100644 --- a/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts +++ b/src/lib/agent/runner/sequence/orchestrator/__tests__/task-notice-timeout.test.ts @@ -19,9 +19,6 @@ const { showTaskNotice, cancelTaskNotice, wizardCapture, captureException } = captureException: vi.fn(), })); -vi.mock('@ui', () => ({ - getUI: () => ({ showTaskNotice, cancelTaskNotice }), -})); vi.mock('@utils/analytics', () => ({ analytics: { wizardCapture, @@ -38,6 +35,12 @@ import { TASK_NOTICE_TIMEOUT_MS, } from '@lib/agent/runner/sequence/orchestrator/orchestrator-runner'; +/** The answerer under test, standing where `getUI()` used to. */ +const interaction = { + taskNotice: (notice: TaskNotice) => showTaskNotice(notice), + cancelTaskNotice: () => cancelTaskNotice(), +}; + const NOTICE: TaskNotice = { title: 'Connect your data sources', body: ['We found some sources.'], @@ -67,7 +70,7 @@ describe('task notice timeout', () => { // Nobody presses anything. showTaskNotice.mockReturnValue(new Promise(() => undefined)); - const promise = offerSeededTask(NOTICE, 1000); + const promise = offerSeededTask(NOTICE, 1000, interaction); vi.advanceTimersByTime(1000); await expect(promise).resolves.toEqual({ keep: false, timedOut: true }); @@ -83,7 +86,7 @@ describe('task notice timeout', () => { try { showTaskNotice.mockResolvedValue(true); - const result = await offerSeededTask(NOTICE, 1000); + const result = await offerSeededTask(NOTICE, 1000, interaction); expect(result).toEqual({ keep: true, timedOut: false }); vi.advanceTimersByTime(5000); @@ -102,10 +105,12 @@ describe('task notice timeout', () => { // Both decline, but only one of them means "the user was not there" — // the run reports them differently, and now skips them differently too. - await expect(offerSeededTask(NOTICE, 1000)).resolves.toEqual({ - keep: false, - timedOut: false, - }); + await expect(offerSeededTask(NOTICE, 1000, interaction)).resolves.toEqual( + { + keep: false, + timedOut: false, + }, + ); expect(cancelTaskNotice).not.toHaveBeenCalled(); } finally { vi.useRealTimers(); @@ -164,7 +169,9 @@ describe('askSeededConsent', () => { it('records an acceptance', async () => { showTaskNotice.mockResolvedValue(true); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, 1000, interaction), + ).resolves.toEqual({ keep: true, timedOut: false, errored: false, @@ -174,7 +181,9 @@ describe('askSeededConsent', () => { it('records an explicit decline', async () => { showTaskNotice.mockResolvedValue(false); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, 1000, interaction), + ).resolves.toEqual({ keep: false, timedOut: false, errored: false, @@ -186,7 +195,7 @@ describe('askSeededConsent', () => { try { showTaskNotice.mockReturnValue(new Promise(() => undefined)); - const promise = askSeededConsent('warehouse', NOTICE, 1000); + const promise = askSeededConsent('warehouse', NOTICE, 1000, interaction); vi.advanceTimersByTime(1000); await expect(promise).resolves.toEqual({ @@ -202,7 +211,7 @@ describe('askSeededConsent', () => { it('reports the answer on one event per offer', async () => { showTaskNotice.mockResolvedValue(true); - await askSeededConsent('warehouse', NOTICE, 1000); + await askSeededConsent('warehouse', NOTICE, 1000, interaction); expect(wizardCapture).toHaveBeenCalledTimes(1); expect(wizardCapture).toHaveBeenCalledWith( @@ -216,7 +225,9 @@ describe('askSeededConsent', () => { // that could not be put to the user must never be read as a yes. showTaskNotice.mockRejectedValue(new Error('UI blew up')); - await expect(askSeededConsent('warehouse', NOTICE, 1000)).resolves.toEqual({ + await expect( + askSeededConsent('warehouse', NOTICE, 1000, interaction), + ).resolves.toEqual({ keep: false, timedOut: false, errored: true, @@ -247,7 +258,9 @@ describe('offering notices from the seed loop', () => { const answers: { keep: boolean; timedOut: boolean; errored: boolean }[] = []; for (const entry of entries) { - answers.push(await askSeededConsent(entry.type, NOTICE, 60_000)); + answers.push( + await askSeededConsent(entry.type, NOTICE, 60_000, interaction), + ); } return answers; }; diff --git a/src/lib/agent/runner/sequence/orchestrator/executor.ts b/src/lib/agent/runner/sequence/orchestrator/executor.ts index 616d29442..e17e8c269 100644 --- a/src/lib/agent/runner/sequence/orchestrator/executor.ts +++ b/src/lib/agent/runner/sequence/orchestrator/executor.ts @@ -10,6 +10,7 @@ * injected: the real one spins up a fresh agent, the tests use a fake. */ import { analytics } from '@utils/analytics'; +import type { AgentFailure } from '../../shared/types'; import { logToFile } from '@utils/debug'; import { TaskStatus, type QueueStore, type QueuedTask } from './queue'; @@ -33,6 +34,19 @@ export type TaskResolver = ( * (via the task agent calling complete_task). */ export type RunTask = (task: QueuedTask) => Promise; +/** + * Thrown by a `RunTask` when its harness returned a decided failure that ends + * the whole run (a 401 the auth screen already reported), not just the task. + * `drainQueue` lets it through so the sequence can return the failure where + * the harness used to exit the process. + */ +export class RunTaskFatal extends Error { + constructor(public readonly failure: AgentFailure) { + super(failure.message ?? 'agent run failed'); + this.name = 'RunTaskFatal'; + } +} + export interface DrainOptions { /** Backstop against a pathological always-one-more-pending loop. */ maxStarts: number; @@ -51,6 +65,7 @@ async function runOne( try { await runTask(task); } catch (error) { + if (error instanceof RunTaskFatal) throw error; // The task threw rather than reporting. The outcome check below handles // the queue; the exception itself should never be silent. logToFile(`[executor] runTask threw for ${task.type}:`, error); diff --git a/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts b/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts index 8778431ad..502b3a02c 100644 --- a/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts +++ b/src/lib/agent/runner/sequence/orchestrator/orchestrator-runner.ts @@ -20,31 +20,28 @@ import { writeFileSync, } from 'fs'; import * as path from 'path'; -import { - OutroKind, - type TaskNotice, - type WizardSession, -} from '@lib/wizard-session'; -import { - POSTHOG_DOCS_URL, - WIZARD_CONTACT_EMAIL, - type Integration, -} from '@lib/constants'; -import { FRAMEWORK_REGISTRY } from '@lib/registry'; +import { OutroKind, type TaskNotice } from '@lib/wizard-session'; +import { POSTHOG_DOCS_URL, WIZARD_CONTACT_EMAIL } from '@lib/constants'; import { installSkillById, fetchSkillMenu, type SkillEntry, } from '@lib/wizard-tools'; -import { getUI } from '@ui'; import { analytics } from '@utils/analytics'; import { ciExcludedTaskTypes } from '@utils/ci-flag-overrides'; import { logToFile } from '@utils/debug'; import { ringTerminalBell } from '@utils/terminal-bell'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; import { ErrorCodes } from '@lib/errors'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { BootstrapResult, ProgramRun } from '../../shared/types'; +import type { AgentInteraction } from '@lib/agent/progress'; +import type { + AgentFailure, + RunResult, + RunSnapshot, + SequenceContext, +} from '../../shared/types'; +import { createEmitSpinner } from '../../shared/progress-collector'; +import { createAskBridge } from '../../shared/ask'; import { areSeededTasksEnabled, getHarness, @@ -61,14 +58,11 @@ import { TaskStatus, type QueuedTask, } from './queue'; -import { drainQueue, type RunTask } from './executor'; +import { drainQueue, RunTaskFatal, type RunTask } from './executor'; import { RunMetrics } from './run-metrics'; import { dependencyClosure, uncoveredBySink } from './queue-tools'; import { deferSeededTasks } from './seeded-deps'; -import { - createWizardAskBridge, - LONGER_ASK_TIMEOUT_MS, -} from '@lib/wizard-ask-bridge'; +import { LONGER_ASK_TIMEOUT_MS } from '@lib/wizard-ask-bridge'; import { shouldDisableAsk } from '../../shared/bootstrap'; import { agentRunTools, @@ -183,7 +177,7 @@ export function resolveSkillVariantId( } /** - * The framework reference is the full `integration` skill. `session.skillId` is + * The framework reference is the full `integration` skill. `input.skillId` is * the bare framework (e.g. `django`), but the skill menu ids it as * `integration-`. */ @@ -222,7 +216,14 @@ export const TASK_NOTICE_TIMEOUT_MS = 5 * 60 * 1000; export async function offerSeededTask( notice: TaskNotice, timeoutMs: number = TASK_NOTICE_TIMEOUT_MS, + interaction?: AgentInteraction, + signal: AbortSignal = new AbortController().signal, ): Promise<{ keep: boolean; timedOut: boolean }> { + // No one to show the notice to: a step nobody can answer for must not run. + // The same answer a non-interactive host gives today. + if (!interaction?.taskNotice) return { keep: false, timedOut: false }; + const { taskNotice, cancelTaskNotice } = interaction; + let timer: ReturnType | undefined; let timedOut = false; const timeout = new Promise((resolve) => { @@ -230,12 +231,12 @@ export async function offerSeededTask( timedOut = true; // Dismisses the overlay and settles the showTaskNotice promise too, so // the losing side of the race cannot leave a modal on screen. - getUI().cancelTaskNotice(); + cancelTaskNotice?.(); resolve(false); }, timeoutMs); }); try { - const keep = await Promise.race([getUI().showTaskNotice(notice), timeout]); + const keep = await Promise.race([taskNotice(notice, { signal }), timeout]); return { keep, timedOut }; } finally { if (timer) clearTimeout(timer); @@ -337,8 +338,15 @@ export async function askSeededConsent( type: string, notice: TaskNotice, timeoutMs?: number, + interaction?: AgentInteraction, + signal?: AbortSignal, ): Promise { - const consent = await offerSeededTask(notice, timeoutMs).then( + const consent = await offerSeededTask( + notice, + timeoutMs, + interaction, + signal, + ).then( (answer): SeededConsent => ({ ...answer, errored: false }), (err: unknown): SeededConsent => { logToFile( @@ -427,33 +435,58 @@ export function displayOrder( .map((entry) => entry.task); } -export async function runOrchestrator( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, -): Promise { +/** The snapshot the sequence returns; `runAgent` fills it from its collector. */ +const PENDING_SNAPSHOT: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, +}; + +const failed = (failure: AgentFailure): RunResult => ({ + outcome: 'failed', + failure, + snapshot: PENDING_SNAPSHOT, +}); + +export async function runOrchestrator({ + config, + input, + boot, + emit, + interaction, + signal, +}: SequenceContext): Promise { const runId = randomUUID(); + const { run } = config; + const programId = config.programId; + + if (signal.aborted) { + return { + outcome: 'cancelled', + failure: { message: 'Wizard setup cancelled.' }, + snapshot: PENDING_SNAPSHOT, + }; + } // Switchboard context — reused for every per-role harness resolution below. - const switchboardCtx = { - program: programConfig.id, - flags: boot.wizardFlags, - flagPayloads: boot.wizardFlagPayloads, - cliHarness: session.harness, - cliSequence: session.sequence, - cliModel: session.model, - }; + // The caller resolved the run-level binding from it; per-task roles overlay + // `binding.contextMillOverride[role]` on the same inputs. + const switchboardCtx = { ...config.switchboard, trace: undefined }; // The WHAT (agent prompts) is served from context-mill. Fetch the registry // once up front: its types drive enqueue validation, and resolving a task to // its run config is then synchronous, with no mid-drain network latency. - const flow = programConfig.agentFlow ?? programConfig.id; + const flow = config.agentFlow ?? programId; const registry = await loadAgentRegistry(boot.skillsBaseUrl, flow, { exclude: ciExcludedTaskTypes(), // Baked into the prompts at load, so enqueue, dispatch, and telemetry all read one effective spec. overrides: resolveStageOverrides( - programConfig.id, + programId, boot.wizardFlags, boot.wizardFlagPayloads, ), @@ -492,7 +525,7 @@ export async function runOrchestrator( ? Date.parse(t.finishedAt) - Date.parse(t.startedAt) : undefined; - const store = new QueueStore(session.installDir, runId, { + const store = new QueueStore(input.installDir, runId, { onTransition: (event, task) => { const pick = resolveHarness(switchboardCtx, task.type); // Mirror dispatch's allow-list fallback so attribution names the model that runs. @@ -565,20 +598,20 @@ export async function runOrchestrator( let commandmentsPath: string | undefined; let referenceInstallPath: string | undefined; const menuSkillEntries = await fetchSkillMenuEntries(boot.skillsBaseUrl); - // The framework key for reference + variant resolution. `session.integration` - // is the detected framework and always wins; `session.skillId` is the - // fallback for the basic-integration path, where bootstrap sets it to the + // The framework key for reference + variant resolution. `input.integration` + // is the detected framework and always wins; `input.skillId` is the + // fallback for the basic-integration path, where the caller sets it to the // framework label. Programs whose run config carries their own skill id // (agent-skill commands like replay-vision) would otherwise leak that id in - // here as a bogus framework after bootstrap overwrites the detect result. - const framework = session.integration ?? session.skillId ?? undefined; + // here as a bogus framework after the caller overwrites the detect result. + const framework = input.integration ?? input.skillId ?? undefined; const referenceSkillId = framework ? resolveReferenceSkillId(menuSkillEntries, framework) : undefined; if (referenceSkillId) { const ref = await installSkillById( referenceSkillId, - session.installDir, + input.installDir, boot.skillsBaseUrl, { skillsRoot: path.join(QUEUE_DIR_NAME, 'reference'), @@ -588,11 +621,11 @@ export async function runOrchestrator( if (ref.kind === 'ok') { referenceInstallPath = ref.path; const example = path.join(ref.path, 'references', 'EXAMPLE.md'); - if (existsSync(path.join(session.installDir, example))) { + if (existsSync(path.join(input.installDir, example))) { examplePath = example; } const commandments = path.join(ref.path, 'references', 'COMMANDMENTS.md'); - if (existsSync(path.join(session.installDir, commandments))) { + if (existsSync(path.join(input.installDir, commandments))) { commandmentsPath = commandments; } } else { @@ -627,11 +660,9 @@ export async function runOrchestrator( } } if (missingVariants.length > 0) { - // The framework's own docs page from its config; generic docs when detection found none. - const docsUrl = framework - ? FRAMEWORK_REGISTRY[framework as Integration]?.metadata.docsUrl - : undefined; - await wizardAbort({ + // The framework's own docs page, resolved by the caller; generic docs when detection found none. + const docsUrl = framework ? input.frameworkDocsUrl : undefined; + return failed({ code: ErrorCodes.AgentOrchestratorSkillVariantMissing, message: 'Setup instructions for this project failed to download.\n' + @@ -662,27 +693,28 @@ export async function runOrchestrator( }; logToFile( - `[orchestrator] START program=${programConfig.id} dir=${session.installDir} run=${runId}`, + `[orchestrator] START program=${programId} dir=${input.installDir} run=${runId}`, ); analytics.wizardCapture('orchestrator started', { - program_id: programConfig.id, + program_id: programId, }); - getUI().startRun(); + emit({ kind: 'lifecycle', phase: 'started' }); // Label precedence: what the orchestrator set at enqueue, then the agent // prompt's default, then the bare type. const labelFor = (t: { type: string; label?: string }) => t.label ?? registry.get(t.type)?.label ?? t.type; const renderQueue = () => - getUI().syncTodos( - displayOrder(store.list(), (t) => + emit({ + kind: 'tasks', + tasks: displayOrder(store.list(), (t) => registry.runnerSeededTypes.includes(t.type), ).map((t) => ({ content: labelFor(t), status: toTodoStatus(t.status), activeForm: labelFor(t), })), - ); + }); // Each task's run binds the wizard-tools MCP server to a per-task // orchestrator context so complete_task / enqueue_task attribute correctly @@ -709,7 +741,7 @@ export async function runOrchestrator( // Kill switch: off (or unset), the wizard queues nothing itself and the run // is byte-identical to a project with no detected sources. const seedEntries = areSeededTasksEnabled(boot.wizardFlags) - ? programConfig.seedTasks?.(session) ?? [] + ? config.seedTasks?.() ?? [] : []; const seededTypes: string[] = []; // Kept so their dependencies can be resolved once the planner has run — they @@ -758,7 +790,13 @@ export async function runOrchestrator( // not, which is why the offer lives here and not there. seededConsent.set( task.id, - await askSeededConsent(seeded.type, seeded.notice), + await askSeededConsent( + seeded.type, + seeded.notice, + undefined, + interaction, + signal, + ), ); } logToFile(`[orchestrator] runner-seeded task ${seeded.type}`); @@ -790,11 +828,11 @@ export async function runOrchestrator( // One bridge for the run, handed only to a task whose prompt allows asking. // Absent in CI and signup, where nobody can answer. - const askBridge = shouldDisableAsk(session) + const askBridge = shouldDisableAsk(input.flags) ? undefined - : createWizardAskBridge({ - getSource: () => session.skillId ?? programConfig.id, - showQuestion: (q) => { + : createAskBridge(interaction, signal, { + getSource: () => input.skillId ?? programId, + beforeShow: () => { // How late the first ask lands is the measure of this run shape: it // should follow the autonomous work, not interrupt it. metrics.recordAsk(Date.now()); @@ -804,16 +842,14 @@ export async function runOrchestrator( // unanswered ask still times out into the deep-link fallback — it just // gives a person who stepped away a chance to come back first. ringTerminalBell(); - return getUI().requestQuestion(q); }, - cancelQuestion: () => getUI().cancelPendingQuestion(), - richLinks: config.richLinks ?? false, + richLinks: run.richLinks ?? false, // A task ask waits on a person, and the drain waits it out — the // executor holds the task's promise — so this is the only real limit. timeoutMs: LONGER_ASK_TIMEOUT_MS, - }); + })?.bridge; - const spinner = getUI().spinner(); + const spinner = createEmitSpinner(emit); // 1. Seed the queue with the orchestrator agent. It is itself an agent prompt // (the WHAT), so its model and tools come from its frontmatter. The seed @@ -826,9 +862,10 @@ export async function runOrchestrator( const seedHarness = requireTaskHarness(seedPick); const seedModel = promptModelFor(seedPrompt, seedPick.harness); const seedResult = await seedHarness.runTask({ - session, - programConfig, + config, + input, boot, + emit, prompt: assembleSeedPrompt(promptContext, seedPrompt.body, store.list()), spinner, model: requireKnownModel(seedModel.model, seedPick.model), @@ -841,6 +878,9 @@ export async function runOrchestrator( requestRemark: false, analyticsProperties: { task_type: 'seed', harness: seedPick.harness }, }); + // A decided failure (a 401 the harness already reported) ends the run here, + // before anything is queued, exactly where the harness used to exit. + if (seedResult.failure) return failed(seedResult.failure); if (seedResult.error) { logToFile( `[orchestrator] seed error: ${seedResult.error} ${ @@ -941,7 +981,7 @@ export async function runOrchestrator( analytics.wizardCapture('orchestrator sink invariant violated', { uncovered_types: unwaited.map((t) => t.type), }); - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorSinkInvariant, message: `The wizard could not plan this setup: the final step would have skipped ${unwaited .map((t) => t.type) @@ -965,7 +1005,7 @@ export async function runOrchestrator( // Task agents can install durable skills mid-run (load_skill), and only the // framework reference docs earn a place — snapshot what was already there so // the sweeps remove exactly what this run added. - const claudeSkillsDir = path.join(session.installDir, '.claude', 'skills'); + const claudeSkillsDir = path.join(input.installDir, '.claude', 'skills'); const preexistingSkills = new Set( existsSync(claudeSkillsDir) ? readdirSync(claudeSkillsDir) : [], ); @@ -998,7 +1038,7 @@ export async function runOrchestrator( } const result = await installSkillById( variantId, - session.installDir, + input.installDir, boot.skillsBaseUrl, { skillsRoot: taskSkillsRoot, triage: boot.triageProvider }, ); @@ -1029,10 +1069,11 @@ export async function runOrchestrator( const taskPick = resolveHarness(switchboardCtx, task.type); const taskHarness = requireTaskHarness(taskPick); const taskModel = taskModelSpec(registry, task, taskPick.harness); - await taskHarness.runTask({ - session, - programConfig, + const taskResult = await taskHarness.runTask({ + config, + input, boot, + emit, prompt: assembleTaskPrompt(promptContext, resolved.prompt, skillPaths), spinner, model: requireKnownModel(taskModel.model, taskPick.model), @@ -1051,6 +1092,10 @@ export async function runOrchestrator( harness: taskPick.harness, }, }); + // A decided failure (a 401 the harness already reported) is the run's, + // not the task's: stop the drain and report it, where the harness used + // to exit the process. + if (taskResult.failure) throw new RunTaskFatal(taskResult.failure); } finally { // Durable skills a task installed are irrelevant to later tasks — and // the sdk harness auto-loads .claude/skills into every agent — so sweep @@ -1074,13 +1119,17 @@ export async function runOrchestrator( renderQueue(); } + let fatal: AgentFailure | undefined; try { await drainQueue(store, runTask); + } catch (error) { + if (!(error instanceof RunTaskFatal)) throw error; + fatal = error.failure; } finally { try { if (referenceSkillId && referenceInstallPath) { promoteReferenceSkill( - path.join(session.installDir, referenceInstallPath), + path.join(input.installDir, referenceInstallPath), claudeSkillsDir, referenceSkillId, ); @@ -1095,7 +1144,7 @@ export async function runOrchestrator( // cache folder (queue, handoffs, reference example, installed task // instructions). The .DELETE-ME.md inside is the fallback if we don't. try { - rmSync(path.join(session.installDir, QUEUE_DIR_NAME), { + rmSync(path.join(input.installDir, QUEUE_DIR_NAME), { recursive: true, force: true, }); @@ -1119,6 +1168,8 @@ export async function runOrchestrator( } } + if (fatal) return failed(fatal); + renderQueue(); const summary = store.summary(); @@ -1172,7 +1223,7 @@ export async function runOrchestrator( ', ', )}.\n\nPlease try again, approving all permissions on the PostHog authorization screen. If it still fails, report it to: ${WIZARD_CONTACT_EMAIL}` : `The wizard was unable to set up PostHog: ${whatFailed}.\n\nPlease report this to: ${WIZARD_CONTACT_EMAIL}`; - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorTasksFailed, message, error: new WizardError( @@ -1193,7 +1244,7 @@ export async function runOrchestrator( // usable model response (e.g. the gateway returned empty completions). // "0/0 completed" is a dead run, not a success with an empty denominator. if (summary.total === 0) { - await wizardAbort({ + return failed({ code: ErrorCodes.AgentOrchestratorHollowRun, message: `The wizard was unable to set up PostHog: the planning step produced no work, so nothing ran.\n\nPlease try again — and if it happens again, report it to: ${WIZARD_CONTACT_EMAIL}`, error: new WizardError( @@ -1219,19 +1270,20 @@ export async function runOrchestrator( } steps completed${ stepNotes.length > 0 ? ` (${stepNotes.join(', ')})` : '' }.`; - getUI().setOutroData({ + const outro = { kind: OutroKind.Success, message, body: conflict ? `⚠ Build conflict: ${conflict}\nFull details are in the setup report.` : undefined, docsUrl: 'https://posthog.com/docs/ai-engineering/ai-wizard', - nextSteps: config.buildOutroNextSteps?.( - session, + nextSteps: config.hooks?.buildOutroNextSteps?.( boot.credentials, completedSeededTypes(store, seededTasks), ), - }); - getUI().outro(message); + }; + emit({ kind: 'completion', outro }); + emit({ kind: 'lifecycle', phase: 'completed', message }); await analytics.shutdown('success'); + return { outcome: 'success', outro, snapshot: PENDING_SNAPSHOT }; } diff --git a/src/lib/agent/runner/shared/ask.ts b/src/lib/agent/runner/shared/ask.ts new file mode 100644 index 000000000..640c0a781 --- /dev/null +++ b/src/lib/agent/runner/shared/ask.ts @@ -0,0 +1,61 @@ +/** + * The ask bridge over an injected answerer. + * + * `createWizardAskBridge` already owns request ids, the timeout race, the + * `__cancelled__` sentinel and the analytics; it only ever needed a + * `showQuestion` and a `cancelQuestion`. Here those come from + * `AgentInteraction` instead of `getUI()`. With no answerer there is no bridge, + * so `wizard_ask` reports its existing "not available" error rather than + * hanging on a question nobody can see. + */ + +import { + createWizardAskBridge, + type WizardAskBridge, +} from '@lib/wizard-ask-bridge'; +import type { PendingQuestion } from '@lib/wizard-session'; +import type { AgentInteraction } from '@lib/agent/progress'; + +export interface AskBridgeHandle { + bridge: WizardAskBridge; + /** + * The question currently on screen, or null. The anthropic harness reads it + * to block Write/Edit while an overlay is open, the same guard pi keeps in + * its security extension. + */ + getPendingQuestion: () => PendingQuestion | null; +} + +export function createAskBridge( + interaction: AgentInteraction | undefined, + signal: AbortSignal, + options: { + getSource: () => string; + richLinks: boolean; + timeoutMs?: number; + /** Runs before each question is shown (the orchestrator's bell and metric). */ + beforeShow?: () => void; + }, +): AskBridgeHandle | undefined { + const ask = interaction?.ask; + if (!ask) return undefined; + + let pending: PendingQuestion | null = null; + const bridge = createWizardAskBridge({ + getSource: options.getSource, + showQuestion: async (question) => { + options.beforeShow?.(); + pending = question; + try { + return await ask(question, { signal }); + } finally { + pending = null; + } + }, + cancelQuestion: interaction?.cancelAsk, + richLinks: options.richLinks, + timeoutMs: options.timeoutMs, + }); + + return { bridge, getPendingQuestion: () => pending }; +} diff --git a/src/lib/agent/runner/shared/bootstrap.ts b/src/lib/agent/runner/shared/bootstrap.ts index dc12a2b60..1b776a87c 100644 --- a/src/lib/agent/runner/shared/bootstrap.ts +++ b/src/lib/agent/runner/shared/bootstrap.ts @@ -1,46 +1,22 @@ /** - * Shared bootstrap for the runner pipeline. + * Shared preparation for the runner pipeline. * - * Runs before the fork into the linear or orchestrator arm: logging, health - * check, settings conflicts, OAuth and credentials, feature flags, variant - * metadata, and MCP url. Sets `session.credentials`, role, and user as side - * effects. Returns the values the arms still need. + * Runs before the fork into the linear or orchestrator arm: logging targets, + * the gateway mint and the scan-triage classifier built on it. Everything the + * caller must decide first — health gates, settings conflicts, authentication, + * the AI opt-in gate, post-auth gates, feature flags, run tags, token refresh — + * arrives already resolved in `RunConfig` and `RunInput`. */ -import type { WizardSession } from '@lib/wizard-session'; import { analytics } from '@utils/analytics'; -import { getUI } from '@ui'; -import { authenticate, refreshAccessTokenIfNeeded } from './authenticate'; -import { maybeStampAiSdkDetected } from '@lib/programs/posthog-integration/detect'; import { createTriageLLMProvider } from '@lib/agent/triage-provider'; import { gatewayAuth } from '@lib/gateway-session'; -import { resolveHarness } from '../switchboard'; -import { buildRunTags } from '@lib/agent/agent-interface'; -import { - checkAllSettingsConflicts, - backupAndFixClaudeSettings, - classifySettingsConflicts, -} from '@lib/agent/claude-settings'; -import { - evaluateWizardReadiness, - WizardReadiness, - SIGNUP_WIZARD_READINESS_CONFIG, - getBlockingServiceKeys, - SERVICE_LABELS, -} from '@lib/health-checks/readiness'; import { enableDebugLogs, logToFile, initLogFile } from '@utils/debug'; -import { wizardAbort } from '@utils/wizard-abort'; -import { ErrorCodes } from '@lib/errors'; -import { isNonInteractiveEnvironment } from '@utils/environment'; -import { CallType, getSkillsBaseUrl, IS_DEV } from '@lib/constants'; +import { CallType, IS_DEV } from '@lib/constants'; import { VERSION } from '@lib/version'; import { mcpUrlFor } from '@lib/host-resolution'; import type { WizardRunOptions } from '@utils/types'; -import { - postAuthGateSteps, - type ProgramConfig, -} from '@lib/programs/program-step'; -import type { ProgramRun, BootstrapResult } from './types'; +import type { BootstrapResult, RunConfig, RunFlags, RunInput } from './types'; // ── Helpers ────────────────────────────────────────────────────────── @@ -51,7 +27,7 @@ import type { ProgramRun, BootstrapResult } from './types'; * the program's `disallowedTools` so the SDK rejects calls outright. * Extracted so the policy can be unit-tested directly. * - * `session.e2eAsk` is the one escape hatch. The e2e harness runs a `ci` + * `e2eAsk` is the one escape hatch. The e2e harness runs a `ci` * session, but it does have an answerer — the driver loop answers each * `wizard_ask` batch from the program's e2e profile. Without the flag the * agent-in-the-loop layer (the ask bridge in both sequence arms, and the @@ -61,254 +37,68 @@ import type { ProgramRun, BootstrapResult } from './types'; * populates it, so plain `--ci` and `--signup` runs behave exactly as before. */ export function shouldDisableAsk( - session: Pick, + flags: Pick, ): boolean { - return (session.ci || session.signup) && !session.e2eAsk; + return (flags.ci || flags.signup) && !flags.e2eAsk; } -export function sessionToOptions(session: WizardSession): WizardRunOptions { +/** The option bag the agent interface and the middleware read. */ +export function runOptions(input: RunInput): WizardRunOptions { return { - installDir: session.installDir, - debug: session.debug, - signup: session.signup, - ci: session.ci, - benchmark: session.benchmark, - projectId: session.projectId, - apiKey: session.apiKey, - yaraReport: session.yaraReport, + installDir: input.installDir, + debug: input.flags.debug, + signup: input.flags.signup, + ci: input.flags.ci, + benchmark: input.flags.benchmark, + projectId: input.host.projectId, + apiKey: input.host.apiKey, + yaraReport: input.flags.yaraReport, }; } -// ── Bootstrap ───────────────────────────────────────────────────────── +// ── Prepare ─────────────────────────────────────────────────────────── /** - * Shared setup for both arms: logging, health check, settings conflicts, OAuth - * and credentials, then the feature flags, variant metadata, and MCP url. Sets - * `session.credentials`, role, and user as a side effect. Returns the values the - * arms still need. + * Shared setup for both arms: logging targets, then the gateway mint and the + * triage classifier. Throws when the mint is refused, so the run fails before + * any agent starts — the caller maps that the way it maps any unexpected error. */ -export async function bootstrapProgram( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, +export async function prepareRun( + config: RunConfig, + input: RunInput, ): Promise { + const { run } = config; + // 1. Init logging + debug initLogFile(); - session.skillId = config.skillId ?? config.integrationLabel; logToFile( - `[agent-runner] START ${config.integrationLabel} build=${analytics.build}` + - `${session.ci ? ' (non-interactive)' : ''}`, + `[agent-runner] START ${run.integrationLabel} build=${analytics.build}` + + `${input.flags.ci ? ' (non-interactive)' : ''}`, ); - if (session.debug) { + if (input.flags.debug) { enableDebugLogs(); } - const skillsBaseUrl = getSkillsBaseUrl(); + const { skillsBaseUrl } = config; // Where this run actually points. The three services switch independently, // so otherwise "why did it use prod skills?" means reading three call sites. logToFile( `[agent-runner] targets build=${VERSION}${IS_DEV ? '/dev' : ''} ` + `skills=${skillsBaseUrl} ` + - `mcp=${mcpUrlFor(session.localMcp)} ` + - `posthog=${session.baseUrl ?? 'region-resolved'}`, - ); - - // 2. Health check (guarded — skip if TUI already ran it). Only - // programs that declare a health-check screen get pre-flight checks; - // for everything else the checks never fire and never block. - const hasHealthCheckScreen = programConfig.steps.some( - (s) => s.screenId === 'health-check', - ); - if (session.readinessResult) { - logToFile( - `[agent-runner] readiness pre-computed by TUI: decision=${session.readinessResult.decision}` + - `${ - session.outageDismissed ? ' (outage dismissed by user)' : '' - } — skipping re-check`, - ); - } - if (hasHealthCheckScreen && !session.readinessResult) { - logToFile('[agent-runner] evaluating wizard readiness'); - const readinessConfig = session.signup - ? SIGNUP_WIZARD_READINESS_CONFIG - : undefined; - const readiness = await evaluateWizardReadiness(readinessConfig); - logToFile(`[agent-runner] readiness=${readiness.decision}`); - if (readiness.decision === WizardReadiness.No) { - const blockingKeys = getBlockingServiceKeys( - readiness.health, - readinessConfig, - ); - const blockingLabels = blockingKeys.map( - (k) => `${SERVICE_LABELS[k]} (${readiness.health[k].status})`, - ); - logToFile(`[agent-runner] blocked by: ${blockingLabels.join(', ')}`); - - await getUI().showBlockingOutage(readiness); - - // The TUI lets the user continue past an outage; non-interactive runs - // (CI) do the same automatically — the degraded services are reported - // above, but we proceed rather than aborting on a transient upstream blip. - if (!isNonInteractiveEnvironment()) { - await wizardAbort({ - code: ErrorCodes.EnvServiceOutage, - message: - 'Cannot start — external services are down:\n' + - blockingLabels.map((l) => ` - ${l}`).join('\n') + - '\n\nPlease try again later.', - }); - } - } else if (readiness.decision === WizardReadiness.YesWithWarnings) { - getUI().setReadinessWarnings(readiness); - } - } - - // 3. Settings conflicts - const settingsConflicts = checkAllSettingsConflicts(session.installDir); - logToFile( - `[agent-runner] settings conflicts: ${ - settingsConflicts.length > 0 - ? settingsConflicts - .map((c) => `${c.source}(${c.keys.join(',')})`) - .join('; ') - : 'none' - }`, + `mcp=${mcpUrlFor(input.flags.localMcp)} ` + + `posthog=${input.host.baseUrl ?? 'region-resolved'}`, ); - if (settingsConflicts.length > 0) { - for (const conflict of settingsConflicts) { - const level = conflict.source === 'managed' ? 'org' : conflict.source; - analytics.wizardCapture('settings conflict detected', { - level, - keys: conflict.keys, - }); - } - - const { autoFix, failClosed, warnOnly } = - classifySettingsConflicts(settingsConflicts); - - // User-global and project-local files are already neutralized — the agent - // runs with settingSources:['project'], so the SDK never reads them. Record - // it and move on; don't make the user act on a setting that can't bite. - for (const conflict of warnOnly) { - logToFile( - `[agent-runner] settings conflict in ${conflict.source} (${conflict.path}) ` + - `neutralized by settingSources:['project'] — not blocking`, - ); - analytics.wizardCapture('settings conflict neutralized', { - level: conflict.source, - keys: conflict.keys, - }); - } - - // Writable project settings.json — the SDK *does* read it, but we can back - // it up and remove it (restored at outro). Neutralize without prompting. - let unfixable = failClosed; - if (autoFix.length > 0) { - const fixed = backupAndFixClaudeSettings(session.installDir); - if (fixed) { - logToFile('[agent-runner] auto-neutralized writable settings conflict'); - analytics.wizardCapture('settings conflict auto-neutralized', { - keys: autoFix.flatMap((c) => c.keys), - }); - } else { - // Couldn't remove it — don't run into the redirect; fail closed instead. - logToFile( - '[agent-runner] could not back up writable settings conflict — failing closed', - ); - unfixable = [...failClosed, ...autoFix]; - } - } - - // What we cannot neutralize (org-managed, always read by the SDK; or a - // writable file we failed to back up) must be fixed by the user. Fail - // closed: the screen names the file + keys and exits. - if (unfixable.length > 0) { - if (isNonInteractiveEnvironment()) { - await wizardAbort({ - code: ErrorCodes.SettingsUnfixableConflict, - message: - 'Cannot start — a Claude settings file redirects the agent away ' + - 'from the PostHog gateway and cannot be neutralized automatically:\n' + - unfixable - .map((c) => ` - ${c.source} (${c.path}): ${c.keys.join(', ')}`) - .join('\n') + - '\n\nRemove the conflicting keys and re-run the wizard.', - }); - } - await getUI().showSettingsOverride(unfixable, () => - backupAndFixClaudeSettings(session.installDir), - ); - logToFile('[agent-runner] settings override resolved'); - } - } - - analytics.wizardCapture('agent started', { - integration: config.integrationLabel, - program_id: programConfig.id, - skill_id: config.skillId ?? null, - }); - - // 4. Authenticate — idempotent within a run (see authenticate()). A second - // agent run in the same invocation (self-driving's integration phase) reuses - // the first login; it does not launch another OAuth. authenticate() also - // identifies the user and sets analytics groups. - await authenticate(session, programConfig.id); - maybeStampAiSdkDetected(session); - const project = session.apiProject; - - // 4.5. AI opt-in enforcement. Parks here while AiOptInRequiredScreen is - // up if the org hasn't approved third-party AI — BEFORE the skill - // install and agent start, so no source leaves the machine. The screen - // alone is cosmetic; this await is the actual gate. Resolves - // immediately when the program declared requiresAi: false or in CI. - // In bootstrapProgram so both the linear and orchestrator arms gate. - logToFile('[agent-runner] checking AI opt-in gate'); - await getUI().waitForAiOptIn(); - logToFile('[agent-runner] AI opt-in gate cleared'); - - // Park for any interactive step the user must complete AFTER authenticating - // but BEFORE the agent runs — e.g. the source-maps project picker, which - // needs credentials to scan and writes its choice to frameworkContext that - // the run prompt reads. Generic: await every gated step between auth and run. - for (const step of postAuthGateSteps(programConfig.steps)) { - logToFile(`[agent-runner] awaiting post-auth gate: ${step.id}`); - await getUI().waitForGate(step.id); - logToFile(`[agent-runner] post-auth gate cleared: ${step.id}`); - } - - // Feature flags. Both arms need these, and the fork decision reads the flags. - // This map is PostHog-side only — CLI `--harness` / `--sequence` precedence - // lives at the resolution sites (`runner/index.ts` for sequence, - // `resolveHarness` for harness), not here. - const wizardFlags = await analytics.getAllFlagsForWizard(); - const wizardFlagPayloads = analytics.getWizardFlagPayloads(); - - // Gateway trace tags for this run. The runner stamps its variant onto this - // after the fork (see runProgram), so the value reflects which arm ran. - const wizardMetadata = buildRunTags({ - programId: programConfig.id, - integration: config.integrationLabel, - runId: analytics.runId, - build: analytics.build, - skillId: config.skillId, - }); - - // The agent can't swap tokens mid-run, so freshness is measured after every park above, right before the mint. - await refreshAccessTokenIfNeeded(session); - - // Credentials (incl. the resolved host family and its MCP url) live on - // `session.credentials`; narrow once at this boundary — `authenticate` above - // set them — so downstream readers get a non-null type without asserting. - const credentials = session.credentials!; + const { credentials } = input; + const { wizardFlags, wizardFlagPayloads, wizardMetadata, programId } = config; // Mint now so a refusal fails the boot before any agent starts. Later // readers re-resolve through the cache, which re-mints past the refresh // point. const currentGatewayAuth = () => - gatewayAuth(credentials.host, credentials.accessToken, programConfig.id); + gatewayAuth(credentials.host, credentials.accessToken, programId); await currentGatewayAuth(); return { @@ -316,33 +106,24 @@ export async function bootstrapProgram( credentials, // Carried so per-task sessions re-resolve against the same program the boot // minted for, rather than digging it back out of the metadata bag. - programId: programConfig.id, + programId, wizardFlags, wizardFlagPayloads, wizardMetadata, - project, - // Resolved once, here: the only place holding both the switchboard inputs + project: input.project, + // Resolved once, here: the only place holding both the run-level harness // and the gateway auth. Every skill install downstream reads it off boot. - triageProvider: createTriageLLMProvider( - async () => { - const auth = await currentGatewayAuth(); - return { - baseURL: auth.gatewayUrl, - authToken: auth.token, - teamId: auth.teamId, - // `call_type` splits scan spend out of the program's agent cost, - // the same tag the in-run triage provider carries. - wizardMetadata: { ...wizardMetadata, call_type: CallType.yaraTriage }, - wizardFlags, - }; - }, - resolveHarness({ - program: programConfig.id, - flags: wizardFlags, - flagPayloads: wizardFlagPayloads, - cliHarness: session.harness, - cliModel: session.model, - }).harness, - ), + triageProvider: createTriageLLMProvider(async () => { + const auth = await currentGatewayAuth(); + return { + baseURL: auth.gatewayUrl, + authToken: auth.token, + teamId: auth.teamId, + // `call_type` splits scan spend out of the program's agent cost, + // the same tag the in-run triage provider carries. + wizardMetadata: { ...wizardMetadata, call_type: CallType.yaraTriage }, + wizardFlags, + }; + }, config.binding.harness), }; } diff --git a/src/lib/agent/runner/shared/errors.ts b/src/lib/agent/runner/shared/errors.ts index 376cb45d4..b3778a9b1 100644 --- a/src/lib/agent/runner/shared/errors.ts +++ b/src/lib/agent/runner/shared/errors.ts @@ -4,14 +4,14 @@ import type { InstallSkillResult } from '@lib/wizard-tools'; import { skillErrorCode } from '@lib/errors'; -import { wizardAbort, WizardError } from '@utils/wizard-abort'; +import { WizardError } from '@lib/errors/wizard-error'; +import type { AgentFailure } from './types'; -export async function abortOnInstallFailure( +/** The failure a skill install error decides. The caller reports and exits. */ +export function installFailure( integrationLabel: string, - result: InstallSkillResult, -): Promise { - if (result.kind === 'ok') return; - + result: Exclude, +): AgentFailure { const code = skillErrorCode(result) ?? undefined; const message = (() => { @@ -25,7 +25,7 @@ export async function abortOnInstallFailure( } })(); - await wizardAbort({ + return { message, code, error: new WizardError( @@ -40,5 +40,5 @@ export async function abortOnInstallFailure( }, code, ), - }); + }; } diff --git a/src/lib/agent/runner/shared/progress-collector.ts b/src/lib/agent/runner/shared/progress-collector.ts new file mode 100644 index 000000000..bdbde70f2 --- /dev/null +++ b/src/lib/agent/runner/shared/progress-collector.ts @@ -0,0 +1,120 @@ +/** + * The agent's own record of what it reported. + * + * `runAgent` accumulates its final `RunSnapshot` here, independently of any + * observer, so a missing or throwing `onProgress` never makes the result + * incomplete. The emitter wraps the caller's callback: it applies the event + * here first, then hands a copy to the observer, and logs rather than + * propagates anything the observer throws. + */ + +import { logToFile } from '@utils/debug'; +import type { + AgentProgress, + ProgressEmitter, + SpinnerHandle, +} from '@lib/agent/progress'; +import type { RunSnapshot } from './types'; + +export interface ProgressCollector { + emit: ProgressEmitter; + snapshot(): RunSnapshot; +} + +export function createProgressCollector( + onProgress?: (event: AgentProgress) => void, +): ProgressCollector { + const snapshot: RunSnapshot = { + tasks: [], + statusMessages: [], + usage: { + inputTokens: 0, + outputTokens: 0, + cacheReadTokens: 0, + cacheCreationTokens: 0, + }, + }; + + const apply = (event: AgentProgress): void => { + switch (event.kind) { + case 'tasks': + snapshot.tasks = event.tasks.map((t) => ({ ...t })); + break; + case 'status': + snapshot.statusMessages.push(event.message); + break; + case 'stage': + snapshot.stage = event.stage; + break; + case 'url': + if (event.which === 'dashboard') snapshot.dashboardUrl = event.url; + else snapshot.notebookUrl = event.url; + break; + case 'usage': + snapshot.usage.inputTokens += event.delta.inputTokens; + snapshot.usage.outputTokens += event.delta.outputTokens; + snapshot.usage.cacheReadTokens += event.delta.cacheReadTokens; + snapshot.usage.cacheCreationTokens += event.delta.cacheCreationTokens; + break; + case 'finalCost': + snapshot.finalCostUsd = event.usd; + break; + case 'handoff': + snapshot.handoffText = event.text; + break; + default: + break; + } + }; + + const emit: ProgressEmitter = (event) => { + apply(event); + if (!onProgress) return; + try { + onProgress(event); + } catch (error) { + // A broken projection is the host's problem, not the run's. Say so in + // the log and carry on; the snapshot above is the source of truth. + logToFile( + `[agent] progress observer threw on ${event.kind}:`, + error instanceof Error ? error.message : error, + ); + } + }; + + return { + emit, + snapshot: () => ({ + ...snapshot, + tasks: snapshot.tasks.map((t) => ({ ...t })), + statusMessages: [...snapshot.statusMessages], + usage: { ...snapshot.usage }, + }), + }; +} + +/** The run spinner as a progress emitter. One per run, like `getUI().spinner()`. */ +export function createEmitSpinner(emit: ProgressEmitter): SpinnerHandle { + return { + start: (message) => emit({ kind: 'spinner', action: 'start', message }), + stop: (message) => emit({ kind: 'spinner', action: 'stop', message }), + message: (message) => emit({ kind: 'spinner', action: 'message', message }), + }; +} + +/** `WizardUI.log` as a progress emitter. */ +export function createEmitLog(emit: ProgressEmitter): { + info(message: string): void; + warn(message: string): void; + error(message: string): void; + success(message: string): void; + step(message: string): void; +} { + return { + info: (message) => emit({ kind: 'log', level: 'info', message }), + warn: (message) => emit({ kind: 'log', level: 'warn', message }), + error: (message) => emit({ kind: 'log', level: 'error', message }), + success: (message) => emit({ kind: 'log', level: 'success', message }), + step: (message) => emit({ kind: 'log', level: 'step', message }), + }; +} diff --git a/src/lib/agent/runner/shared/types.ts b/src/lib/agent/runner/shared/types.ts index 4f4292f83..ae0123ce0 100644 --- a/src/lib/agent/runner/shared/types.ts +++ b/src/lib/agent/runner/shared/types.ts @@ -1,16 +1,30 @@ /** - * Shared types for the runner pipeline. + * The agent's run contracts. + * + * `runAgent(config, input, options)` takes resolved execution data and an + * invocation snapshot, reports through `options.onProgress`, asks through + * `options.interaction`, and returns a `RunResult`. Nothing here names a UI, + * a store, a session or a program registry: the caller resolves those and + * hands over plain data. `src/lib/programs/run-agent-legacy.ts` is the caller + * that rebuilds today's session-driven behavior on top of this contract. */ import type { - Credentials, AdditionalFeature, - WizardSession, + CloudRegion, + Credentials, + OutroData, + TaskNotice, } from '@lib/wizard-session'; import type { PromptContext } from '@lib/agent/agent-prompt'; import type { PackageManagerDetector } from '@lib/detection/package-manager'; -import type { ApiProject } from '@lib/api'; +import type { ApiProject, ApiUser } from '@lib/api'; +import type { Harness, Integration, Sequence } from '@lib/constants'; +import type { ErrorCode } from '@lib/errors'; import type { LLMProvider } from '@posthog/warlock'; +import type { AgentInteraction, ProgressEmitter } from '@lib/agent/progress'; +import type { EffortLevel } from '../switchboard/models'; +import type { SwitchboardCtx } from '../switchboard'; export type { PromptContext, Credentials }; @@ -23,17 +37,18 @@ export interface AbortCase { message: string; body: string; docsUrl?: string; - errorCode?: import('@lib/errors').ErrorCode; + errorCode?: ErrorCode; } /** - * Unified agent run configuration. + * What the agent reads of a program's run definition. * - * Every program provides one of these — either as a static object - * or via a function that builds one from the session. The runner - * assembles the final prompt from `prompt` + `skillId`. + * A program's `ProgramRun` (`src/lib/programs/program-run.ts`) extends this + * with hooks that take the wizard session; the caller binds those and hands + * the agent `RunConfig.hooks` instead. Everything below is plain data or a + * pure function of agent-side context. */ -export interface ProgramRun { +export interface AgentRunDefinition { /** Analytics label (e.g. 'revenue-analytics-setup', 'nextjs') */ integrationLabel: string; /** Skill ID to pre-install. Omit for agent-driven skill discovery. */ @@ -53,33 +68,6 @@ export interface ProgramRun { additionalFeatureQueue?: readonly AdditionalFeature[]; /** Known `[ABORT] ` cases this program can render. */ abortCases?: AbortCase[]; - /** Runs after agent completes, before outro (e.g. env var upload). */ - postRun?: (session: WizardSession, credentials: Credentials) => Promise; - /** Custom outro data. Omit for default built from successMessage/reportFile/docsUrl. */ - buildOutroData?: ( - session: WizardSession, - credentials: Credentials, - ) => WizardSession['outroData']; - /** - * Outro bullets for a sequence that composes its own outro data. - * - * `buildOutroData` is the linear sequence's seam: it hands the program the - * whole outro. The orchestrated sequence cannot, because its message is the - * drain's result — how many steps ran, what was skipped, which conflict the - * review step left. So a program with next steps to offer had nowhere to put - * them there, and the integration's data-source links were built and then - * dropped on every orchestrated run. This hook keeps the message with the - * sequence and the bullets with the program. - * - * `completedSeededTypes` names the runner-seeded task types that finished - * successfully, so a program can leave out a step its own seeded task - * already did — the sequence stays ignorant of what any type means. - */ - buildOutroNextSteps?: ( - session: WizardSession, - credentials: Credentials, - completedSeededTypes: readonly string[], - ) => { heading: string; items: string[] } | undefined; /** * Per-run cap on `wizard_ask` invocations. Defaults to 10. The 4th call * always returns a "batch your questions" error regardless of the cap. @@ -116,15 +104,132 @@ export interface ProgramRun { resolveStepKey?: (stepName: string | undefined) => string | undefined; } +/** A task the caller queues itself before the orchestrator's planner runs. */ +export interface SeedTaskEntry { + type: string; + label?: string; + inputs?: Record; + /** Shown before the run starts, letting the user decline the task. */ + notice?: TaskNotice; +} + +/** + * Completion hooks the caller binds for the agent. Each receives the run's + * resolved credentials, exactly what the linear and orchestrator sequences + * passed alongside the session before. + */ +export interface RunHooks { + /** Runs after the agent completes, before the outro (linear only). */ + postRun?: (credentials: Credentials) => Promise; + /** Custom outro data (linear only). Omit for the default outro. */ + buildOutroData?: (credentials: Credentials) => OutroData | undefined; + /** Outro bullets for the orchestrated sequence. */ + buildOutroNextSteps?: ( + credentials: Credentials, + completedSeededTypes: readonly string[], + ) => { heading: string; items: string[] } | undefined; +} + +/** The run-level routing decision the caller made. */ +export interface ResolvedBinding { + sequence: Sequence; + harness: Harness; + /** Gateway model id. */ + model: string; + /** Reasoning-effort override. Absent → the model's table default. */ + thinkingLevel?: EffortLevel; +} + +/** + * Resolved execution data for one agent run. The caller has already decided + * which program this is, how it is routed and which flags apply; the agent + * treats every label as opaque. + */ +export interface RunConfig { + /** Program id: gateway spend pin, analytics label, commandments axis. */ + programId: string; + /** The program's run definition, minus the session-bound hooks. */ + run: AgentRunDefinition; + /** A composed sub-run skips the terminal outro and the analytics shutdown. */ + composed: boolean; + /** Run-level sequence, harness and model. */ + binding: ResolvedBinding; + /** + * The inputs the run-level binding was resolved from. The orchestrator + * re-resolves the harness per task role from these; nothing else reads them. + */ + switchboard: SwitchboardCtx; + /** Primary skills origin (context-mill dev or GitHub Releases). */ + skillsBaseUrl: string; + /** Feature flag key → variant, evaluated before the run. */ + wizardFlags: Record; + /** Flag payloads from the same snapshot. */ + wizardFlagPayloads: Record; + /** Gateway trace tags for this run, already stamped with sequence and harness. */ + wizardMetadata: Record; + /** Extra tools added on top of BASE_ALLOWED_TOOLS for this run. */ + allowedTools?: readonly string[]; + /** Tools removed from BASE_ALLOWED_TOOLS for this run. */ + disallowedTools?: readonly string[]; + /** Context-mill flow the orchestrator loads. Defaults to `programId`. */ + agentFlow?: string; + /** Tasks to queue before the orchestrator's planner runs. */ + seedTasks?: () => SeedTaskEntry[]; + /** Completion hooks, bound by the caller. */ + hooks?: RunHooks; +} + +/** Invocation flags the agent reads. */ +export interface RunFlags { + ci: boolean; + signup: boolean; + debug: boolean; + /** Harness-only: keep the ask bridge in a `ci` run that has an answerer. */ + e2eAsk: boolean; + localMcp: boolean; + captureAio: boolean; + benchmark: boolean; + yaraReport: boolean; +} + +/** + * The invocation snapshot for one agent run. Taken once by the caller; the + * agent never refreshes these from a higher layer. + */ +export interface RunInput { + installDir: string; + /** Resolved credentials, including the host family and its MCP url. */ + credentials: Credentials; + /** Project payload resolved at authentication, for prompt context. */ + project: ApiProject | null; + /** User payload resolved at authentication, for the AI opt-in prompt line. */ + apiUser: ApiUser | null; + /** The skill this run is for: the run's skill id, else its integration label. */ + skillId?: string; + /** Detected framework, when the caller has one. */ + integration?: Integration | null; + /** Docs page for the detected framework, for the orchestrator's preflight message. */ + frameworkDocsUrl?: string; + flags: RunFlags; + /** Where PostHog is, as the CLI was told. */ + host: { + baseUrl?: string; + region?: CloudRegion; + email?: string; + /** `--project-id`, for the auth-error classifier. */ + projectId?: number; + /** `--api-key`, for the auth-error classifier. */ + apiKey?: string; + }; +} + /** - * Result of the shared bootstrap, consumed by both the linear and the - * orchestrator arm. `bootstrapProgram` runs `authenticate` before returning, so - * `credentials` is guaranteed non-null here — the single narrowing point owns - * the invariant, and downstream readers get a properly non-null type for free. + * The values `prepareRun` resolves from `RunConfig` + `RunInput` and hands to + * both sequences: the gateway mint and the scan-triage classifier built on it. */ export interface BootstrapResult { skillsBaseUrl: string; - /** Auth outputs (incl. the resolved host family and its MCP url), narrowed at the boundary. */ + /** Resolved credentials (incl. the host family and its MCP url). */ credentials: Credentials; /** Program this run is, and the node its gateway spend pins to. */ programId: string; @@ -137,3 +242,88 @@ export interface BootstrapResult { /** Scan-triage classifier on this run's harness. Undefined → skill scans fail closed. */ triageProvider: LLMProvider | undefined; } + +/** + * A decided failure. The same fields `wizardAbort` takes, so the legacy + * adapter passes it through untouched and the exit sequence, codes and + * messages stay exactly what they were. + */ +export interface AgentFailure { + message?: string; + /** Structured error data. Renders via `outroError` instead of `outro`. */ + outroData?: OutroData; + error?: Error; + exitCode?: number; + code?: ErrorCode; + detail?: Record; +} + +export type RunOutcome = + /** The run finished; `outro` holds what to show. */ + | 'success' + /** The agent chose to stop (`[ABORT]`); `failure` holds the rendered case. */ + | 'aborted' + /** A coded failure the agent decided; `failure` holds it. */ + | 'failed' + /** `options.signal` aborted the run. */ + | 'cancelled' + /** + * Something threw that the agent did not decide (a refused gateway mint, an + * SDK crash). `failure.error` is the original error and `failure.code` its + * classification, so a caller can report it or rethrow it as it did before. + */ + | 'crashed'; + +/** Totals of every `usage` event the run emitted. */ +export interface TokenUsageTotals { + inputTokens: number; + outputTokens: number; + cacheReadTokens: number; + cacheCreationTokens: number; +} + +/** What the agent reported, accumulated independently of any observer. */ +export interface RunSnapshot { + tasks: import('@lib/agent/progress').TaskSnapshot[]; + statusMessages: string[]; + stage?: string; + handoffText?: string; + usage: TokenUsageTotals; + finalCostUsd?: number; + dashboardUrl?: string; + notebookUrl?: string; +} + +/** + * `runAgent` never rejects: every ending is one of these, and every ending + * that is not `success` carries a `failure` the caller can act on. + */ +export interface RunResult { + outcome: RunOutcome; + /** The skill this run installed or was for. */ + skillId?: string; + /** Success outro. Absent for a composed sub-run, which has no terminal outro. */ + outro?: OutroData; + /** Present whenever `outcome` is not `success`. */ + failure?: AgentFailure; + snapshot: RunSnapshot; +} + +export interface RunAgentOptions { + /** Receives every progress event in emission order. Never awaited. */ + onProgress?: (event: import('@lib/agent/progress').AgentProgress) => void; + /** Answers the agent's questions. Absent → no ask bridge, notices declined. */ + interaction?: AgentInteraction; + /** Cancels the run at its next phase boundary and any pending question. */ + signal?: AbortSignal; +} + +/** What a sequence receives: the contracts plus the prepared run. */ +export interface SequenceContext { + config: RunConfig; + input: RunInput; + boot: BootstrapResult; + emit: ProgressEmitter; + interaction: AgentInteraction | undefined; + signal: AbortSignal; +} diff --git a/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts b/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts index ebddc32e2..501a29006 100644 --- a/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts +++ b/src/lib/agent/runner/switchboard/flags/__tests__/binding-cases.ts @@ -11,6 +11,7 @@ import { type SwitchboardTrace, } from '@lib/agent/runner/switchboard'; import type { EffortLevel } from '@lib/agent/runner/switchboard/models'; +import { bindingFor } from '@lib/programs/bindings'; /** The complete resolved binding — every axis stated, nothing implicit. */ export interface ExpectedBinding { @@ -38,7 +39,12 @@ export function runBindingCases( it(c.name, () => { if (c.surface) setSurface?.(c.surface); try { - const ctx: SwitchboardCtx = { ...c.ctx }; + // The program's declared binding rides along, as the legacy adapter + // supplies it; a case may pin its own. + const ctx: SwitchboardCtx = { + binding: bindingFor(c.ctx.program), + ...c.ctx, + }; expect(resolveBinding(ctx)).toEqual(c.binding); if (c.trace) expect(ctx.trace).toEqual(c.trace); } finally { diff --git a/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts b/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts index f0540ab5b..cf509b9ed 100644 --- a/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts +++ b/src/lib/agent/runner/switchboard/flags/__tests__/flags.test.ts @@ -27,6 +27,7 @@ import { } from '@lib/agent/runner/switchboard/flags/orchestrator'; import { SELF_DRIVING_EXPERIMENT } from '@lib/agent/runner/switchboard/flags/self-driving'; import { runBindingCases } from './binding-cases'; +import { bindingFor } from '@lib/programs/bindings'; const envState = vi.hoisted(() => ({ runSurface: 'local' as 'cloud' | 'local', @@ -261,7 +262,12 @@ describe('isolation — everything on at once', () => { it('only the two covered programs move; each lands exactly on its own row', () => { for (const program of PROGRAM_IDS) { - const ctx: SwitchboardCtx = { program, flags, flagPayloads }; + const ctx: SwitchboardCtx = { + program, + flags, + flagPayloads, + binding: bindingFor(program), + }; const resolved = resolveBinding(ctx); if (program === 'posthog-integration') { expect(resolved).toEqual(ORCHESTRATOR_PI_DEFAULT); @@ -325,6 +331,7 @@ describe('isolation — everything on at once', () => { it('regression (2026-07-17): self-driving never rides the global orchestrator flag into the orchestrator', () => { const binding = resolveBinding({ program: 'self-driving', + binding: bindingFor('self-driving'), flags: { [ORCH]: 'true', [SD]: 'true' }, flagPayloads: { [SD]: { model: 'gpt-5-6-terra' } }, }); diff --git a/src/lib/agent/runner/switchboard/flags/index.ts b/src/lib/agent/runner/switchboard/flags/index.ts index b9d3d03a2..4df468611 100644 --- a/src/lib/agent/runner/switchboard/flags/index.ts +++ b/src/lib/agent/runner/switchboard/flags/index.ts @@ -7,7 +7,6 @@ import { RUN_SURFACE } from '@env'; import { logToFile } from '@utils/debug'; import type { Sequence } from '@lib/constants'; -import type { ProgramId } from '@lib/programs/program-registry'; import { ORCHESTRATOR_HARNESS_ROUTE, ORCHESTRATOR_SEQUENCE_ROUTE, @@ -34,7 +33,7 @@ export const SEQUENCE_EXPERIMENTS: readonly SequenceExperiment[] = [ /** The flag-driven route for a program, or undefined when no experiment covers it or its flags don't validly route. */ export function resolveFlagRoute( - program: ProgramId, + program: string, flags: Record, flagPayloads?: Record, ): FlagRoute | undefined { @@ -46,7 +45,7 @@ export function resolveFlagRoute( /** The flag-driven sequence for a program, or undefined when no sequence experiment covers it with its flag on. Surface/build scoping is the flag's own job (see `flagPersonProperties`). */ export function resolveFlagSequence( - program: ProgramId, + program: string, flags: Record, ): Sequence | undefined { return SEQUENCE_EXPERIMENTS.find( @@ -56,7 +55,7 @@ export function resolveFlagSequence( /** The per-stage overrides for a program's run, or undefined (prompt frontmatter stays). Applied once, where the agent prompts are loaded. */ export function resolveStageOverrides( - program: ProgramId, + program: string, flags: Record, flagPayloads?: Record, ): Record | undefined { diff --git a/src/lib/agent/runner/switchboard/flags/schemes.ts b/src/lib/agent/runner/switchboard/flags/schemes.ts index f55f30141..ee074d0ec 100644 --- a/src/lib/agent/runner/switchboard/flags/schemes.ts +++ b/src/lib/agent/runner/switchboard/flags/schemes.ts @@ -13,7 +13,6 @@ import { Sequence, SONNET_5_MODEL, } from '@lib/constants'; -import type { ProgramId } from '@lib/programs/program-registry'; import { logToFile } from '@utils/debug'; import type { EffortLevel } from '../models'; @@ -68,13 +67,13 @@ export type ConfigFlag = HarnessConfigFlag | PayloadConfigFlag; * program. */ export interface HarnessExperiment { - program: ProgramId; + program: string; flags: ConfigFlag; } /** A sequence-axis experiment: one boolean flag, inert outside its listed programs. */ export interface SequenceExperiment { - programs: readonly ProgramId[]; + programs: readonly string[]; flag: string; /** Sequence the flag routes covered programs to. */ sequence: Sequence; diff --git a/src/lib/agent/runner/switchboard/harness.ts b/src/lib/agent/runner/switchboard/harness.ts index 8710a166f..837a4cbe6 100644 --- a/src/lib/agent/runner/switchboard/harness.ts +++ b/src/lib/agent/runner/switchboard/harness.ts @@ -10,8 +10,7 @@ import { piBackend } from '../harness/pi'; import type { AgentHarness } from '../harness/types'; import { resolveFlagRoute } from './flags'; import { - DEFAULT_BINDING, - PROGRAM_BINDINGS, + bindingOf, runChain, type HarnessPick, type Middleware, @@ -86,7 +85,7 @@ export function resolveHarness( const pick = runChain(HARNESS_MIDDLEWARE, ctx, () => { if (ctx.trace) Object.assign(ctx.trace, { harness: 'binding', model: 'binding' }); - const binding = PROGRAM_BINDINGS[ctx.program] ?? DEFAULT_BINDING; + const binding = bindingOf(ctx); return { harness: binding.harness, model: binding.model, diff --git a/src/lib/agent/runner/switchboard/index.ts b/src/lib/agent/runner/switchboard/index.ts index f6d641a66..31d645bd1 100644 --- a/src/lib/agent/runner/switchboard/index.ts +++ b/src/lib/agent/runner/switchboard/index.ts @@ -1,13 +1,6 @@ // Resolves routing; model additions also require mint allowlists and gateway prompt/transport support. -import { - DEFAULT_AGENT_MODEL, - GPT5_6_SOL_MODEL, - GPT5_6_TERRA_MODEL, - Harness, - Sequence, -} from '@lib/constants'; -import type { ProgramId } from '@lib/programs/program-registry'; +import { GPT5_6_SOL_MODEL, Harness, Sequence } from '@lib/constants'; import { resolveHarness } from './harness'; import type { EffortLevel } from './models'; import { resolveSequence } from './sequence'; @@ -29,7 +22,14 @@ export interface SwitchboardTrace { /** Everything a resolver middleware may branch on. Built once per run. */ export interface SwitchboardCtx { - program: ProgramId; + /** Program id. An opaque label here: experiments match on it, nothing else reads it. */ + program: string; + /** + * The program's declared binding, resolved by the caller (see + * `src/lib/programs/bindings.ts`). Absent → `DEFAULT_BINDING`, for + * standalone and skill-only runs that belong to no registered program. + */ + binding?: ProgramBinding; /** Composed sub-run (a dependency inside a parent program). Structurally linear — no override can orchestrate it. */ composed?: boolean; flags: Record; @@ -110,60 +110,10 @@ export const DEFAULT_BINDING: ProgramBinding = { thinkingLevel: 'medium', }; -/** - * Per-program routing. Kept in lockstep with `PROGRAM_REGISTRY` by the - * switchboard test. Anything absent falls back to `DEFAULT_BINDING`. - */ -export const PROGRAM_BINDINGS: Partial> = { - 'posthog-integration': DEFAULT_BINDING, - 'revenue-analytics-setup': DEFAULT_BINDING, - 'warehouse-source': DEFAULT_BINDING, - 'error-tracking-upload-source-maps': { - sequence: Sequence.linear, - harness: Harness.pi, - model: GPT5_6_SOL_MODEL, - thinkingLevel: 'medium', - }, - audit: DEFAULT_BINDING, - 'events-audit': DEFAULT_BINDING, - 'posthog-doctor': DEFAULT_BINDING, - 'web-analytics-doctor': DEFAULT_BINDING, - migration: DEFAULT_BINDING, - 'self-driving': DEFAULT_BINDING, - 'agent-skill': DEFAULT_BINDING, - 'mcp-add': DEFAULT_BINDING, - 'mcp-remove': DEFAULT_BINDING, - 'mcp-tutorial': DEFAULT_BINDING, - 'mcp-analytics': DEFAULT_BINDING, - // Orchestrator on pi. The binding routes only; every stage's model and - // effort are pinned context-mill side in the flow's frontmatter - // (`model_pi`/`effort_pi`: terra seed, sol tasks, luna report). - metrics: { - sequence: Sequence.orchestrator, - harness: Harness.pi, - model: DEFAULT_AGENT_MODEL, - }, - 'replay-vision': { - sequence: Sequence.orchestrator, - harness: Harness.anthropic, - model: DEFAULT_AGENT_MODEL, - }, - // Orchestrator on pi, like metrics. The binding routes only; every stage's - // model and effort are pinned context-mill side in the flow's frontmatter - // (`model_pi`/`effort_pi`: terra seed, install and init, sol tasks, luna report). - 'error-tracking': { - sequence: Sequence.orchestrator, - harness: Harness.pi, - model: DEFAULT_AGENT_MODEL, - }, - 'ai-observability': { - sequence: Sequence.linear, - harness: Harness.pi, - model: GPT5_6_TERRA_MODEL, - thinkingLevel: 'high', - }, - slack: DEFAULT_BINDING, -}; +/** The binding a context resolves from: the caller's, else the agent default. */ +export function bindingOf(ctx: SwitchboardCtx): ProgramBinding { + return ctx.binding ?? DEFAULT_BINDING; +} // ── Unified resolver ──────────────────────────────────────────────────── diff --git a/src/lib/agent/runner/switchboard/sequence.ts b/src/lib/agent/runner/switchboard/sequence.ts index e39e228e8..8e76c8599 100644 --- a/src/lib/agent/runner/switchboard/sequence.ts +++ b/src/lib/agent/runner/switchboard/sequence.ts @@ -12,43 +12,27 @@ import { resolveFlagSequence, } from './flags'; import { getHarness, resolveHarness } from './harness'; -import type { WizardSession } from '@lib/wizard-session'; -import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun, BootstrapResult } from '../shared/types'; +import type { RunResult, SequenceContext } from '../shared/types'; import { runLinearProgram } from '../sequence/linear'; import { runOrchestrator } from '../sequence/orchestrator/orchestrator-runner'; -import { - DEFAULT_BINDING, - PROGRAM_BINDINGS, - runChain, - type Middleware, - type SwitchboardCtx, -} from '.'; +import { bindingOf, runChain, type Middleware, type SwitchboardCtx } from '.'; // ── Registry ──────────────────────────────────────────────────────────── export interface SequenceRunner { readonly name: Sequence; - run( - session: WizardSession, - config: ProgramRun, - programConfig: ProgramConfig, - boot: BootstrapResult, - /** Composed sub-run (integration inside self-driving); linear-only. */ - composed: boolean, - ): Promise; + /** Run one program to a decided result. Unexpected errors propagate. */ + run(ctx: SequenceContext): Promise; } export const SEQUENCE_OPTIONS: Partial> = { [Sequence.linear]: { name: Sequence.linear, - run: (session, config, programConfig, boot, composed) => - runLinearProgram(session, config, programConfig, boot, composed), + run: (ctx) => runLinearProgram(ctx), }, [Sequence.orchestrator]: { name: Sequence.orchestrator, - run: (session, config, programConfig, boot, _composed) => - runOrchestrator(session, config, programConfig, boot), + run: (ctx) => runOrchestrator(ctx), }, }; @@ -129,8 +113,7 @@ const SEQUENCE_MIDDLEWARE: Middleware[] = [ export function resolveSequence(ctx: SwitchboardCtx): Sequence { const sequence = runChain(SEQUENCE_MIDDLEWARE, ctx, () => { if (ctx.trace) ctx.trace.sequence = 'binding'; - const binding = PROGRAM_BINDINGS[ctx.program] ?? DEFAULT_BINDING; - return binding.sequence; + return bindingOf(ctx).sequence; }); logToFile( `[switchboard] resolved: program=${ctx.program} sequence=${sequence}` + diff --git a/src/lib/detection/__tests__/project-scope.test.ts b/src/lib/detection/__tests__/project-scope.test.ts index 46aedaf04..18f9173dc 100644 --- a/src/lib/detection/__tests__/project-scope.test.ts +++ b/src/lib/detection/__tests__/project-scope.test.ts @@ -10,12 +10,12 @@ import { AGENTIC_DETECTION_TIMEOUT_MS, WIZARD_BASIC_INTEGRATION_AGENTIC_DETECTION_FLAG_KEY, } from '@lib/constants'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { authenticate } from '@lib/programs/authenticate'; import { buildSession } from '@lib/wizard-session'; import { analytics } from '@utils/analytics'; // Mock only the two network edges of scopeInstallDirToProject; everything else runs real. -vi.mock('@lib/agent/runner/shared/authenticate', () => ({ +vi.mock('@lib/programs/authenticate', () => ({ authenticate: vi.fn().mockResolvedValue(undefined), })); vi.mock('@lib/detection/agentic', async (importOriginal) => ({ diff --git a/src/lib/detection/project-scope.ts b/src/lib/detection/project-scope.ts index 73279b075..da4ffaf0a 100644 --- a/src/lib/detection/project-scope.ts +++ b/src/lib/detection/project-scope.ts @@ -8,7 +8,7 @@ import { type DetectEvent, type DetectTarget, } from './agentic.js'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { authenticate } from '@lib/programs/authenticate'; import { FRAMEWORK_REGISTRY } from '@lib/registry'; import { AGENTIC_DETECTION_TIMEOUT_MS, diff --git a/src/lib/errors/index.ts b/src/lib/errors/index.ts index 5558c7942..30e98f4ec 100644 --- a/src/lib/errors/index.ts +++ b/src/lib/errors/index.ts @@ -18,3 +18,4 @@ export { } from './emit'; export { sanitizeErrorDetail } from './sanitize'; export { classifyRunFailure, type RunFailure } from './run-failure'; +export { WizardError } from './wizard-error'; diff --git a/src/lib/errors/wizard-error.ts b/src/lib/errors/wizard-error.ts new file mode 100644 index 000000000..fb4690749 --- /dev/null +++ b/src/lib/errors/wizard-error.ts @@ -0,0 +1,20 @@ +import type { ErrorCode } from './codes'; + +/** + * A data carrier for a decided failure: message, analytics context and the + * catalog code. Passed to `wizardAbort()` by the caller that owns the exit; + * never thrown by the agent. + */ +export class WizardError extends Error { + readonly code?: ErrorCode; + + constructor( + message: string, + public readonly context?: Record, + code?: ErrorCode, + ) { + super(message); + this.name = 'WizardError'; + this.code = code; + } +} diff --git a/src/lib/middleware/benchmark.ts b/src/lib/middleware/benchmark.ts index 20017ad75..1e6c0937b 100644 --- a/src/lib/middleware/benchmark.ts +++ b/src/lib/middleware/benchmark.ts @@ -7,7 +7,7 @@ * pipeline.finalize(resultMessage, durationMs); */ -import { getUI, type SpinnerHandle } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; import { logToFile, getLogFilePath, configureLogFile } from '@utils/debug'; import { MiddlewarePipeline } from './pipeline'; import { PhaseDetector } from './phase-detector'; @@ -68,8 +68,10 @@ export function createBenchmarkPipeline( spinner: SpinnerHandle, options: WizardRunOptions, configOverride?: BenchmarkConfig, + reporting: { log?: (message: string) => void } = {}, ): MiddlewarePipeline { const config = configOverride ?? loadBenchmarkConfig(options.installDir); + const log = reporting.log ?? (() => undefined); configureLogFile({ path: config.output.logPath, @@ -80,13 +82,12 @@ export function createBenchmarkPipeline( spinner, phased: false, outputPath: config.output.benchmarkPath, + log, }); if (!config.output.suppressWizardLogs) { - getUI().log.info( - `${AgentSignals.BENCHMARK} Verbose logs: ${getLogFilePath()}`, - ); - getUI().log.info( + log(`${AgentSignals.BENCHMARK} Verbose logs: ${getLogFilePath()}`); + log( `${AgentSignals.BENCHMARK} Benchmark data will be written to: ${config.output.benchmarkPath}`, ); } diff --git a/src/lib/middleware/benchmarks/index.ts b/src/lib/middleware/benchmarks/index.ts index 470dbd82b..61f845e3e 100644 --- a/src/lib/middleware/benchmarks/index.ts +++ b/src/lib/middleware/benchmarks/index.ts @@ -30,8 +30,8 @@ const PLUGIN_REGISTRY: Record = { contextSize: () => new ContextSizeTrackerPlugin(), cost: () => new CostTrackerPlugin(), duration: () => new DurationTrackerPlugin(), - summary: (opts) => new SummaryPlugin(opts.spinner!), - jsonWriter: (opts) => new JsonWriterPlugin(opts.outputPath!), + summary: (opts) => new SummaryPlugin(opts.spinner!, opts.log), + jsonWriter: (opts) => new JsonWriterPlugin(opts.outputPath!, opts.log), }; /** diff --git a/src/lib/middleware/benchmarks/json-writer.ts b/src/lib/middleware/benchmarks/json-writer.ts index 253225335..43d94dbf9 100644 --- a/src/lib/middleware/benchmarks/json-writer.ts +++ b/src/lib/middleware/benchmarks/json-writer.ts @@ -6,7 +6,6 @@ */ import fs from 'fs'; -import { getUI } from '@ui'; import { logToFile } from '@utils/debug'; import { AgentSignals } from '@lib/agent/agent-interface'; import type { @@ -57,7 +56,13 @@ export class JsonWriterPlugin implements Middleware { private outputPath: string; - constructor(outputPath: string) { + private readonly log: (message: string) => void; + + constructor( + outputPath: string, + log: (message: string) => void = () => undefined, + ) { + this.log = log; this.outputPath = outputPath; } @@ -170,7 +175,7 @@ export class JsonWriterPlugin implements Middleware { try { fs.writeFileSync(this.outputPath, JSON.stringify(data, null, 2)); logToFile(`Benchmark data written to ${this.outputPath}`); - getUI().log.info( + this.log( `● ${AgentSignals.BENCHMARK} Results written to ${this.outputPath}`, ); } catch (error) { diff --git a/src/lib/middleware/benchmarks/summary.ts b/src/lib/middleware/benchmarks/summary.ts index 45c3b0a8e..161b515a0 100644 --- a/src/lib/middleware/benchmarks/summary.ts +++ b/src/lib/middleware/benchmarks/summary.ts @@ -1,4 +1,4 @@ -import { getUI, type SpinnerHandle } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; import { AgentSignals } from '@lib/agent/agent-interface'; import type { Middleware, @@ -97,8 +97,14 @@ export class SummaryPlugin implements Middleware { private spinner: SpinnerHandle; - constructor(spinner: SpinnerHandle) { + private readonly log: (message: string) => void; + + constructor( + spinner: SpinnerHandle, + log: (message: string) => void = () => undefined, + ) { this.spinner = spinner; + this.log = log; } onPhaseTransition( @@ -117,7 +123,7 @@ export class SummaryPlugin implements Middleware { this.spinner.stop(`${AgentSignals.BENCHMARK} ${fromPhase}`); } - getUI().log.info(`${AgentSignals.BENCHMARK} Starting phase: ${toPhase}`); + this.log(`${AgentSignals.BENCHMARK} Starting phase: ${toPhase}`); this.spinner.start(`Integrating PostHog (${toPhase})...`); } @@ -135,31 +141,31 @@ export class SummaryPlugin implements Middleware { const phaseCount = duration?.phaseSnapshots.length ?? 0; const totalCost = cost?.totalCost ?? 0; - getUI().log.info(''); - getUI().log.info( + this.log(''); + this.log( `◇ ${AgentSignals.BENCHMARK} ${phaseCount} phases in ${fmtDuration( totalDurationMs, )}, cost: ${fmtCost(totalCost)}`, ); - getUI().log.info( + this.log( ` total in: ${fmtTok(tokens?.totalInput ?? 0)}, out: ${fmtTok( tokens?.totalOutput ?? 0, )}, cache_read: ${fmtTok(cache?.totalRead ?? 0)}, cache_5m: ${fmtTok( cache?.totalCreation5m ?? 0, )}, cache_1h: ${fmtTok(cache?.totalCreation1h ?? 0)}`, ); - getUI().log.info(''); - getUI().log.info(`● ${AgentSignals.BENCHMARK} Summary by phase:`); + this.log(''); + this.log(`● ${AgentSignals.BENCHMARK} Summary by phase:`); if (duration?.phaseSnapshots) { for (let i = 0; i < duration.phaseSnapshots.length; i++) { const stats = getPhaseStats(i, ctx); if (stats) { - getUI().log.info(printPhase(stats)); + this.log(printPhase(stats)); } } } - getUI().log.info(''); + this.log(''); } } diff --git a/src/lib/middleware/types.ts b/src/lib/middleware/types.ts index ea1715700..3b8ab8c46 100644 --- a/src/lib/middleware/types.ts +++ b/src/lib/middleware/types.ts @@ -5,7 +5,7 @@ * and can publish data to a shared store for downstream middleware to read. */ -import type { SpinnerHandle } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; export type SDKMessage = any; @@ -57,4 +57,6 @@ export interface MiddlewareFactoryOptions { spinner?: SpinnerHandle; outputPath?: string; phased?: boolean; + /** Where the summary and writer plugins print their lines. */ + log?: (message: string) => void; } diff --git a/src/lib/programs/__tests__/agent-skill.test.ts b/src/lib/programs/__tests__/agent-skill.test.ts index 132586e1d..b836f0486 100644 --- a/src/lib/programs/__tests__/agent-skill.test.ts +++ b/src/lib/programs/__tests__/agent-skill.test.ts @@ -3,7 +3,7 @@ import { AGENT_SKILL_STEPS, type SkillProgramOptions, } from '@lib/programs/agent-skill/index'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { buildSession, RunPhase } from '@lib/wizard-session'; import { HostResolution } from '@lib/host-resolution'; diff --git a/src/lib/programs/__tests__/error-tracking.test.ts b/src/lib/programs/__tests__/error-tracking.test.ts index 7cf0561d8..5ecef5279 100644 --- a/src/lib/programs/__tests__/error-tracking.test.ts +++ b/src/lib/programs/__tests__/error-tracking.test.ts @@ -1,6 +1,6 @@ import { beforeEach, describe, expect, test, vi } from 'vitest'; -import type { ProgramRun } from '@lib/agent/runner/shared/types'; +import type { ProgramRun } from '@lib/programs/program-run'; import { Integration } from '@lib/constants'; import type { AgenticDetectionReport } from '@lib/detection/agentic'; import { detectFramework } from '@lib/detection/index'; diff --git a/src/lib/programs/__tests__/metrics-program.test.ts b/src/lib/programs/__tests__/metrics-program.test.ts index ae5936408..46f381471 100644 --- a/src/lib/programs/__tests__/metrics-program.test.ts +++ b/src/lib/programs/__tests__/metrics-program.test.ts @@ -1,7 +1,7 @@ import { AGENT_SKILL_STEPS } from '@lib/programs/agent-skill/index'; import { getProgramConfig, Program } from '@lib/programs/program-registry'; import { metricsConfig } from '@lib/programs/metrics/index'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { metricsCommand } from '../../../commands/metrics'; diff --git a/src/lib/agent/runner/shared/__tests__/refresh-access-token-if-needed.test.ts b/src/lib/programs/__tests__/refresh-access-token-if-needed.test.ts similarity index 100% rename from src/lib/agent/runner/shared/__tests__/refresh-access-token-if-needed.test.ts rename to src/lib/programs/__tests__/refresh-access-token-if-needed.test.ts diff --git a/src/lib/programs/agent-skill/index.ts b/src/lib/programs/agent-skill/index.ts index dbd207ae0..b2f6198a7 100644 --- a/src/lib/programs/agent-skill/index.ts +++ b/src/lib/programs/agent-skill/index.ts @@ -20,7 +20,7 @@ */ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun, AbortCase } from '@lib/agent/agent-runner'; +import type { ProgramRun, AbortCase } from '@lib/programs/program-run'; import { AGENT_SKILL_STEPS } from './steps.js'; import { getContentBlocks } from './content/index.js'; diff --git a/src/lib/programs/audit/detect.ts b/src/lib/programs/audit/detect.ts index 2cfa299a5..6ffbb69c8 100644 --- a/src/lib/programs/audit/detect.ts +++ b/src/lib/programs/audit/detect.ts @@ -1,4 +1,4 @@ -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; /** `[ABORT] ` cases the audit skill can emit. Reason strings are diff --git a/src/lib/programs/audit/index.ts b/src/lib/programs/audit/index.ts index fc5f60297..261fbaefc 100644 --- a/src/lib/programs/audit/index.ts +++ b/src/lib/programs/audit/index.ts @@ -3,7 +3,7 @@ import { createSkillProgram, } from '@lib/programs/agent-skill/index'; import type { ProgramStep, ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { OutroKind } from '@lib/wizard-session'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; diff --git a/src/lib/agent/runner/shared/authenticate.ts b/src/lib/programs/authenticate.ts similarity index 98% rename from src/lib/agent/runner/shared/authenticate.ts rename to src/lib/programs/authenticate.ts index 8644a328c..1c55453dd 100644 --- a/src/lib/agent/runner/shared/authenticate.ts +++ b/src/lib/programs/authenticate.ts @@ -11,7 +11,7 @@ */ import type { Credentials, WizardSession } from '@lib/wizard-session'; -import type { ProgramId } from '@lib/programs/program-registry'; +import type { ProgramId } from './program-registry'; import { getOrAskForProjectData } from '@utils/setup-utils'; import { refreshAccessToken } from '@utils/oauth'; import { OAuthError } from '@utils/oauth-errors'; diff --git a/src/lib/programs/bindings.ts b/src/lib/programs/bindings.ts new file mode 100644 index 000000000..1d445e08f --- /dev/null +++ b/src/lib/programs/bindings.ts @@ -0,0 +1,77 @@ +/** + * Per-program routing: which sequence, harness and model run each program. + * + * Kept in lockstep with `PROGRAM_REGISTRY` by the switchboard test. Anything + * absent falls back to the agent's `DEFAULT_BINDING`. The agent never reads + * this table: `run-agent-legacy.ts` looks a program up here and hands the + * binding to the switchboard as `SwitchboardCtx.binding`. + */ + +import { + DEFAULT_AGENT_MODEL, + GPT5_6_SOL_MODEL, + GPT5_6_TERRA_MODEL, + Harness, + Sequence, +} from '@lib/constants'; +import { + DEFAULT_BINDING, + type ProgramBinding, +} from '@lib/agent/runner/switchboard'; +import type { ProgramId } from './program-registry'; + +export const PROGRAM_BINDINGS: Partial> = { + 'posthog-integration': DEFAULT_BINDING, + 'revenue-analytics-setup': DEFAULT_BINDING, + 'warehouse-source': DEFAULT_BINDING, + 'error-tracking-upload-source-maps': { + sequence: Sequence.linear, + harness: Harness.pi, + model: GPT5_6_SOL_MODEL, + thinkingLevel: 'medium', + }, + audit: DEFAULT_BINDING, + 'events-audit': DEFAULT_BINDING, + 'posthog-doctor': DEFAULT_BINDING, + 'web-analytics-doctor': DEFAULT_BINDING, + migration: DEFAULT_BINDING, + 'self-driving': DEFAULT_BINDING, + 'agent-skill': DEFAULT_BINDING, + 'mcp-add': DEFAULT_BINDING, + 'mcp-remove': DEFAULT_BINDING, + 'mcp-tutorial': DEFAULT_BINDING, + 'mcp-analytics': DEFAULT_BINDING, + // Orchestrator on pi. The binding routes only; every stage's model and + // effort are pinned context-mill side in the flow's frontmatter + // (`model_pi`/`effort_pi`: terra seed, sol tasks, luna report). + metrics: { + sequence: Sequence.orchestrator, + harness: Harness.pi, + model: DEFAULT_AGENT_MODEL, + }, + 'replay-vision': { + sequence: Sequence.orchestrator, + harness: Harness.anthropic, + model: DEFAULT_AGENT_MODEL, + }, + // Orchestrator on pi, like metrics. The binding routes only; every stage's + // model and effort are pinned context-mill side in the flow's frontmatter + // (`model_pi`/`effort_pi`: terra seed, install and init, sol tasks, luna report). + 'error-tracking': { + sequence: Sequence.orchestrator, + harness: Harness.pi, + model: DEFAULT_AGENT_MODEL, + }, + 'ai-observability': { + sequence: Sequence.linear, + harness: Harness.pi, + model: GPT5_6_TERRA_MODEL, + thinkingLevel: 'high', + }, + slack: DEFAULT_BINDING, +}; + +/** The binding for a program, or the agent default when none is declared. */ +export function bindingFor(programId: string): ProgramBinding { + return PROGRAM_BINDINGS[programId] ?? DEFAULT_BINDING; +} diff --git a/src/lib/programs/error-tracking-upload-source-maps/detect.ts b/src/lib/programs/error-tracking-upload-source-maps/detect.ts index 2d81b0061..f6ddbfb09 100644 --- a/src/lib/programs/error-tracking-upload-source-maps/detect.ts +++ b/src/lib/programs/error-tracking-upload-source-maps/detect.ts @@ -17,7 +17,7 @@ import { safeReadFile, } from '@utils/bounded-fs'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; /** diff --git a/src/lib/programs/error-tracking-upload-source-maps/index.ts b/src/lib/programs/error-tracking-upload-source-maps/index.ts index c79f245e2..afdc17007 100644 --- a/src/lib/programs/error-tracking-upload-source-maps/index.ts +++ b/src/lib/programs/error-tracking-upload-source-maps/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { OutroKind } from '@lib/wizard-session'; import { ERROR_TRACKING_UPLOAD_SOURCE_MAPS_PROGRAM } from './steps.js'; diff --git a/src/lib/programs/error-tracking/index.ts b/src/lib/programs/error-tracking/index.ts index 2fd6a7cb1..68e7986b5 100644 --- a/src/lib/programs/error-tracking/index.ts +++ b/src/lib/programs/error-tracking/index.ts @@ -2,7 +2,7 @@ import { Integration } from '@lib/constants'; import { detectFramework } from '@lib/detection/index'; import { scopeInstallDirToProject } from '@lib/detection/project-scope'; import { FRAMEWORK_REGISTRY } from '@lib/registry'; -import type { ProgramRun } from '@lib/agent/runner/shared/types'; +import type { ProgramRun } from '@lib/programs/program-run'; import { AGENT_SKILL_STEPS } from '@lib/programs/agent-skill/steps'; import { getContentBlocks } from '@lib/programs/error-tracking/content/index'; import { getTips } from '@lib/programs/error-tracking/content/tips'; diff --git a/src/lib/programs/events-audit/index.ts b/src/lib/programs/events-audit/index.ts index 080e59d69..4cfa397f0 100644 --- a/src/lib/programs/events-audit/index.ts +++ b/src/lib/programs/events-audit/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { OutroKind } from '@lib/wizard-session'; import { SPINNER_MESSAGE } from '@lib/framework-config'; diff --git a/src/lib/programs/mcp-analytics/index.ts b/src/lib/programs/mcp-analytics/index.ts index f9199d76e..5231a1d0a 100644 --- a/src/lib/programs/mcp-analytics/index.ts +++ b/src/lib/programs/mcp-analytics/index.ts @@ -1,4 +1,4 @@ -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; import { createSkillProgram } from '@lib/programs/agent-skill/index'; diff --git a/src/lib/programs/migration/index.ts b/src/lib/programs/migration/index.ts index 27cf0b3d0..9c0ddbad2 100644 --- a/src/lib/programs/migration/index.ts +++ b/src/lib/programs/migration/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import { MIGRATION_PROGRAM } from './steps.js'; import { getContentBlocks } from './content/index.js'; diff --git a/src/lib/programs/posthog-integration/index.ts b/src/lib/programs/posthog-integration/index.ts index c3ac0b913..433db9e7a 100644 --- a/src/lib/programs/posthog-integration/index.ts +++ b/src/lib/programs/posthog-integration/index.ts @@ -1,5 +1,6 @@ import type { ProgramConfig, ProgramStep } from '@lib/programs/program-step'; -import { runAgent, type ProgramRun } from '@lib/agent/agent-runner'; +import { runAgent } from '@lib/programs/run-agent-legacy'; +import type { ProgramRun } from '@lib/programs/program-run'; import { WIZARD_TOOL_NAMES } from '@lib/wizard-tools'; import type { WizardSession } from '@lib/wizard-session'; import { mayReportScanResults, OutroKind, RunPhase } from '@lib/wizard-session'; @@ -21,7 +22,7 @@ import { requestDeepLink } from '@utils/provisioning'; import { openTrackedLink, withUtm } from '@utils/links'; import type { HostResolution } from '@lib/host-resolution'; import { getDetectedWarehouseSources } from '@lib/programs/warehouse-source/detect'; -import { shouldDisableAsk } from '@lib/agent/runner/shared/bootstrap'; +import { shouldDisableAsk } from '@lib/agent/agent-runner'; import { POSTHOG_INTEGRATION_PROGRAM } from './steps.js'; import { getContentBlocks } from './content/index.js'; import { buildCodingAgentPrompt } from './handoff.js'; diff --git a/src/lib/programs/program-run.ts b/src/lib/programs/program-run.ts new file mode 100644 index 000000000..a2cb5ff3a --- /dev/null +++ b/src/lib/programs/program-run.ts @@ -0,0 +1,48 @@ +/** + * A program's agent run definition. + * + * Every program provides one of these — either as a static object or via a + * function that builds one from the session. The agent reads the + * `AgentRunDefinition` part; the three hooks below take the wizard session and + * are bound by `run-agent-legacy.ts` before the agent sees them. + */ + +import type { Credentials, WizardSession } from '@lib/wizard-session'; +import type { AgentRunDefinition } from '@lib/agent/runner/shared/types'; + +export type { + AbortCase, + AgentRunDefinition, + PromptContext, +} from '@lib/agent/runner/shared/types'; +export type { Credentials }; + +export interface ProgramRun extends AgentRunDefinition { + /** Runs after agent completes, before outro (e.g. env var upload). */ + postRun?: (session: WizardSession, credentials: Credentials) => Promise; + /** Custom outro data. Omit for default built from successMessage/reportFile/docsUrl. */ + buildOutroData?: ( + session: WizardSession, + credentials: Credentials, + ) => WizardSession['outroData']; + /** + * Outro bullets for a sequence that composes its own outro data. + * + * `buildOutroData` is the linear sequence's seam: it hands the program the + * whole outro. The orchestrated sequence cannot, because its message is the + * drain's result — how many steps ran, what was skipped, which conflict the + * review step left. So a program with next steps to offer had nowhere to put + * them there, and the integration's data-source links were built and then + * dropped on every orchestrated run. This hook keeps the message with the + * sequence and the bullets with the program. + * + * `completedSeededTypes` names the runner-seeded task types that finished + * successfully, so a program can leave out a step its own seeded task + * already did — the sequence stays ignorant of what any type means. + */ + buildOutroNextSteps?: ( + session: WizardSession, + credentials: Credentials, + completedSeededTypes: readonly string[], + ) => { heading: string; items: string[] } | undefined; +} diff --git a/src/lib/programs/program-step.ts b/src/lib/programs/program-step.ts index 81e55a655..5c24decf4 100644 --- a/src/lib/programs/program-step.ts +++ b/src/lib/programs/program-step.ts @@ -4,7 +4,7 @@ import type { TaskNotice, } from '@lib/wizard-session'; import type { WizardReadinessResult } from '@lib/health-checks/readiness'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from './program-run.js'; import type { Integration } from '@lib/constants'; import type { FrameworkConfig } from '@lib/framework-config'; import type { ContentBlock } from '@ui/tui/primitives/index'; diff --git a/src/lib/programs/replay-vision/index.ts b/src/lib/programs/replay-vision/index.ts index 046f76e25..540436b0a 100644 --- a/src/lib/programs/replay-vision/index.ts +++ b/src/lib/programs/replay-vision/index.ts @@ -1,4 +1,4 @@ -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { Integration } from '@lib/constants'; import { detectFramework, gatherFrameworkContext } from '@lib/detection/index'; import { scopeInstallDirToProject } from '@lib/detection/project-scope'; diff --git a/src/lib/programs/revenue-analytics/detect.ts b/src/lib/programs/revenue-analytics/detect.ts index de75eae0b..8ff8528dd 100644 --- a/src/lib/programs/revenue-analytics/detect.ts +++ b/src/lib/programs/revenue-analytics/detect.ts @@ -7,7 +7,7 @@ import { existsSync, statSync } from 'fs'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { findPackageJsons } from '@lib/programs/shared/package-scanning'; export { diff --git a/src/lib/programs/run-agent-legacy.ts b/src/lib/programs/run-agent-legacy.ts new file mode 100644 index 000000000..92f264570 --- /dev/null +++ b/src/lib/programs/run-agent-legacy.ts @@ -0,0 +1,539 @@ +/** + * The session-driven agent runner every existing caller uses. + * + * `runAgent(programConfig, session)` rebuilds today's behavior on top of the + * functional `runAgent(config, input, options)` in `@lib/agent/runner`: it + * runs the gates the TUI owns (health, settings, AI opt-in, post-auth steps), + * authenticates, resolves the program's binding, builds the agent's inputs + * from the session, maps every progress event back onto `getUI()` one call + * per event, answers the agent's questions through `getUI()`, and applies the + * result — `wizardAbort` for a decided failure, nothing more for success. + * + * This is the only file that knows about `getUI()`, the session and + * `wizardAbort` on the agent's behalf. Programs replace it in Release B. + */ + +import type { WizardSession } from '@lib/wizard-session'; +import { analytics } from '@utils/analytics'; +import { getUI, type WizardUI } from '@ui'; +import type { SpinnerHandle } from '@lib/agent/progress'; +import type { AgentInteraction, AgentProgress } from '@lib/agent/progress'; +import { + runAgent as runAgentFunctional, + resolveBinding, + type RunConfig, + type RunInput, + type SwitchboardCtx, +} from '@lib/agent/runner'; +import type { ProgramBinding } from '@lib/agent/runner/switchboard'; +import { buildRunTags } from '@lib/agent/agent-interface'; +import { + backupAndFixClaudeSettings, + checkAllSettingsConflicts, + classifySettingsConflicts, + restoreClaudeSettings, +} from '@lib/agent/claude-settings'; +import { flushScanReport } from '@lib/yara-hooks'; +import { + evaluateWizardReadiness, + WizardReadiness, + SIGNUP_WIZARD_READINESS_CONFIG, + getBlockingServiceKeys, + SERVICE_LABELS, +} from '@lib/health-checks/readiness'; +import { enableDebugLogs, logToFile, initLogFile } from '@utils/debug'; +import { registerCleanup, wizardAbort } from '@utils/wizard-abort'; +import { ErrorCodes } from '@lib/errors'; +import { isNonInteractiveEnvironment } from '@utils/environment'; +import { + getSkillsBaseUrl, + Sequence, + WIZARD_ORCHESTRATOR_FLAG_KEY, + WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY, + type Integration, +} from '@lib/constants'; +import { FRAMEWORK_REGISTRY } from '@lib/registry'; +import { postAuthGateSteps, type ProgramConfig } from './program-step'; +import type { ProgramRun } from './program-run'; +import { authenticate, refreshAccessTokenIfNeeded } from './authenticate'; +import { maybeStampAiSdkDetected } from './posthog-integration/detect'; +import { startAuditLedgerWatcher } from './audit/ledger-watcher'; +import { bindingFor } from './bindings'; + +/** + * Resolve a ProgramConfig's agent run definition and execute the pipeline. + * Entry point for the runners and for composed run steps. + */ +export async function runAgent( + programConfig: ProgramConfig, + session: WizardSession, + options: { composed?: boolean } = {}, +): Promise { + if (!programConfig.run) { + throw new Error(`Program "${programConfig.id}" has no run configuration.`); + } + + // Before `run()` resolves: an audit seeds the ledger from inside its recipe, + // and a watcher started later would ignore that write as pre-existing. + const ledger = programConfig.auditLedgerFile + ? startAuditLedgerWatcher(session.installDir, programConfig.auditLedgerFile) + : null; + if (ledger) registerCleanup(() => ledger.stop()); + + try { + const runDef = + typeof programConfig.run === 'function' + ? await programConfig.run(session) + : programConfig.run; + + await runProgram(session, runDef, programConfig, options.composed ?? false); + } finally { + ledger?.stop(); + } +} + +/** + * Gates → authenticate → flags → binding → the functional run → apply result. + * Every step happens in the order it did inside the agent's bootstrap. + */ +async function runProgram( + session: WizardSession, + run: ProgramRun, + programConfig: ProgramConfig, + composed: boolean, +): Promise { + // 1. Init logging + debug + initLogFile(); + session.skillId = run.skillId ?? run.integrationLabel; + logToFile( + `[agent-runner] START ${run.integrationLabel} build=${analytics.build}` + + `${session.ci ? ' (non-interactive)' : ''}`, + ); + if (session.debug) { + enableDebugLogs(); + } + + // 2. Health check (guarded — skip if TUI already ran it). Only + // programs that declare a health-check screen get pre-flight checks; + // for everything else the checks never fire and never block. + await runHealthGate(session, programConfig); + + // 3. Settings conflicts + await runSettingsGate(session); + + analytics.wizardCapture('agent started', { + integration: run.integrationLabel, + program_id: programConfig.id, + skill_id: run.skillId ?? null, + }); + + // 4. Authenticate — idempotent within a run (see authenticate()). A second + // agent run in the same invocation (self-driving's integration phase) reuses + // the first login; it does not launch another OAuth. authenticate() also + // identifies the user and sets analytics groups. + await authenticate(session, programConfig.id); + maybeStampAiSdkDetected(session); + + // 4.5. AI opt-in enforcement. Parks here while AiOptInRequiredScreen is + // up if the org hasn't approved third-party AI — BEFORE the skill + // install and agent start, so no source leaves the machine. The screen + // alone is cosmetic; this await is the actual gate. Resolves + // immediately when the program declared requiresAi: false or in CI. + logToFile('[agent-runner] checking AI opt-in gate'); + await getUI().waitForAiOptIn(); + logToFile('[agent-runner] AI opt-in gate cleared'); + + // Park for any interactive step the user must complete AFTER authenticating + // but BEFORE the agent runs — e.g. the source-maps project picker, which + // needs credentials to scan and writes its choice to frameworkContext that + // the run prompt reads. Generic: await every gated step between auth and run. + for (const step of postAuthGateSteps(programConfig.steps)) { + logToFile(`[agent-runner] awaiting post-auth gate: ${step.id}`); + await getUI().waitForGate(step.id); + logToFile(`[agent-runner] post-auth gate cleared: ${step.id}`); + } + + // Feature flags. Both arms need these, and the routing decision reads them. + const wizardFlags = await analytics.getAllFlagsForWizard(); + const wizardFlagPayloads = analytics.getWizardFlagPayloads(); + + // Gateway trace tags for this run; the binding below stamps its axes on. + const wizardMetadata = buildRunTags({ + programId: programConfig.id, + integration: run.integrationLabel, + runId: analytics.runId, + build: analytics.build, + skillId: run.skillId, + }); + + // The agent can't swap tokens mid-run, so freshness is measured after every + // park above, right before the agent mints. + await refreshAccessTokenIfNeeded(session); + + // Credentials (incl. the resolved host family and its MCP url) live on + // `session.credentials`; narrow once at this boundary — `authenticate` above + // set them — so downstream readers get a non-null type without asserting. + const credentials = session.credentials!; + + // Resolve which sequence and harness will run a program (CLI → PostHog flag → + // per-program binding → default), tag both axes onto analytics, and hand the + // binding to the agent for dispatch. + const switchboard: SwitchboardCtx = { + program: programConfig.id, + composed, + flags: wizardFlags, + flagPayloads: wizardFlagPayloads, + cliHarness: session.harness, + cliSequence: session.sequence, + cliModel: session.model, + binding: bindingFor(programConfig.id), + }; + const binding = resolveBinding(switchboard); + analytics.setTag('sequence', binding.sequence); + analytics.setTag('harness', binding.harness); + wizardMetadata.SEQUENCE = binding.sequence; + wizardMetadata.HARNESS = binding.harness; + captureSwitchboardDecision(switchboard, binding); + + const ui = getUI(); + + // Cleanup coverage for the abort/cancel path: `wizardAbort` runs the + // registered cleanups, and the agent's own `finally` covers completion. + // flushScanReport is idempotent, so the overlap is a harmless no-op. + registerCleanup(() => + flushScanReport({ yaraReport: session.yaraReport }, (message) => + ui.log.info(message), + ), + ); + + // Linear settings restoration fires on entry to the outro screen, so it + // is registered before the run can reach that screen. Same owner, same + // timing as before; the abort path still restores through the cleanup + // `backupAndFixClaudeSettings` registered. + if (binding.sequence === Sequence.linear) { + ui.onEnterScreen('outro', () => restoreClaudeSettings(session.installDir)); + } + + const framework = session.integration ?? session.skillId ?? undefined; + const config: RunConfig = { + programId: programConfig.id, + run, + composed, + binding, + switchboard, + skillsBaseUrl: getSkillsBaseUrl(), + wizardFlags, + wizardFlagPayloads, + wizardMetadata, + allowedTools: programConfig.allowedTools, + disallowedTools: programConfig.disallowedTools, + agentFlow: programConfig.agentFlow, + seedTasks: programConfig.seedTasks + ? () => programConfig.seedTasks!(session) + : undefined, + hooks: { + postRun: run.postRun + ? (creds) => run.postRun!(session, creds) + : undefined, + buildOutroData: run.buildOutroData + ? (creds) => run.buildOutroData!(session, creds) ?? undefined + : undefined, + buildOutroNextSteps: run.buildOutroNextSteps + ? (creds, completed) => + run.buildOutroNextSteps!(session, creds, completed) + : undefined, + }, + }; + const input: RunInput = { + installDir: session.installDir, + credentials, + project: session.apiProject, + apiUser: session.apiUser, + skillId: session.skillId ?? undefined, + integration: session.integration, + frameworkDocsUrl: framework + ? FRAMEWORK_REGISTRY[framework as Integration]?.metadata.docsUrl + : undefined, + flags: { + ci: session.ci, + signup: session.signup, + debug: session.debug, + e2eAsk: session.e2eAsk, + localMcp: session.localMcp, + captureAio: session.captureAio, + benchmark: session.benchmark, + yaraReport: session.yaraReport, + }, + host: { + baseUrl: session.baseUrl, + region: session.region, + email: session.email, + projectId: session.projectId, + apiKey: session.apiKey, + }, + }; + + const result = await runAgentFunctional(config, input, { + onProgress: createUiReducer(ui), + interaction: uiInteraction(ui), + }); + + // Success already rendered through the reducer (outro data, outro line) and + // shut analytics down. A decided failure exits exactly as it always has. An + // error the agent did not decide used to propagate out of this call, and the + // runners handle it (mint-failure screen, classifyRunFailure): keep throwing + // the original error for them. + if (result.outcome === 'crashed') { + throw result.failure?.error ?? new Error(result.failure?.message); + } + if (result.outcome !== 'success') { + await wizardAbort(result.failure); + } +} + +// ── Gates ───────────────────────────────────────────────────────────── + +async function runHealthGate( + session: WizardSession, + programConfig: ProgramConfig, +): Promise { + const hasHealthCheckScreen = programConfig.steps.some( + (s) => s.screenId === 'health-check', + ); + if (session.readinessResult) { + logToFile( + `[agent-runner] readiness pre-computed by TUI: decision=${session.readinessResult.decision}` + + `${ + session.outageDismissed ? ' (outage dismissed by user)' : '' + } — skipping re-check`, + ); + } + if (!hasHealthCheckScreen || session.readinessResult) return; + + logToFile('[agent-runner] evaluating wizard readiness'); + const readinessConfig = session.signup + ? SIGNUP_WIZARD_READINESS_CONFIG + : undefined; + const readiness = await evaluateWizardReadiness(readinessConfig); + logToFile(`[agent-runner] readiness=${readiness.decision}`); + if (readiness.decision === WizardReadiness.No) { + const blockingKeys = getBlockingServiceKeys( + readiness.health, + readinessConfig, + ); + const blockingLabels = blockingKeys.map( + (k) => `${SERVICE_LABELS[k]} (${readiness.health[k].status})`, + ); + logToFile(`[agent-runner] blocked by: ${blockingLabels.join(', ')}`); + + await getUI().showBlockingOutage(readiness); + + // The TUI lets the user continue past an outage; non-interactive runs + // (CI) do the same automatically — the degraded services are reported + // above, but we proceed rather than aborting on a transient upstream blip. + if (!isNonInteractiveEnvironment()) { + await wizardAbort({ + code: ErrorCodes.EnvServiceOutage, + message: + 'Cannot start — external services are down:\n' + + blockingLabels.map((l) => ` - ${l}`).join('\n') + + '\n\nPlease try again later.', + }); + } + } else if (readiness.decision === WizardReadiness.YesWithWarnings) { + getUI().setReadinessWarnings(readiness); + } +} + +async function runSettingsGate(session: WizardSession): Promise { + const settingsConflicts = checkAllSettingsConflicts(session.installDir); + logToFile( + `[agent-runner] settings conflicts: ${ + settingsConflicts.length > 0 + ? settingsConflicts + .map((c) => `${c.source}(${c.keys.join(',')})`) + .join('; ') + : 'none' + }`, + ); + if (settingsConflicts.length === 0) return; + + for (const conflict of settingsConflicts) { + const level = conflict.source === 'managed' ? 'org' : conflict.source; + analytics.wizardCapture('settings conflict detected', { + level, + keys: conflict.keys, + }); + } + + const { autoFix, failClosed, warnOnly } = + classifySettingsConflicts(settingsConflicts); + + // User-global and project-local files are already neutralized — the agent + // runs with settingSources:['project'], so the SDK never reads them. Record + // it and move on; don't make the user act on a setting that can't bite. + for (const conflict of warnOnly) { + logToFile( + `[agent-runner] settings conflict in ${conflict.source} (${conflict.path}) ` + + `neutralized by settingSources:['project'] — not blocking`, + ); + analytics.wizardCapture('settings conflict neutralized', { + level: conflict.source, + keys: conflict.keys, + }); + } + + // Writable project settings.json — the SDK *does* read it, but we can back + // it up and remove it (restored at outro). Neutralize without prompting. + let unfixable = failClosed; + if (autoFix.length > 0) { + const fixed = backupAndFixClaudeSettings(session.installDir); + if (fixed) { + logToFile('[agent-runner] auto-neutralized writable settings conflict'); + analytics.wizardCapture('settings conflict auto-neutralized', { + keys: autoFix.flatMap((c) => c.keys), + }); + } else { + // Couldn't remove it — don't run into the redirect; fail closed instead. + logToFile( + '[agent-runner] could not back up writable settings conflict — failing closed', + ); + unfixable = [...failClosed, ...autoFix]; + } + } + + // What we cannot neutralize (org-managed, always read by the SDK; or a + // writable file we failed to back up) must be fixed by the user. Fail + // closed: the screen names the file + keys and exits. + if (unfixable.length > 0) { + if (isNonInteractiveEnvironment()) { + await wizardAbort({ + code: ErrorCodes.SettingsUnfixableConflict, + message: + 'Cannot start — a Claude settings file redirects the agent away ' + + 'from the PostHog gateway and cannot be neutralized automatically:\n' + + unfixable + .map((c) => ` - ${c.source} (${c.path}): ${c.keys.join(', ')}`) + .join('\n') + + '\n\nRemove the conflicting keys and re-run the wizard.', + }); + } + await getUI().showSettingsOverride(unfixable, () => + backupAndFixClaudeSettings(session.installDir), + ); + logToFile('[agent-runner] settings override resolved'); + } +} + +// ── Switchboard telemetry ───────────────────────────────────────────── + +/** + * One event + one log line per run: what entered the switchboard, which + * precedence rung decided each axis, and the final pick. + */ +function captureSwitchboardDecision( + ctx: SwitchboardCtx, + binding: ProgramBinding, +): void { + const trace = ctx.trace ?? {}; + // Unpinned orchestrator runs choose a model per task from the context-mill agent prompts; the orchestrator logs that map once the prompts load. + const perTaskModel = + binding.sequence === Sequence.orchestrator && trace.model === 'binding'; + const model = perTaskModel ? 'chosen-per-task' : binding.model; + const modelSource = perTaskModel ? 'agent-prompts' : trace.model; + analytics.wizardCapture('switchboard resolved', { + program: ctx.program, + flag_self_driving_use_pi_harness: + ctx.flags[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY], + flag_self_driving_pi_payload: JSON.stringify( + ctx.flagPayloads?.[WIZARD_SELF_DRIVING_USE_PI_HARNESS_FLAG_KEY] ?? null, + ), + flag_orchestrator: ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY], + cli_harness: ctx.cliHarness, + cli_sequence: ctx.cliSequence, + cli_model: ctx.cliModel, + harness_source: trace.harness, + model_source: modelSource, + sequence_source: trace.sequence, + harness: binding.harness, + model, + thinking_level: binding.thinkingLevel, + sequence: binding.sequence, + }); + logToFile( + `[switchboard] decision: program=${ctx.program}` + + ` in(orchestrator=${ctx.flags[WIZARD_ORCHESTRATOR_FLAG_KEY] ?? '-'},` + + ` cli=${ctx.cliHarness ?? '-'}/${ctx.cliSequence ?? '-'}/${ + ctx.cliModel ?? '-' + })` + + ` → harness=${binding.harness} (${trace.harness ?? '?'})` + + ` model=${model} (${modelSource ?? '?'})` + + ` sequence=${binding.sequence} (${trace.sequence ?? '?'})`, + ); +} + +// ── Progress → WizardUI, one call per event ─────────────────────────── + +/** + * The inverse of the agent's former `getUI()` calls: one event, one + * `WizardUI` method, synchronous, in emission order. Because each case maps + * back to exactly the call the agent used to make, the frame and flow goldens + * hold without regeneration. + */ +export function createUiReducer(ui: WizardUI): (event: AgentProgress) => void { + let spinner: SpinnerHandle | undefined; + return (event) => { + switch (event.kind) { + case 'lifecycle': + if (event.phase === 'started') ui.startRun(); + else ui.outro(event.message); + break; + case 'spinner': { + const handle = (spinner ??= ui.spinner()); + handle[event.action](event.message); + break; + } + case 'log': + ui.log[event.level](event.message); + break; + case 'status': + ui.pushStatus(event.message); + break; + case 'tasks': + ui.syncTodos(event.tasks); + break; + case 'stage': + ui.setStage(event.stage); + break; + case 'url': + if (event.which === 'dashboard') ui.setDashboardUrl(event.url); + else ui.setNotebookUrl(event.url); + break; + case 'usage': + ui.addTokenUsage(event.delta); + break; + case 'finalCost': + ui.setFinalTokenCostUsd(event.usd); + break; + case 'handoff': + ui.setHandoffText(event.text); + break; + case 'authError': + ui.showAuthError(event.detail); + break; + case 'completion': + ui.setOutroData(event.outro); + break; + } + }; +} + +/** The agent's questions, answered wherever `getUI()` answers them today. */ +export function uiInteraction(ui: WizardUI): AgentInteraction { + return { + ask: (question) => ui.requestQuestion(question), + cancelAsk: () => ui.cancelPendingQuestion(), + taskNotice: (notice) => ui.showTaskNotice(notice), + cancelTaskNotice: () => ui.cancelTaskNotice(), + }; +} diff --git a/src/lib/programs/self-driving/detect.ts b/src/lib/programs/self-driving/detect.ts index d9c74f055..804fcd245 100644 --- a/src/lib/programs/self-driving/detect.ts +++ b/src/lib/programs/self-driving/detect.ts @@ -29,7 +29,7 @@ import { import { join } from 'path'; import { analytics } from '@utils/analytics'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; import { detectWarehouseSources } from '@lib/warehouse-sources/detect'; import type { DetectedSource } from '@lib/warehouse-sources/types'; diff --git a/src/lib/programs/self-driving/index.ts b/src/lib/programs/self-driving/index.ts index b507cc18e..38a0ad930 100644 --- a/src/lib/programs/self-driving/index.ts +++ b/src/lib/programs/self-driving/index.ts @@ -1,7 +1,7 @@ import { join } from 'path'; import { access, rm } from 'node:fs/promises'; import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import { OutroKind, type WizardSession } from '@lib/wizard-session'; import { createSkillProgram } from '../agent-skill/index.js'; import { SELF_DRIVING_PROGRAM } from './steps.js'; diff --git a/src/lib/programs/self-driving/prompt.ts b/src/lib/programs/self-driving/prompt.ts index 5bb0a6af2..5bfe59839 100644 --- a/src/lib/programs/self-driving/prompt.ts +++ b/src/lib/programs/self-driving/prompt.ts @@ -1,5 +1,5 @@ import { AgentSignals } from '@lib/agent/agent-interface'; -import type { PromptContext } from '@lib/agent/agent-runner'; +import type { PromptContext } from '@lib/programs/program-run'; import type { DetectedSource } from '@lib/warehouse-sources/types'; /** diff --git a/src/lib/programs/warehouse-source/detect.ts b/src/lib/programs/warehouse-source/detect.ts index 9e97b7d0c..0386593e1 100644 --- a/src/lib/programs/warehouse-source/detect.ts +++ b/src/lib/programs/warehouse-source/detect.ts @@ -9,7 +9,7 @@ import { existsSync, statSync } from 'fs'; import { analytics } from '@utils/analytics'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { detectWarehouseSources } from '@lib/warehouse-sources/detect'; import type { DetectedSource } from '@lib/warehouse-sources/types'; diff --git a/src/lib/programs/warehouse-source/index.ts b/src/lib/programs/warehouse-source/index.ts index 059e240a0..4b31f3851 100644 --- a/src/lib/programs/warehouse-source/index.ts +++ b/src/lib/programs/warehouse-source/index.ts @@ -1,5 +1,5 @@ import type { ProgramConfig } from '@lib/programs/program-step'; -import type { ProgramRun } from '@lib/agent/agent-runner'; +import type { ProgramRun } from '@lib/programs/program-run'; import type { WizardSession } from '@lib/wizard-session'; import { LONGER_ASK_TIMEOUT_MS } from '@lib/wizard-ask-bridge'; import { WAREHOUSE_SOURCE_PROGRAM } from './steps.js'; diff --git a/src/lib/programs/web-analytics-doctor/detect.ts b/src/lib/programs/web-analytics-doctor/detect.ts index 749f4e75d..ec224425c 100644 --- a/src/lib/programs/web-analytics-doctor/detect.ts +++ b/src/lib/programs/web-analytics-doctor/detect.ts @@ -1,6 +1,6 @@ import { existsSync, statSync } from 'fs'; import type { WizardSession } from '@lib/wizard-session'; -import type { AbortCase } from '@lib/agent/agent-runner'; +import type { AbortCase } from '@lib/programs/program-run'; import { ErrorCodes } from '@lib/errors'; import { findPackageJsons } from '@lib/programs/shared/package-scanning'; diff --git a/src/lib/runners/__tests__/mint-recovery.test.ts b/src/lib/runners/__tests__/mint-recovery.test.ts index 24f9b5d83..076bf9a20 100644 --- a/src/lib/runners/__tests__/mint-recovery.test.ts +++ b/src/lib/runners/__tests__/mint-recovery.test.ts @@ -1,6 +1,6 @@ import { vi, it, expect, afterEach } from 'vitest'; import { runWizard } from '../run-wizard'; -import { runAgent } from '@lib/agent/agent-runner'; +import { runAgent } from '@lib/programs/run-agent-legacy'; import { startTUI } from '@ui/tui/start-tui'; import { WizardStore } from '@ui/tui/store'; import { InkUI } from '@ui/tui/ink-ui'; @@ -10,7 +10,7 @@ import { ScreenId } from '@ui/tui/router'; import { HostResolution } from '@lib/host-resolution'; import { analytics } from '@utils/analytics'; -vi.mock('@lib/agent/agent-runner', () => ({ runAgent: vi.fn() })); +vi.mock('@lib/programs/run-agent-legacy', () => ({ runAgent: vi.fn() })); vi.mock('@ui/tui/start-tui', () => ({ startTUI: vi.fn() })); vi.mock('@lib/local-dev', async (original) => ({ ...(await original()), diff --git a/src/lib/runners/run-non-interactive.ts b/src/lib/runners/run-non-interactive.ts index 7230c3ec0..a4fd1f037 100644 --- a/src/lib/runners/run-non-interactive.ts +++ b/src/lib/runners/run-non-interactive.ts @@ -335,7 +335,7 @@ export function runNonInteractive( } } - const { runAgent } = await import('@lib/agent/agent-runner'); + const { runAgent } = await import('@lib/programs/run-agent-legacy'); await runAgent(config, session); await settleStream(RunPhase.Completed); } catch (error) { diff --git a/src/lib/runners/run-wizard.ts b/src/lib/runners/run-wizard.ts index 1ee2e7120..a0e0879ff 100644 --- a/src/lib/runners/run-wizard.ts +++ b/src/lib/runners/run-wizard.ts @@ -1,7 +1,7 @@ import { VERSION } from '@lib/version'; import { logToFile, getLogFilePath } from '@utils/debug'; -import { runAgent } from '@lib/agent/agent-runner'; -import { authenticate } from '@lib/agent/runner/shared/authenticate'; +import { runAgent } from '@lib/programs/run-agent-legacy'; +import { authenticate } from '@lib/programs/authenticate'; import { getProgramConfig } from '@lib/programs/program-registry'; import { getAuditChecks } from '@lib/programs/audit/types'; import { maybeStampAiSdkDetected } from '@lib/programs/posthog-integration/detect'; diff --git a/src/lib/wizard-tools/__tests__/handoff.test.ts b/src/lib/wizard-tools/__tests__/handoff.test.ts index bbe2aae14..ea773abe8 100644 --- a/src/lib/wizard-tools/__tests__/handoff.test.ts +++ b/src/lib/wizard-tools/__tests__/handoff.test.ts @@ -11,34 +11,31 @@ import { } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from 'node:path'; -import { getUI, setUI } from '@ui'; -import type { WizardUI } from '@ui/wizard-ui'; -import { MAX_HANDOFF_TEXT_CHARS, publishHandoff } from '../handoff'; +import { + MAX_HANDOFF_TEXT_CHARS, + publishHandoff as publishHandoffTo, +} from '../handoff'; describe('publishHandoff', () => { const captured: string[] = []; - let previousUI: WizardUI; let temporaryDirectory: string; let ambientOutputPath: string | undefined; + // The reporting seam: what the agent's `handoff` progress event carries. + const publishHandoff = (content: string) => + publishHandoffTo(content, (text) => { + captured.push(text); + }); beforeEach(() => { captured.length = 0; - previousUI = getUI(); temporaryDirectory = mkdtempSync(join(tmpdir(), 'wizard-handoff-')); // Save the ambient value so a developer running these tests with the var // exported doesn't have their environment silently clobbered. ambientOutputPath = process.env.POSTHOG_HANDOFF_OUTPUT_PATH; delete process.env.POSTHOG_HANDOFF_OUTPUT_PATH; - setUI({ - ...previousUI, - setHandoffText: (text: string) => { - captured.push(text); - }, - } as WizardUI); }); afterEach(() => { - setUI(previousUI); rmSync(temporaryDirectory, { recursive: true, force: true }); if (ambientOutputPath === undefined) { delete process.env.POSTHOG_HANDOFF_OUTPUT_PATH; @@ -47,7 +44,7 @@ describe('publishHandoff', () => { } }); - it('publishes the content through the UI seam', () => { + it('publishes the content through the reporting seam', () => { const result = publishHandoff('# Setup report\n\nAll done.'); expect(result.ok).toBe(true); expect(captured).toEqual(['# Setup report\n\nAll done.']); diff --git a/src/lib/wizard-tools/handoff.ts b/src/lib/wizard-tools/handoff.ts index 4dedf0428..4f3b3217d 100644 --- a/src/lib/wizard-tools/handoff.ts +++ b/src/lib/wizard-tools/handoff.ts @@ -3,7 +3,6 @@ * markdown) in one explicit call, replacing the report file + watcher path. */ -import { getUI } from '@ui'; import { analytics } from '@utils/analytics'; import { logToFile } from '@utils/debug'; import { runtimeEnv } from '@env'; @@ -134,7 +133,10 @@ function writeHandoffFileAtomically( } } -export function publishHandoff(content: string): PublishHandoffResult { +export function publishHandoff( + content: string, + onHandoffText?: (text: string) => void, +): PublishHandoffResult { if (content.trim() === '') { analytics.wizardCapture('handoff published', { handoff_ok: false, @@ -148,7 +150,7 @@ export function publishHandoff(content: string): PublishHandoffResult { } const truncated = content.length > MAX_HANDOFF_TEXT_CHARS; const text = truncated ? content.slice(0, MAX_HANDOFF_TEXT_CHARS) : content; - getUI().setHandoffText(text); + onHandoffText?.(text); const handoffOutputPath = runtimeEnv('POSTHOG_HANDOFF_OUTPUT_PATH'); let handoffOutputWritten: boolean | undefined; diff --git a/src/lib/wizard-tools/mcp.ts b/src/lib/wizard-tools/mcp.ts index 4ad43e235..493724991 100644 --- a/src/lib/wizard-tools/mcp.ts +++ b/src/lib/wizard-tools/mcp.ts @@ -147,6 +147,9 @@ export interface WizardToolsOptions { /** Scan-triage classifier for install_skill's scan, resolved by the caller. */ triageProvider: LLMProvider; + + /** Receives the handoff document `publish_handoff` accepted. */ + onHandoffText?: (text: string) => void; } /** Default per-run cap on wizard_ask calls when no override is provided. */ @@ -168,6 +171,7 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { secretVault = createSecretVault(), orchestrator, triageProvider, + onHandoffText, } = options; const sdk = await getSDKModule(); const { tool, createSdkMcpServer } = sdk; @@ -755,7 +759,7 @@ export async function createWizardToolsServer(options: WizardToolsOptions) { content: z.string().describe(PUBLISH_HANDOFF_CONTENT_DESCRIPTION), }, (args: { content: string }) => { - const result = publishHandoff(args.content); + const result = publishHandoff(args.content, onHandoffText); logToFile(`publish_handoff: ${result.message}`); return { content: [{ type: 'text' as const, text: result.message }], diff --git a/src/lib/yara-hooks.ts b/src/lib/yara-hooks.ts index 3ec2cb266..63dadafe3 100644 --- a/src/lib/yara-hooks.ts +++ b/src/lib/yara-hooks.ts @@ -35,8 +35,6 @@ import type { import { logToFile } from '@utils/debug'; import { readFileHead } from '@utils/bounded-fs'; import { analytics } from '@utils/analytics'; -import { getUI } from '@ui'; -import type { WizardSession } from '@lib/wizard-session'; import { isSkillInstallCommand } from './skill-install'; import { highestSeverityMatch, @@ -431,13 +429,14 @@ export function captureScanReport(): void { * scanCount === 0 and every step no-ops. */ export function flushScanReport( - session: Pick, + options: { yaraReport: boolean }, + log: (message: string) => void, ): void { - if (session.yaraReport) { + if (options.yaraReport) { const reportPath = writeScanReport(); if (reportPath) { const summary = formatScanReport(); - getUI().log.info(`YARA scan report: ${reportPath}${summary ?? ''}`); + log(`YARA scan report: ${reportPath}${summary ?? ''}`); } } captureScanReport(); diff --git a/src/ui/tui/services/mcp-suggested-prompts-services.ts b/src/ui/tui/services/mcp-suggested-prompts-services.ts index 8b41505ca..f5ba6cd66 100644 --- a/src/ui/tui/services/mcp-suggested-prompts-services.ts +++ b/src/ui/tui/services/mcp-suggested-prompts-services.ts @@ -21,22 +21,8 @@ import { } from '@lib/mcp-project-profile'; import { seedDemoEvents as runSeed } from '@lib/mcp-seed-events'; -/** - * Discriminated union covering every kind of streamed event the screen - * needs to render. Production yields these from Claude SDK messages; - * the playground yields them from canned scripts. - */ -export type AgentChunk = - | { kind: 'text'; text: string } - /** `command` carries CLI mode's exec command string (`call …`) so the - * screen can recover the inner tool for context-aware follow-ups. */ - | { kind: 'tool-call'; toolName: string; detail: string; command?: string } - | { kind: 'tool-result'; toolName: string; detail: string } - | { kind: 'error'; text: string } - /** Stream completed. `sessionId` is the SDK session ID of the just- - * completed turn; pass it back as `resumeSessionId` on a follow-up - * call to continue the conversation with full history. */ - | { kind: 'done'; sessionId?: string }; +export type { AgentChunk } from '@lib/agent/mcp-prompt-streaming'; +import type { AgentChunk } from '@lib/agent/mcp-prompt-streaming'; export interface McpSuggestedPromptsServices { /** diff --git a/src/ui/wizard-ui.ts b/src/ui/wizard-ui.ts index 38ad29235..7a96fea1b 100644 --- a/src/ui/wizard-ui.ts +++ b/src/ui/wizard-ui.ts @@ -9,6 +9,11 @@ */ import type { SettingsConflict } from '@lib/agent/claude-settings'; +import type { + AuthErrorDetail, + SpinnerHandle, + TokenUsageDelta, +} from '@lib/agent/progress'; import type { WizardReadinessResult } from '@lib/health-checks/readiness'; import type { ApiUser } from '@lib/api'; import type { Credentials, TaskNotice } from '@lib/wizard-session'; @@ -29,60 +34,11 @@ export function isTaskStatus(value: string): value is TaskStatus { return (Object.values(TaskStatus) as string[]).includes(value); } -/** - * One assistant turn's token usage, for the hidden Ctrl+T token/cost HUD. - * `model` is the model that produced *this* turn (e.g. the SDK's - * `message.message.model`) — a subagent can run on a different model than - * the main session, and some programs override to Haiku, so pricing must key - * off the per-turn model rather than a single run-wide assumption. Omit only - * when the caller genuinely has no model context (falls back to Sonnet - * pricing — see `pricePerMtokForModel` in `@lib/agent/token-pricing`). - */ -export interface TokenUsageDelta { - inputTokens: number; - outputTokens: number; - cacheReadTokens: number; - cacheCreationTokens: number; - cacheCreation5m: number; - cacheCreation1h: number; - model?: string; -} - -export interface SpinnerHandle { - start(message?: string): void; - stop(message?: string): void; - message(msg?: string): void; -} - -/** - * Context passed to `showAuthError` so the screen can pick the right copy. - * - * `hasSettingsConflict` is true when a Claude Code settings file (project, - * project-local, the user's global config, or managed) actually overrides the - * LLM Gateway auth. `conflicts` carries the exact files and keys so the screen - * can name them. When there is no conflict, the 401 has a different cause (bad - * PAT prefix, missing scope, expired key, region mismatch) and we should not - * advise the user to log out of Claude Code. - */ -export interface AuthErrorDetail { - hasSettingsConflict: boolean; - conflicts?: SettingsConflict[]; - /** - * True when the agent SDK authenticated from a stored Claude login - * (`apiKeySource: "/login managed key"`) instead of the wizard's gateway - * token — conflicting Anthropic credentials. Takes priority in the screen. - */ - usingManagedLogin?: boolean; - /** Human-readable places a conflicting Anthropic credential may live. */ - credentialPlaces?: string[]; - /** - * True when a pre-run refresh already failed on a dead grant. The login is - * gone and re-running is the only fix, so this outranks every other branch — - * none of the usual advice (key type, scopes, region) applies. - */ - sessionExpired?: boolean; - logFilePath: string; -} +export type { + AuthErrorDetail, + SpinnerHandle, + TokenUsageDelta, +} from '@lib/agent/progress'; export interface WizardUI { // ── Lifecycle messages ──────────────────────────────────────────── diff --git a/src/utils/wizard-abort.ts b/src/utils/wizard-abort.ts index 7805edcc4..bb3731a8e 100644 --- a/src/utils/wizard-abort.ts +++ b/src/utils/wizard-abort.ts @@ -13,20 +13,9 @@ import { LoggingUI } from '@ui/logging-ui'; import { OutroKind, type OutroData } from '@lib/wizard-session'; import type { ErrorCode } from '@lib/errors'; import { emitWizardError, sanitizeErrorDetail } from '@lib/errors'; +import { WizardError } from '@lib/errors/wizard-error'; -export class WizardError extends Error { - readonly code?: ErrorCode; - - constructor( - message: string, - public readonly context?: Record, - code?: ErrorCode, - ) { - super(message); - this.name = 'WizardError'; - this.code = code; - } -} +export { WizardError }; interface WizardAbortOptions { message?: string;