From 535facf9d75df7375746401222244b03434f9a1a Mon Sep 17 00:00:00 2001
From: "Vincent (Wen Yu) Ge"
Date: Sat, 19 Sep 2026 01:19:11 -0400
Subject: [PATCH 1/8] chore(harness): drive the e2e routes over the control
socket
The snapshot route and the wizard-ci MCP server spawn the real binary with
`--ci --control-socket` in a PTY and drive it through the control API: the
run is released with POST /run, state is long polled, and every commit goes
through POST /actions. The in-process host, its store driver, and the action
registry are gone; the harness keeps the launcher, the detection picks, the
profiles, and the result payload. Harness tests run against the store's
ControlDriver and per-flow actions with unchanged goldens, and two process
specs exercise each surface end to end. Docs describe the API and the
launcher; the workbench environment contract is unchanged.
Generated-By: PostHog Desktop
Task-Id: d14e92bb-6ee1-49b5-8502-39cb80079589
---
.claude/skills/exploring-the-wizard/SKILL.md | 61 +-
docs/benchmarking.md | 167 ++--
docs/local-dev.md | 115 +--
e2e-harness/ARCHITECTURE.md | 286 ++++---
.../__tests__/control-actions-parity.test.ts | 54 --
...-driver.test.ts => control-driver.test.ts} | 59 +-
.../__tests__/control-socket-headless.test.ts | 77 ++
.../__tests__/control-socket-tui.test.ts | 82 ++
.../__tests__/e2e-flow-snapshot.test.ts | 7 +-
e2e-harness/__tests__/e2e-profile-ask.test.ts | 10 +-
e2e-harness/__tests__/e2e-result.test.ts | 4 +-
.../__tests__/keyboard-equivalence.test.tsx | 4 +-
e2e-harness/__tests__/launch.test.ts | 99 +++
e2e-harness/action-registry.ts | 402 ----------
e2e-harness/e2e-profile.ts | 10 +-
e2e-harness/e2e-result.ts | 33 +-
e2e-harness/launch.ts | 109 +++
e2e-harness/picks.ts | 101 +++
e2e-harness/wizard-ci-driver.ts | 220 -----
scripts/README.md | 75 +-
scripts/tui-host.no-jest.ts | 754 ------------------
scripts/tui-snapshots.no-jest.ts | 401 +++++++++-
scripts/wizard-ci-mcp.no-jest.ts | 144 ++--
src/store/control/__tests__/state.test.ts | 10 +-
src/store/control/state.ts | 1 +
src/store/control/types.ts | 1 +
src/store/programs/index.ts | 1 +
src/store/state/store-api.ts | 2 -
28 files changed, 1343 insertions(+), 1946 deletions(-)
delete mode 100644 e2e-harness/__tests__/control-actions-parity.test.ts
rename e2e-harness/__tests__/{wizard-ci-driver.test.ts => control-driver.test.ts} (89%)
create mode 100644 e2e-harness/__tests__/control-socket-headless.test.ts
create mode 100644 e2e-harness/__tests__/control-socket-tui.test.ts
create mode 100644 e2e-harness/__tests__/launch.test.ts
delete mode 100644 e2e-harness/action-registry.ts
create mode 100644 e2e-harness/launch.ts
create mode 100644 e2e-harness/picks.ts
delete mode 100644 e2e-harness/wizard-ci-driver.ts
delete mode 100644 scripts/tui-host.no-jest.ts
diff --git a/.claude/skills/exploring-the-wizard/SKILL.md b/.claude/skills/exploring-the-wizard/SKILL.md
index 4310f08d1..23d64733f 100644
--- a/.claude/skills/exploring-the-wizard/SKILL.md
+++ b/.claude/skills/exploring-the-wizard/SKILL.md
@@ -9,7 +9,7 @@ compatibility:
wizard-ci MCP server.
metadata:
author: posthog
- version: '5.0'
+ version: '6.0'
---
# Exploring the wizard as an agent
@@ -23,8 +23,8 @@ sequence, and gateway policy. For new exploration, launch the server with
`SNAP_HARNESS=pi`; prefer `SNAP_SEQUENCE=orchestrator` for the integration flow.
These are server environment variables, not MCP arguments. Restart an existing
server to change its environment. See the
-[host architecture](../../../e2e-harness/ARCHITECTURE.md) for other programs,
-overrides, and current limitations.
+[harness architecture](../../../e2e-harness/ARCHITECTURE.md) for the control
+API, other programs, and overrides.
## Prepare the run
@@ -35,16 +35,14 @@ wizard, so finish recording one app before opening another.
- **Detection only:** pass `appDir` and `projectId` (both required strings),
with no key. Stop at `auth` without calling `run_agent`.
- **Full integration:** reuse the authorized phx key file path, separate gateway
- token file path, and project id;
- ask only for missing inputs. Prefer `keyFile` so the key stays out of tool
- arguments. Set `WIZARD_CI_GATEWAY_TOKEN_FILE` in the MCP server environment
- before launch (restart an existing server); it is not an `open_app` argument.
- The file must contain an already-issued gateway bearer, not the phx key.
- CI does not mint or refresh it. Never print or commit either secret. See
- [local credential setup](../../../docs/local-dev.md#credentials-for-local-ci-and-headless-runs). Read the
- [credential and region limitations](../../../e2e-harness/ARCHITECTURE.md#current-host-limitations)
- before starting: an inherited key can shadow `keyFile`, and the host currently
- hardcodes the US region.
+ token file path, and project id; ask only for missing inputs. Prefer `keyFile`
+ so the key stays out of tool arguments. Set `WIZARD_CI_GATEWAY_TOKEN_FILE` in
+ the MCP server environment before launch (restart an existing server); it is
+ not an `open_app` argument. The file must contain an already-issued gateway
+ bearer, not the phx key. CI does not mint or refresh it. Never print or commit
+ either secret. See
+ [local credential setup](../../../docs/local-dev.md#credentials-for-local-ci-and-headless-runs).
+ `region` selects the PostHog region for auth and the gateway.
- **Questions during the run:** launch the server with `E2E_ASK=true` to keep
`wizard_ask` available in this CI session. Handle questions yourself through
the actions below; fixed-route answer profiles do not drive the MCP route.
@@ -79,31 +77,33 @@ own decisions through the same state and action contract.
judging detection. Inspect `session.integration` and `setupQuestions`.
2. Capture `render_screen` before each decision and during task or phase
changes. Save numbered frames such as `/tmp/wz-explore-snaps/01-intro.txt`.
-3. Commit only actions currently offered. Common choices are below; the
- [action registry](../../../e2e-harness/action-registry.ts) defines the full
- set.
+3. Commit only actions currently offered. Common choices are below; the generic
+ set lives in
+ [`src/store/control/actions.ts`](../../../src/store/control/actions.ts) and a
+ program adds its own through `controlActions` on its steps.
4. For a full run, confirm setup and call `run_agent` at `auth`. Continue
reading state and handling overlays while it runs; polling alone cannot
answer them.
5. Check `runPhase` (`idle`, `running`, `completed`, `error`), background
status, and the rendered outro. On error, capture the frame and reason before
dismissing it. An error outro can wait for dismissal while `integration`
- still says `running`; a host exit can instead surface as a socket error.
+ still says `running`; a wizard exit can instead surface as a socket error.
6. After successful agent completion, finish the offered outro and follow-up
actions. For the integration flow, `session.skillsComplete` marks the tail's
completion. Other programs can have a terminal outro or exit screen.
-| Decision | Action and `params` |
-| ---------------------------------------- | -------------------------------------------------------------------------------------------- |
-| Confirm intro / dismiss blocking outage | `confirm_setup` / `dismiss_outage` |
-| Answer setup question | `choose`, `{ key, value }` from `setupQuestions` |
-| Answer every question in a pending batch | `answer_question`, `{ answers: { questionId: value } }`; values are strings or string arrays |
-| Cancel a question batch | `cancel_question` |
-| Accept or decline an optional task | `resolve_notice`, `{ keep: true }` or `{ keep: false }` |
-| Finish outro | `dismiss_outro` |
-| Record MCP outcome | `set_mcp_outcome`, `{ outcome: "skipped" }` or `{ outcome: "installed", clients: [...] }` |
-| Dismiss suggested prompts / Slack step | `dismiss` / `dismiss_slack` |
-| Record keep-skills choice | `keep_skills`, `{ kept: true }` or `{ kept: false }` |
+| Decision | Action and `params` |
+| ---------------------------------------- | ---------------------------------------------------------------------------------------------------------------- |
+| Confirm intro / dismiss blocking outage | `confirm_setup` / `dismiss_outage` |
+| Answer setup question | `choose`, `{ key, value }` from `setupQuestions` |
+| Answer every question in a pending batch | `answer_question`, `{ answers: { questionId: value } }`; values are strings or string arrays |
+| Cancel a question batch | `cancel_question` |
+| Accept or decline an optional task | `resolve_notice`, `{ keep: true }` or `{ keep: false }` |
+| Finish outro | `dismiss_outro` |
+| Record MCP outcome | `set_mcp_outcome`, `{ outcome: "skipped" }` or `{ outcome: "installed", clients: [...] }` |
+| Dismiss suggested prompts / Slack step | `dismiss` / `dismiss_slack` |
+| Record keep-skills choice | `keep_skills`, `{ kept: true }` or `{ kept: false }` |
+| Pick the project on a detect screen | `pick_integration_target`, `{ path, integration }`; source maps: `pick_source_maps_project`, `{ variant, path }` |
MCP and keep-skills actions commit store state; recording an outcome does not
perform the corresponding installation or cleanup. Report which outcomes were
@@ -129,5 +129,6 @@ The shared log is `/tmp/posthog-wizard.log`. Record its byte count before a run
and read from that count plus one afterward. Run sweeps serially so their logs
remain attributable. `read_state` omits `frameworkContext`; an empty
`setupQuestions` list alone does not prove a router mode. When necessary,
-inspect the detector under [`src/store/frameworks/`](../../../src/store/frameworks/) against
-the same fixture.
+inspect the detector under
+[`src/store/frameworks/`](../../../src/store/frameworks/) against the same
+fixture.
diff --git a/docs/benchmarking.md b/docs/benchmarking.md
index e5fadf4f7..17a4bc6e8 100644
--- a/docs/benchmarking.md
+++ b/docs/benchmarking.md
@@ -29,29 +29,29 @@ Selection criteria, checked in this order:
1. **Real product, in production** — an open-source app people actually run
(stars are a proxy; a hosted instance is better evidence).
2. **Single-app repo** — reject monorepos: fetch the repo's top-level listing
- and reject on `pnpm-workspace.yaml`, `turbo.json`, `lerna.json`, or
- top-level `apps/`/`packages/` directories.
+ and reject on `pnpm-workspace.yaml`, `turbo.json`, `lerna.json`, or top-level
+ `apps/`/`packages/` directories.
3. **Greenfield** — grep the repo for `posthog` (manifest and source). An app
that already integrates PostHog measures augmentation discipline, not
integration quality; keep at most one such app and exclude it from quality
scoring.
4. **Framework coverage** — spread picks across the frameworks the wizard
supports; results do not transfer between them.
-5. **Locally installable** — its toolchain (node/python/php/ruby/gradle)
- exists on the bench machine, or its runs will fail for reasons that are
- yours, not the model's.
+5. **Locally installable** — its toolchain (node/python/php/ruby/gradle) exists
+ on the bench machine, or its runs will fail for reasons that are yours, not
+ the model's.
Apps used in the 2026-07 benchmark, as worked examples of the spread:
-| app | upstream | stack |
-|---|---|---|
-| Maybe | `maybe-finance/maybe` | Rails |
-| Outline | `outline/outline` | React + Koa / TS |
-| WordPress-Android | `wordpress-mobile/WordPress-Android` | native Kotlin |
-| healthchecks | `healthchecks/healthchecks` | Django |
-| Firefly III | `firefly-iii/firefly-iii` | Laravel, server-rendered |
-| Monica | `monicahq/monica` | Laravel + Inertia/Vue |
-| Papermark | `mfts/papermark` | Next.js — already shipped posthog-js; kept as the augment-existing case, excluded from quality scoring |
+| app | upstream | stack |
+| ----------------- | ------------------------------------ | ------------------------------------------------------------------------------------------------------ |
+| Maybe | `maybe-finance/maybe` | Rails |
+| Outline | `outline/outline` | React + Koa / TS |
+| WordPress-Android | `wordpress-mobile/WordPress-Android` | native Kotlin |
+| healthchecks | `healthchecks/healthchecks` | Django |
+| Firefly III | `firefly-iii/firefly-iii` | Laravel, server-rendered |
+| Monica | `monicahq/monica` | Laravel + Inertia/Vue |
+| Papermark | `mfts/papermark` | Next.js — already shipped posthog-js; kept as the augment-existing case, excluded from quality scoring |
## Setup (from a bare machine)
@@ -90,21 +90,21 @@ Define each config as a `WIZARD_CI_FLAG_OVERRIDES` JSON, plus the baseline as
Everything below ships in this repo (`wizard/`) and its workbench
(`wizard-workbench/`); paths are from each repo's root.
-- **Headless run (snapshotting CI harness):** `wizard/scripts/tui-snapshots.no-jest.ts`
- spawns the real TUI (`wizard/scripts/tui-host.no-jest.ts`) in a PTY via
+- **Headless run (snapshotting CI harness):**
+ `wizard/scripts/tui-snapshots.no-jest.ts` spawns the real wizard
+ (`bin.ts --ci --control-socket`) in a PTY via
`wizard/e2e-harness/tui-capture.ts`, self-drives the fixed e2e profile
- (`wizard/e2e-harness/wizard-ci-driver.ts`, `wizard/e2e-harness/profiles.ts`)
- through auth, the agent run, and the outro, and writes each screen as an
- `NN-.ans` frame. An `NN-outro.ans` frame is the flow-completion
- signal. `tsx` runs source — no build step. Invocation: see the run-cell
- recipe below.
+ (`wizard/e2e-harness/profiles.ts`) over the control socket through auth, the
+ agent run, and the outro, and writes each screen as an `NN-.ans`
+ frame. An `NN-outro.ans` frame is the flow-completion signal. `tsx` runs
+ source — no build step. Invocation: see the run-cell recipe below.
- **Config selection:** the flag axis is `wizard-orchestrator` (on → the
orchestrator on pi, per-task models from context-mill frontmatter; off → the
linear anthropic default). Per-stage variations ride
- `wizard-orchestrator-override` payloads (`{stage: {model?, effort?}}`,
- variant keys in `wizard/src/agent/runner/switchboard/flags/schemes.ts`).
- The baseline is `{"wizard-orchestrator":"false"}` — never an empty override,
- or live remote flags leak into the baseline.
+ `wizard-orchestrator-override` payloads (`{stage: {model?, effort?}}`, variant
+ keys in `wizard/src/agent/runner/switchboard/flags/schemes.ts`). The baseline
+ is `{"wizard-orchestrator":"false"}` — never an empty override, or live remote
+ flags leak into the baseline.
## Running one cell
@@ -147,19 +147,19 @@ files=$(git -C "$WORK" diff --name-only main integ | wc -l | tr -d ' ')" \
| tee "$OUT/result.txt"
```
-Both commits are `--no-verify` (see Traps). The diff, frames, stdout, and
-result line are the cell's complete artifact set — everything else (the shared
-debug log) is unreliable under parallelism.
+Both commits are `--no-verify` (see Traps). The diff, frames, stdout, and result
+line are the cell's complete artifact set — everything else (the shared debug
+log) is unreliable under parallelism.
## Running the matrix
-- One app at a time; per app, launch its configs in parallel (≤4 on one
- machine) and `wait`. Contention inflates absolute times roughly uniformly.
+- One app at a time; per app, launch its configs in parallel (≤4 on one machine)
+ and `wait`. Contention inflates absolute times roughly uniformly.
- Cost: anthropic-harness cells report `modelUsage.costUSD` in
- `/tmp/posthog-wizard.log` — zero the log before each app's wave and slice
- the block per baseline run. pi-harness cells do not persist token totals;
- add a temporary hook in the pi harness success path that writes the session
- token stats to a per-run file, and price them at list rates.
+ `/tmp/posthog-wizard.log` — zero the log before each app's wave and slice the
+ block per baseline run. pi-harness cells do not persist token totals; add a
+ temporary hook in the pi harness success path that writes the session token
+ stats to a per-run file, and price them at list rates.
- Rerun any anomalous cell solo (zeroed log, no parallelism) before drawing a
conclusion from it.
@@ -167,61 +167,61 @@ debug log) is unreliable under parallelism.
- **Target-app git hooks.** Your `git commit` runs the app's husky/lint-staged
hooks if a prior install activated them; a failing hook silently rolls the
- tree back and the run measures as zero-diff. Always commit `--no-verify`.
- On any zero-diff run, check `git stash list` before believing it.
-- **Zero-diff has many causes.** Distinguish: the agent honestly declined
- (read its setup report), the agent's tool calls failed, your harness ate the
- work, or `.gitignore` hid it (env files never show in diffs). Attribute
- before you blame the model.
+ tree back and the run measures as zero-diff. Always commit `--no-verify`. On
+ any zero-diff run, check `git stash list` before believing it.
+- **Zero-diff has many causes.** Distinguish: the agent honestly declined (read
+ its setup report), the agent's tool calls failed, your harness ate the work,
+ or `.gitignore` hid it (env files never show in diffs). Attribute before you
+ blame the model.
- **"Reached the outro" is not success.** The flow completes even when nothing
was integrated. Treat completion as outro + a non-trivial diff.
- **Parallel runs interleave shared state.** The shared debug log cannot be
attributed per-run; capture everything per-run or run solo when attribution
matters.
- **Sandbox/allowlist gaps look like model failures.** If a config produces
- empty or thin work, check whether a blocked command (package-manager
- install, formatter) caused it, and whether other models worked around the
- same block. File the gap; exclude the affected cells.
-- **Repo-wide format scripts.** An agent running the app's `format`/`lint
- --fix` buries its real diff under hundreds of churn files. Count "real
- files" excluding scaffolding, lockfiles, env files — and read a sample of
- the churn before scoring.
-- **A stale credential fails silently mid-batch.** Read the key per run, not
- per session.
+ empty or thin work, check whether a blocked command (package-manager install,
+ formatter) caused it, and whether other models worked around the same block.
+ File the gap; exclude the affected cells.
+- **Repo-wide format scripts.** An agent running the app's `format`/`lint --fix`
+ buries its real diff under hundreds of churn files. Count "real files"
+ excluding scaffolding, lockfiles, env files — and read a sample of the churn
+ before scoring.
+- **A stale credential fails silently mid-batch.** Read the key per run, not per
+ session.
## Judging
Use the wizard-workbench PR evaluator's rubric — do not invent your own. It
lives at `wizard-workbench/services/pr-evaluator/`:
-- **Rubric criteria:** `wizard-workbench/services/pr-evaluator/prompts/evaluation.md`
- — per-item YES/NO/N-A checks grouped into four dimensions.
-- **Scoring math:** `wizard-workbench/services/pr-evaluator/evaluator.ts` —
- each dimension scores `max(1, round(pass_rate × 5))` over its applicable
- items; confidence = `min(app_sanity, round(mean of the four))`.
+- **Rubric criteria:**
+ `wizard-workbench/services/pr-evaluator/prompts/evaluation.md` — per-item
+ YES/NO/N-A checks grouped into four dimensions.
+- **Scoring math:** `wizard-workbench/services/pr-evaluator/evaluator.ts` — each
+ dimension scores `max(1, round(pass_rate × 5))` over its applicable items;
+ confidence = `min(app_sanity, round(mean of the four))`.
- **Automated run:** from `wizard-workbench/`,
`pnpm run evaluate --branch --base --test-run` (needs
`POSTHOG_PERSONAL_API_KEY`; judge model via `EVALUATOR_MODEL`). Output lands
in `wizard-workbench/test-evaluations//` as `rubric.json` +
`scores.json`.
-- **Manual run:** an agent applies the same rubric directly to each cell's
- diff — faster for many cells, and what the 2026-07 benchmark did. Either
- way, report the four dimensions under their full names, 1–5 each
- (5 production-ready, 3 works with real issues, 1 broken or empty):
+- **Manual run:** an agent applies the same rubric directly to each cell's diff
+ — faster for many cells, and what the 2026-07 benchmark did. Either way,
+ report the four dimensions under their full names, 1–5 each (5
+ production-ready, 3 works with real issues, 1 broken or empty):
-- **Files** (`file_analysis`) — right files touched, nothing unrelated,
- imports valid
+- **Files** (`file_analysis`) — right files touched, nothing unrelated, imports
+ valid
- **App** (`app_sanity`) — nothing broken: builds, existing code and configs
preserved, changes minimal
- **PostHog** (`posthog_implementation`) — SDK installed, initialized at the
- right entry points, env-based keys, real distinct id, identify, error
- tracking
+ right entry points, env-based keys, real distinct id, identify, error tracking
- **Events** (`event_quality`) — real user actions, useful properties, no PII,
consistent names
-Verify claims against the diff (grep for `capture`/`identify` call sites,
-check the init file, check the manifest), and build or typecheck where cheap.
-Judge the same subset of apps for every config you compare.
+Verify claims against the diff (grep for `capture`/`identify` call sites, check
+the init file, check the manifest), and build or typecheck where cheap. Judge
+the same subset of apps for every config you compare.
## Publishing evidence
@@ -230,48 +230,55 @@ inspectable:
1. Fork each app to the operator's account.
2. Pin a `bench-base` branch at the exact commit the runs used.
-3. Per cell: branch from `bench-base`, apply the **sanitized** patch, push,
- open a draft PR against `bench-base`.
+3. Per cell: branch from `bench-base`, apply the **sanitized** patch, push, open
+ a draft PR against `bench-base`.
4. Sanitize before anything touches a public fork: drop env files, wizard
- scaffolding, and lockfiles from the patch; redact every token literal.
- Verify zero secrets in the pushed diff before opening the PR.
+ scaffolding, and lockfiles from the patch; redact every token literal. Verify
+ zero secrets in the pushed diff before opening the PR.
## Report template
```markdown
# — model benchmark
-
+
## Summary — configs that completed everywhere
-| config | completed | median time | median cost | quality (judged on) |
-
## Results
-
-| config | | | … |
-
-
-> have any feedback, please drop an email to **[wizard@posthog.com](mailto:wizard@posthog.com)**.
+> have any feedback, please drop an email to
+> **[wizard@posthog.com](mailto:wizard@posthog.com)**.
PostHog wizard ✨
@@ -19,22 +19,36 @@ To use the wizard, you can run it directly using:
npx @posthog/wizard@latest
```
-Currently the wizard can be used for over 16+ frameworks for frontend, backend, and mobile applications. If you have other integrations you would like the wizard to
-support, please open a [GitHub issue](https://github.com/posthog/wizard/issues)!
+Currently the wizard can be used for over 16+ frameworks for frontend, backend,
+and mobile applications. If you have other integrations you would like the
+wizard to support, please open a
+[GitHub issue](https://github.com/posthog/wizard/issues)!
-Visit our [docs](https://posthog.com/docs/ai-engineering/ai-wizard) to learn more.
+Visit our [docs](https://posthog.com/docs/ai-engineering/ai-wizard) to learn
+more.
## Privacy & data usage
-The wizard uses **AI models from Anthropic or OpenAI**, routed through PostHog's AI gateway, to read your project's source files and integrate PostHog. A few things worth knowing up front:
-
-- **Source files** are sent to the selected model provider as part of the agent's context.
-- **`.env*` files and secrets** stay on your machine. The wizard's security scanner blocks anything it identifies as a secret from being read by the agent.
-- **Telemetry** (run metadata — phase, task list, planned events) is sent to PostHog by default. Pass `--no-telemetry` (or set `POSTHOG_WIZARD_NO_TELEMETRY=1`) to disable.
-- **AI opt-in**: for existing organizations in interactive runs, the wizard checks `is_ai_data_processing_approved` and waits for approval before agent work. CI and signup runs bypass this interactive gate.
-- **Prefer your own AI?** The wizard's integration knowledge ships as a context-mill skill you can download and run inside your own agent.
-
-The wizard's "Privacy & data usage" menu (intro screen) and the `[I]` shortcut on the auth screen surface the same information in-terminal.
+The wizard uses **AI models from Anthropic or OpenAI**, routed through PostHog's
+AI gateway, to read your project's source files and integrate PostHog. A few
+things worth knowing up front:
+
+- **Source files** are sent to the selected model provider as part of the
+ agent's context.
+- **`.env*` files and secrets** stay on your machine. The wizard's security
+ scanner blocks anything it identifies as a secret from being read by the
+ agent.
+- **Telemetry** (run metadata — phase, task list, planned events) is sent to
+ PostHog by default. Pass `--no-telemetry` (or set
+ `POSTHOG_WIZARD_NO_TELEMETRY=1`) to disable.
+- **AI opt-in**: for existing organizations in interactive runs, the wizard
+ checks `is_ai_data_processing_approved` and waits for approval before agent
+ work. CI and signup runs bypass this interactive gate.
+- **Prefer your own AI?** The wizard's integration knowledge ships as a
+ context-mill skill you can download and run inside your own agent.
+
+The wizard's "Privacy & data usage" menu (intro screen) and the `[I]` shortcut
+on the auth screen surface the same information in-terminal.
## MCP Commands
@@ -51,33 +65,43 @@ npx @posthog/wizard@latest mcp remove
## Wizard programs
-The wizard's commands are grouped into **programs** — self-contained agentic jobs that install, audit, or wire up a specific piece of PostHog. They're powered by skills from the [context mill](https://github.com/PostHog/context-mill).
+The wizard's commands are grouped into **programs** — self-contained agentic
+jobs that install, audit, or wire up a specific piece of PostHog. They're
+powered by skills from the
+[context mill](https://github.com/PostHog/context-mill).
### PostHog integration (default)
-Running the wizard with no arguments installs PostHog into your project. It detects your framework, wires up initialization, instruments a starter set of events, and walks you through a first dashboard:
+Running the wizard with no arguments installs PostHog into your project. It
+detects your framework, wires up initialization, instruments a starter set of
+events, and walks you through a first dashboard:
```bash
npx @posthog/wizard@latest
```
-Powered by the `posthog-integration` program. Most other programs below build on it (they declare `requires: ['posthog-integration']`) and will offer to run it first if PostHog isn't already set up.
+Powered by the `posthog-integration` program. Most other programs below build on
+it (they declare `requires: ['posthog-integration']`) and will offer to run it
+first if PostHog isn't already set up.
### Self-driving
-Autonomously sets up PostHog self-driving end-to-end. It connects GitHub, enables Session Replay and Error Tracking, wires up signal sources, and configures a Signals scout troop that watches your project for you.
+Autonomously sets up PostHog self-driving end-to-end. It connects GitHub,
+enables Session Replay and Error Tracking, wires up signal sources, and
+configures a Signals scout troop that watches your project for you.
```bash
npx @posthog/wizard@latest self-driving
```
-If PostHog isn't already installed, the wizard runs the default integration first (composed run) before starting the self-driving setup.
+If PostHog isn't already installed, the wizard runs the default integration
+first (composed run) before starting the self-driving setup.
### Audit
Audit an existing PostHog integration for correctness and best practices. The
-`audit` command is a **family**. With no subcommand it runs the **events**
-audit (the default); pass a subcommand to run a specific one:
+`audit` command is a **family**. With no subcommand it runs the **events** audit
+(the default); pass a subcommand to run a specific one:
```bash
# Runs the events audit (the default) — no subcommand needed
@@ -96,12 +120,12 @@ npx @posthog/wizard@latest audit web-analytics # web analytics setup
Most audit subcommands resolve at runtime from the published skill registry, so
new audits appear without a wizard release (`web-analytics` is wizard-native).
-> **`audit ` chooses an audit area — it does not take a skill name.**
-> The audit subcommands above *are* context-mill skills promoted to commands (via
-> a `cli: role: command` block); [`wizard skill `](#run-a-single-skill)
-> runs a skill that hasn't been promoted. Same machinery, two surfaces.
-> (`wizard audit --help` still labels the positional `[skill]` — read it as "pick
-> a subcommand.")
+> **`audit ` chooses an audit area — it does not take a skill
+> name.** The audit subcommands above _are_ context-mill skills promoted to
+> commands (via a `cli: role: command` block);
+> [`wizard skill `](#run-a-single-skill) runs a skill that hasn't
+> been promoted. Same machinery, two surfaces. (`wizard audit --help` still
+> labels the positional `[skill]` — read it as "pick a subcommand.")
### Revenue Analytics
@@ -129,7 +153,8 @@ OAuth sources open the PostHog app's new-source flow in your browser.
### Upload source maps
-Upload JavaScript source maps to PostHog error tracking so stack traces are symbolicated back to your original code:
+Upload JavaScript source maps to PostHog error tracking so stack traces are
+symbolicated back to your original code:
```bash
npx @posthog/wizard@latest upload-source-maps
@@ -149,36 +174,35 @@ npx @posthog/wizard@latest skill # run one by name
Reviews are auto-requested via [`.github/CODEOWNERS`](.github/CODEOWNERS) — the
file is the source of truth; this table just mirrors it for readability.
-`team-wizard-docs` is the default reviewer; the team-owned programs below
-route review to their owning team instead.
-
-| Path | Owning team |
-|---|---|
-| `*` (everything else, including all other programs) | `@PostHog/team-wizard-docs` |
-| `src/agent/` | `@PostHog/team-wizard-docs` |
-| `src/store/programs/posthog-integration/` | `@PostHog/team-wizard-docs` |
-| `src/store/programs/error-tracking-upload-source-maps/` | `@PostHog/team-error-tracking` |
-| `src/store/programs/mcp-analytics/` | `@PostHog/team-mcp-analytics` |
-| `src/store/programs/revenue-analytics/` | `@PostHog/team-web-analytics` |
-| `src/store/programs/self-driving/` | `@PostHog/team-self-driving` |
-| `src/store/programs/warehouse-source/` | `@PostHog/team-warehouse-sources` |
-| `src/store/programs/web-analytics-doctor/` | `@PostHog/team-web-analytics` |
-
-Ownership is by directory. Programs not listed above
-(`agent-skill`, `audit`, `events-audit`, `mcp`, `migration`, `posthog-doctor`,
-`shared`, `slack`) fall through the default and are owned by
-`team-wizard-docs`. Today CODEOWNERS only auto-requests review — approval is
-not a merge gate.
+`team-wizard-docs` is the default reviewer; the team-owned programs below route
+review to their owning team instead.
+
+| Path | Owning team |
+| ------------------------------------------------------- | --------------------------------- |
+| `*` (everything else, including all other programs) | `@PostHog/team-wizard-docs` |
+| `src/agent/` | `@PostHog/team-wizard-docs` |
+| `src/store/programs/posthog-integration/` | `@PostHog/team-wizard-docs` |
+| `src/store/programs/error-tracking-upload-source-maps/` | `@PostHog/team-error-tracking` |
+| `src/store/programs/mcp-analytics/` | `@PostHog/team-mcp-analytics` |
+| `src/store/programs/revenue-analytics/` | `@PostHog/team-web-analytics` |
+| `src/store/programs/self-driving/` | `@PostHog/team-self-driving` |
+| `src/store/programs/warehouse-source/` | `@PostHog/team-warehouse-sources` |
+| `src/store/programs/web-analytics-doctor/` | `@PostHog/team-web-analytics` |
+
+Ownership is by directory. Programs not listed above (`agent-skill`, `audit`,
+`events-audit`, `mcp`, `migration`, `posthog-doctor`, `shared`, `slack`) fall
+through the default and are owned by `team-wizard-docs`. Today CODEOWNERS only
+auto-requests review — approval is not a merge gate.
## Headless signup + install (agents / CI)
-> ⚠️ `--ci` is **not currently supported in published builds** (see [CI Mode](#ci-mode)).
-> This flow works in development builds only.
+> ⚠️ `--ci` is **not currently supported in published builds** (see
+> [CI Mode](#ci-mode)). This flow works in development builds only.
-For a fully non-interactive first-run (no existing PostHog account, no TTY,
-no browser), combine `--ci --signup --email`. The wizard provisions a new
-account, uses the returned personal API key to run the normal CI install,
-and wires PostHog into the project at `--install-dir`:
+For a fully non-interactive first-run (no existing PostHog account, no TTY, no
+browser), combine `--ci --signup --email`. The wizard provisions a new account,
+uses the returned personal API key to run the normal CI install, and wires
+PostHog into the project at `--install-dir`:
```bash
npx @posthog/wizard@latest --ci --signup \
@@ -204,38 +228,37 @@ npx @posthog/wizard@latest provision --email user@example.com --region eu --json
```
Success prints the full `ProvisioningResult` (`projectApiKey`, `host`,
-`projectId`, `accountId`, `accessToken`, `refreshToken`, and
-`personalApiKey` if present). Failure exits 1; in `--json` mode the error
-is emitted to stderr as `{"error":"...","code":"..."}`, with `code` set to
-`email_exists` when the address is already registered.
+`projectId`, `accountId`, `accessToken`, `refreshToken`, and `personalApiKey` if
+present). Failure exits 1; in `--json` mode the error is emitted to stderr as
+`{"error":"...","code":"..."}`, with `code` set to `email_exists` when the
+address is already registered.
-> ⚠️ **Output contains live credentials.** Pipe it into a secrets store —
-> do not let it be captured by shared CI logs. Mask the step output or
-> redirect stdout to a file your job reads and discards.
+> ⚠️ **Output contains live credentials.** Pipe it into a secrets store — do not
+> let it be captured by shared CI logs. Mask the step output or redirect stdout
+> to a file your job reads and discards.
# Options
The following CLI arguments are available:
-| Option | Description | Type | Default | Choices | Environment Variable |
-| ----------------- | ---------------------------------------------------------------- | ------- | ------- | ---------------------------------------------------- | ------------------------------ |
-| `--help` | Show help | boolean | | | |
-| `--version` | Show version number | boolean | | | |
-| `--debug` | Enable verbose logging | boolean | `false` | | `POSTHOG_WIZARD_DEBUG` |
-| `--signup` | Create a new PostHog account during setup | boolean | `false` | | `POSTHOG_WIZARD_SIGNUP` |
-| `--install-dir` | Directory to install PostHog in | string | | | `POSTHOG_WIZARD_INSTALL_DIR` |
-| `--ci` | Enable CI mode for non-interactive execution | boolean | `false` | | `POSTHOG_WIZARD_CI` |
-| `--api-key` | PostHog personal API key (phx_xxx) for authentication | string | | | `POSTHOG_WIZARD_API_KEY` |
-| `--no-telemetry` | Disable wizard run-state telemetry | boolean | `false` | | `POSTHOG_WIZARD_NO_TELEMETRY` |
-
+| Option | Description | Type | Default | Choices | Environment Variable |
+| ---------------- | ----------------------------------------------------- | ------- | ------- | ------- | ----------------------------- |
+| `--help` | Show help | boolean | | | |
+| `--version` | Show version number | boolean | | | |
+| `--debug` | Enable verbose logging | boolean | `false` | | `POSTHOG_WIZARD_DEBUG` |
+| `--signup` | Create a new PostHog account during setup | boolean | `false` | | `POSTHOG_WIZARD_SIGNUP` |
+| `--install-dir` | Directory to install PostHog in | string | | | `POSTHOG_WIZARD_INSTALL_DIR` |
+| `--ci` | Enable CI mode for non-interactive execution | boolean | `false` | | `POSTHOG_WIZARD_CI` |
+| `--api-key` | PostHog personal API key (phx_xxx) for authentication | string | | | `POSTHOG_WIZARD_API_KEY` |
+| `--no-telemetry` | Disable wizard run-state telemetry | boolean | `false` | | `POSTHOG_WIZARD_NO_TELEMETRY` |
# CI Mode
**CI mode is available only in development/test builds.** Published builds
reject `--ci`; use an interactive terminal for `npx @posthog/wizard@latest`.
-Local CI runs require a PostHog personal API key **and a separate gateway
-token file**, plus the target project ID. See
+Local CI runs require a PostHog personal API key **and a separate gateway token
+file**, plus the target project ID. See
[local credentials](docs/local-dev.md#credentials-for-local-ci-and-headless-runs)
for setup and the CI secret names. With both secrets configured:
@@ -257,8 +280,10 @@ The CLI args override environment variables in CI mode.
### Required Flags for CI Mode
-- `--api-key`: Personal API key (`phx_xxx`) from your [PostHog settings](https://app.posthog.com/settings/user-api-keys)
-- `--install-dir`: Directory to install PostHog in (e.g., `.` for current directory)
+- `--api-key`: Personal API key (`phx_xxx`) from your
+ [PostHog settings](https://app.posthog.com/settings/user-api-keys)
+- `--install-dir`: Directory to install PostHog in (e.g., `.` for current
+ directory)
### Required API Key Scopes
@@ -270,8 +295,8 @@ dashboard:write insight:write notebook:write event_definition:write
health_issue:read wizard_session:read wizard_session:write
```
-The source of truth is `WIZARD_OAUTH_SCOPES` in `src/store/shared/constants.ts`, which
-documents why each scope is needed — if this block drifts, trust the code.
+The source of truth is `WIZARD_OAUTH_SCOPES` in `src/store/shared/constants.ts`,
+which documents why each scope is needed — if this block drifts, trust the code.
Some programs request more on top (`PROGRAM_SCOPE_ADDITIONS` in
`src/store/services/oauth/program-scopes.ts`); the default integration flow adds
`integration:read` and `external_data_source:read` /
@@ -279,23 +304,23 @@ Some programs request more on top (`PROGRAM_SCOPE_ADDITIONS` in
### OAuth app scope ceiling
-The wizard's OAuth app on the PostHog side caps the scopes its tokens may
-carry (`OAuthApplication.scopes`). Any scope requested in this repo (see
-`src/store/services/oauth/program-scopes.ts`) must be grantable under that ceiling, or
-`/authorize` drops it and the call that needs it 403s.
+The wizard's OAuth app on the PostHog side caps the scopes its tokens may carry
+(`OAuthApplication.scopes`). Any scope requested in this repo (see
+`src/store/services/oauth/program-scopes.ts`) must be grantable under that
+ceiling, or `/authorize` drops it and the call that needs it 403s.
**A granted token can be narrower than the request even with a correct
ceiling.** The consent screen lets the user deselect any scope the app doesn't
mark required (`OAuthApplication.required_scopes`), and out-of-ceiling scopes
are clamped silently (`clamp_scopes_to_ceiling`) — neither path errors;
-`/oauth/token` just returns a smaller `scope`. So never assume the token
-carries what was requested: the token response's `scope` field is the truth.
-The wizard diffs granted vs requested at login (`missingOAuthScopes` in
-`src/store/shared/oauth.ts`), warns the user which permissions are missing, and emits
-`wizard: oauth grant narrowed` so narrowed runs are countable in analytics.
-The diff also rides on the session (`credentials.missingScopes`), so when a
-run does fail on a scope-gated step, the error names the missing permission
-and the fix instead of the generic report-a-bug line.
+`/oauth/token` just returns a smaller `scope`. So never assume the token carries
+what was requested: the token response's `scope` field is the truth. The wizard
+diffs granted vs requested at login (`missingOAuthScopes` in
+`src/store/shared/oauth.ts`), warns the user which permissions are missing, and
+emits `wizard: oauth grant narrowed` so narrowed runs are countable in
+analytics. The diff also rides on the session (`credentials.missingScopes`), so
+when a run does fail on a scope-gated step, the error names the missing
+permission and the fix instead of the generic report-a-bug line.
**To make scopes impossible to deselect, list them explicitly in the app's
`scopes`.** `required_scopes` is not a separate field — it is derived
@@ -305,8 +330,8 @@ consent POST 400s with `invalid_scope` if the grant misses one), while scopes
covered only by `@default` stay deselectable and `optional_scopes` are
declinable extras. That is why `llm_gateway:read` and `wizard_session:*` are
already un-deselectable today, and everything else is not. To pin the base set
-the wizard cannot run without, seed each region's app with `@default` plus
-every scope in `WIZARD_OAUTH_SCOPES`:
+the wizard cannot run without, seed each region's app with `@default` plus every
+scope in `WIZARD_OAUTH_SCOPES`:
```
python manage.py seed_oauth_app_scopes --client-id --dry-run \
@@ -314,10 +339,10 @@ python manage.py seed_oauth_app_scopes --client-id --dry-run \
```
then re-run without `--dry-run`. Keep `@default` in the list — dropping it
-narrows the ceiling to only the explicit entries and strips the
-program-specific additions. The wizard has no client-side lever for any of
-this; the login diff and prompt-threaded degrade above handle a narrowed
-grant, but only pinning prevents one.
+narrows the ceiling to only the explicit entries and strips the program-specific
+additions. The wizard has no client-side lever for any of this; the login diff
+and prompt-threaded degrade above handle a narrowed grant, but only pinning
+prevents one.
**The live wizard apps use the `@default` sentinel, so most net-new scopes need
no ceiling edit.** The prod US app's `scopes` is:
@@ -362,18 +387,18 @@ used to silently add permissions.
The CLI was overhauled to consolidate commands into a smaller, extensible
surface. If you used an older command, here's where it went:
-| Old command | New command | What changed |
-|---|---|---|
-| `wizard integrate` | `wizard` (default flow) | Command removed; the default flow runs the integration |
-| `wizard events-audit` | `wizard audit events` | Now an `audit`-family subcommand |
-| `wizard audit` (single audit) | `wizard audit ` | Now a family; see [Audit](#audit) for the subcommands |
-| `wizard audit-3000` | *removed* | Retired |
-| `wizard revenue` | `wizard revenue-analytics` | Renamed (old `revenue` removed) |
-| `wizard upload-sourcemaps` | `wizard upload-source-maps` | Renamed; `upload-sourcemaps` still works as an alias |
-
-> **Commands vs. programs:** `integrate` was the *command*; the program behind it
-> is `posthog-integration`, which still exists and now powers the default flow.
-> Other commands depend on it via `requires: ['posthog-integration']`. The
+| Old command | New command | What changed |
+| ----------------------------- | --------------------------- | ------------------------------------------------------ |
+| `wizard integrate` | `wizard` (default flow) | Command removed; the default flow runs the integration |
+| `wizard events-audit` | `wizard audit events` | Now an `audit`-family subcommand |
+| `wizard audit` (single audit) | `wizard audit ` | Now a family; see [Audit](#audit) for the subcommands |
+| `wizard audit-3000` | _removed_ | Retired |
+| `wizard revenue` | `wizard revenue-analytics` | Renamed (old `revenue` removed) |
+| `wizard upload-sourcemaps` | `wizard upload-source-maps` | Renamed; `upload-sourcemaps` still works as an alias |
+
+> **Commands vs. programs:** `integrate` was the _command_; the program behind
+> it is `posthog-integration`, which still exists and now powers the default
+> flow. Other commands depend on it via `requires: ['posthog-integration']`. The
> program id is internal — it was never a command you typed.
# Steal this code
@@ -396,8 +421,8 @@ and set up the general flow of the application.
## Analytics
Did you know you can capture PostHog events even for smaller, supporting
-products like a command line tool? `src/store/shared/analytics.ts` is a great example
-of how to do it.
+products like a command line tool? `src/store/shared/analytics.ts` is a great
+example of how to do it.
This file wraps `posthog-node` with some convenience functions to set up an
analytics session and log events. We can see the usage and outcomes of this
@@ -409,8 +434,8 @@ When the user authenticates, the wizard also streams live run state — current
phase, task list, planned events — to `POST /api/projects/{id}/wizard/sessions/`
so the PostHog web app can render real-time progress. Updates are debounced
(250ms) with phase changes flushed immediately; failures fall back silently to
-the wizard's debug log without disturbing the TUI. Pass `--no-telemetry` (or
-set `POSTHOG_WIZARD_NO_TELEMETRY=1`) to disable.
+the wizard's debug log without disturbing the TUI. Pass `--no-telemetry` (or set
+`POSTHOG_WIZARD_NO_TELEMETRY=1`) to disable.
## Leave rules behind
@@ -445,17 +470,16 @@ users of the wizard, no training delays or other ambiguity.
## Keep secrets out of the LLM
-The wizard somtimes needs to move a secret. The agent
-orchestrates that journey, but the raw value should _never_ enter the LLM
-conversation, where it would be sent to the model provider, written to
-transcripts, and captured in logs.
+The wizard somtimes needs to move a secret. The agent orchestrates that journey,
+but the raw value should _never_ enter the LLM conversation, where it would be
+sent to the model provider, written to transcripts, and captured in logs.
-`src/store/session/secret-vault.ts` is a small, reusable pattern for exactly this. It's a
-session-scoped, in-memory vault: a tool that handles a secret calls `put()` to
-store the raw value and hands the agent an opaque `secret:` reference
-instead. The agent passes that ref between tools as if it were the value; the
-host resolves it back to the real secret only at the last moment, inside the
-process, when it writes the file.
+`src/store/session/secret-vault.ts` is a small, reusable pattern for exactly
+this. It's a session-scoped, in-memory vault: a tool that handles a secret calls
+`put()` to store the raw value and hands the agent an opaque `secret:`
+reference instead. The agent passes that ref between tools as if it were the
+value; the host resolves it back to the real secret only at the last moment,
+inside the process, when it writes the file.
Two tools in `src/store/tools/tools.ts` form the ends of that pipe:
@@ -471,48 +495,59 @@ drive the work end to end, but the only thing it ever sees is an opaque handle.
## Build system
-Built with [tsdown](https://tsdown.dev/) (Rolldown). `pnpm build` bundles `bin.ts` into ESM chunks in `dist/`, inlining all local source and keeping npm dependencies external.
+Built with [tsdown](https://tsdown.dev/) (Rolldown). `pnpm build` bundles
+`bin.ts` into ESM chunks in `dist/`, inlining all local source and keeping npm
+dependencies external.
### Environment variables
-**Build-time (locked).** `NODE_ENV` is replaced with `"production"` at compile time. It cannot be overridden at runtime. All URLs, OAuth client IDs, and dev-mode code paths resolve to their production values unconditionally.
+**Build-time (locked).** `NODE_ENV` is replaced with `"production"` at compile
+time. It cannot be overridden at runtime. All URLs, OAuth client IDs, and
+dev-mode code paths resolve to their production values unconditionally.
-To add a new build-time constant, add it to `env` in `tsdown.config.ts` and export it from `src/env.ts`.
+To add a new build-time constant, add it to `env` in `tsdown.config.ts` and
+export it from `src/env.ts`.
-**Runtime (allowlisted).** Runtime env reads go through `runtimeEnv()` in `src/env.ts`, which only accepts keys in the `RuntimeEnvKey` union:
+**Runtime (allowlisted).** Runtime env reads go through `runtimeEnv()` in
+`src/env.ts`, which only accepts keys in the `RuntimeEnvKey` union:
-| Variable | Purpose |
-|---|---|
-| `POSTHOG_WIZARD_BENCHMARK_CONFIG` | Path to benchmark config file |
-| `POSTHOG_WIZARD_BENCHMARK_FILE` | Output path for benchmark results |
-| `POSTHOG_WIZARD_LOG_DIR` | Log directory override |
-| `POSTHOG_WIZARD_DEBUG` / `DEBUG` | Enable debug output |
-| `MCP_URL` | Override MCP server URL |
-| `POSTHOG_API_KEY` | API key for MCP subprocess auth |
-| `TERM`, `TERM_PROGRAM`, `CI`, etc. | Terminal/platform detection |
-| `APPDATA`, `XDG_CONFIG_HOME` | Platform path resolution |
+| Variable | Purpose |
+| ---------------------------------- | --------------------------------- |
+| `POSTHOG_WIZARD_BENCHMARK_CONFIG` | Path to benchmark config file |
+| `POSTHOG_WIZARD_BENCHMARK_FILE` | Output path for benchmark results |
+| `POSTHOG_WIZARD_LOG_DIR` | Log directory override |
+| `POSTHOG_WIZARD_DEBUG` / `DEBUG` | Enable debug output |
+| `MCP_URL` | Override MCP server URL |
+| `POSTHOG_API_KEY` | API key for MCP subprocess auth |
+| `TERM`, `TERM_PROGRAM`, `CI`, etc. | Terminal/platform detection |
+| `APPDATA`, `XDG_CONFIG_HOME` | Platform path resolution |
To add a new runtime env var, add its key to `RuntimeEnvKey` in `src/env.ts`.
-**Direct `process.env` access** is only used for subprocess environment writes (e.g. `agent-interface.ts` setting `ANTHROPIC_BASE_URL`), vendored code, and tests.
+**Direct `process.env` access** is only used for subprocess environment writes
+(e.g. `agent-interface.ts` setting `ANTHROPIC_BASE_URL`), vendored code, and
+tests.
### Import aliases
Path aliases defined in `tsconfig.build.json`, resolved by tsdown:
-| Alias | Maps to |
-|---|---|
-| `@env` | `src/env.ts` |
-| `@store`, `@store/types`, `@store/programs` | `src/store/{index,types,programs/index}.ts` |
-| `@agent`, `@agent/types` | `src/agent/{index,types}.ts` |
-| `@tui`, `@tui/types`, `@tui/console` | `src/tui/{index,types,console/index}.ts` |
-| `@cli/*` | `src/cli/*` (composition root, internal) |
-| `@store/*`, `@agent/*`, `@tui/*` | surface internals; tests only, never across surfaces |
+| Alias | Maps to |
+| ------------------------------------------- | -------------------------------------------------------------------------- |
+| `@env` | `src/env.ts` |
+| `@store`, `@store/types`, `@store/programs` | `src/store/{index,types,programs/index}.ts` |
+| `@agent`, `@agent/types` | `src/agent/{index,types}.ts` |
+| `@tui`, `@tui/types`, `@tui/console` | `src/tui/{index,types,console/index}.ts` |
+| `@cli/*` | `src/cli/*` (composition root, internal) |
+| `@store/control` | `src/store/control/index.ts`; the cli runners only, by dynamic import |
+| `@e2e-harness/*` | `e2e-harness/*` (harness, scripts, tests) |
+| `@store/*`, `@agent/*`, `@tui/*` | surface internals; harness, scripts, and tests only, never across surfaces |
## Running locally
For `--ci`, smoke tests, and full headless runs, configure both the personal API
-key and gateway token file first: [local credentials](docs/local-dev.md#credentials-for-local-ci-and-headless-runs).
+key and gateway token file first:
+[local credentials](docs/local-dev.md#credentials-for-local-ci-and-headless-runs).
Interactive runs mint their gateway token after authentication.
### Quick test without linking
@@ -527,7 +562,8 @@ pnpm try --install-dir=[a path]
pnpm run dev
```
-This builds, links globally, and watches for changes. Leave it running - any `.ts` file changes will auto-rebuild. Then from any project:
+This builds, links globally, and watches for changes. Leave it running - any
+`.ts` file changes will auto-rebuild. Then from any project:
```bash
wizard --integration=nextjs
@@ -538,8 +574,8 @@ wizard --integration=nextjs --local-mcp # MCP from localhost:8787
wizard --integration=nextjs --local-dev # context-mill + MCP + PostHog
```
-See [`docs/local-dev.md`](docs/local-dev.md) for the full catalog.
-`--local-mcp` selects the MCP server; `--local-context-mill` selects the skills server.
+See [`docs/local-dev.md`](docs/local-dev.md) for the full catalog. `--local-mcp`
+selects the MCP server; `--local-context-mill` selects the skills server.
### Testing
@@ -564,8 +600,8 @@ You can hand the wizard to an AI agent and have it drive the real flow itself
deciding each screen and snapshotting the TUI to see what happened. The agent
drives through the `wizard-ci` MCP tools (`open_app` / `read_state` /
`perform_action` / `render_screen` / `run_agent`), which are registered in this
-repo's `.mcp.json` and bound in every session here — approve `wizard-ci` the first
-time you're prompted. The how-to is the `exploring-the-wizard` skill
+repo's `.mcp.json` and bound in every session here — approve `wizard-ci` the
+first time you're prompted. The how-to is the `exploring-the-wizard` skill
(`.claude/skills/exploring-the-wizard/SKILL.md`), which an agent discovers
automatically.
@@ -573,12 +609,12 @@ Example prompt — explore against
[open-saas](https://github.com/wasp-lang/open-saas):
> Explore the PostHog wizard against open-saas, following the
-> `exploring-the-wizard` skill. Reuse my phx key file path, gateway token file path, and project id,
-> asking only for missing inputs. Launch the MCP server with
-> `WIZARD_CI_GATEWAY_TOKEN_FILE` set to the gateway token file path;
-> then clone `https://github.com/wasp-lang/open-saas` into a throwaway `/tmp`
-> copy. Drive the whole flow yourself through the `wizard-ci` MCP tools, deciding
-> each screen:
+> `exploring-the-wizard` skill. Reuse my phx key file path, gateway token file
+> path, and project id, asking only for missing inputs. Launch the MCP server
+> with `WIZARD_CI_GATEWAY_TOKEN_FILE` set to the gateway token file path; then
+> clone `https://github.com/wasp-lang/open-saas` into a throwaway `/tmp` copy.
+> Drive the whole flow yourself through the `wizard-ci` MCP tools, deciding each
+> screen:
>
> 1. `open_app` on the `/tmp` copy, then `read_state` to see the screen and the
> actions legal right now.
@@ -605,23 +641,24 @@ To make your version of a tool usable with a one-line `npx` command:
# Health checks
-`src/store/health-checks/` checks skills download origins before the wizard runs.
-The entry point is `evaluateWizardReadiness()`, which only blocks on skill downloads:
+`src/store/health-checks/` checks skills download origins before the wizard
+runs. The entry point is `evaluateWizardReadiness()`, which only blocks on skill
+downloads:
-| Decision | Meaning |
-| ------------------- | --------------------------------------------------------------- |
-| `yes` | Skills are reachable — proceed without outage warnings. |
-| `no` | Neither skills origin is reachable — do not run. |
+| Decision | Meaning |
+| -------- | ------------------------------------------------------- |
+| `yes` | Skills are reachable — proceed without outage warnings. |
+| `no` | Neither skills origin is reachable — do not run. |
### Module layout
-| File | Responsibility |
-| --- | --- |
-| `types.ts` | Enums, interfaces (`ServiceHealthStatus`, `AllServicesHealth`, etc.) |
+| File | Responsibility |
+| -------------- | ----------------------------------------------------------------------- |
+| `types.ts` | Enums, interfaces (`ServiceHealthStatus`, `AllServicesHealth`, etc.) |
| `endpoints.ts` | Direct gateway (`/readyz`) and skills origin (`skill-menu.json`) checks |
| `readiness.ts` | `checkAllExternalServices`, `evaluateWizardReadiness`, readiness config |
-| `index.ts` | Barrel re-export |
-| `testme.md` | Test running instructions and endpoint reference |
+| `index.ts` | Barrel re-export |
+| `testme.md` | Test running instructions and endpoint reference |
## What blocks a run
@@ -643,29 +680,35 @@ The same policy applies during signup. Third-party status pages are not queried.
After minting a token, `gateway-session.ts` checks `/readyz` on the returned
gateway URL and reports an unavailable gateway through the existing error path.
-`skillsOrigin` is one entry covering two origins: skills are published to
-GitHub Releases and an AWS mirror under the same filenames, and downloads fail
-over between them (`src/store/fetch-retry.ts`). Both are probed in parallel, so
-the key only reports **Down** when neither origin answers — a GitHub Releases
-outage on its own doesn't block a run, including a 403 or 404, which is as
-often about the origin (expired asset redirect, blocked region, a publish that
-reached one origin and not the other) as about the asset.
+`skillsOrigin` is one entry covering two origins: skills are published to GitHub
+Releases and an AWS mirror under the same filenames, and downloads fail over
+between them (`src/store/fetch-retry.ts`). Both are probed in parallel, so the
+key only reports **Down** when neither origin answers — a GitHub Releases outage
+on its own doesn't block a run, including a 403 or 404, which is as often about
+the origin (expired asset redirect, blocked region, a publish that reached one
+origin and not the other) as about the asset.
## Smoke test helper (`scripts/smoke-test-ci.sh`)
-This repo includes a helper script to run a full end‑to‑end smoke test of the wizard packaged in a tarball against a real app from [`posthog/wizard-workbench`](https://github.com/PostHog/wizard-workbench). This will catch certain packaging issues that might not be caught by other tests.
+This repo includes a helper script to run a full end‑to‑end smoke test of the
+wizard packaged in a tarball against a real app from
+[`posthog/wizard-workbench`](https://github.com/PostHog/wizard-workbench). This
+will catch certain packaging issues that might not be caught by other tests.
**Prerequisites**
- Point to a `wizard-workbench` checkout either by:
- Setting `WIZARD_WORKBENCH_ROOT=/absolute/path/to/wizard-workbench`, or
- - Cloning `wizard-workbench` next to this repo (so it lives at `../wizard-workbench`).
-- Set `POSTHOG_PERSONAL_API_KEY` either in your shell or in `../wizard-workbench/.env`.
+ - Cloning `wizard-workbench` next to this repo (so it lives at
+ `../wizard-workbench`).
+- Set `POSTHOG_PERSONAL_API_KEY` either in your shell or in
+ `../wizard-workbench/.env`.
- Set `WIZARD_CI_GATEWAY_TOKEN_FILE` to an absolute path containing the separate
- AI gateway token. See [local credentials](docs/local-dev.md#credentials-for-local-ci-and-headless-runs).
+ AI gateway token. See
+ [local credentials](docs/local-dev.md#credentials-for-local-ci-and-headless-runs).
- Set `POSTHOG_WIZARD_PROJECT_ID` to the intended test project and
- `POSTHOG_WIZARD_REGION` to `us` or `eu` (CI uses `us`). The helper also accepts
- `POSTHOG_PROJECT_ID` and `POSTHOG_REGION` as fallback names.
+ `POSTHOG_WIZARD_REGION` to `us` or `eu` (CI uses `us`). The helper also
+ accepts `POSTHOG_PROJECT_ID` and `POSTHOG_REGION` as fallback names.
**Usage**
@@ -695,8 +738,10 @@ The script will:
- Copy the selected app into a temp directory
- Install dependencies for the app
- Install the packed wizard tarball into an isolated temp project
-- Run `wizard` in `--ci` mode against the copied app and perform basic post‑install checks
+- Run `wizard` in `--ci` mode against the copied app and perform basic
+ post‑install checks
## Contributing
-Start with [AGENTS.md](AGENTS.md) for the development skills and execution policy.
+Start with [AGENTS.md](AGENTS.md) for the development skills and execution
+policy.
diff --git a/e2e-harness/ARCHITECTURE.md b/e2e-harness/ARCHITECTURE.md
index 1d4bfa378..cecd792d9 100644
--- a/e2e-harness/ARCHITECTURE.md
+++ b/e2e-harness/ARCHITECTURE.md
@@ -24,8 +24,12 @@ src/store/control/
state.ts the secret-free projection GET /state returns
driver.ts ControlDriver: read and act on one store in process
runs.ts the run ledger behind POST /runs and GET /runs
+ params.ts body and param validation; every bad input is a 400
+ marker.ts the string the bundle audit looks for
+ index.ts the barrel the cli runners import dynamically
e2e-harness/
- launch.ts buildLaunch: the argv and env that start a controlled wizard
+ launch.ts launchWords, buildLaunch, waitForSocket: how a controlled wizard starts
+ run-status.ts the MCP route's integration status, read off the run phase
picks.ts the detection picks a headless run supplies to picker screens
e2e-profile.ts WizardE2eProfile and decideE2eAction: the scripted walk policy
profiles.ts per-program profiles, profileFor(programId), resolveE2eProfile(env)
@@ -34,6 +38,8 @@ e2e-harness/
scripts/
tui-snapshots.no-jest.ts CI route: the wizard in a PTY, driven over the socket, per-screen snapshots
wizard-ci-mcp.no-jest.ts agent route: an MCP server that proxies the same API
+ controlled-headless-smoke.no-jest.ts headless surface: detect, runs, ledger, shutdown
+ wizard-ci-explore.no-jest.ts open an app, confirm setup, print one frame
```
The server reads and mutates the **real** `WizardStore` the TUI renders from.
@@ -48,7 +54,8 @@ the TUI's input.
its own command word (`self-driving`, `audit all`, `upload-source-maps`); the
default flow has none. The flags are `--ci` (API-key auth, no browser),
`--control-socket `, `--install-dir`, `--project-id`, `--region`, and
-`--e2e-ask` when the parent answers the agent's questions. Switchboard overrides
+`--e2e-ask` when the parent answers the agent's questions, `--integrate` for
+self-driving, and `--task-stream-log` to dump the stream. Switchboard overrides
map to `--harness`, `--sequence`, and `--model`.
The personal API key travels as `POSTHOG_WIZARD_API_KEY` in the child's
@@ -67,18 +74,18 @@ JSON in and out. Success is `{ ok: true, ... }`; failure is
program), 404 (unknown route), 409 (a run is in flight), 413 (body over 64 KB),
415 (not JSON), 500 (a hook failed), or 501 (the other surface's route).
-| Route | Behavior |
-| --------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- |
-| `GET /health` | `{ ok, version, surface, pid, program }` |
-| `GET /state` | `{ ok, state }`, the projection below |
-| `GET /state?wait=&since=` | Long poll: resolves on the first commit with `version > since`, or after `wait` ms |
-| `POST /actions/` body `{ params }` | Applies one action legal on the current screen through its store setter, returns the state |
-| `POST /credentials` | Resolves the API key into project credentials and commits them, advancing `auth` |
-| `POST /run` | TUI surface: `requestRun` on the store, releasing the runner's agent start. Idempotent |
-| `POST /detect` body `{ programId?, installDir? }` | Headless surface: runs detection through the store's setters |
-| `POST /runs` body `{ programId, installDir?, frameworkContext?, skillId? }` | Headless surface: one independent agent run; 409 while one runs |
-| `GET /runs` | The run ledger: `runId`, `programId`, `installDir`, `status`, `error`, timestamps, `result` |
-| `POST /shutdown` | Flushes and exits. Idempotent |
+| Route | Behavior |
+| --------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- |
+| `GET /health` | `{ ok, version, surface, pid, program }` |
+| `GET /state` | `{ ok, state }`, the projection below |
+| `GET /state?wait=&since=` | Long poll: resolves on the first commit with `version > since`, or after `wait` ms, capped at 10 min |
+| `POST /actions/` body `{ params }` | Applies one action legal on the current screen through its store setter, returns the state |
+| `POST /credentials` | Resolves the API key into project credentials and commits them, advancing `auth` |
+| `POST /run` | TUI surface: `requestRun` on the store, releasing the runner's agent start. Idempotent |
+| `POST /detect` body `{ programId?, installDir? }` | Headless surface: runs detection through the store's setters |
+| `POST /runs` body `{ programId, installDir?, frameworkContext?, skillId? }` | Headless surface: one independent agent run; 409 while one runs |
+| `GET /runs` | The run ledger: `runId`, `programId`, `installDir`, `status`, `error`, timestamps, `result` |
+| `POST /shutdown` | Flushes and exits. Idempotent |
`state` mirrors the store: `version`, `currentScreen`, `session`, `tasks`,
`statusMessages`, `eventPlan`, `handoffText`, the unanswered `setupQuestions`,
@@ -137,8 +144,9 @@ APP_DIR=/tmp/app PROJECT_ID= POSTHOG_KEY_FILE=/path/to/phx-key.txt \
npx tsx scripts/wizard-ci-explore.no-jest.ts
# Controlled headless: detect, independent runs, the ledger, shutdown
-POSTHOG_WIZARD_API_KEY=phx_... WIZARD_CI_GATEWAY_TOKEN_FILE=/path/to/token.txt \
-npx tsx scripts/controlled-headless-smoke.no-jest.ts --app /tmp/app --project-id posthog-integration metrics
+APP_DIR=/tmp/app PROJECT_ID= POSTHOG_KEY_FILE=/path/to/phx-key.txt \
+WIZARD_CI_GATEWAY_TOKEN_FILE=/path/to/token.txt \
+npx tsx scripts/controlled-headless-smoke.no-jest.ts posthog-integration metrics
# Process specs: the real binary on both surfaces, no credentials, no agent run
pnpm test:harness # WIZARD_PTY_TESTS=0 skips the PTY spec
diff --git a/e2e-harness/__fixtures__/control-frame-auth.txt b/e2e-harness/__fixtures__/control-frame-auth.txt
deleted file mode 100644
index 644e86286..000000000
--- a/e2e-harness/__fixtures__/control-frame-auth.txt
+++ /dev/null
@@ -1,49 +0,0 @@
- PostHog Wizard v2.76.0 Feedback: wizard@posthog.com
-
- PostHog Setup Wizard
- ✔ Framework: Node.js
-
- Privacy & data
- • Source files are read by Claude for AI context
- • .env* and secrets stay on your machine
- • Press [I] for full privacy & usage info
-
- ⠴ Waiting for authentication...
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
- I privacy & data
diff --git a/e2e-harness/__fixtures__/control-frame-intro.txt b/e2e-harness/__fixtures__/control-frame-intro.txt
deleted file mode 100644
index 493bdcb50..000000000
--- a/e2e-harness/__fixtures__/control-frame-intro.txt
+++ /dev/null
@@ -1,49 +0,0 @@
- PostHog Wizard v2.76.0 Feedback: wizard@posthog.com
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
- ███ PostHog Wizard 🦔
-
- We'll use AI to analyze your project and complete work.
- Review what data is shared in "Privacy & data."
- .env* values stay on your machine.
-
- Let's do two hours of work in eight minutes.
-
- Directory ✔ /wz-baseline-express-todo
- Framework ✔ Node.js (detected)
-
- ▸ Continue
- Change framework
- More info
- Privacy & data
- Cancel
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
- ↑↓ navigate enter select
diff --git a/e2e-harness/__fixtures__/control-state-baseline.json b/e2e-harness/__fixtures__/control-state-baseline.json
deleted file mode 100644
index 9797a94c0..000000000
--- a/e2e-harness/__fixtures__/control-state-baseline.json
+++ /dev/null
@@ -1,107 +0,0 @@
-{
- "recordedFrom": "scripts/tui-host.no-jest.ts MODE=serve via scripts/wizard-ci-mcp.no-jest.ts, wizard 2.76.0",
- "app": "wizard-workbench/apps/basic-integration/javascript-node/express-todo (throwaway copy, detection only, no key)",
- "projectId": "228144",
- "steps": [
- {
- "call": "open_app",
- "state": {
- "currentScreen": "intro",
- "hasOverlay": false,
- "runPhase": "idle",
- "session": {
- "installDir": "",
- "integration": "javascript_node",
- "detectedFrameworkLabel": "Node.js",
- "detectionComplete": true,
- "setupConfirmed": false,
- "integrate": null,
- "hasCredentials": false,
- "projectId": null,
- "mcpComplete": false,
- "slackStepDismissed": false,
- "skillsComplete": false,
- "outroDismissed": false,
- "llmOptIn": false,
- "discoveredFeatures": []
- },
- "tasks": [],
- "statusMessages": [],
- "eventPlan": [],
- "pendingQuestion": null,
- "taskNotice": null,
- "setupQuestions": [],
- "actions": [
- { "id": "confirm_setup", "description": "Confirm the intro and continue (sets setupConfirmed)." }
- ],
- "integration": "idle",
- "integrationError": null
- }
- },
- {
- "call": "perform_action",
- "action": "confirm_setup",
- "state": {
- "currentScreen": "auth",
- "hasOverlay": false,
- "runPhase": "idle",
- "session": {
- "installDir": "",
- "integration": "javascript_node",
- "detectedFrameworkLabel": "Node.js",
- "detectionComplete": true,
- "setupConfirmed": true,
- "integrate": null,
- "hasCredentials": false,
- "projectId": null,
- "mcpComplete": false,
- "slackStepDismissed": false,
- "skillsComplete": false,
- "outroDismissed": false,
- "llmOptIn": false,
- "discoveredFeatures": []
- },
- "tasks": [],
- "statusMessages": [],
- "eventPlan": [],
- "pendingQuestion": null,
- "taskNotice": null,
- "setupQuestions": [],
- "actions": []
- }
- },
- {
- "call": "read_state",
- "state": {
- "currentScreen": "auth",
- "hasOverlay": false,
- "runPhase": "idle",
- "session": {
- "installDir": "",
- "integration": "javascript_node",
- "detectedFrameworkLabel": "Node.js",
- "detectionComplete": true,
- "setupConfirmed": true,
- "integrate": null,
- "hasCredentials": false,
- "projectId": null,
- "mcpComplete": false,
- "slackStepDismissed": false,
- "skillsComplete": false,
- "outroDismissed": false,
- "llmOptIn": false,
- "discoveredFeatures": []
- },
- "tasks": [],
- "statusMessages": [],
- "eventPlan": [],
- "pendingQuestion": null,
- "taskNotice": null,
- "setupQuestions": [],
- "actions": [],
- "integration": "idle",
- "integrationError": null
- }
- }
- ]
-}
diff --git a/e2e-harness/__tests__/__snapshots__/keyboard-equivalence.test.tsx.snap b/e2e-harness/__tests__/__snapshots__/keyboard-equivalence.test.tsx.snap
index c9b7a24f5..8ec99706e 100644
--- a/e2e-harness/__tests__/__snapshots__/keyboard-equivalence.test.tsx.snap
+++ b/e2e-harness/__tests__/__snapshots__/keyboard-equivalence.test.tsx.snap
@@ -6,7 +6,6 @@ exports[`keyboard commit vs control action commit > audit-outro: any key vs dism
"outroDismissed": true,
"screen": "keep-skills",
},
- "equal": false,
"keyboard": {
"outroDismissed": true,
"screen": "keep-skills",
@@ -18,10 +17,11 @@ exports[`keyboard commit vs control action commit > audit-outro: any key vs dism
exports[`keyboard commit vs control action commit > intro: enter on Continue vs confirm_setup 1`] = `
{
"action": {
+ "scanConsent": "granted",
"screen": "health-check",
"setupConfirmed": true,
+ "warehouseSourcesReported": true,
},
- "equal": false,
"keyboard": {
"scanConsent": "granted",
"screen": "health-check",
@@ -36,7 +36,6 @@ exports[`keyboard commit vs control action commit > keep-skills: mount with no s
"action": {
"skillsComplete": true,
},
- "equal": false,
"keyboard": {},
}
`;
@@ -47,7 +46,6 @@ exports[`keyboard commit vs control action commit > manual-auth-code: escape vs
"overlay": false,
"screen": "auth",
},
- "equal": true,
"keyboard": {
"overlay": false,
"screen": "auth",
@@ -61,7 +59,6 @@ exports[`keyboard commit vs control action commit > manual-auth-code: paste code
"overlay": false,
"screen": "auth",
},
- "equal": true,
"keyboard": {
"overlay": false,
"screen": "auth",
@@ -76,7 +73,6 @@ exports[`keyboard commit vs control action commit > mcp: decline install vs set_
"mcpOutcome": "skipped",
"screen": "slack-connect",
},
- "equal": false,
"keyboard": {
"mcpComplete": true,
"mcpOutcome": "skipped",
@@ -92,7 +88,6 @@ exports[`keyboard commit vs control action commit > outro: any key vs dismiss_ou
"outroDismissed": true,
"screen": "mcp",
},
- "equal": true,
"keyboard": {
"outroDismissed": true,
"screen": "mcp",
@@ -107,7 +102,6 @@ exports[`keyboard commit vs control action commit > port-conflict: enter vs reso
"portConflictProcess": null,
"screen": "auth",
},
- "equal": true,
"keyboard": {
"overlay": false,
"portConflictProcess": null,
@@ -122,7 +116,6 @@ exports[`keyboard commit vs control action commit > self-driving-handoff: enter
"screen": "self-driving-github",
"selfDrivingHandoffConfirmed": true,
},
- "equal": true,
"keyboard": {
"screen": "self-driving-github",
"selfDrivingHandoffConfirmed": true,
@@ -136,7 +129,6 @@ exports[`keyboard commit vs control action commit > self-driving-integration-che
"integrate": true,
"screen": "health-check",
},
- "equal": true,
"keyboard": {
"integrate": true,
"screen": "health-check",
@@ -152,7 +144,6 @@ exports[`keyboard commit vs control action commit > setup: enter on first router
},
"screen": "auth",
},
- "equal": true,
"keyboard": {
"frameworkContext": {
"router": "app-router",
@@ -168,7 +159,6 @@ exports[`keyboard commit vs control action commit > slack-connect: skip vs dismi
"screen": "keep-skills",
"slackStepDismissed": true,
},
- "equal": false,
"keyboard": {
"screen": "keep-skills",
"skillsComplete": true,
@@ -183,7 +173,6 @@ exports[`keyboard commit vs control action commit > source-maps-outro: any key v
"outroDismissed": true,
"screen": "keep-skills",
},
- "equal": false,
"keyboard": {
"outroDismissed": true,
"screen": "keep-skills",
@@ -199,7 +188,6 @@ exports[`keyboard commit vs control action commit > task-notice: enter vs resolv
"screen": "intro",
"taskNotice": null,
},
- "equal": true,
"keyboard": {
"overlay": false,
"screen": "intro",
@@ -215,7 +203,6 @@ exports[`keyboard commit vs control action commit > task-notice: escape vs resol
"screen": "intro",
"taskNotice": null,
},
- "equal": true,
"keyboard": {
"overlay": false,
"screen": "intro",
@@ -231,7 +218,6 @@ exports[`keyboard commit vs control action commit > wizard-ask: type answer and
"pendingQuestion": null,
"screen": "intro",
},
- "equal": true,
"keyboard": {
"overlay": false,
"pendingQuestion": null,
diff --git a/e2e-harness/__tests__/control-socket-headless.test.ts b/e2e-harness/__tests__/control-socket-headless.test.ts
index f88f56918..1eb4c0206 100644
--- a/e2e-harness/__tests__/control-socket-headless.test.ts
+++ b/e2e-harness/__tests__/control-socket-headless.test.ts
@@ -7,10 +7,10 @@ import { spawn, type ChildProcess } from 'node:child_process';
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
-import { afterEach, describe, expect, it } from 'vitest';
import { ControlClient } from '@store/control';
import { HEADLESS_FLAG } from '@env';
import { waitForSocket } from '@e2e-harness/launch';
+import { expectNoSecrets } from '@store/testing';
const REPO = path.resolve(__dirname, '../..');
let child: ChildProcess | null = null;
@@ -63,7 +63,7 @@ describe('controlled headless surface', () => {
const state = await client.state();
expect(state.currentScreen).toBe('intro');
expect(state.session.runPhase).toBe('idle');
- expect(JSON.stringify(state)).not.toContain('phx_test_only');
+ expectNoSecrets(JSON.stringify(state));
await expect(client.startRun({ programId: 'nope' })).rejects.toMatchObject({
status: 400,
});
diff --git a/e2e-harness/__tests__/control-socket-tui.test.ts b/e2e-harness/__tests__/control-socket-tui.test.ts
index d96c94e60..6311545bc 100644
--- a/e2e-harness/__tests__/control-socket-tui.test.ts
+++ b/e2e-harness/__tests__/control-socket-tui.test.ts
@@ -6,7 +6,6 @@
import * as fs from 'node:fs';
import * as os from 'node:os';
import * as path from 'node:path';
-import { afterEach, describe, expect, it } from 'vitest';
import { ControlClient, ControlClientError } from '@store/control';
import { Program } from '@store/programs';
import { buildLaunch, waitForSocket } from '@e2e-harness/launch';
diff --git a/e2e-harness/__tests__/docs-routes.test.ts b/e2e-harness/__tests__/docs-routes.test.ts
new file mode 100644
index 000000000..e8a58728c
--- /dev/null
+++ b/e2e-harness/__tests__/docs-routes.test.ts
@@ -0,0 +1,19 @@
+import fs from 'fs';
+import path from 'path';
+import { ROUTES } from '@store/control';
+
+/** The route table in ARCHITECTURE.md is the contract the workbench reads; it must name what the server answers. */
+describe('the documented control API', () => {
+ it('lists exactly the routes the server serves', () => {
+ const doc = fs.readFileSync(
+ path.resolve(__dirname, '../ARCHITECTURE.md'),
+ 'utf8',
+ );
+ const documented = new Set(
+ [...doc.matchAll(/^\| `((?:GET|POST) \/[^`?]*)/gm)].map((m) =>
+ m[1].replace('', ':id').trim(),
+ ),
+ );
+ expect([...documented].sort()).toEqual([...ROUTES].sort());
+ });
+});
diff --git a/e2e-harness/__tests__/e2e-profile-ask.test.ts b/e2e-harness/__tests__/e2e-profile-ask.test.ts
index 346653106..ca4a2fe93 100644
--- a/e2e-harness/__tests__/e2e-profile-ask.test.ts
+++ b/e2e-harness/__tests__/e2e-profile-ask.test.ts
@@ -494,6 +494,29 @@ describe('E2E_DRIVABLE_SCREENS', () => {
expect(E2E_DRIVABLE_SCREENS).toContain(Overlay.TaskNotice);
});
+ it('is exactly the set of screens decideE2eAction commits on', () => {
+ // A rich state lets every conditional case act; screens that still wait
+ // are not drivable and must not be listed.
+ const rich: StateOverrides = {
+ setupQuestions: [
+ {
+ key: 'router',
+ message: 'router?',
+ options: [{ label: 'a', value: 'a' }],
+ },
+ ],
+ pendingQuestion: { id: 'a', source: 's', questions: [text('q')] },
+ taskNotice: { title: 't', items: [], prompt: 'p' },
+ };
+ const every = [...Object.values(ScreenId), ...Object.values(Overlay)];
+ const acting = every.filter(
+ (screen) =>
+ decideE2eAction(state({ currentScreen: screen, ...rich }), profile())
+ .action !== undefined,
+ );
+ expect([...acting].sort()).toEqual([...E2E_DRIVABLE_SCREENS].sort());
+ });
+
it('has a decideE2eAction case for every screen it lists', () => {
// A listed screen with no case would return `{ wait: true }` forever,
// stalling the run instead of failing it.
diff --git a/e2e-harness/__tests__/e2e-profile-failure.test.ts b/e2e-harness/__tests__/e2e-profile-failure.test.ts
index c5dd48482..97ef9b1bd 100644
--- a/e2e-harness/__tests__/e2e-profile-failure.test.ts
+++ b/e2e-harness/__tests__/e2e-profile-failure.test.ts
@@ -1,4 +1,3 @@
-import { describe, expect, it } from 'vitest';
import { ScreenId } from '@tui/router';
import type { ControlState } from '@store/types';
import { DEFAULT_E2E_PROFILE, decideE2eAction } from '@e2e-harness/e2e-profile';
diff --git a/e2e-harness/__tests__/keyboard-equivalence.test.tsx b/e2e-harness/__tests__/keyboard-equivalence.test.tsx
index 1aefd7f28..3db83fc8c 100644
--- a/e2e-harness/__tests__/keyboard-equivalence.test.tsx
+++ b/e2e-harness/__tests__/keyboard-equivalence.test.tsx
@@ -3,7 +3,7 @@
* keyboard on one store and apply the control action on another, then golden
* both session diffs. Pairs whose diffs differ today are recorded, not hidden.
*/
-import { vi, describe, it, expect, afterEach, beforeAll } from 'vitest';
+import { afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
import { render, cleanup } from 'ink-testing-library';
import {
WizardStore,
@@ -436,6 +436,18 @@ function applyAction(pair: Pair): Record {
return diff(before, snap(store));
}
+/**
+ * Keys the rendered walk commits from the *next* screen's mount effect (keep-skills
+ * completes with no skills dir; slack-connect records "not connected"), which no
+ * control action produces. Everything else must match exactly.
+ */
+const MOUNT_EFFECTS: Record = {
+ 'mcp: decline install vs set_mcp_outcome skipped': ['slackConnected'],
+ 'slack-connect: skip vs dismiss_slack': ['skillsComplete'],
+ 'audit-outro: any key vs dismiss_outro': ['skillsComplete'],
+ 'source-maps-outro: any key vs dismiss_outro': ['skillsComplete'],
+};
+
describe('keyboard commit vs control action commit', () => {
beforeAll(() => {
vi.spyOn(process, 'exit').mockImplementation(() => undefined as never);
@@ -446,11 +458,19 @@ describe('keyboard commit vs control action commit', () => {
it(pair.name, async () => {
const keyboard = await driveKeyboard(pair);
const action = applyAction(pair);
- expect({
- keyboard,
- action,
- equal: JSON.stringify(keyboard) === JSON.stringify(action),
- }).toMatchSnapshot();
+ const mountOnly = MOUNT_EFFECTS[pair.name] ?? [];
+ const keyboardOwn = Object.fromEntries(
+ Object.entries(keyboard).filter(([k]) => !mountOnly.includes(k)),
+ );
+ if (pair.keys.length === 0) {
+ // A bare mount commits nothing in this harness; the action still must.
+ expect(keyboard).toEqual({});
+ expect(Object.keys(action).length).toBeGreaterThan(0);
+ } else {
+ expect(action).toEqual(keyboardOwn);
+ for (const key of mountOnly) expect(keyboard).toHaveProperty(key);
+ }
+ expect({ keyboard, action }).toMatchSnapshot();
});
}
});
diff --git a/e2e-harness/__tests__/launch.test.ts b/e2e-harness/__tests__/launch.test.ts
index bb23c48ca..dfb206d6a 100644
--- a/e2e-harness/__tests__/launch.test.ts
+++ b/e2e-harness/__tests__/launch.test.ts
@@ -1,7 +1,7 @@
-import { describe, expect, it } from 'vitest';
-import { Program } from '@store/programs';
+import { HEADLESS_FLAG } from '@env';
+import { getProgramConfig, Program, PROGRAM_REGISTRY } from '@store/programs';
import type { ProgramId } from '@store/types';
-import { buildLaunch, PROGRAM_COMMANDS } from '@e2e-harness/launch';
+import { buildLaunch, launchWords } from '@e2e-harness/launch';
import { hasProfile } from '@e2e-harness/profiles';
const base = {
@@ -10,18 +10,28 @@ const base = {
projectId: '228144',
};
-describe('buildLaunch', () => {
- it('launches every profiled program through its command words', () => {
- for (const id of Object.keys(PROGRAM_COMMANDS)) {
- expect(hasProfile(id), id).toBe(true);
- const { args } = buildLaunch({ ...base, programId: id }, '/repo');
- expect(args.slice(0, 1 + PROGRAM_COMMANDS[id]!.length)).toEqual([
- 'bin.ts',
- ...PROGRAM_COMMANDS[id]!,
- ]);
+describe('launchWords', () => {
+ it('launches every profiled program through the command its config declares', () => {
+ const profiled = PROGRAM_REGISTRY.map((c) => c.id).filter((id) =>
+ hasProfile(id),
+ );
+ expect(profiled.length).toBeGreaterThanOrEqual(9);
+ for (const id of profiled) {
+ const words = launchWords(id);
+ if (id === Program.PostHogIntegration) expect(words).toEqual([]);
+ else if (id === Program.Audit) expect(words).toEqual(['audit', 'all']);
+ else expect(words, id).toEqual([getProgramConfig(id).command]);
}
});
+ it('refuses a program with no command', () => {
+ expect(() => launchWords('mcp-add' as ProgramId)).toThrow(
+ /no launch command/,
+ );
+ });
+});
+
+describe('buildLaunch', () => {
it('maps the run inputs to flags and keeps the key in the environment', () => {
const { cmd, args, env } = buildLaunch(
{
@@ -78,6 +88,29 @@ describe('buildLaunch', () => {
});
});
+ it('launches the headless surface under its flag, without the ask flag', () => {
+ const { cmd, args } = buildLaunch(
+ {
+ ...base,
+ programId: Program.PostHogIntegration,
+ surface: 'headless',
+ e2eAsk: true,
+ bin: 'dist/bin.js',
+ env: {},
+ },
+ '/repo',
+ );
+ expect(cmd).toBe('node');
+ expect(args.slice(0, 4)).toEqual([
+ 'dist/bin.js',
+ `--${HEADLESS_FLAG}`,
+ '--control-socket',
+ '/tmp/w/w.sock',
+ ]);
+ expect(args).not.toContain('--ci');
+ expect(args).not.toContain('--e2e-ask');
+ });
+
it('drops an inherited key when the run has none', () => {
const { env, args } = buildLaunch(
{
@@ -90,10 +123,4 @@ describe('buildLaunch', () => {
expect(env.POSTHOG_WIZARD_API_KEY).toBeUndefined();
expect(args[1]).toBe('--ci');
});
-
- it('refuses a program with no launch command', () => {
- expect(() =>
- buildLaunch({ ...base, programId: 'mcp-add' as ProgramId }, '/repo'),
- ).toThrow(/no launch command/);
- });
});
diff --git a/e2e-harness/__tests__/picks.test.ts b/e2e-harness/__tests__/picks.test.ts
new file mode 100644
index 000000000..2812206c0
--- /dev/null
+++ b/e2e-harness/__tests__/picks.test.ts
@@ -0,0 +1,77 @@
+import fs from 'fs';
+import os from 'os';
+import path from 'path';
+import { nativeVariantFor, pickIntegrationTarget } from '@e2e-harness/picks';
+
+const dirs: string[] = [];
+afterEach(() => {
+ for (const d of dirs.splice(0)) {
+ fs.rmSync(d, { recursive: true, force: true });
+ }
+});
+
+function fixture(files: Record): string {
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), 'wz-picks-'));
+ dirs.push(root);
+ for (const [rel, text] of Object.entries(files)) {
+ fs.mkdirSync(path.dirname(path.join(root, rel)), { recursive: true });
+ fs.writeFileSync(path.join(root, rel), text);
+ }
+ return root;
+}
+
+const pkg = (deps: Record) =>
+ JSON.stringify({ name: 'x', dependencies: deps });
+
+describe('pickIntegrationTarget', () => {
+ it('prefers the first instrumentable app under apps/ over packages/ and the root', async () => {
+ const root = fixture({
+ 'package.json': pkg({ express: '4.19.0' }),
+ 'apps/web/package.json': pkg({ next: '15.0.0', react: '19.0.0' }),
+ 'packages/api/package.json': pkg({ express: '4.19.0' }),
+ });
+ expect(await pickIntegrationTarget(root)).toEqual({
+ integration: 'nextjs',
+ path: 'apps/web',
+ });
+ });
+
+ it('falls back to the root for a single app, and to null for nothing detectable', async () => {
+ expect(
+ await pickIntegrationTarget(
+ fixture({ 'package.json': pkg({ express: '4.19.0' }) }),
+ ),
+ ).toEqual({ integration: 'javascript_node', path: '.' });
+ expect(
+ await pickIntegrationTarget(fixture({ 'README.md': '' })),
+ ).toBeNull();
+ });
+});
+
+describe('nativeVariantFor', () => {
+ it('trusts what the detector recognised before reading manifests', () => {
+ const root = fixture({ 'go.mod': 'module x' });
+ expect(nativeVariantFor(root, { detected: 'flutter' })).toBe('flutter');
+ expect(nativeVariantFor(root, { detected: 'unknown' })).toBe('go');
+ expect(nativeVariantFor(root, undefined)).toBe('go');
+ });
+
+ it('names the platform from its manifest, or an Xcode project, else null', () => {
+ expect(nativeVariantFor(fixture({ 'Cargo.toml': '' }), undefined)).toBe(
+ 'rust',
+ );
+ expect(
+ nativeVariantFor(fixture({ 'settings.gradle': '' }), undefined),
+ ).toBe('android');
+ expect(
+ nativeVariantFor(
+ fixture({ 'App.xcodeproj/project.pbxproj': '' }),
+ undefined,
+ ),
+ ).toBe('ios');
+ expect(
+ nativeVariantFor(fixture({ 'README.md': '' }), undefined),
+ ).toBeNull();
+ expect(nativeVariantFor('/nonexistent/root', undefined)).toBeNull();
+ });
+});
diff --git a/e2e-harness/__tests__/run-status.test.ts b/e2e-harness/__tests__/run-status.test.ts
new file mode 100644
index 000000000..7a82cd64d
--- /dev/null
+++ b/e2e-harness/__tests__/run-status.test.ts
@@ -0,0 +1,40 @@
+import { OutroKind, RunPhase } from '@store';
+import { runStatus } from '@e2e-harness/run-status';
+
+const session = (
+ runPhase: RunPhase,
+ over: { runRequested?: boolean; message?: string } = {},
+) => ({
+ runPhase,
+ runRequested: over.runRequested ?? false,
+ outroData: over.message
+ ? { kind: OutroKind.Error, message: over.message }
+ : null,
+});
+
+describe('the MCP run status', () => {
+ it.each([
+ [RunPhase.Idle, 'idle'],
+ [RunPhase.Running, 'running'],
+ [RunPhase.Completed, 'done'],
+ [RunPhase.Error, 'failed'],
+ ])('maps %s to %s', (phase, status) => {
+ expect(runStatus(session(phase)).integration).toBe(status);
+ });
+
+ it('reports an armed but not yet started run as running', () => {
+ expect(
+ runStatus(session(RunPhase.Idle, { runRequested: true })).integration,
+ ).toBe('running');
+ });
+
+ it('carries the failure reason only for a failed run', () => {
+ expect(
+ runStatus(session(RunPhase.Error, { message: 'gateway refused' })),
+ ).toEqual({ integration: 'failed', integrationError: 'gateway refused' });
+ expect(
+ runStatus(session(RunPhase.Completed, { message: 'stale' }))
+ .integrationError,
+ ).toBeNull();
+ });
+});
diff --git a/e2e-harness/e2e-profile.ts b/e2e-harness/e2e-profile.ts
index 5039c3857..1c4ea95de 100644
--- a/e2e-harness/e2e-profile.ts
+++ b/e2e-harness/e2e-profile.ts
@@ -403,14 +403,27 @@ export function decideE2eAction(
}
}
-/** Screens this profile knows how to act on — for completeness checks/tests. */
+/** Every screen `decideE2eAction` commits on; a test derives the same set from the function. */
export const E2E_DRIVABLE_SCREENS: readonly ScreenName[] = [
ScreenId.Intro,
+ ScreenId.RevenueIntro,
+ ScreenId.MigrationIntro,
+ ScreenId.AgentSkillIntro,
+ ScreenId.AiObservabilityIntro,
+ ScreenId.MetricsIntro,
+ ScreenId.ErrorTrackingIntro,
+ ScreenId.AuditIntro,
+ ScreenId.SourceMapsIntro,
+ ScreenId.DoctorIntro,
+ ScreenId.WarehouseIntro,
+ ScreenId.SelfDrivingIntro,
ScreenId.HealthCheck,
ScreenId.Setup,
ScreenId.SelfDrivingIntegrationCheck,
+ ScreenId.SelfDrivingHandoff,
ScreenId.Outro,
ScreenId.SourceMapsOutro,
+ ScreenId.AuditOutro,
ScreenId.Mcp,
ScreenId.McpSuggestedPrompts,
ScreenId.SlackConnect,
diff --git a/e2e-harness/launch.ts b/e2e-harness/launch.ts
index f3aaf3635..c9804fb82 100644
--- a/e2e-harness/launch.ts
+++ b/e2e-harness/launch.ts
@@ -1,25 +1,31 @@
-/**
- * How the harness starts a controlled wizard: the real binary, `--ci` for
- * API-key auth, `--control-socket` for the parent, one program per command.
- */
-import * as fs from 'node:fs';
-import * as path from 'node:path';
-import { Program } from '@store/programs';
+/** How the harness starts a controlled wizard: the real binary, API-key auth, a control socket, one program per command. */
+import fs from 'fs';
+import path from 'path';
+import { HEADLESS_FLAG } from '@env';
+import { ControlClient } from '@store/control';
+import { getProgramConfig, Program } from '@store/programs';
import type { ProgramId } from '@store/types';
-/** The command words that launch each program; the default flow has none. */
-export const PROGRAM_COMMANDS: Partial> = {
+/** Programs whose launch words are not their `command`: the default flow has none, audit runs `audit all`. */
+const COMMAND_OVERRIDES: Partial> = {
[Program.PostHogIntegration]: [],
- [Program.AiObservability]: ['ai-observability'],
- [Program.Metrics]: ['metrics'],
- [Program.ReplayVision]: ['replay-vision'],
- [Program.SelfDriving]: ['self-driving'],
- [Program.ErrorTrackingUploadSourceMaps]: ['upload-source-maps'],
- [Program.ErrorTracking]: ['error-tracking'],
- [Program.WarehouseSource]: ['warehouse'],
[Program.Audit]: ['audit', 'all'],
};
+/** The command words that launch a program: the `command` its config declares, unless overridden. */
+export function launchWords(programId: ProgramId): readonly string[] {
+ const override = COMMAND_OVERRIDES[programId];
+ if (override) return override;
+ let command: string | undefined;
+ try {
+ command = getProgramConfig(programId).command;
+ } catch {
+ command = undefined;
+ }
+ if (!command) throw new Error(`no launch command for program ${programId}`);
+ return [command];
+}
+
export interface LaunchOptions {
programId: ProgramId;
appDir: string;
@@ -28,7 +34,9 @@ export interface LaunchOptions {
region?: 'us' | 'eu';
/** The personal API key. Travels in the environment, never in argv. */
apiKey?: string;
- /** Keep `wizard_ask` wired so the parent answers the agent's questions. */
+ /** `tui` (default) drives the real screens with `--ci`; `headless` uses the published headless flag. */
+ surface?: 'tui' | 'headless';
+ /** Keep `wizard_ask` wired so the parent answers the agent's questions (TUI surface). */
e2eAsk?: boolean;
/** Self-driving only: skip the integration check and integrate first. */
integrate?: boolean;
@@ -37,6 +45,8 @@ export interface LaunchOptions {
model?: string;
/** Dump task-stream payloads; '' means the default path. */
taskStreamLog?: string;
+ /** The entry to run; a `.js` file runs under node, anything else under tsx. */
+ bin?: string;
env?: NodeJS.ProcessEnv;
}
@@ -50,12 +60,12 @@ export interface Launch {
const STRIPPED_ENV = /^(CLAUDE|ANTHROPIC|AI_AGENT)/;
export function buildLaunch(o: LaunchOptions, cwd = process.cwd()): Launch {
- const words = PROGRAM_COMMANDS[o.programId];
- if (!words) throw new Error(`no launch command for program ${o.programId}`);
+ const bin = o.bin ?? 'bin.ts';
+ const headless = o.surface === 'headless';
const args = [
- 'bin.ts',
- ...words,
- '--ci',
+ bin,
+ ...launchWords(o.programId),
+ headless ? `--${HEADLESS_FLAG}` : '--ci',
'--control-socket',
o.socketPath,
'--install-dir',
@@ -65,14 +75,16 @@ export function buildLaunch(o: LaunchOptions, cwd = process.cwd()): Launch {
'--region',
o.region ?? 'us',
];
- if (o.e2eAsk) args.push('--e2e-ask');
- if (o.integrate === true && o.programId === Program.SelfDriving)
+ if (o.e2eAsk && !headless) args.push('--e2e-ask');
+ if (o.integrate === true && o.programId === Program.SelfDriving) {
args.push('--integrate');
+ }
if (o.harness) args.push('--harness', o.harness);
if (o.sequence) args.push('--sequence', o.sequence);
if (o.model) args.push('--model', o.model);
- if (o.taskStreamLog !== undefined)
+ if (o.taskStreamLog !== undefined) {
args.push('--task-stream-log', o.taskStreamLog);
+ }
const env: NodeJS.ProcessEnv = {};
for (const [k, v] of Object.entries(o.env ?? process.env)) {
@@ -82,21 +94,32 @@ export function buildLaunch(o: LaunchOptions, cwd = process.cwd()): Launch {
env.WIZARD_ASK_AUTODRIVE = '1';
if (o.apiKey) env.POSTHOG_WIZARD_API_KEY = o.apiKey;
else delete env.POSTHOG_WIZARD_API_KEY;
- return { cmd: path.join(cwd, 'node_modules/.bin/tsx'), args, env };
+ const cmd = bin.endsWith('.js')
+ ? 'node'
+ : path.join(cwd, 'node_modules/.bin/tsx');
+ return { cmd, args, env };
}
-/** Resolve once the wizard listens on its socket, or reject after `timeoutMs`. */
+/** Resolve once the wizard answers on its socket, or reject after `timeoutMs`. */
export async function waitForSocket(
socketPath: string,
timeoutMs: number,
): Promise {
const deadline = Date.now() + timeoutMs;
+ const client = new ControlClient(socketPath);
while (Date.now() < deadline) {
- if (fs.existsSync(socketPath)) return;
+ if (fs.existsSync(socketPath)) {
+ try {
+ await client.health();
+ return;
+ } catch {
+ // A stale file or a server still binding; try again.
+ }
+ }
await new Promise((r) => setTimeout(r, 150));
}
throw new Error(
- `the wizard did not open ${socketPath} within ${timeoutMs}ms`,
+ `the wizard did not answer on ${socketPath} within ${timeoutMs}ms`,
);
}
diff --git a/e2e-harness/picks.ts b/e2e-harness/picks.ts
index ba407cf97..5fb679eb4 100644
--- a/e2e-harness/picks.ts
+++ b/e2e-harness/picks.ts
@@ -4,10 +4,9 @@
* Computed here, on the parent's side of the socket, and committed through the
* program's control actions.
*/
-import * as fs from 'node:fs';
-import { join } from 'node:path';
-import { FRAMEWORK_REGISTRY, buildSession } from '@store';
-import { detectFramework } from '@store/detection';
+import fs from 'fs';
+import { join } from 'path';
+import { FRAMEWORK_REGISTRY, buildSession, detectFramework } from '@store';
import {
detectSourceMapsPrerequisites,
SOURCE_MAPS_CONTEXT_KEYS,
diff --git a/e2e-harness/profiles.ts b/e2e-harness/profiles.ts
index 38f8271f7..a6af5072a 100644
--- a/e2e-harness/profiles.ts
+++ b/e2e-harness/profiles.ts
@@ -3,7 +3,7 @@
* program's flow.
*
* Each program declares its test path as JSON next to it
- * (`src/lib/programs//test/e2e.json`): a `profile` (the options the run
+ * (`src/store/programs//test/e2e.json`): a `profile` (the options the run
* auto-takes) plus a documented `path`. {@link profileFor} loads the `profile`
* and maps it by program id.
*
diff --git a/e2e-harness/run-status.ts b/e2e-harness/run-status.ts
new file mode 100644
index 000000000..cb358b341
--- /dev/null
+++ b/e2e-harness/run-status.ts
@@ -0,0 +1,31 @@
+import { RunPhase } from '@store';
+import type { ControlState } from '@store/types';
+
+const RUN_STATUS: Record = {
+ [RunPhase.Idle]: 'idle',
+ [RunPhase.Running]: 'running',
+ [RunPhase.Completed]: 'done',
+ [RunPhase.Error]: 'failed',
+};
+
+/** The MCP route's background run status, read off the store's run phase. */
+export function runStatus(
+ session: Pick<
+ ControlState['session'],
+ 'runPhase' | 'runRequested' | 'outroData'
+ >,
+): { integration: string; integrationError: string | null } {
+ const armed = session.runPhase === RunPhase.Idle && session.runRequested;
+ return {
+ integration: armed ? 'running' : RUN_STATUS[session.runPhase],
+ integrationError:
+ session.runPhase === RunPhase.Error
+ ? session.outroData?.message ?? null
+ : null,
+ };
+}
+
+/** `read_state` adds the run status under its historical names. */
+export function withRunStatus(state: ControlState): Record {
+ return { ...state, ...runStatus(state.session) };
+}
diff --git a/scripts/controlled-headless-smoke.no-jest.ts b/scripts/controlled-headless-smoke.no-jest.ts
index 318050691..94ed2339d 100644
--- a/scripts/controlled-headless-smoke.no-jest.ts
+++ b/scripts/controlled-headless-smoke.no-jest.ts
@@ -3,86 +3,59 @@
* one independent run per program named on the command line, then shutdown.
* Prints every request and a redacted view of every response.
*
- * POSTHOG_WIZARD_API_KEY=… npx tsx scripts/controlled-headless-smoke.no-jest.ts \
- * --app /tmp/app --project-id 228144 [--region us] [--bin dist/bin.js] posthog-integration [metrics …]
+ * APP_DIR=/tmp/app PROJECT_ID=228144 POSTHOG_REGION=us POSTHOG_KEY_FILE=… \
+ * WIZARD_CI_GATEWAY_TOKEN_FILE=… npx tsx scripts/controlled-headless-smoke.no-jest.ts \
+ * posthog-integration [metrics …]
+ *
+ * WIZARD_BIN=dist/bin.js runs a built binary instead of the source tree.
*/
-import { spawn } from 'node:child_process';
-import fs from 'node:fs';
-import os from 'node:os';
-import path from 'node:path';
-import { HEADLESS_FLAG } from '@env';
+import { spawn } from 'child_process';
+import fs from 'fs';
+import os from 'os';
+import path from 'path';
import { RunPhase } from '@store';
import { ControlClient } from '@store/control';
-import type { ControlState } from '@store/types';
+import type { ControlState, ProgramId } from '@store/types';
+import { buildLaunch, readApiKey, waitForSocket } from '@e2e-harness/launch';
-async function waitForSocket(p: string, timeoutMs: number): Promise {
- const deadline = Date.now() + timeoutMs;
- while (Date.now() < deadline) {
- if (fs.existsSync(p)) return;
- await new Promise((r) => setTimeout(r, 150));
- }
- throw new Error(`the wizard did not open ${p} within ${timeoutMs}ms`);
-}
-
-const args = process.argv.slice(2);
-const opt = (name: string, fallback?: string): string | undefined => {
- const i = args.indexOf(`--${name}`);
- return i >= 0 ? args[i + 1] : fallback;
-};
-const programs = args.filter(
- (a, i) => !a.startsWith('--') && (i === 0 || !args[i - 1].startsWith('--')),
-);
-const app = opt('app');
-const projectId = opt('project-id');
-const region = opt('region', 'us')!;
-const bin = opt('bin', 'bin.ts')!;
-if (!app || !projectId || programs.length === 0) {
+const programs = process.argv.slice(2) as ProgramId[];
+const appDir = process.env.APP_DIR;
+const projectId = process.env.PROJECT_ID;
+const apiKey = readApiKey();
+if (!appDir || !projectId || programs.length === 0 || !apiKey) {
process.stderr.write(
- 'usage: --app --project-id [--region us|eu] [--bin dist/bin.js] [program…]\n',
- );
- process.exit(2);
-}
-if (!process.env.POSTHOG_WIZARD_API_KEY) {
- process.stderr.write(
- 'POSTHOG_WIZARD_API_KEY must be set in the environment\n',
+ 'usage: APP_DIR= PROJECT_ID= [POSTHOG_REGION=us|eu] POSTHOG_KEY_FILE= ' +
+ 'npx tsx scripts/controlled-headless-smoke.no-jest.ts [program…]\n',
);
process.exit(2);
}
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'wizard-ctl-'));
const socketPath = path.join(dir, 'w.sock');
-const cmd = bin.endsWith('.ts')
- ? path.join(process.cwd(), 'node_modules/.bin/tsx')
- : 'node';
-const argv = [
- bin,
- `--${HEADLESS_FLAG}`,
- '--control-socket',
+const launch = buildLaunch({
+ programId: programs[0],
+ appDir,
socketPath,
- '--project-id',
projectId,
- '--region',
- region,
- '--install-dir',
- app,
-];
+ region: process.env.POSTHOG_REGION === 'eu' ? 'eu' : 'us',
+ apiKey,
+ surface: 'headless',
+ bin: process.env.WIZARD_BIN,
+});
const log = (line: string) => process.stdout.write(`${line}\n`);
log(
- `$ ${cmd === 'node' ? 'node' : 'npx tsx'} ${argv.join(
+ `$ ${launch.cmd.endsWith('tsx') ? 'npx tsx' : launch.cmd} ${launch.args.join(
' ',
)} # POSTHOG_WIZARD_API_KEY in env`,
);
-const env = { ...process.env };
-for (const k of Object.keys(env))
- if (/^(CLAUDE|ANTHROPIC|AI_AGENT)/.test(k)) delete env[k];
-const child = spawn(cmd, argv, {
+const child = spawn(launch.cmd, launch.args, {
cwd: process.cwd(),
- env,
+ env: launch.env,
stdio: ['ignore', 'pipe', 'pipe'],
});
-const wizardOut: string[] = [];
-child.stdout.on('data', (d: Buffer) => wizardOut.push(d.toString()));
-child.stderr.on('data', (d: Buffer) => wizardOut.push(d.toString()));
+const childOutput: string[] = [];
+child.stdout.on('data', (d: Buffer) => childOutput.push(d.toString()));
+child.stderr.on('data', (d: Buffer) => childOutput.push(d.toString()));
const exit = new Promise((resolve) =>
child.once('exit', (code) => resolve(code)),
);
@@ -136,11 +109,11 @@ async function main(): Promise {
const code = await exit;
log(`wizard exit code: ${code}`);
log(`socket removed: ${!fs.existsSync(socketPath)}`);
- const console = wizardOut
+ const output = childOutput
.join('')
.replace(/phx_[A-Za-z0-9_]+/g, 'phx_')
.trim();
- if (console) log(`wizard console:\n${console}`);
+ if (output) log(`wizard console:\n${output}`);
process.exit(code === 0 ? 0 : 1);
}
diff --git a/scripts/mcp-install-smoke-test.ts b/scripts/mcp-install-smoke-test.ts
index 633a57f3a..fc7eb76c7 100644
--- a/scripts/mcp-install-smoke-test.ts
+++ b/scripts/mcp-install-smoke-test.ts
@@ -100,7 +100,7 @@ function wizard(...args: string[]): Run {
/**
* The non-interactive install. Reuses the run pipeline's headless flag, so the
- * name is imported rather than spelled out — see @lib/headless-mode.
+ * name is imported rather than spelled out — see src/store/shared/headless-mode.ts.
*/
function mcpAdd(...extra: string[]): Run {
return wizard('mcp', 'add', `--${HEADLESS_FLAG}`, ...extra);
diff --git a/scripts/smoke-test.sh b/scripts/smoke-test.sh
index 32ca4e36b..ae62d0b15 100755
--- a/scripts/smoke-test.sh
+++ b/scripts/smoke-test.sh
@@ -11,6 +11,8 @@
# non-interactive published-build path) — it must not be rejected as an
# unknown argument. It is intentionally undocumented; this check only keeps
# the published binary from silently dropping the flag the cloud runs need.
+# 5. The control server ships in its own chunk that the TUI path never imports,
+# and --control-socket is refused without the headless flag.
#
# Runs from the wizard repo root via `pnpm test:smoke` (postbuild hook).
set -e
@@ -24,7 +26,7 @@ node --input-type=module -e "import '$DIST_BIN'" 2>&1 | head -5 | grep -q 'PostH
}
# ── 2. CI flag overrides physically absent from production builds ───────────
-# The override path (src/utils/ci-flag-overrides.ts) is dead code in published
+# The override path (src/store/shared/ci-flag-overrides.ts) is dead code in published
# builds and tsdown strips it; its env var name appearing in dist/*.js means
# dead-code elimination regressed and a prod surface leaked. Sourcemaps keep
# the original source, so only .js output counts.
@@ -86,7 +88,7 @@ fi
# not reject the flag, and it must not fall through to the --ci rejection. With
# no api-key the run exits fast on "Headless mode requires --api-key" — all this
# asserts is that the flag is recognized and live in the published binary. The
-# flag name is intentionally undocumented; keep it in sync with @lib/headless-mode.
+# flag name is intentionally undocumented; keep it in sync with src/store/shared/headless-mode.ts.
HEADLESS_FLAG='--headless-DONOTUSE-EXPERIMENTAL'
hl_output=$(node "$DIST_BIN" "$HEADLESS_FLAG" --install-dir /tmp/wizard-smoke-probe 2>&1) || true
if echo "$hl_output" | grep -qiE 'unknown argument|not currently supported'; then
@@ -106,6 +108,7 @@ fi
# The control API ships for headless runs only. Exactly one chunk carries the
# server; the TUI entry chunk and bin.js never import it; the published binary
# refuses --control-socket without the headless flag and accepts it with it.
+# Defined in src/store/control/marker.ts and src/tui/start-tui.ts.
CONTROL_MARKER='wizard-control-server'
TUI_MARKER='wizard-tui-entry'
control_chunks=$(grep -l "$CONTROL_MARKER" ./dist/*.js || true)
diff --git a/scripts/tui-replay.no-jest.ts b/scripts/tui-replay.no-jest.ts
index 55b2fa4b8..cc9f2f907 100644
--- a/scripts/tui-replay.no-jest.ts
+++ b/scripts/tui-replay.no-jest.ts
@@ -1,6 +1,6 @@
/**
* Replay captured real-TUI snapshots in the terminal — step through or auto-play
- * the `NN-.txt` frames a snapshot run dropped in SNAP_OUT.
+ * the `NN-.ans` frames a snapshot run dropped in SNAP_OUT.
*
* npx tsx scripts/tui-replay.no-jest.ts [--step | --delay ]
* pnpm wizard-ci-replay /tmp/snaps # Enter ▸ advance (default)
@@ -35,10 +35,10 @@ async function main() {
}
const frames = fs
.readdirSync(dir)
- .filter((f) => f.endsWith('.txt') && f !== 'latest.txt')
+ .filter((f) => f.endsWith('.ans'))
.sort();
if (frames.length === 0) {
- console.error(`✖ no NN-.txt snapshots in ${dir}`);
+ console.error(`✖ no NN-.ans snapshots in ${dir}`);
process.exit(1);
}
// Step (Enter to advance) is the default; fall back to a timed play when not a
diff --git a/scripts/tui-snapshots.no-jest.ts b/scripts/tui-snapshots.no-jest.ts
index c5ac38b86..cec42e04c 100644
--- a/scripts/tui-snapshots.no-jest.ts
+++ b/scripts/tui-snapshots.no-jest.ts
@@ -18,7 +18,7 @@ import path from 'path';
import { spawnSync } from 'child_process';
import { logToFile } from '@store';
import { getProgramConfig, Program } from '@store/programs';
-import { ControlClient } from '@store/control';
+import { ControlClient, ControlClientError } from '@store/control';
import type { ControlState, ProgramId } from '@store/types';
import { captureTui } from '@e2e-harness/tui-capture';
import { buildLaunch, readApiKey, waitForSocket } from '@e2e-harness/launch';
@@ -252,7 +252,8 @@ async function main(): Promise {
let state: ControlState;
try {
state = await client.state();
- } catch {
+ } catch (e) {
+ if (e instanceof ControlClientError) throw e; // a server fault, not an exit
break; // the wizard exited between polls
}
lastState = state;
@@ -342,7 +343,8 @@ async function main(): Promise {
try {
if (acted && (await client.state()).currentScreen !== before) continue;
state = await client.waitForChange(state.version, 600_000);
- } catch {
+ } catch (e) {
+ if (e instanceof ControlClientError) throw e;
break;
}
}
diff --git a/scripts/wizard-ci-mcp.no-jest.ts b/scripts/wizard-ci-mcp.no-jest.ts
index 38326c0c1..cbbb06658 100644
--- a/scripts/wizard-ci-mcp.no-jest.ts
+++ b/scripts/wizard-ci-mcp.no-jest.ts
@@ -14,12 +14,12 @@ import { z } from 'zod';
import fs from 'fs';
import os from 'os';
import path from 'path';
-import { RunPhase } from '@store';
import { ControlClient } from '@store/control';
import { Program } from '@store/programs';
-import type { ControlState, ProgramId } from '@store/types';
+import type { ProgramId } from '@store/types';
import { captureTui, type TuiCapture } from '@e2e-harness/tui-capture';
import { buildLaunch, waitForSocket } from '@e2e-harness/launch';
+import { withRunStatus } from '@e2e-harness/run-status';
const text = (data: unknown) => ({
content: [
@@ -48,25 +48,6 @@ function active(): ControlClient {
return client;
}
-const RUN_STATUS: Record = {
- [RunPhase.Idle]: 'idle',
- [RunPhase.Running]: 'running',
- [RunPhase.Completed]: 'done',
- [RunPhase.Error]: 'failed',
-};
-
-/** read_state adds the background run status under its historical names. */
-function withRunStatus(state: ControlState): Record {
- const { runPhase, runRequested, outroData } = state.session;
- const armed = runPhase === RunPhase.Idle && runRequested;
- return {
- ...state,
- integration: armed ? 'running' : RUN_STATUS[runPhase],
- integrationError:
- runPhase === RunPhase.Error ? outroData?.message ?? null : null,
- };
-}
-
async function waitFor(cond: () => boolean, ms: number): Promise {
const end = Date.now() + ms;
while (Date.now() < end) {
diff --git a/src/agent/README.md b/src/agent/README.md
index 7a189021a..e19da06a0 100644
--- a/src/agent/README.md
+++ b/src/agent/README.md
@@ -13,7 +13,9 @@ Runs exactly one independent agent run per `runAgent` call.
## Never contains
-Ink, screens, program definitions, or session mutation outside `WizardUI`.
+Ink, screens, program definitions, or session mutation outside `WizardUI`. One
+exception: `runner/shared/bootstrap.ts` stamps `skillId` on the run's own
+session copy before the run starts.
## May import
diff --git a/src/agent/__tests__/auth-error-context.test.ts b/src/agent/__tests__/auth-error-context.test.ts
index 580a5b4b5..001702033 100644
--- a/src/agent/__tests__/auth-error-context.test.ts
+++ b/src/agent/__tests__/auth-error-context.test.ts
@@ -1,7 +1,6 @@
import * as fs from 'fs';
import * as os from 'os';
import * as path from 'path';
-import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
vi.mock('@store/shared/analytics', () => ({
analytics: { captureException: vi.fn(), wizardCapture: vi.fn() },
diff --git a/src/cli/README.md b/src/cli/README.md
index 1ac77c7e2..899dfcb69 100644
--- a/src/cli/README.md
+++ b/src/cli/README.md
@@ -23,10 +23,21 @@ surfaces' public entries. Holds no domain logic.
`scripts/controlled-headless-smoke.no-jest.ts` drives the surface end to end
and prints every request and response.
+## Never contains
+
+Domain logic, Ink, or an agent run of its own: every function parses input,
+picks a surface, or forwards to a surface's public entry.
+
## May import
Every surface, but only through `@store`, `@store/types`, `@store/programs`,
-`@agent`, `@agent/types`, `@tui`, `@tui/types`, and `@tui/console`.
+`@agent`, `@agent/types`, `@tui`, `@tui/types`, and `@tui/console`; the two
+runners import `@store/control` dynamically.
+
+## Public entries
+
+None for other surfaces: `bin.ts` imports `main.ts`, and nothing else imports
+`@cli/*`.
## Tests
diff --git a/src/cli/__tests__/bin-preflight.test.ts b/src/cli/__tests__/bin-preflight.test.ts
index fa4b771b2..e1eb1cfa5 100644
--- a/src/cli/__tests__/bin-preflight.test.ts
+++ b/src/cli/__tests__/bin-preflight.test.ts
@@ -1,7 +1,6 @@
import * as fs from 'fs';
import * as path from 'path';
import { fileURLToPath } from 'url';
-import { describe, expect, it } from 'vitest';
import { ErrorCodes } from '@store';
import { PHW_ERROR_PREFIX } from '@store/shared/errors/emit';
diff --git a/src/cli/__tests__/control-refusal.test.ts b/src/cli/__tests__/control-refusal.test.ts
index 0657a5ef0..f090c4c1f 100644
--- a/src/cli/__tests__/control-refusal.test.ts
+++ b/src/cli/__tests__/control-refusal.test.ts
@@ -1,5 +1,3 @@
-import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
-
vi.mock('@env', async (importOriginal) => ({
...(await importOriginal()),
IS_PRODUCTION_BUILD: true,
diff --git a/src/cli/__tests__/control-surface.test.ts b/src/cli/__tests__/control-surface.test.ts
index 3e78442bc..5802222e4 100644
--- a/src/cli/__tests__/control-surface.test.ts
+++ b/src/cli/__tests__/control-surface.test.ts
@@ -1,4 +1,3 @@
-import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
import type { Arguments } from 'yargs';
const runners = vi.hoisted(() => ({
@@ -11,9 +10,11 @@ vi.mock('../runners/index.js', () => runners);
import { HEADLESS_FLAG } from '@env';
import {
errorTrackingUploadSourceMapsConfig,
+ posthogDoctorConfig,
posthogIntegrationConfig,
selfDrivingConfig,
} from '@store/programs';
+import { doctorCommand } from '../commands/doctor.js';
import { selfDrivingCommand } from '../commands/self-driving.js';
import { uploadSourcemapsCommand } from '../commands/upload-sourcemaps.js';
import { dispatchProgram } from '../commands/factories/shared.js';
@@ -99,4 +100,12 @@ describe('commands with their own handlers follow the same dispatch', () => {
expect(runners.runWizardCI).toHaveBeenCalledTimes(1);
},
);
+
+ it('doctor drives the real TUI when a parent holds the socket', () => {
+ doctorCommand.handler?.(argv({ ci: true, controlSocket: '/tmp/c.sock' }));
+ expect(runners.runWizard).toHaveBeenCalledWith(
+ posthogDoctorConfig,
+ expect.objectContaining({ ci: true, controlSocket: '/tmp/c.sock' }),
+ );
+ });
});
diff --git a/src/cli/commands/doctor.ts b/src/cli/commands/doctor.ts
index e389c6a32..764f41998 100644
--- a/src/cli/commands/doctor.ts
+++ b/src/cli/commands/doctor.ts
@@ -1,6 +1,7 @@
import {
getUI,
setUI,
+ isControlledTui,
readApiKeyFromEnv,
ErrorCodes,
emitWizardError,
@@ -27,10 +28,9 @@ export const doctorCommand: Command = {
posthogDoctorConfig.mapCliOptions?.(argv as Record) ??
{};
const options = { ...argv, ...extras };
- // doctor is otherwise a TUI-only diagnostic (it has no agent run); in CI we
- // fetch the project's health issues headlessly and report them instead. A
- // control socket means a parent drives the real TUI.
- if (options.ci && !options.controlSocket) {
+ // doctor has no agent run: `--ci` reports the project's health issues
+ // headlessly, unless a parent on the socket drives the real TUI.
+ if (options.ci && !isControlledTui(options)) {
void runDoctorCI(options);
} else {
runWizard(posthogDoctorConfig, options);
diff --git a/src/cli/commands/self-driving.ts b/src/cli/commands/self-driving.ts
index 130f5108d..cfeecedf5 100644
--- a/src/cli/commands/self-driving.ts
+++ b/src/cli/commands/self-driving.ts
@@ -1,3 +1,4 @@
+import { isControlledTui } from '@store';
import { dispatchProgram } from './factories/shared.js';
import { selfDrivingConfig } from '@store/programs';
import { skillProgramOptions } from './skill-program-options.js';
@@ -29,9 +30,8 @@ export const selfDrivingCommand: Command = {
'(no flag needed).',
);
}
- // A controlling parent on the socket answers those steps, so `--ci` is
- // fine when a control socket is attached.
- if (argv.ci && !argv.controlSocket) {
+ // A controlling parent on the socket answers those steps.
+ if (argv.ci && !isControlledTui(argv)) {
throw new Error(
'`self-driving` cannot run in CI mode — it requires interactive steps ' +
'(GitHub connect, issue-tracker selection, custom-scout approval).',
diff --git a/src/cli/index.ts b/src/cli/index.ts
index 7a07c7eb7..85b1d9cb9 100644
--- a/src/cli/index.ts
+++ b/src/cli/index.ts
@@ -1,2 +1,2 @@
-/** Entry of the cli composition root. Populated by the surface split (P3). */
+/** The cli composition root has no public runtime API; `bin.ts` imports `main.ts`. */
export {};
diff --git a/src/cli/main.ts b/src/cli/main.ts
index bdedaf416..b946f595f 100644
--- a/src/cli/main.ts
+++ b/src/cli/main.ts
@@ -29,7 +29,7 @@ import { skillCommand } from './commands/skill.js';
import { cliCommand } from './commands/cli/index.js';
import { LoggingUI } from '@tui/console';
-// The entry point owns the default renderer; @ui ships with none.
+// The entry point owns the default renderer; the store ships with none.
setUI(new LoggingUI());
// The store and the TUI never import the agent; the entry point installs it,
// lazily, so `--version` and `--help` never load the SDK or Ink.
diff --git a/src/cli/wizard.ts b/src/cli/wizard.ts
index 1a0823686..2852c3a79 100644
--- a/src/cli/wizard.ts
+++ b/src/cli/wizard.ts
@@ -104,7 +104,7 @@ export class Wizard {
// flag. init() additionally detects it up front to print a clearer message.
// The published-build, non-interactive path is the experimental headless
// flag, declared per-command on basic integration + audit via
- // `headlessOption` (see @lib/headless-mode), so no other command accepts
+ // `headlessOption` (see src/store/shared/headless-mode.ts), so no other command accepts
// it. CI needs `region` globally because the workbench passes it to every
// command. --ci and headless stay separate so their behavior can diverge.
if (!IS_PRODUCTION_BUILD) {
diff --git a/src/env.ts b/src/env.ts
index 8a35cd70a..24a28345d 100644
--- a/src/env.ts
+++ b/src/env.ts
@@ -49,7 +49,7 @@ export const RUN_SURFACE: 'cloud' | 'local' = process.argv.some(
* Add new keys here when a new runtime dependency is needed.
*/
type RuntimeEnvKey =
- // CI-build-only flag overrides (see utils/ci-flag-overrides.ts).
+ // CI-build-only flag overrides (see src/store/shared/ci-flag-overrides.ts).
// Deliberately NOT POSTHOG_WIZARD_-prefixed: yargs .env('POSTHOG_WIZARD')
// would claim it as an unknown CLI option and strict-reject the run.
| 'WIZARD_CI_FLAG_OVERRIDES'
diff --git a/src/store/README.md b/src/store/README.md
index 92a66a7e9..75c5d6171 100644
--- a/src/store/README.md
+++ b/src/store/README.md
@@ -25,7 +25,9 @@ Render-agnostic state and the contract between the agent and whatever renders.
## Never contains
-Ink, console output, or any import of `@agent`, `@tui`, or `@cli`.
+Ink, rendering, or any import of `@agent`, `@tui`, or `@cli`. Two shared helpers
+write to stderr on purpose: the terminal bell and the machine-readable error
+line.
## May import
diff --git a/src/store/control/__tests__/actions.test.ts b/src/store/control/__tests__/actions.test.ts
index 3c2027978..71650ebe9 100644
--- a/src/store/control/__tests__/actions.test.ts
+++ b/src/store/control/__tests__/actions.test.ts
@@ -1,4 +1,3 @@
-import { describe, expect, it } from 'vitest';
import { ERROR_TRACKING_PROJECT_PATH_KEY } from '../../programs/error-tracking/detect-agentic.js';
import { SOURCE_MAPS_CONTEXT_KEYS } from '../../programs/error-tracking-upload-source-maps/detect.js';
import { flowFor } from '../../programs/flow-for.js';
diff --git a/src/store/index.ts b/src/store/index.ts
index 1b716db11..ad459dafc 100644
--- a/src/store/index.ts
+++ b/src/store/index.ts
@@ -196,6 +196,7 @@ export {
logToFile,
} from './shared/debug.js';
export { readApiKeyFromEnv } from './shared/env-api-key.js';
+export { detectFramework } from './detection/index.js';
export { isTemplateEnvFileName } from './shared/env-scan.js';
export {
isNonInteractiveEnvironment,
diff --git a/src/store/ui/wizard-ui.ts b/src/store/ui/wizard-ui.ts
index 0c68f9c86..898b8a18d 100644
--- a/src/store/ui/wizard-ui.ts
+++ b/src/store/ui/wizard-ui.ts
@@ -36,7 +36,7 @@ export function isTaskStatus(value: string): value is TaskStatus {
* the main session, and some programs override to Haiku, so pricing must key
* off the per-turn model rather than a single run-wide assumption. Omit only
* when the caller genuinely has no model context (falls back to Sonnet
- * pricing — see `pricePerMtokForModel` in `@lib/agent/token-pricing`).
+ * pricing — see `pricePerMtokForModel` in `store/agent-protocol/token-pricing.ts`).
*/
export interface TokenUsageDelta {
inputTokens: number;
diff --git a/src/tui/__tests__/audit-checks-viewer-layout.test.ts b/src/tui/__tests__/audit-checks-viewer-layout.test.ts
index b97f35230..2f23d1fb5 100644
--- a/src/tui/__tests__/audit-checks-viewer-layout.test.ts
+++ b/src/tui/__tests__/audit-checks-viewer-layout.test.ts
@@ -1,4 +1,3 @@
-import { describe, expect, it } from 'vitest';
import { AUDIT_SEED_CHECKS } from '@store/programs/audit/seed';
import { COL_AREA_WIDTH } from '../screens/audit/AuditChecksViewer/layout.js';
diff --git a/src/tui/__tests__/error-tracking-tips.test.ts b/src/tui/__tests__/error-tracking-tips.test.ts
index acfdb80b3..1c17f1646 100644
--- a/src/tui/__tests__/error-tracking-tips.test.ts
+++ b/src/tui/__tests__/error-tracking-tips.test.ts
@@ -1,4 +1,3 @@
-import { describe, expect, test } from 'vitest';
import { Integration } from '@store';
import { ERROR_TRACKING_TIPS } from '../programs/error-tracking/content/tips.js';
diff --git a/src/tui/router.ts b/src/tui/router.ts
index 8af34dda2..0cc6b6897 100644
--- a/src/tui/router.ts
+++ b/src/tui/router.ts
@@ -1,6 +1,6 @@
/**
* Screen name vocabulary for the TUI. Resolution of the active screen lives in
- * the store (`WizardStore.currentScreen` over `@lib/flow-resolution`).
+ * the store (`WizardStore.currentScreen` over `state/flow-resolution.ts`).
*/
import { ScreenId } from './screen-sequences.js';
@@ -10,6 +10,6 @@ import type { ProgramId } from '@store/types';
export { ScreenId, Program };
export type { ProgramId };
-/** Interrupts, under the name the TUI has always used for them. */
+/** Interrupts, exported as `Overlay` for the screens. */
export { Interrupt as Overlay };
export type ScreenName = ScreenId | Interrupt;
From 917199e155efef07914c9e4a7b522e6d49baa648 Mon Sep 17 00:00:00 2001
From: "Vincent (Wen Yu) Ge"
Date: Sun, 20 Sep 2026 23:03:44 -0400
Subject: [PATCH 8/8] chore(build): refresh the chunk manifest fixtures
The bundle audit compares these with the manifests CI builds; the new modules
and the regrouped chunks belong in them. A failed comparison now uploads its
manifests and reports both files.
Generated-By: PostHog Desktop
Task-Id: d14e92bb-6ee1-49b5-8502-39cb80079589
---
scripts/__fixtures__/chunk-manifest.ci.json | 87 +++++++------------
scripts/__fixtures__/chunk-manifest.prod.json | 87 ++++++++++++++-----
2 files changed, 99 insertions(+), 75 deletions(-)
diff --git a/scripts/__fixtures__/chunk-manifest.ci.json b/scripts/__fixtures__/chunk-manifest.ci.json
index 2e2ade035..1f1d937b5 100644
--- a/scripts/__fixtures__/chunk-manifest.ci.json
+++ b/scripts/__fixtures__/chunk-manifest.ci.json
@@ -23,9 +23,9 @@
],
"imports": [
"src/agent/runner/sequence/orchestrator/queue-tools.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts"
+ "src/store/auth-session-state.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
},
"src/agent/agent-prompt-loader.ts": {
@@ -78,10 +78,10 @@
"src/agent/agent-prompt-loader.ts",
"src/agent/aio-capture.ts",
"src/agent/runner/sequence/orchestrator/queue-tools.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
- "src/store/programs/ai-opt-in-gate.ts"
+ "src/store/programs/ai-opt-in-gate.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
},
"src/agent/aio-capture.ts": {
@@ -98,8 +98,9 @@
"src/agent/runner/harness/pi/task.ts",
"src/agent/runner/harness/pi/tasks.ts",
"src/agent/runner/harness/pi/tools.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
- "src/store/agent-protocol/agent-signals.ts"
+ "src/store/agent-protocol/agent-signals.ts",
+ "src/store/auth-session-state.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
},
"src/agent/runner/harness/pi/mcp.ts": {
@@ -144,8 +145,9 @@
"src/agent/runner/harness/pi/security.ts",
"src/agent/runner/harness/pi/tools.ts",
"src/agent/runner/sequence/orchestrator/queue-tools.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
- "src/store/agent-protocol/agent-signals.ts"
+ "src/store/agent-protocol/agent-signals.ts",
+ "src/store/auth-session-state.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
},
"src/agent/runner/harness/pi/tasks.ts": {
@@ -160,8 +162,9 @@
],
"imports": [
"src/agent/aio-capture.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
- "src/store/auth-session-state.ts"
+ "src/store/agent-protocol/agent-signals.ts",
+ "src/store/auth-session-state.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
},
"src/agent/runner/sequence/orchestrator/queue-tools.ts": {
@@ -216,10 +219,11 @@
"src/cli/commands/basic-integration/non-interactive.ts",
"src/cli/commands/basic-integration/playground.ts",
"src/cli/control-hooks.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
- "src/tui/App.tsx"
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts",
+ "src/tui/App.tsx",
+ "src/tui/console/headless-ui.ts"
]
},
"src/cli/commands/basic-integration/ci-install.ts": {
@@ -229,7 +233,8 @@
"imports": [
"src/cli/control-hooks.ts",
"src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts"
+ "src/store/auth-session-state.ts",
+ "src/tui/console/headless-ui.ts"
]
},
"src/cli/commands/basic-integration/interactive.ts": {
@@ -268,49 +273,13 @@
],
"imports": [
"src/agent/agent-prompt.ts",
- "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
"src/store/control/actions.ts",
"src/store/programs/ai-opt-in-gate.ts",
- "src/tui/App.tsx"
- ]
- },
- "src/store/agent-protocol/agent-env-isolation.ts": {
- "sources": [
- "src/store/agent-protocol/agent-env-isolation.ts",
- "src/store/agent-protocol/agent-phase.ts",
- "src/store/agent-protocol/mcp-prompt.ts",
- "src/store/agent-protocol/token-pricing.ts",
- "src/store/file-watcher.ts",
- "src/store/mcp-project-profile.ts",
- "src/store/mcp-role-prompts.copy.json",
- "src/store/mcp-role-prompts.ts",
- "src/store/mcp-seed-events.ts",
- "src/store/safe-tools.ts",
- "src/store/services/claude-settings.ts",
- "src/store/services/steps/add-mcp-server-to-clients/browser-client.ts",
- "src/store/services/steps/add-mcp-server-to-clients/login-client.ts",
- "src/store/session/secret-vault.ts",
- "src/store/shared/clipboard.ts",
- "src/store/shared/custom-headers.ts",
- "src/store/shared/env-api-key.ts",
- "src/store/shared/environment.ts",
- "src/store/shared/terminal-bell.ts",
- "src/store/state/store.ts",
- "src/store/task-stream/audit-areas.ts",
- "src/store/task-stream/destinations/file.ts",
- "src/store/task-stream/destinations/posthog.ts",
- "src/store/task-stream/event-plan-watcher.ts",
- "src/store/task-stream/task-stream-push.ts",
- "src/store/tools/handoff.ts",
- "src/store/ui/store-ui.ts",
- "src/store/ui/wizard-ui.ts",
- "src/store/wizard-spellbook.ts"
- ],
- "imports": [
- "src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts"
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts",
+ "src/tui/App.tsx",
+ "src/tui/console/headless-ui.ts"
]
},
"src/store/agent-protocol/agent-signals.ts": {
@@ -501,9 +470,9 @@
"src/store/programs/run-config.ts"
],
"imports": [
- "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts"
+ "src/store/auth-session-state.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
},
"src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts": {
@@ -718,11 +687,19 @@
"src/tui/ui-store.ts"
],
"imports": [
- "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
"src/store/programs/ai-opt-in-gate.ts",
"src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
+ },
+ "src/tui/console/headless-ui.ts": {
+ "sources": [
+ "src/tui/console/headless-ui.ts",
+ "src/tui/console/logging-ui.ts"
+ ],
+ "imports": [
+ "src/store/auth-session-state.ts"
+ ]
}
}
diff --git a/scripts/__fixtures__/chunk-manifest.prod.json b/scripts/__fixtures__/chunk-manifest.prod.json
index 3e48dd1ac..262963df6 100644
--- a/scripts/__fixtures__/chunk-manifest.prod.json
+++ b/scripts/__fixtures__/chunk-manifest.prod.json
@@ -23,9 +23,9 @@
],
"imports": [
"src/agent/runner/sequence/orchestrator/queue-tools.ts",
+ "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
+ "src/store/auth-session-state.ts"
]
},
"src/agent/agent-prompt-loader.ts": {
@@ -78,10 +78,10 @@
"src/agent/agent-prompt-loader.ts",
"src/agent/aio-capture.ts",
"src/agent/runner/sequence/orchestrator/queue-tools.ts",
+ "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
- "src/store/programs/ai-opt-in-gate.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
+ "src/store/programs/ai-opt-in-gate.ts"
]
},
"src/agent/aio-capture.ts": {
@@ -98,9 +98,8 @@
"src/agent/runner/harness/pi/task.ts",
"src/agent/runner/harness/pi/tasks.ts",
"src/agent/runner/harness/pi/tools.ts",
- "src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
+ "src/store/agent-protocol/agent-env-isolation.ts",
+ "src/store/agent-protocol/agent-signals.ts"
]
},
"src/agent/runner/harness/pi/mcp.ts": {
@@ -145,9 +144,8 @@
"src/agent/runner/harness/pi/security.ts",
"src/agent/runner/harness/pi/tools.ts",
"src/agent/runner/sequence/orchestrator/queue-tools.ts",
- "src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
+ "src/store/agent-protocol/agent-env-isolation.ts",
+ "src/store/agent-protocol/agent-signals.ts"
]
},
"src/agent/runner/harness/pi/tasks.ts": {
@@ -162,9 +160,8 @@
],
"imports": [
"src/agent/aio-capture.ts",
- "src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
+ "src/store/agent-protocol/agent-env-isolation.ts",
+ "src/store/auth-session-state.ts"
]
},
"src/agent/runner/sequence/orchestrator/queue-tools.ts": {
@@ -219,10 +216,11 @@
"src/cli/commands/basic-integration/non-interactive.ts",
"src/cli/commands/basic-integration/playground.ts",
"src/cli/control-hooks.ts",
+ "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts",
- "src/tui/App.tsx"
+ "src/tui/App.tsx",
+ "src/tui/console/headless-ui.ts"
]
},
"src/cli/commands/basic-integration/ci-install.ts": {
@@ -232,7 +230,8 @@
"imports": [
"src/cli/control-hooks.ts",
"src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts"
+ "src/store/auth-session-state.ts",
+ "src/tui/console/headless-ui.ts"
]
},
"src/cli/commands/basic-integration/interactive.ts": {
@@ -271,12 +270,50 @@
],
"imports": [
"src/agent/agent-prompt.ts",
+ "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
"src/store/control/actions.ts",
"src/store/programs/ai-opt-in-gate.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts",
- "src/tui/App.tsx"
+ "src/tui/App.tsx",
+ "src/tui/console/headless-ui.ts"
+ ]
+ },
+ "src/store/agent-protocol/agent-env-isolation.ts": {
+ "sources": [
+ "src/store/agent-protocol/agent-env-isolation.ts",
+ "src/store/agent-protocol/agent-phase.ts",
+ "src/store/agent-protocol/mcp-prompt.ts",
+ "src/store/agent-protocol/token-pricing.ts",
+ "src/store/file-watcher.ts",
+ "src/store/mcp-project-profile.ts",
+ "src/store/mcp-role-prompts.copy.json",
+ "src/store/mcp-role-prompts.ts",
+ "src/store/mcp-seed-events.ts",
+ "src/store/safe-tools.ts",
+ "src/store/services/claude-settings.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/browser-client.ts",
+ "src/store/services/steps/add-mcp-server-to-clients/login-client.ts",
+ "src/store/session/secret-vault.ts",
+ "src/store/shared/clipboard.ts",
+ "src/store/shared/custom-headers.ts",
+ "src/store/shared/env-api-key.ts",
+ "src/store/shared/environment.ts",
+ "src/store/shared/terminal-bell.ts",
+ "src/store/state/store.ts",
+ "src/store/task-stream/audit-areas.ts",
+ "src/store/task-stream/destinations/file.ts",
+ "src/store/task-stream/destinations/posthog.ts",
+ "src/store/task-stream/event-plan-watcher.ts",
+ "src/store/task-stream/task-stream-push.ts",
+ "src/store/tools/handoff.ts",
+ "src/store/ui/store-ui.ts",
+ "src/store/ui/wizard-ui.ts",
+ "src/store/wizard-spellbook.ts"
+ ],
+ "imports": [
+ "src/store/agent-protocol/agent-signals.ts",
+ "src/store/auth-session-state.ts"
]
},
"src/store/agent-protocol/agent-signals.ts": {
@@ -467,9 +504,9 @@
"src/store/programs/run-config.ts"
],
"imports": [
+ "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
- "src/store/auth-session-state.ts",
- "src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
+ "src/store/auth-session-state.ts"
]
},
"src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts": {
@@ -684,10 +721,20 @@
"src/tui/ui-store.ts"
],
"imports": [
+ "src/store/agent-protocol/agent-env-isolation.ts",
"src/store/agent-protocol/agent-signals.ts",
"src/store/auth-session-state.ts",
"src/store/programs/ai-opt-in-gate.ts",
"src/store/services/steps/add-mcp-server-to-clients/MCPClient.ts"
]
+ },
+ "src/tui/console/headless-ui.ts": {
+ "sources": [
+ "src/tui/console/headless-ui.ts",
+ "src/tui/console/logging-ui.ts"
+ ],
+ "imports": [
+ "src/store/auth-session-state.ts"
+ ]
}
}