Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -97,8 +97,8 @@ aliases.

| Subcommand | What it audits |
| ----------------------------- | ---------------------------------------------------- |
| `wizard audit events` | event capture quality + cost (**default** leaf) |
| `wizard audit all` | comprehensive audit across every area |
| `wizard audit events` | event capture quality + cost |
| `wizard audit all` | comprehensive audit across every area (**default**) |
| `wizard audit autocapture` | autocapture setup + cost |
| `wizard audit feature-flags` | feature flag usage + cost |
| `wizard audit identify` | `$identify` implementation |
Expand Down
9 changes: 9 additions & 0 deletions docs/local-dev.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,10 +76,19 @@ These flags are available in dev/test builds. Published builds reject them.
| `--local-context-mill` | `POSTHOG_WIZARD_LOCAL_CONTEXT_MILL` | skills → `:8765` |
| `--local-mcp` | `POSTHOG_WIZARD_LOCAL_MCP` | MCP → `:8787` |
| `--local-posthog` | `POSTHOG_WIZARD_LOCAL_POSTHOG` | PostHog origins → `:8010` |
| `--task-stream-log[=path]` | `POSTHOG_WIZARD_TASK_STREAM_LOG` | dump every attempted task-stream sync as JSONL (default `/tmp/posthog-wizard-task-stream.jsonl`) |

`--local-posthog` is sugar over `--base-url`. It pins the API host, app host,
OAuth server, and the LLM gateway derived from them.

`--task-stream-log` records what the run published, one JSON line per push,
truncated per run. It rides beside the PostHog destination rather than
replacing it, so a logged run is the same run the backend sees. A line means
the payload was attempted, not accepted — `[task-stream] wizard/sessions push
ok: 201` in the debug log is the delivery signal. `--ci` dumps to the default
path on every run and never pushes, since a synthetic run would otherwise
create a session row in a real project.

### Precedence

Most specific wins:
Expand Down
13 changes: 10 additions & 3 deletions e2e-harness/action-registry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -67,19 +67,18 @@ function requireString(
* - the runner or agent advances them: auth (runner sets credentials), run
* (agent sets runPhase), ai-opt-in (org approval / ci auto-consent), exit,
* and the no-dismiss terminal overlays.
* - screens of programs the integration e2e profile never enters (audit,
* doctor).
* - screens of programs the integration e2e profile never enters (doctor).
*/
export const NO_ACTION_SCREENS: ReadonlySet<ScreenName> = new Set<ScreenName>([
ScreenId.Auth,
ScreenId.Run,
ScreenId.AiOptIn,
ScreenId.Exit,
// The agent advances the audit run, the same way it advances `run`.
ScreenId.AuditRun,
ScreenId.DoctorReport,
// The detector + picker are interactive; no headless e2e drives this screen.
ScreenId.SelfDrivingIntegrationDetect,
ScreenId.AuditOutro,
ScreenId.SelfDrivingIntegrationCheck,
ScreenId.SelfDrivingIntegrationDetect,
ScreenId.SelfDrivingHandoff,
Expand Down Expand Up @@ -210,6 +209,14 @@ export const ACTION_REGISTRY: Partial<Record<ScreenName, DriverAction[]>> = {
apply: (store) => store.setOutroDismissed(),
},
],
[ScreenId.AuditOutro]: [
{
id: 'dismiss_outro',
description:
'Dismiss the audit outro, which carries the report, dashboard, and notebook links.',
apply: (store) => store.setOutroDismissed(),
},
],
[ScreenId.MintFailure]: [
{
id: 'continue_setup',
Expand Down
1 change: 1 addition & 0 deletions e2e-harness/e2e-profile.ts
Original file line number Diff line number Diff line change
Expand Up @@ -321,6 +321,7 @@ export function decideE2eAction(

case ScreenId.Outro:
case ScreenId.SourceMapsOutro:
case ScreenId.AuditOutro:
return { action: { id: 'dismiss_outro' } };

case ScreenId.Mcp:
Expand Down
3 changes: 3 additions & 0 deletions e2e-harness/profiles.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,7 @@ import selfDrivingE2e from '@lib/programs/self-driving/test/e2e.json';
import sourceMapsE2e from '@lib/programs/error-tracking-upload-source-maps/test/e2e.json';
import errorTrackingE2e from '@lib/programs/error-tracking/test/e2e.json';
import warehouseSourceE2e from '@lib/programs/warehouse-source/test/e2e.json';
import auditE2e from '@lib/programs/audit/test/e2e.json';

const PROFILES: Partial<Record<ProgramId, WizardE2eProfile>> = {
[Program.PostHogIntegration]:
Expand All @@ -39,6 +40,7 @@ const PROFILES: Partial<Record<ProgramId, WizardE2eProfile>> = {
sourceMapsE2e.profile as WizardE2eProfile,
[Program.ErrorTracking]: errorTrackingE2e.profile as WizardE2eProfile,
[Program.WarehouseSource]: warehouseSourceE2e.profile as WizardE2eProfile,
[Program.Audit]: auditE2e.profile as WizardE2eProfile,
};

const VARIATIONS: Partial<Record<ProgramId, WizardE2eVariation[]>> = {
Expand All @@ -51,6 +53,7 @@ const VARIATIONS: Partial<Record<ProgramId, WizardE2eVariation[]>> = {
[Program.ErrorTracking]: errorTrackingE2e.variations as WizardE2eVariation[],
[Program.WarehouseSource]:
warehouseSourceE2e.variations as WizardE2eVariation[],
[Program.Audit]: auditE2e.variations as WizardE2eVariation[],
};

/** The e2e profile for a program, or the happy-path default if none is set. */
Expand Down
38 changes: 34 additions & 4 deletions scripts/tui-host.no-jest.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,8 @@ import { buildSession } from '@lib/wizard-session';
import { initLocalDev } from '@lib/local-dev';
import { configureGatewayFromCIEnvironment } from '@lib/gateway-session';
import { runAgent } from '@lib/agent/agent-runner';
import { TaskStreamPush, createFileDestination } from '@lib/task-stream/index';
import { getAuditChecks } from '@lib/programs/audit/types';
import { authenticate } from '@lib/agent/runner/shared/authenticate';
import { getOrAskForProjectData } from '@utils/setup-utils';
import { logToFile } from '@utils/debug';
Expand Down Expand Up @@ -55,6 +57,16 @@ import {
readReportFile,
} from '@e2e-harness/e2e-result';

/** Cheap 32-bit FNV-1a, to fold framework-context values into a signature. */
function digest(s: string): string {
let h = 0x811c9dc5;
for (let i = 0; i < s.length; i++) {
h ^= s.charCodeAt(i);
h = Math.imul(h, 0x01000193);
}
return (h >>> 0).toString(36);
}

const sleep = (ms: number) => new Promise((r) => setTimeout(r, ms));
const mark = (m: string) => logToFile(`[tui-host] ${m}`);

Expand Down Expand Up @@ -233,6 +245,25 @@ async function main() {
sequence: (process.env.SNAP_SEQUENCE || undefined) as Sequence | undefined,
model: process.env.SNAP_MODEL || undefined,
});
// Dumped, never pushed: an e2e run is synthetic, like `--ci`.
const streamLog = createFileDestination(process.env.TASK_STREAM_LOG ?? '');
if (streamLog) {
const stream = new TaskStreamPush({
store,
programId,
destinations: [streamLog],
eventPlanPath: programConfig.eventPlanFile
? join(store.session.installDir, programConfig.eventPlanFile)
: undefined,
auditChecks: programConfig.auditLedgerFile
? () => getAuditChecks(store.session)
: undefined,
});
stream.attach();
process.on('exit', () => void stream.shutdown(0));
mark(`task stream dump → ${streamLog.path}`);
}

// Optional skip-ahead: pre-resolve the self-driving integration check so its
// screen never shows (INTEGRATE=true integrates first; false = already set up).
if (process.env.INTEGRATE === 'true' || process.env.INTEGRATE === 'false') {
Expand Down Expand Up @@ -429,10 +460,9 @@ async function main() {
overlay: store.router.hasOverlay,
tasks: store.tasks.map((t) => [t.label, t.status, t.done]),
phase: store.session.runPhase,
// Snap on within-screen state too: when a screen publishes new
// framework-context (e.g. the detector's projects), so the picker frame
// is captured, not just the loading state. Generic — keys, not values.
ctx: Object.keys(store.session.frameworkContext).sort().join(','),
// Values, not just keys: a screen rerendering from an artifact updated
// in place (the audit ledger) keeps its key and would snap once, empty.
ctx: digest(JSON.stringify(store.session.frameworkContext)),
});
const snap = (): Promise<void> => {
const sig = signature();
Expand Down
32 changes: 25 additions & 7 deletions src/__tests__/cli.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,15 +10,22 @@ const { mockBuildSessionCli, mockProvisionNewAccountCli } = vi.hoisted(() => ({
// Headless-only machinery, stubbed so the headless path doesn't construct a
// real WizardStore (which would re-call the mocked buildSession) or open a real
// network stream. The spies assert the stream is wired in headless and not CI.
const { mockStreamAttach, mockStreamShutdown } = vi.hoisted(() => ({
mockStreamAttach: vi.fn(),
mockStreamShutdown: vi.fn(),
}));
const { mockStreamAttach, mockStreamShutdown, mockStreamDestinations } =
vi.hoisted(() => ({
mockStreamAttach: vi.fn(),
mockStreamShutdown: vi.fn(),
// Which destinations each run wired up, by name. The CI contract is about
// destinations, not about whether a stream exists.
mockStreamDestinations: vi.fn(),
}));
vi.mock('../lib/task-stream/index', () => ({
// shutdown() hardcodes a resolved Promise (not a bare vi.fn) so the
// interactive runWizard's dangling SIGTERM handler — which calls
// shutdown().catch() and outlives these tests — never hits undefined.catch.
TaskStreamPush: class {
constructor(opts: { destinations: Array<{ name: string }> }) {
mockStreamDestinations(opts.destinations.map((d) => d.name));
}
attach() {
mockStreamAttach();
}
Expand All @@ -27,7 +34,13 @@ vi.mock('../lib/task-stream/index', () => ({
return Promise.resolve();
}
},
PostHogDestination: class {},
PostHogDestination: class {
readonly name = 'posthog';
},
createFileDestination: (value: unknown) =>
value === undefined || value === null || value === false
? null
: { name: 'file', path: '/tmp/task-stream.jsonl' },
}));
vi.mock('../ui/tui/store', async (importOriginal) => ({
...(await importOriginal<typeof import('../ui/tui/store')>()),
Expand Down Expand Up @@ -435,7 +448,9 @@ describe('CLI argument parsing', () => {
expect(analytics.setTag).toHaveBeenCalledWith('build', 'ci');
});

test('does not stream wizard-session state in CI', async () => {
// CI dumps the stream to a local file and never pushes: a CI run is
// synthetic, so a push would create a session row in a real project.
test('dumps the wizard-session stream locally and never pushes in CI', async () => {
await runCLI([
'--ci',
'--api-key',
Expand All @@ -444,7 +459,8 @@ describe('CLI argument parsing', () => {
'/tmp/test',
]);

expect(mockStreamAttach).not.toHaveBeenCalled();
expect(mockStreamAttach).toHaveBeenCalled();
expect(mockStreamDestinations).toHaveBeenCalledWith(['file']);
});

// The CI bot authenticates with a wizard-app pha_ token, the same
Expand Down Expand Up @@ -544,6 +560,8 @@ describe('CLI argument parsing', () => {

expect(mockStreamAttach).toHaveBeenCalled();
expect(mockStreamShutdown).toHaveBeenCalled();
// Headless is the surface the web app watches, so it pushes.
expect(mockStreamDestinations).toHaveBeenCalledWith(['posthog']);
});

test('does not require --region when headless is set', async () => {
Expand Down
19 changes: 19 additions & 0 deletions src/__tests__/programs-cli.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -127,6 +127,25 @@
expect(opts).toMatchObject({ debug: true });
});

test('an audit leaf publishes under the family, not under agent-skill', async () => {
// A leaf runs on the generic skill program, whose id is `agent-skill`, so
// without the override every audit would share one indistinguishable
// channel with every other `wizard skill` run. skill_id discriminates.
mockMenu([
entry({
skillId: 'audit-events',
command: 'events',
parentCommand: 'audit',
}),
]);
await dispatchFamily('audit', makeArgv({ skill: 'events' }));
const [config] = mockRunWizard.mock.calls[0] as [
{ id?: string; streamWorkflowId?: string },
];
expect(config.id).toBe('agent-skill');
expect(config.streamWorkflowId).toBe('audit');
});

test('routes through runWizardCI when --ci is set', async () => {
mockMenu([
entry({
Expand Down Expand Up @@ -169,7 +188,7 @@
});

test('migrate dispatches with migrate-statsig skillId', () => {
migrateCommand.handler!(makeArgv({ installDir: '/tmp/some-app' }));

Check warning on line 191 in src/__tests__/programs-cli.test.ts

View workflow job for this annotation

GitHub Actions / Lint

Forbidden non-null assertion
const [config, opts] = mockRunWizard.mock.calls[0] as [
{ skillId?: string },
Record<string, unknown>,
Expand All @@ -179,13 +198,13 @@
});

test('revenue-analytics dispatches with revenue-analytics-setup skillId', () => {
revenueCommand.handler!(makeArgv({ debug: true }));

Check warning on line 201 in src/__tests__/programs-cli.test.ts

View workflow job for this annotation

GitHub Actions / Lint

Forbidden non-null assertion
const [config] = mockRunWizard.mock.calls[0] as [{ skillId?: string }];
expect(config.skillId).toBe('revenue-analytics-setup');
});

test('mcp-analytics dispatches with mcp-analytics skillId', () => {
mcpAnalyticsCommand.handler!(makeArgv({ debug: true }));

Check warning on line 207 in src/__tests__/programs-cli.test.ts

View workflow job for this annotation

GitHub Actions / Lint

Forbidden non-null assertion
const [config] = mockRunWizard.mock.calls[0] as [{ skillId?: string }];
expect(config.skillId).toBe('mcp-analytics');
});
Expand Down
14 changes: 14 additions & 0 deletions src/lib/agent/runner/harness/pi/__tests__/task-tools.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -115,3 +115,17 @@ describe('fenceDisallowList', () => {
expect(fenceDisallowList(undefined)).toEqual([]);
});
});

describe('audit ledger tools on pi', () => {
it('grants them only to a task that asked, under either name form', () => {
expect(allowedPiWizardTools(['Read']).has('audit_resolve_checks')).toBe(
false,
);
const granted = allowedPiWizardTools([
'mcp__wizard-tools__audit_seed_checks',
'audit_add_checks',
'mcp__wizard-tools__audit_resolve_checks',
]);
expect([...granted].filter((t) => t.startsWith('audit_'))).toHaveLength(3);
});
});
47 changes: 47 additions & 0 deletions src/lib/agent/runner/harness/pi/__tests__/tools.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
*/
import { mkdtempSync } from 'node:fs';
import { mkdir, readFile, writeFile } from 'node:fs/promises';
import { readFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { describe, it, expect, vi } from 'vitest';
Expand Down Expand Up @@ -657,3 +658,49 @@ describe('pi task tool grant — the names the inventory shows', () => {
]);
});
});

/** pi mounts no MCP server, so these native tools are the only ledger writer. */
describe('audit ledger tools', () => {
it('seeds, resolves by id, and appends, at the path the watcher reads', async () => {
const workingDirectory = mkdtempSync(join(tmpdir(), 'pi-audit-ledger-'));
const tools = createWizardPiTools({
workingDirectory,
skillsBaseUrl: 'http://localhost:0',
triageProvider: undefined,
});
const tool = (name: string) => {
const found = tools.find((t) => t.name === name);
if (!found) throw new Error(`${name} not registered`);
return found;
};
const ledger = () =>
JSON.parse(
readFileSync(
join(workingDirectory, '.posthog-audit-checks.json'),
'utf8',
),
) as Array<{ id: string; status: string }>;
const check = (id: string) => ({
id,
area: 'Installation',
label: id,
status: 'pending' as const,
});

await tool('audit_seed_checks').execute('1', {
checks: [check('sdk-installed'), check('init-correct')],
} as never);
await tool('audit_resolve_checks').execute('2', {
updates: [{ id: 'sdk-installed', status: 'pass' }],
} as never);
await tool('audit_add_checks').execute('3', {
checks: [check('live-data-source-maps')],
} as never);

expect(ledger().map((c) => [c.id, c.status])).toEqual([
['sdk-installed', 'pass'],
['init-correct', 'pending'],
['live-data-source-maps', 'pending'],
]);
});
});
10 changes: 9 additions & 1 deletion src/lib/agent/runner/harness/pi/task.ts
Original file line number Diff line number Diff line change
Expand Up @@ -107,7 +107,15 @@ const ALWAYS_ON_WIZARD_TOOLS = [
'publish_handoff',
];

const OPT_IN_WIZARD_TOOLS = ['wizard_ask', 'load_skill_menu', 'install_skill'];
const OPT_IN_WIZARD_TOOLS = [
'wizard_ask',
'load_skill_menu',
'install_skill',
// Audit programs only: they declare the ledger tools on `allowedTools`.
'audit_seed_checks',
'audit_add_checks',
'audit_resolve_checks',
];

export function allowedPiWizardTools(
allowedTools: readonly string[] | undefined,
Expand Down
Loading
Loading