Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
133 changes: 133 additions & 0 deletions .agents/skills/senpi-qa/scripts/gpt-6-family-preset-mock-loop.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,133 @@
/**
* Channel 3 proof: GPT-6 Sol and GPT-6 Luna reach the wire on the GPT-6 preset.
*
* Boots the real senpi CLI in --print mode against a fake OpenAI Responses
* server that serves the built-in catalog rows for `gpt-6-sol` and `gpt-6-luna`,
* then inspects each captured request: the model id and the requested reasoning
* effort must be the ones the CLI was asked for, and the developer/system message
* must carry the GPT-6 core (its "Asynchronous Work" and "Instructions From Files"
* sections) and none of the GPT-5.6 core's own sections.
*
* Runs under bun (`bun gpt-6-family-preset-mock-loop.mjs`): the CLI is spawned
* from TypeScript source with the current runtime, so no tsx shim is needed.
*/
import { spawn } from "node:child_process";
import { writeFileSync } from "node:fs";
import { join } from "node:path";
import { cliEntry, evidenceDir, guardRealAuth, installCleanupHooks, makeSandbox, repoRoot, track } from "./lib/common.mjs";
import { startFakeModelServer } from "./lib/fake-model-server.mjs";
import { hermeticEnv, writeMockModelsJson } from "./lib/mock-loop-support.mjs";

const EVIDENCE_SLUG = "gpt-6-family-preset-mock-loop";
const FINAL_MARKER = "SENPI-QA-GPT6-FAMILY-PRESET-APPLIED-7d2a";
const GPT6_ONLY_SECTIONS = ["## Asynchronous Work", "## Instructions From Files"];
const GPT56_ONLY_SECTIONS = ["## Manual QA Gate", "## Pragmatism & Scope"];
const SCENARIOS = [
{ id: "gpt-6-sol", name: "GPT-6 Sol", thinking: "medium", expectedEffort: "medium" },
{ id: "gpt-6-luna", name: "GPT-6 Luna", thinking: "off", expectedEffort: "none" },
];

const checks = [];
function check(label, condition) {
checks.push({ label, pass: !!condition });
console.log(`[${condition ? "PASS" : "FAIL"}] ${label}`);
}

function runCliWithCurrentRuntime(args, { env, cwd, timeoutMs }) {
return new Promise((resolve) => {
const root = repoRoot();
const child = track(spawn(process.execPath, [cliEntry(root), ...args], { cwd, env, stdio: ["pipe", "pipe", "pipe"] }));
let stdout = "";
let stderr = "";
child.stdout.on("data", (chunk) => {
stdout += chunk.toString();
});
child.stderr.on("data", (chunk) => {
stderr += chunk.toString();
});
const timer = setTimeout(() => {
child.kill("SIGKILL");
resolve({ code: null, stdout, stderr, timedOut: true });
}, timeoutMs);
child.on("close", (code) => {
clearTimeout(timer);
resolve({ code, stdout, stderr, timedOut: false });
});
child.stdin.end();
});
}

function systemTextOf(body) {
if (typeof body?.instructions === "string") return body.instructions;
const input = Array.isArray(body?.input) ? body.input : [];
const systemItems = input.filter((item) => item?.role === "developer" || item?.role === "system");
return systemItems
.map((item) => (typeof item.content === "string" ? item.content : (item.content ?? []).map((part) => part?.text ?? "").join("\n")))
.join("\n");
}

async function runScenario(scenario, evidence, authGuard) {
const sandbox = makeSandbox(`senpi-qa-${scenario.id}`);
const env = hermeticEnv(sandbox.env);
const server = await startFakeModelServer({ turns: [{ text: FINAL_MARKER }] });
try {
writeMockModelsJson(sandbox.agentDir, server, "openai-responses", { id: scenario.id, name: scenario.name, reasoning: true });

const result = await runCliWithCurrentRuntime(
["--print", "--provider", "openai", "--model", scenario.id, "--thinking", scenario.thinking, "Say hello"],
{ env, cwd: sandbox.cwd, timeoutMs: 60000 },
);

check(`${scenario.id}: --print exits 0`, result.code === 0);
check(`${scenario.id}: --print output contains the final marker`, result.stdout.includes(FINAL_MARKER));
check(`${scenario.id}: fake server captured exactly 1 request`, server.requests.length === 1);

const request = server.requests[0];
if (request) {
check(`${scenario.id}: request names ${scenario.id}`, request.body?.model === scenario.id);
check(
`${scenario.id}: request carries reasoning.effort=${scenario.expectedEffort}`,
request.body?.reasoning?.effort === scenario.expectedEffort,
);
const systemText = systemTextOf(request.body);
check(`${scenario.id}: request carries a developer/system message`, systemText.length > 0);
for (const section of GPT6_ONLY_SECTIONS) {
check(`${scenario.id}: system prompt carries the GPT-6 section ${section}`, systemText.includes(section));
}
for (const section of GPT56_ONLY_SECTIONS) {
check(`${scenario.id}: system prompt does NOT carry the GPT-5.6 section ${section}`, !systemText.includes(section));
}
check(`${scenario.id}: system prompt does not name Astra`, !/\bAstra\b/.test(systemText));
writeFileSync(join(evidence, `${scenario.id}-system-prompt.txt`), systemText);
writeFileSync(join(evidence, `${scenario.id}-request.json`), JSON.stringify(request.body, null, 2));
}

writeFileSync(
join(evidence, `${scenario.id}-stdout.txt`),
`exit=${result.code}\n---STDOUT---\n${result.stdout}\n---STDERR---\n${result.stderr}\n`,
);
check(`${scenario.id}: real auth store unchanged`, authGuard.assertUnchanged());
} finally {
await server.stop();
sandbox.cleanup();
}
}

async function main() {
const evidence = evidenceDir(EVIDENCE_SLUG);
installCleanupHooks();
const authGuard = guardRealAuth();

for (const scenario of SCENARIOS) {
await runScenario(scenario, evidence, authGuard);
}

const passed = checks.filter((entry) => entry.pass).length;
console.log(`\n${EVIDENCE_SLUG}: ${passed}/${checks.length} passed (evidence: ${evidence})`);
process.exitCode = passed === checks.length ? 0 : 1;
}

main().catch((error) => {
console.error(error);
process.exitCode = 1;
});
4 changes: 4 additions & 0 deletions packages/ai/CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -6,10 +6,14 @@

### Added

- GPT-6 Sol (`gpt-6-sol`) and GPT-6 Luna (`gpt-6-luna`) join the catalog on OpenAI, ChatGPT Subscription, Azure OpenAI, OpenCode Zen, OpenRouter, Venice and Vercel AI Gateway, with `-fast` Priority-tier variants on OpenAI and ChatGPT Subscription. Both carry their published prices (Sol \$2/\$10 per 1M tokens with \$0.20 cache reads, Luna \$0.10/\$0.50 with \$0.01 cache reads, both doubling input and 1.5x output past 272k), 128k output, text and image input, tool search and additional-tools support, and the documented effort ladder `none`/`low`/`medium`/`high`/`xhigh`/`max`. Project prompt budgets: Luna ships the full 922k input cap, Sol ships 400k, on every provider that lists the model.

### Changed

### Fixed

- OpenRouter's passthrough rows for `openai/gpt-6-sol` and `openai/gpt-6-luna` (and their `-pro` / `:batch` siblings) shipped in 2026.9.22-4 with no effort ladder at all, so `xhigh` and `max` were not selectable there and the rows sat at the raw 922k window instead of the tier budget. They now carry the same GPT-6 ladder and budget as the first-party rows.

### Removed

## [2026.9.22-4] - 2026-09-22
Expand Down
21 changes: 21 additions & 0 deletions packages/ai/changes.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,24 @@
## 2026-09-23 - GPT-6 Sol and GPT-6 Luna catalog rows

### What changed

- `packages/ai/scripts/generate-models.ts`: `gpt-6-sol` and `gpt-6-luna` join every OpenAI id table that already carried `gpt-6-astra` (tool search + additional tools on `openai` and `chatgpt-subscription`, short-context cap, long-context pricing tiers, `none` reasoning on the direct provider, Priority `-fast` variants on both first-party providers). Hand-added rows for both tiers land in the `openai` fallback list (used only when models.dev lacks the id) and in the `chatgpt-subscription` list (models.dev never carries the Codex backend), priced from the new `OPENAI_GPT_6_STANDARD_COSTS` table (Sol 2/10/0.2/2.5, Luna 0.1/0.5/0.01/0.125 per MTok) through `withOpenAiLongContextPricing`; `OPENAI_STANDARD_COSTS` merges the 5.6 and 6 tables for the direct-provider and Cloudflare price overrides. `supportsOpenAiXhigh` / `supportsOpenAiMax` and the former Astra-only final pass now key on `isGpt6FamilyId` (astra | sol | luna markers, prefixed or suffixed ids): `applyGpt6ContextWindow` stamps the tier budget on every provider row (Astra 600,000 unchanged, Sol 400,000, Luna 922,000 = the documented input cap) and `applyGpt6ThinkingLevels` stamps the documented ladder (`minimal: null`, low..max) and forces `off: null` for Astra only, since Sol and Luna document `none`.
- `packages/ai/src/providers/data/` regenerated with `--strict`: +4 rows each on `openai.json` and `chatgpt-subscription.json` (base + `-fast`), +2 on `azure-openai-responses.json`, `opencode.json` (plus incidental upstream additions `claude-opus-5-5`, `grok-4.7`), `venice.json`, +4 on `vercel-ai-gateway.json`; the eight pre-existing `openrouter.json` GPT-6 Sol/Luna rows gain the ladder and the Sol budget. Incidental upstream drift: OpenRouter pricing/context refreshes (aion, deepseek, hy3, glm, `~latest` aliases), Vercel gemini metadata. No model id was removed.
- Tests: `test/gpt-6-family-catalog.test.ts` (first-party rows, pricing tiers, ladder incl. `off`, tool metadata, `-fast` variants, map-less id inference, family-wide budget across every catalog); `test/openai-input-cap-catalog.test.ts` exempts the deliberate `gpt-6-sol @ 400,000` pairing from the documented-total check (a 1,050,000 Sol row would still fail) and pins the new direct-provider defaults.

### Why

OpenAI's GPT-6 family is Astra, Sol and Luna (developers.openai.com/api/docs/guides/latest-model). 2026.9.22-4 shipped Astra rows only; models.dev had since added Sol/Luna rows for openai, opencode, azure, openrouter and vercel, and 2026.9.22-4's regeneration had already pulled the OpenRouter passthrough rows without any effort ladder, so `xhigh` / `max` were unreachable there and the Codex backend could not select either tier at all. Budgets follow the user-stated defaults (Luna full window, Sol 400k) rather than the 272k short-context tier.

### Why an extension could not handle it

The catalog shards ship inside this package; nothing loaded at runtime can add a first-party row or change what the generator wrote.

### Expected merge conflict zones

- `packages/ai/scripts/generate-models.ts`: the OpenAI id tables near the top, `supportsOpenAiXhigh` / `supportsOpenAiMax`, the hand-added `openai` and `chatgpt-subscription` row lists, and the final metadata pass.
- `packages/ai/src/providers/data/*.json` + `.manifest.json`: regenerate rather than merge.

## 2026-09-22 - Claude Opus 5.5 catalog rows and request compat

### What changed
Expand Down
Loading
Loading