From 1229c28346ded9ec9d002ed2f6183094be1e2616 Mon Sep 17 00:00:00 2001 From: Sergei Tarassov Date: Sat, 5 Sep 2026 18:51:35 +0300 Subject: [PATCH] test: pin the translation layer with an offline suite The CI gate was a parse check standing in for tests. Replace it with 82 node:test cases over the part of the bridge that can actually be wrong: the two translation directions and the SSE rebuild. The suite never opens a socket or calls the upstream API. Streaming is driven by a recorded Responses stream in test/fixtures/, re-fed at chunk sizes 1, 7, 64 and 997 to prove that events spanning a socket boundary reassemble identically. What the tests hold in place, beyond the happy path: - tool_result leaves the user turn and becomes a top-level function_call_output; call ids stay paired across a full tool round - an empty errored tool_result still sends "error" -- Responses rejects an empty output string - the encrypted reasoning blob survives a round trip through a thinking block's signature, and the rebuilt reasoning item is hoisted ahead of the call it produced - a foreign thinking block or a corrupt signature decodes to nothing instead of throwing - reasoning.effort and include are sent only to gpt-5.6-*; the cheap fallback 400s on them - max_tokens: 1 is floored to 16; temperature and top_p are dropped - an unmapped model lands on the fallback rather than a 400 - truncated tool arguments degrade to {} instead of crashing the response - a block left open by a dropped upstream connection is still closed, so the client cannot hang Structural change, needed to test any of it without a live server: Options considered - spawn the proxy on a port and drive it over HTTP: covers the same code but adds sockets, an upstream stub and flakiness to every case - move the translators into a second module: clean, but the package's one claim is that it is a single file Picked: export the pure functions from codex-proxy.ts and start the server only when the file is the process entry point. Importing it is now side-effect free -- no port, no exit on a missing key. A child-process test pins that contract, and a smoke run confirms `npm start` still exits 1 with FATAL without a key and still prints the ready line with one. Two conversions were inlined in the request handler and had to come out to be reachable: toAnthropicMessage for the non-streaming response, and createStreamTranslator, which takes raw SSE text and emits events through a callback so the socket and the fixtures drive the same code. Neither changes behaviour. ci: run npm test instead of `node --check`. The job name stays `test`. --- .github/workflows/ci.yml | 8 +- README.md | 15 +- codex-proxy.ts | 559 +++++++++++++++---------- package.json | 3 +- test/fixtures/import-probe.ts | 9 + test/fixtures/responses-stream.sse.txt | 35 ++ test/server-shape.test.ts | 107 +++++ test/streaming.test.ts | 198 +++++++++ test/translate-request.test.ts | 383 +++++++++++++++++ test/translate-response.test.ts | 188 +++++++++ 10 files changed, 1267 insertions(+), 238 deletions(-) create mode 100644 test/fixtures/import-probe.ts create mode 100644 test/fixtures/responses-stream.sse.txt create mode 100644 test/server-shape.test.ts create mode 100644 test/streaming.test.ts create mode 100644 test/translate-request.test.ts create mode 100644 test/translate-response.test.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a58f4be..04023d0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -17,7 +17,7 @@ jobs: - uses: actions/setup-node@v4 with: node-version: '22' - # There is no test suite yet. Importing the module would start the proxy, - # so the gate is a parse-and-strip check: it catches broken syntax and - # type-stripping errors without binding a port. - - run: node --experimental-strip-types --check codex-proxy.ts + # `npm test` runs node:test over the translation layer. There is nothing + # to install -- the bridge has no dependencies and no build step, and the + # suite never opens a socket or calls the upstream API. + - run: npm test diff --git a/README.md b/README.md index 1cfb96e..6dc9d85 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ Responses API** — run Claude Code (or any Anthropic-API client) on `gpt-5.6-*` models while keeping the client's own tools, skills and MCP servers. -461 lines, zero dependencies, Node ≥ 22, no build step: +One file, zero dependencies, Node ≥ 22, no build step: ```sh OPENAI_API_KEY=sk-... PORT=4001 ./codex-proxy.ts @@ -58,6 +58,19 @@ upstream. Disable with `CODEX_CARRY_REASONING=0`. Model aliases: `codex-luna` → `gpt-5.6-luna`, `codex-sol` → `gpt-5.6-sol`, `codex-terra` → `gpt-5.6-terra`; any `gpt-*` name passes through unchanged. +## Tests + +```sh +npm test +``` + +The translation layer is exported from `codex-proxy.ts`, and the server only +starts when the file is the process entry point — so `node --test` can pin +every conversion without opening a socket or calling the upstream API. The +streaming tests replay a recorded Responses SSE stream from +`test/fixtures/`, re-fed at several chunk sizes to prove events that span a +socket boundary are reassembled. + ## Non-goals Auth beyond a bearer key, retries, rate limiting, multi-tenant use. It is a diff --git a/codex-proxy.ts b/codex-proxy.ts index b7612cd..2d6acd1 100755 --- a/codex-proxy.ts +++ b/codex-proxy.ts @@ -13,45 +13,64 @@ * * PORT=0 picks a free port and prints "codex-proxy ready port=" on stderr; * codex-agent parses that line. + * + * Everything above the server is a pure translation layer and is exported, so + * the test suite can pin it without opening a socket. The server only starts + * when this file is the process entry point -- importing it is side-effect + * free. */ import http from 'node:http' +import { pathToFileURL } from 'node:url' const PORT = Number(process.env.PORT ?? 4001) -const OPENAI_KEY = process.env.OPENAI_API_KEY const UPSTREAM = process.env.CODEX_UPSTREAM ?? 'https://api.openai.com/v1/responses' -const EFFORT = process.env.CODEX_EFFORT ?? 'medium' -/** Carry reasoning across tool rounds via encrypted_content (phase 2). */ -const CARRY_REASONING = process.env.CODEX_CARRY_REASONING !== '0' -const MODEL_MAP: Record = { +export const MODEL_MAP: Record = { 'codex-luna': 'gpt-5.6-luna', 'codex-sol': 'gpt-5.6-sol', 'codex-terra': 'gpt-5.6-terra', } + +type Json = any + /** - * CC fires background calls (conversation titles, quota probes) on haiku no - * matter what --model says; route them somewhere cheap instead of 400-ing. + * Knobs read from the environment once at startup. Passed explicitly through + * the translation layer so a caller (or a test) can pin one without touching + * process.env. */ -const FALLBACK_MODEL = process.env.CODEX_FALLBACK_MODEL ?? 'gpt-5.4-mini' +export interface Config { + /** reasoning.effort sent upstream for gpt-5.6-* models. */ + effort: string + /** Round-trip encrypted reasoning items across tool rounds. */ + carryReasoning: boolean + /** + * CC fires background calls (conversation titles, quota probes) on haiku no + * matter what --model says; route them somewhere cheap instead of 400-ing. + */ + fallbackModel: string +} -type Json = any +export function configFromEnv(env: NodeJS.ProcessEnv = process.env): Config { + return { + effort: env.CODEX_EFFORT ?? 'medium', + carryReasoning: env.CODEX_CARRY_REASONING !== '0', + fallbackModel: env.CODEX_FALLBACK_MODEL ?? 'gpt-5.4-mini', + } +} + +const CONFIG: Config = configFromEnv() const DEBUG = process.env.CODEX_DEBUG === '1' const log = (...a: unknown[]) => console.error('[codex-proxy]', ...a) const debug = (...a: unknown[]) => { if (DEBUG) log(...a) } -if (!OPENAI_KEY) { - log('FATAL: OPENAI_API_KEY is not set') - process.exit(1) -} - -function mapModel(m: string): string { +export function mapModel(m: string, cfg: Config = CONFIG): string { if (MODEL_MAP[m]) return MODEL_MAP[m] if (m?.startsWith('gpt-')) return m - return FALLBACK_MODEL + return cfg.fallbackModel } -function textOf(content: Json): string { +export function textOf(content: Json): string { if (typeof content === 'string') return content if (!Array.isArray(content)) return '' return content.filter((b: Json) => b?.type === 'text').map((b: Json) => b.text).join('\n') @@ -63,10 +82,11 @@ function textOf(content: Json): string { * only round-trips what it received, so we smuggle the blob through a * thinking block's signature and rebuild the item on the way back. */ -const REASONING_MARK = '​' +/** Leading U+200B keeps the marker invisible if a client ever renders it. */ +export const REASONING_MARK = '​' -function encodeReasoning(item: Json): Json | null { - if (!CARRY_REASONING || !item?.encrypted_content) return null +export function encodeReasoning(item: Json, cfg: Config = CONFIG): Json | null { + if (!cfg.carryReasoning || !item?.encrypted_content) return null return { type: 'thinking', thinking: REASONING_MARK, @@ -74,8 +94,8 @@ function encodeReasoning(item: Json): Json | null { } } -function decodeReasoning(block: Json): Json | null { - if (!CARRY_REASONING) return null +export function decodeReasoning(block: Json, cfg: Config = CONFIG): Json | null { + if (!cfg.carryReasoning) return null if (block?.type !== 'thinking' || block.thinking !== REASONING_MARK) return null try { const { id, ec } = JSON.parse(block.signature) @@ -86,7 +106,7 @@ function decodeReasoning(block: Json): Json | null { } /** Anthropic messages[] -> Responses input[] */ -function toResponsesInput(messages: Json[]): Json[] { +export function toResponsesInput(messages: Json[], cfg: Config = CONFIG): Json[] { const input: Json[] = [] for (const msg of messages) { @@ -120,7 +140,7 @@ function toResponsesInput(messages: Json[]): Json[] { if (parts.length) { input.push({ role: 'assistant', content: [...parts] }); parts.length = 0 } } for (const b of blocks) { - const reasoning = decodeReasoning(b) + const reasoning = decodeReasoning(b, cfg) if (reasoning) { // Reasoning must precede the call it produced. flush() @@ -143,7 +163,7 @@ function toResponsesInput(messages: Json[]): Json[] { return input } -function toResponsesTools(tools: Json[] | undefined): Json[] | undefined { +export function toResponsesTools(tools: Json[] | undefined): Json[] | undefined { if (!tools?.length) return undefined const out = tools .filter((t: Json) => t?.input_schema || t?.custom?.input_schema) @@ -158,7 +178,7 @@ function toResponsesTools(tools: Json[] | undefined): Json[] | undefined { return out.length ? out : undefined } -function toToolChoice(tc: Json): Json | undefined { +export function toToolChoice(tc: Json): Json | undefined { if (!tc) return undefined if (tc.type === 'auto') return 'auto' if (tc.type === 'any') return 'required' @@ -167,11 +187,12 @@ function toToolChoice(tc: Json): Json | undefined { return undefined } -function buildUpstream(body: Json) { - const model = mapModel(body.model) +/** Anthropic /v1/messages request body -> Responses API request body. */ +export function buildUpstream(body: Json, cfg: Config = CONFIG) { + const model = mapModel(body.model, cfg) const req: Json = { model, - input: toResponsesInput(body.messages ?? []), + input: toResponsesInput(body.messages ?? [], cfg), store: false, stream: Boolean(body.stream), } @@ -188,27 +209,236 @@ function buildUpstream(body: Json) { // Only reasoning models accept the knob; the cheap fallback does not. if (model.startsWith('gpt-5.6')) { - req.reasoning = { effort: EFFORT } - if (CARRY_REASONING) req.include = ['reasoning.encrypted_content'] + req.reasoning = { effort: cfg.effort } + if (cfg.carryReasoning) req.include = ['reasoning.encrypted_content'] } return req } -const sse = (res: http.ServerResponse, event: string, data: Json) => { - res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`) -} - -function stopReasonFrom(resp: Json): string { +export function stopReasonFrom(resp: Json): string { const hasCall = (resp?.output ?? []).some((i: Json) => i.type === 'function_call') if (hasCall) return 'tool_use' if (resp?.status === 'incomplete') return 'max_tokens' return 'end_turn' } -function safeParse(s: string): Json { +export function safeParse(s: string): Json { try { return JSON.parse(s || '{}') } catch { return {} } } +export interface MessageIdentity { + /** Anthropic message id echoed back to the client. */ + id: string + /** The model name the client asked for, not the one we mapped to. */ + model: string +} + +/** Responses API response body -> Anthropic non-streaming message. */ +export function toAnthropicMessage(data: Json, who: MessageIdentity, cfg: Config = CONFIG): Json { + const content: Json[] = [] + for (const item of data?.output ?? []) { + if (item.type === 'reasoning') { + const enc = encodeReasoning(item, cfg) + if (enc) content.push(enc) + } else if (item.type === 'message') { + for (const c of item.content ?? []) { + if (c.type === 'output_text') content.push({ type: 'text', text: c.text }) + } + } else if (item.type === 'function_call') { + content.push({ type: 'tool_use', id: item.call_id, name: item.name, input: safeParse(item.arguments) }) + } + } + return { + id: who.id, + type: 'message', + role: 'assistant', + model: who.model, + content, + stop_reason: stopReasonFrom(data), + stop_sequence: null, + usage: { + input_tokens: data?.usage?.input_tokens ?? 0, + output_tokens: data?.usage?.output_tokens ?? 0, + }, + } +} + +/** Sink for translated Anthropic SSE events: (eventName, payload). */ +export type EmitEvent = (event: string, data: Json) => void + +export interface StreamTranslator { + /** Emit message_start. Call once, before the first push. */ + start(): void + /** Feed a raw upstream SSE chunk; partial events are buffered. */ + push(text: string): void + /** Close any open blocks and emit message_delta + message_stop. */ + end(): void +} + +/** + * Rebuilds an Anthropic SSE stream from an OpenAI Responses SSE stream. + * + * Transport-free on purpose: it takes raw text in and hands events to `emit`, + * so both the live socket and the test fixtures drive the same code. + */ +export function createStreamTranslator( + emit: EmitEvent, + who: MessageIdentity, + cfg: Config = CONFIG, +): StreamTranslator { + let blockIndex = -1 + const openBlocks = new Map() // upstream item id -> anthropic index + let stopReason = 'end_turn' + let outputTokens = 0 + let buf = '' + + const closeBlock = (idx: number) => emit('content_block_stop', { type: 'content_block_stop', index: idx }) + + const onEvent = (ev: Json) => { + switch (ev.type) { + case 'response.output_item.added': { + const item = ev.item + if (item?.type === 'message') { + blockIndex++ + openBlocks.set(item.id, blockIndex) + emit('content_block_start', { + type: 'content_block_start', + index: blockIndex, + content_block: { type: 'text', text: '' }, + }) + } else if (item?.type === 'function_call') { + blockIndex++ + openBlocks.set(item.id, blockIndex) + stopReason = 'tool_use' + emit('content_block_start', { + type: 'content_block_start', + index: blockIndex, + content_block: { type: 'tool_use', id: item.call_id, name: item.name, input: {} }, + }) + } + break + } + case 'response.output_text.delta': { + const idx = openBlocks.get(ev.item_id) + if (idx === undefined) break + emit('content_block_delta', { + type: 'content_block_delta', + index: idx, + delta: { type: 'text_delta', text: ev.delta }, + }) + break + } + case 'response.function_call_arguments.delta': { + const idx = openBlocks.get(ev.item_id) + if (idx === undefined) break + emit('content_block_delta', { + type: 'content_block_delta', + index: idx, + delta: { type: 'input_json_delta', partial_json: ev.delta }, + }) + break + } + case 'response.output_item.done': { + const item = ev.item + const idx = openBlocks.get(item?.id) + if (idx !== undefined) { + closeBlock(idx) + openBlocks.delete(item.id) + } else if (item?.type === 'reasoning') { + // Reasoning arrives complete (never streamed); emit as one block so + // CC hands the encrypted blob back to us next turn. + const enc = encodeReasoning(item, cfg) + if (enc) { + blockIndex++ + emit('content_block_start', { + type: 'content_block_start', + index: blockIndex, + content_block: { type: 'thinking', thinking: '' }, + }) + emit('content_block_delta', { + type: 'content_block_delta', + index: blockIndex, + delta: { type: 'thinking_delta', thinking: enc.thinking }, + }) + emit('content_block_delta', { + type: 'content_block_delta', + index: blockIndex, + delta: { type: 'signature_delta', signature: enc.signature }, + }) + closeBlock(blockIndex) + } + } + break + } + case 'response.completed': + case 'response.incomplete': { + outputTokens = ev.response?.usage?.output_tokens ?? 0 + stopReason = stopReasonFrom(ev.response) + break + } + case 'response.failed': + case 'error': { + log('stream error', JSON.stringify(ev).slice(0, 300)) + break + } + } + } + + return { + start() { + emit('message_start', { + type: 'message_start', + message: { + id: who.id, + type: 'message', + role: 'assistant', + model: who.model, + content: [], + stop_reason: null, + stop_sequence: null, + usage: { input_tokens: 0, output_tokens: 0 }, + }, + }) + }, + + push(text: string) { + buf += text + const chunks = buf.split('\n\n') + buf = chunks.pop() ?? '' + + for (const chunk of chunks) { + const dataLine = chunk.split('\n').find((l) => l.startsWith('data:')) + if (!dataLine) continue + const payload = dataLine.slice(5).trim() + if (!payload || payload === '[DONE]') continue + + let ev: Json + try { ev = JSON.parse(payload) } catch { continue } + onEvent(ev) + } + }, + + end() { + for (const idx of openBlocks.values()) closeBlock(idx) + + emit('message_delta', { + type: 'message_delta', + delta: { stop_reason: stopReason, stop_sequence: null }, + usage: { output_tokens: outputTokens }, + }) + emit('message_stop', { type: 'message_stop' }) + }, + } +} + +const sse = (res: http.ServerResponse, event: string, data: Json) => { + res.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`) +} + +export function newMessageId(): string { + return `msg_${Math.abs(Date.now() % 1e9)}` +} + async function handleMessages(reqBody: Json, res: http.ServerResponse) { const upstreamReq = buildUpstream(reqBody) const reasoningIn = upstreamReq.input.filter((i: Json) => i.type === 'reasoning').length @@ -217,7 +447,7 @@ async function handleMessages(reqBody: Json, res: http.ServerResponse) { const upstream = await fetch(UPSTREAM, { method: 'POST', - headers: { 'Authorization': `Bearer ${OPENAI_KEY}`, 'Content-Type': 'application/json' }, + headers: { 'Authorization': `Bearer ${process.env.OPENAI_API_KEY}`, 'Content-Type': 'application/json' }, body: JSON.stringify(upstreamReq), }) @@ -229,37 +459,12 @@ async function handleMessages(reqBody: Json, res: http.ServerResponse) { return } - const msgId = `msg_${Math.abs(Date.now() % 1e9)}` + const who: MessageIdentity = { id: newMessageId(), model: reqBody.model } if (!reqBody.stream) { const data: Json = await upstream.json() - const content: Json[] = [] - for (const item of data.output ?? []) { - if (item.type === 'reasoning') { - const enc = encodeReasoning(item) - if (enc) content.push(enc) - } else if (item.type === 'message') { - for (const c of item.content ?? []) { - if (c.type === 'output_text') content.push({ type: 'text', text: c.text }) - } - } else if (item.type === 'function_call') { - content.push({ type: 'tool_use', id: item.call_id, name: item.name, input: safeParse(item.arguments) }) - } - } res.writeHead(200, { 'Content-Type': 'application/json' }) - res.end(JSON.stringify({ - id: msgId, - type: 'message', - role: 'assistant', - model: reqBody.model, - content, - stop_reason: stopReasonFrom(data), - stop_sequence: null, - usage: { - input_tokens: data.usage?.input_tokens ?? 0, - output_tokens: data.usage?.output_tokens ?? 0, - }, - })) + res.end(JSON.stringify(toAnthropicMessage(data, who))) return } @@ -270,192 +475,82 @@ async function handleMessages(reqBody: Json, res: http.ServerResponse) { 'Connection': 'keep-alive', }) - sse(res, 'message_start', { - type: 'message_start', - message: { - id: msgId, - type: 'message', - role: 'assistant', - model: reqBody.model, - content: [], - stop_reason: null, - stop_sequence: null, - usage: { input_tokens: 0, output_tokens: 0 }, - }, - }) - - let blockIndex = -1 - const openBlocks = new Map() // upstream item id -> anthropic index - let stopReason = 'end_turn' - let outputTokens = 0 - - const closeBlock = (idx: number) => sse(res, 'content_block_stop', { type: 'content_block_stop', index: idx }) + const stream = createStreamTranslator((event, data) => sse(res, event, data), who) + stream.start() const reader = upstream.body!.getReader() const decoder = new TextDecoder() - let buf = '' while (true) { const { done, value } = await reader.read() if (done) break - buf += decoder.decode(value, { stream: true }) + stream.push(decoder.decode(value, { stream: true })) + } - const chunks = buf.split('\n\n') - buf = chunks.pop() ?? '' + stream.end() + res.end() +} - for (const chunk of chunks) { - const dataLine = chunk.split('\n').find((l) => l.startsWith('data:')) - if (!dataLine) continue - const payload = dataLine.slice(5).trim() - if (!payload || payload === '[DONE]') continue +/** Route one collected request. Split out from the server so it is callable without a socket. */ +export async function handleRequest(url: string, body: string, res: http.ServerResponse): Promise { + // CC probes this before anything else; a 404 here surfaces to the user as + // "Not logged in - Please run /login". + if (url.startsWith('/api/hello')) { + res.writeHead(200, { 'Content-Type': 'application/json' }) + res.end('{}') + return + } - let ev: Json - try { ev = JSON.parse(payload) } catch { continue } + if (url.startsWith('/v1/messages/count_tokens')) { + // Rough estimate; CC uses it only for context-pressure hints. + res.writeHead(200, { 'Content-Type': 'application/json' }) + res.end(JSON.stringify({ input_tokens: Math.ceil(body.length / 4) })) + return + } - switch (ev.type) { - case 'response.output_item.added': { - const item = ev.item - if (item?.type === 'message') { - blockIndex++ - openBlocks.set(item.id, blockIndex) - sse(res, 'content_block_start', { - type: 'content_block_start', - index: blockIndex, - content_block: { type: 'text', text: '' }, - }) - } else if (item?.type === 'function_call') { - blockIndex++ - openBlocks.set(item.id, blockIndex) - stopReason = 'tool_use' - sse(res, 'content_block_start', { - type: 'content_block_start', - index: blockIndex, - content_block: { type: 'tool_use', id: item.call_id, name: item.name, input: {} }, - }) - } - break - } - case 'response.output_text.delta': { - const idx = openBlocks.get(ev.item_id) - if (idx === undefined) break - sse(res, 'content_block_delta', { - type: 'content_block_delta', - index: idx, - delta: { type: 'text_delta', text: ev.delta }, - }) - break - } - case 'response.function_call_arguments.delta': { - const idx = openBlocks.get(ev.item_id) - if (idx === undefined) break - sse(res, 'content_block_delta', { - type: 'content_block_delta', - index: idx, - delta: { type: 'input_json_delta', partial_json: ev.delta }, - }) - break - } - case 'response.output_item.done': { - const item = ev.item - const idx = openBlocks.get(item?.id) - if (idx !== undefined) { - closeBlock(idx) - openBlocks.delete(item.id) - } else if (item?.type === 'reasoning') { - // Reasoning arrives complete (never streamed); emit as one block so - // CC hands the encrypted blob back to us next turn. - const enc = encodeReasoning(item) - if (enc) { - blockIndex++ - sse(res, 'content_block_start', { - type: 'content_block_start', - index: blockIndex, - content_block: { type: 'thinking', thinking: '' }, - }) - sse(res, 'content_block_delta', { - type: 'content_block_delta', - index: blockIndex, - delta: { type: 'thinking_delta', thinking: enc.thinking }, - }) - sse(res, 'content_block_delta', { - type: 'content_block_delta', - index: blockIndex, - delta: { type: 'signature_delta', signature: enc.signature }, - }) - closeBlock(blockIndex) - } - } - break - } - case 'response.completed': - case 'response.incomplete': { - outputTokens = ev.response?.usage?.output_tokens ?? 0 - stopReason = stopReasonFrom(ev.response) - break - } - case 'response.failed': - case 'error': { - log('stream error', JSON.stringify(ev).slice(0, 300)) - break - } + if (url.startsWith('/v1/messages')) { + try { + await handleMessages(JSON.parse(body || '{}'), res) + } catch (e: unknown) { + log('handler crash', e) + if (!res.headersSent) { + res.writeHead(500, { 'Content-Type': 'application/json' }) + res.end(JSON.stringify({ type: 'error', error: { type: 'api_error', message: String(e) } })) + } else { + res.end() } } + return } - for (const idx of openBlocks.values()) closeBlock(idx) + res.writeHead(404, { 'Content-Type': 'application/json' }) + res.end(JSON.stringify({ error: 'not found' })) +} - sse(res, 'message_delta', { - type: 'message_delta', - delta: { stop_reason: stopReason, stop_sequence: null }, - usage: { output_tokens: outputTokens }, +export function createProxyServer(): http.Server { + return http.createServer((req, res) => { + let body = '' + req.on('data', (c) => { body += c }) + req.on('end', () => { + const url = req.url ?? '' + debug('<-', req.method, url) + void handleRequest(url, body, res) + }) }) - sse(res, 'message_stop', { type: 'message_stop' }) - res.end() } -const server = http.createServer((req, res) => { - let body = '' - req.on('data', (c) => { body += c }) - req.on('end', async () => { - const url = req.url ?? '' - debug('<-', req.method, url) - - // CC probes this before anything else; a 404 here surfaces to the user as - // "Not logged in - Please run /login". - if (url.startsWith('/api/hello')) { - res.writeHead(200, { 'Content-Type': 'application/json' }) - res.end('{}') - return - } - - if (url.startsWith('/v1/messages/count_tokens')) { - // Rough estimate; CC uses it only for context-pressure hints. - res.writeHead(200, { 'Content-Type': 'application/json' }) - res.end(JSON.stringify({ input_tokens: Math.ceil(body.length / 4) })) - return - } - - if (url.startsWith('/v1/messages')) { - try { - await handleMessages(JSON.parse(body || '{}'), res) - } catch (e: unknown) { - log('handler crash', e) - if (!res.headersSent) { - res.writeHead(500, { 'Content-Type': 'application/json' }) - res.end(JSON.stringify({ type: 'error', error: { type: 'api_error', message: String(e) } })) - } else { - res.end() - } - } - return - } +function main() { + if (!process.env.OPENAI_API_KEY) { + log('FATAL: OPENAI_API_KEY is not set') + process.exit(1) + } - res.writeHead(404, { 'Content-Type': 'application/json' }) - res.end(JSON.stringify({ error: 'not found' })) + const server = createProxyServer() + server.listen(PORT, '127.0.0.1', () => { + const port = (server.address() as { port: number }).port + log(`ready port=${port} effort=${CONFIG.effort} carry_reasoning=${CONFIG.carryReasoning ? 'on' : 'off'}`) }) -}) +} -server.listen(PORT, '127.0.0.1', () => { - const port = (server.address() as { port: number }).port - log(`ready port=${port} effort=${EFFORT} carry_reasoning=${CARRY_REASONING ? 'on' : 'off'}`) -}) +const entry = process.argv[1] ? pathToFileURL(process.argv[1]).href : '' +if (entry === import.meta.url) main() diff --git a/package.json b/package.json index 0fb7c0c..6cf72ac 100644 --- a/package.json +++ b/package.json @@ -9,7 +9,8 @@ "node": ">=22" }, "scripts": { - "start": "node --experimental-strip-types --disable-warning=ExperimentalWarning codex-proxy.ts" + "start": "node --experimental-strip-types --disable-warning=ExperimentalWarning codex-proxy.ts", + "test": "node --test --experimental-strip-types --disable-warning=ExperimentalWarning test/*.test.ts" }, "keywords": [ "anthropic", diff --git a/test/fixtures/import-probe.ts b/test/fixtures/import-probe.ts new file mode 100644 index 0000000..fd128a9 --- /dev/null +++ b/test/fixtures/import-probe.ts @@ -0,0 +1,9 @@ +/** + * Probe for the "importing the bridge starts nothing" contract. + * + * Run as its own process with OPENAI_API_KEY unset: the module must import + * cleanly, without exiting on the missing key and without binding a port. + */ +import { mapModel } from '../../codex-proxy.ts' + +console.log(`imported-without-side-effects ${mapModel('codex-sol')}`) diff --git a/test/fixtures/responses-stream.sse.txt b/test/fixtures/responses-stream.sse.txt new file mode 100644 index 0000000..3e29f45 --- /dev/null +++ b/test/fixtures/responses-stream.sse.txt @@ -0,0 +1,35 @@ +event: response.created +data: {"type":"response.created","response":{"id":"resp_fixture_0001","status":"in_progress","output":[]}} + +event: response.output_item.done +data: {"type":"response.output_item.done","output_index":0,"item":{"id":"rs_fixture_0001","type":"reasoning","summary":[],"encrypted_content":"ZW5jcnlwdGVkLXJlYXNvbmluZy1ibG9i"}} + +event: response.output_item.added +data: {"type":"response.output_item.added","output_index":1,"item":{"id":"msg_fixture_0001","type":"message","status":"in_progress","role":"assistant","content":[]}} + +event: response.output_text.delta +data: {"type":"response.output_text.delta","item_id":"msg_fixture_0001","output_index":1,"content_index":0,"delta":"Reading "} + +event: response.output_text.delta +data: {"type":"response.output_text.delta","item_id":"msg_fixture_0001","output_index":1,"content_index":0,"delta":"the file."} + +event: response.output_item.done +data: {"type":"response.output_item.done","output_index":1,"item":{"id":"msg_fixture_0001","type":"message","status":"completed","role":"assistant","content":[{"type":"output_text","text":"Reading the file."}]}} + +event: response.output_item.added +data: {"type":"response.output_item.added","output_index":2,"item":{"id":"fc_fixture_0001","type":"function_call","status":"in_progress","call_id":"call_fixture_0001","name":"Read","arguments":""}} + +event: response.function_call_arguments.delta +data: {"type":"response.function_call_arguments.delta","item_id":"fc_fixture_0001","output_index":2,"delta":"{\"file_path\":"} + +event: response.function_call_arguments.delta +data: {"type":"response.function_call_arguments.delta","item_id":"fc_fixture_0001","output_index":2,"delta":"\"README.md\"}"} + +event: response.output_item.done +data: {"type":"response.output_item.done","output_index":2,"item":{"id":"fc_fixture_0001","type":"function_call","status":"completed","call_id":"call_fixture_0001","name":"Read","arguments":"{\"file_path\":\"README.md\"}"}} + +event: response.completed +data: {"type":"response.completed","response":{"id":"resp_fixture_0001","status":"completed","output":[{"type":"reasoning"},{"type":"message"},{"type":"function_call"}],"usage":{"input_tokens":1204,"output_tokens":48}}} + +data: [DONE] + diff --git a/test/server-shape.test.ts b/test/server-shape.test.ts new file mode 100644 index 0000000..92f34b5 --- /dev/null +++ b/test/server-shape.test.ts @@ -0,0 +1,107 @@ +/** + * The thin HTTP layer: routing decisions and the module's import contract. + * The routes exercised here answer locally, so no socket and no upstream call + * is involved -- the response object is a recorder. + */ +import assert from 'node:assert/strict' +import { execFile } from 'node:child_process' +import { describe, test } from 'node:test' +import { promisify } from 'node:util' +import type { ServerResponse } from 'node:http' + +import { createProxyServer, handleRequest, newMessageId } from '../codex-proxy.ts' + +const execFileAsync = promisify(execFile) + +interface Recorded { + status: number + headers: Record + body: string +} + +function recorder(): { rec: Recorded; res: ServerResponse } { + const rec: Recorded = { status: 0, headers: {}, body: '' } + const res = { + headersSent: false, + writeHead(status: number, headers: Record) { + rec.status = status + rec.headers = headers + this.headersSent = true + return this + }, + write(chunk: string) { rec.body += chunk; return true }, + end(chunk?: string) { if (chunk) rec.body += chunk; return this }, + } + return { rec, res: res as unknown as ServerResponse } +} + +describe('routing', () => { + test('/api/hello answers 200 so the client does not report a logged-out state', async () => { + // A 404 here surfaces to the user as "Not logged in - Please run /login". + const { rec, res } = recorder() + await handleRequest('/api/hello', '', res) + assert.equal(rec.status, 200) + assert.equal(rec.headers['Content-Type'], 'application/json') + assert.deepEqual(JSON.parse(rec.body), {}) + }) + + test('/api/hello matches with a query string attached', async () => { + const { rec, res } = recorder() + await handleRequest('/api/hello?client=cli', '', res) + assert.equal(rec.status, 200) + }) + + test('count_tokens answers locally with a length-based estimate', async () => { + const body = JSON.stringify({ messages: [{ role: 'user', content: 'hello' }] }) + const { rec, res } = recorder() + await handleRequest('/v1/messages/count_tokens', body, res) + assert.equal(rec.status, 200) + assert.deepEqual(JSON.parse(rec.body), { input_tokens: Math.ceil(body.length / 4) }) + }) + + test('count_tokens is matched before the /v1/messages prefix it starts with', async () => { + // Falling through to /v1/messages would try to reach the upstream API. + const { rec, res } = recorder() + await handleRequest('/v1/messages/count_tokens', '', res) + assert.equal(rec.status, 200) + assert.deepEqual(JSON.parse(rec.body), { input_tokens: 0 }) + }) + + test('an unknown path is a JSON 404, not an HTML error page', async () => { + const { rec, res } = recorder() + await handleRequest('/v1/complete', '', res) + assert.equal(rec.status, 404) + assert.equal(rec.headers['Content-Type'], 'application/json') + assert.deepEqual(JSON.parse(rec.body), { error: 'not found' }) + }) +}) + +describe('message ids', () => { + test('carry the msg_ prefix the client expects', () => { + assert.match(newMessageId(), /^msg_\d+$/) + }) +}) + +describe('module contract', () => { + test('the server is built on demand, not at import time', () => { + const server = createProxyServer() + assert.equal(server.listening, false) + server.close() + }) + + test('importing the bridge without OPENAI_API_KEY neither exits nor binds a port', async () => { + // The CLI still refuses to start without a key; only the import is quiet. + const env = { ...process.env } + delete env.OPENAI_API_KEY + delete env.PORT + + const { stdout, stderr } = await execFileAsync( + process.execPath, + ['--experimental-strip-types', '--disable-warning=ExperimentalWarning', 'test/fixtures/import-probe.ts'], + { env, cwd: new URL('..', import.meta.url) }, + ) + assert.match(stdout, /imported-without-side-effects gpt-5\.6-sol/) + assert.doesNotMatch(stderr, /ready port=/) + assert.doesNotMatch(stderr, /FATAL/) + }) +}) diff --git a/test/streaming.test.ts b/test/streaming.test.ts new file mode 100644 index 0000000..750525e --- /dev/null +++ b/test/streaming.test.ts @@ -0,0 +1,198 @@ +/** + * SSE translation, driven by a recorded upstream stream in + * fixtures/responses-stream.sse.txt. The translator takes raw text and hands + * events to a callback, so these tests exercise the same code the socket does + * without opening one. + */ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { describe, test } from 'node:test' + +import { createStreamTranslator, REASONING_MARK, type Config } from '../codex-proxy.ts' + +const CFG: Config = { effort: 'medium', carryReasoning: true, fallbackModel: 'gpt-5.4-mini' } +const NO_CARRY: Config = { ...CFG, carryReasoning: false } +const WHO = { id: 'msg_fixture_0001', model: 'codex-sol' } + +const FIXTURE = readFileSync(new URL('./fixtures/responses-stream.sse.txt', import.meta.url), 'utf8') + +interface Emitted { event: string; data: any } + +function run(chunks: string[], cfg: Config = CFG): Emitted[] { + const out: Emitted[] = [] + const stream = createStreamTranslator((event, data) => out.push({ event, data }), WHO, cfg) + stream.start() + for (const c of chunks) stream.push(c) + stream.end() + return out +} + +/** Split a string into fixed-size pieces, to fake arbitrary socket framing. */ +function slice(text: string, size: number): string[] { + const out: string[] = [] + for (let i = 0; i < text.length; i += size) out.push(text.slice(i, i + size)) + return out +} + +const names = (events: Emitted[]) => events.map((e) => e.event) + +describe('a recorded reasoning + text + tool-call stream', () => { + const events = run([FIXTURE]) + + test('opens with message_start carrying the client-facing id and model', () => { + assert.equal(events[0].event, 'message_start') + assert.equal(events[0].data.message.id, 'msg_fixture_0001') + assert.equal(events[0].data.message.model, 'codex-sol') + assert.equal(events[0].data.message.stop_reason, null) + }) + + test('produces the full Anthropic event sequence, blocks balanced', () => { + assert.deepEqual(names(events), [ + 'message_start', + // reasoning: never streamed, so it arrives as one complete block + 'content_block_start', 'content_block_delta', 'content_block_delta', 'content_block_stop', + // assistant text + 'content_block_start', 'content_block_delta', 'content_block_delta', 'content_block_stop', + // tool call + 'content_block_start', 'content_block_delta', 'content_block_delta', 'content_block_stop', + 'message_delta', 'message_stop', + ]) + const starts = events.filter((e) => e.event === 'content_block_start').length + const stops = events.filter((e) => e.event === 'content_block_stop').length + assert.equal(starts, stops) + }) + + test('block indices are contiguous from zero and never reused', () => { + const indices = events + .filter((e) => e.event === 'content_block_start') + .map((e) => e.data.index) + assert.deepEqual(indices, [0, 1, 2]) + for (const e of events) { + if (e.event === 'content_block_delta' || e.event === 'content_block_stop') { + assert.ok(indices.includes(e.data.index), `index ${e.data.index} was never opened`) + } + } + }) + + test('the reasoning block carries the marker and the encrypted blob', () => { + const [thinking, signature] = events.filter((e) => e.event === 'content_block_delta' && e.data.index === 0) + assert.equal(events[1].data.content_block.type, 'thinking') + assert.equal(thinking.data.delta.type, 'thinking_delta') + assert.equal(thinking.data.delta.thinking, REASONING_MARK) + assert.equal(signature.data.delta.type, 'signature_delta') + assert.deepEqual(JSON.parse(signature.data.delta.signature), { + id: 'rs_fixture_0001', + ec: 'ZW5jcnlwdGVkLXJlYXNvbmluZy1ibG9i', + }) + }) + + test('text deltas reassemble into the sentence the model wrote', () => { + const text = events + .filter((e) => e.event === 'content_block_delta' && e.data.delta.type === 'text_delta') + .map((e) => e.data.delta.text) + .join('') + assert.equal(text, 'Reading the file.') + }) + + test('the tool block announces call_id and name up front, arguments as JSON deltas', () => { + const start = events.filter((e) => e.event === 'content_block_start')[2] + assert.deepEqual(start.data.content_block, { + type: 'tool_use', + id: 'call_fixture_0001', + name: 'Read', + input: {}, + }) + const partial = events + .filter((e) => e.event === 'content_block_delta' && e.data.delta.type === 'input_json_delta') + .map((e) => e.data.delta.partial_json) + .join('') + assert.deepEqual(JSON.parse(partial), { file_path: 'README.md' }) + }) + + test('closes with tool_use and the upstream output-token count', () => { + const delta = events.at(-2)! + assert.equal(delta.event, 'message_delta') + assert.deepEqual(delta.data.delta, { stop_reason: 'tool_use', stop_sequence: null }) + assert.deepEqual(delta.data.usage, { output_tokens: 48 }) + assert.equal(events.at(-1)!.event, 'message_stop') + }) +}) + +describe('chunk framing', () => { + test('a stream split mid-event yields exactly the same output', () => { + // The socket hands over arbitrary byte runs; SSE events span them. + const whole = run([FIXTURE]) + for (const size of [1, 7, 64, 997]) { + assert.deepEqual(run(slice(FIXTURE, size)), whole, `framing broke at chunk size ${size}`) + } + }) + + test('a chunk boundary inside a JSON string is not treated as an event end', () => { + const cut = FIXTURE.indexOf('README.md') + 4 + assert.deepEqual(run([FIXTURE.slice(0, cut), FIXTURE.slice(cut)]), run([FIXTURE])) + }) +}) + +describe('stream edge cases', () => { + test('unparseable and terminator payloads are skipped, not fatal', () => { + const events = run([ + 'event: error\ndata: not json at all\n\n', + 'data: [DONE]\n\n', + ': keep-alive comment\n\n', + 'event: response.completed\ndata: {"type":"response.completed","response":{"status":"completed","output":[],"usage":{"output_tokens":3}}}\n\n', + ]) + assert.deepEqual(names(events), ['message_start', 'message_delta', 'message_stop']) + assert.equal(events[1].data.delta.stop_reason, 'end_turn') + assert.equal(events[1].data.usage.output_tokens, 3) + }) + + test('a stream that only errors still terminates the message', () => { + const events = run(['event: response.failed\ndata: {"type":"response.failed","response":{"error":{"message":"upstream boom"}}}\n\n']) + assert.deepEqual(names(events), ['message_start', 'message_delta', 'message_stop']) + assert.equal(events[1].data.delta.stop_reason, 'end_turn') + }) + + test('response.incomplete reports the token cap downstream', () => { + const events = run([ + 'event: response.incomplete\ndata: {"type":"response.incomplete","response":{"status":"incomplete","output":[],"usage":{"output_tokens":16}}}\n\n', + ]) + assert.equal(events[1].data.delta.stop_reason, 'max_tokens') + assert.equal(events[1].data.usage.output_tokens, 16) + }) + + test('a block left open by a dropped connection is still closed', () => { + // Upstream died after output_item.added; the client would hang forever on + // an unclosed block. + const events = run([ + 'event: response.output_item.added\ndata: {"type":"response.output_item.added","item":{"id":"msg_upstream_0001","type":"message"}}\n\n', + 'event: response.output_text.delta\ndata: {"type":"response.output_text.delta","item_id":"msg_upstream_0001","delta":"half a sen"}\n\n', + ]) + assert.deepEqual(names(events), [ + 'message_start', 'content_block_start', 'content_block_delta', + 'content_block_stop', 'message_delta', 'message_stop', + ]) + assert.equal(events.at(-3)!.data.index, 0) + }) + + test('a delta for an item that was never opened is ignored', () => { + const events = run([ + 'event: response.output_text.delta\ndata: {"type":"response.output_text.delta","item_id":"msg_unknown","delta":"orphan"}\n\n', + ]) + assert.deepEqual(names(events), ['message_start', 'message_delta', 'message_stop']) + }) + + test('an empty stream is still a well-formed Anthropic message', () => { + assert.deepEqual(names(run([])), ['message_start', 'message_delta', 'message_stop']) + assert.deepEqual(names(run([''])), ['message_start', 'message_delta', 'message_stop']) + }) + + test('with carrying off no thinking block is streamed at all', () => { + const events = run([FIXTURE], NO_CARRY) + assert.equal(events.some((e) => e.event === 'content_block_start' && e.data.content_block.type === 'thinking'), false) + // Indices stay contiguous: text and tool call shift down to 0 and 1. + assert.deepEqual( + events.filter((e) => e.event === 'content_block_start').map((e) => e.data.index), + [0, 1], + ) + }) +}) diff --git a/test/translate-request.test.ts b/test/translate-request.test.ts new file mode 100644 index 0000000..8686057 --- /dev/null +++ b/test/translate-request.test.ts @@ -0,0 +1,383 @@ +/** + * Downstream -> upstream: an Anthropic /v1/messages body becomes a Responses + * API request body. Pure functions only; nothing here opens a socket. + */ +import assert from 'node:assert/strict' +import { describe, test } from 'node:test' + +import { + buildUpstream, + configFromEnv, + decodeReasoning, + encodeReasoning, + mapModel, + REASONING_MARK, + textOf, + toResponsesInput, + toResponsesTools, + toToolChoice, + type Config, +} from '../codex-proxy.ts' + +const CFG: Config = { effort: 'medium', carryReasoning: true, fallbackModel: 'gpt-5.4-mini' } +const NO_CARRY: Config = { ...CFG, carryReasoning: false } + +describe('mapModel', () => { + test('resolves the codex-* aliases to gpt-5.6-* names', () => { + assert.equal(mapModel('codex-luna', CFG), 'gpt-5.6-luna') + assert.equal(mapModel('codex-sol', CFG), 'gpt-5.6-sol') + assert.equal(mapModel('codex-terra', CFG), 'gpt-5.6-terra') + }) + + test('passes any gpt-* name through unchanged', () => { + assert.equal(mapModel('gpt-5.6-sol', CFG), 'gpt-5.6-sol') + assert.equal(mapModel('gpt-4.1-mini', CFG), 'gpt-4.1-mini') + }) + + test('routes an unknown model to the fallback instead of failing', () => { + // This is the whole reason the fallback exists: the client fires + // background calls (titles, quota probes) on a model we never mapped. + assert.equal(mapModel('claude-haiku-4-5-20251001', CFG), 'gpt-5.4-mini') + assert.equal(mapModel('', CFG), 'gpt-5.4-mini') + assert.equal(mapModel('claude-haiku-4-5-20251001', { ...CFG, fallbackModel: 'gpt-4.1-nano' }), 'gpt-4.1-nano') + }) +}) + +describe('configFromEnv', () => { + test('defaults with an empty environment', () => { + assert.deepEqual(configFromEnv({}), { + effort: 'medium', + carryReasoning: true, + fallbackModel: 'gpt-5.4-mini', + }) + }) + + test('CODEX_CARRY_REASONING is off only for the exact string "0"', () => { + assert.equal(configFromEnv({ CODEX_CARRY_REASONING: '0' }).carryReasoning, false) + assert.equal(configFromEnv({ CODEX_CARRY_REASONING: 'false' }).carryReasoning, true) + assert.equal(configFromEnv({ CODEX_CARRY_REASONING: '' }).carryReasoning, true) + }) + + test('reads effort and fallback model from the environment', () => { + const cfg = configFromEnv({ CODEX_EFFORT: 'high', CODEX_FALLBACK_MODEL: 'gpt-4.1-nano' }) + assert.equal(cfg.effort, 'high') + assert.equal(cfg.fallbackModel, 'gpt-4.1-nano') + }) +}) + +describe('textOf', () => { + test('accepts both content shapes and ignores non-text blocks', () => { + assert.equal(textOf('plain'), 'plain') + assert.equal(textOf([{ type: 'text', text: 'a' }, { type: 'text', text: 'b' }]), 'a\nb') + assert.equal(textOf([{ type: 'image' }, { type: 'text', text: 'only' }]), 'only') + assert.equal(textOf(undefined), '') + assert.equal(textOf({ type: 'text', text: 'not an array' }), '') + }) +}) + +describe('system -> instructions', () => { + test('a string system prompt becomes instructions', () => { + const req = buildUpstream({ model: 'codex-sol', system: 'Be terse.', messages: [] }, CFG) + assert.equal(req.instructions, 'Be terse.') + }) + + test('the client sends system as text blocks; they join with newlines', () => { + const req = buildUpstream({ + model: 'codex-sol', + system: [ + { type: 'text', text: 'You are a CLI assistant.' }, + { type: 'text', text: 'Never guess file contents.' }, + ], + messages: [], + }, CFG) + assert.equal(req.instructions, 'You are a CLI assistant.\nNever guess file contents.') + }) + + test('no system prompt leaves instructions unset', () => { + const req = buildUpstream({ model: 'codex-sol', messages: [] }, CFG) + assert.equal('instructions' in req, false) + }) +}) + +describe('messages -> input items', () => { + test('a string user message becomes one input_text part', () => { + assert.deepEqual( + toResponsesInput([{ role: 'user', content: 'hello' }], CFG), + [{ role: 'user', content: [{ type: 'input_text', text: 'hello' }] }], + ) + }) + + test('a base64 image becomes an input_image data URL', () => { + const input = toResponsesInput([{ + role: 'user', + content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'aGVsbG8=' } }], + }], CFG) + assert.deepEqual(input, [{ + role: 'user', + content: [{ type: 'input_image', image_url: 'data:image/png;base64,aGVsbG8=' }], + }]) + }) + + test('a URL image is dropped rather than sent as a broken part', () => { + const input = toResponsesInput([{ + role: 'user', + content: [{ type: 'image', source: { type: 'url', url: 'https://example.invalid/a.png' } }], + }], CFG) + assert.deepEqual(input, []) + }) + + test('an assistant turn with text and a tool call splits into two items in order', () => { + const input = toResponsesInput([{ + role: 'assistant', + content: [ + { type: 'text', text: 'Let me look.' }, + { type: 'tool_use', id: 'toolu_fixture_0001', name: 'Read', input: { file_path: 'README.md' } }, + ], + }], CFG) + assert.deepEqual(input, [ + { role: 'assistant', content: [{ type: 'output_text', text: 'Let me look.' }] }, + { + type: 'function_call', + call_id: 'toolu_fixture_0001', + name: 'Read', + arguments: '{"file_path":"README.md"}', + }, + ]) + }) + + test('a tool_use with no input serialises to an empty object, not undefined', () => { + const input = toResponsesInput([{ + role: 'assistant', + content: [{ type: 'tool_use', id: 'toolu_fixture_0002', name: 'ListDir' }], + }], CFG) + assert.equal(input[0].arguments, '{}') + }) + + test('empty assistant text emits no message item', () => { + const input = toResponsesInput([{ role: 'assistant', content: [{ type: 'text', text: '' }] }], CFG) + assert.deepEqual(input, []) + }) + + test('a tool_result leaves the user message and becomes a standalone item', () => { + // Anthropic nests tool results inside a user turn; Responses wants them + // as top-level function_call_output items. This is the shape that breaks + // the round trip if it regresses. + const input = toResponsesInput([{ + role: 'user', + content: [ + { type: 'tool_result', tool_use_id: 'toolu_fixture_0001', content: '# codex-proxy' }, + { type: 'text', text: 'and now summarise it' }, + ], + }], CFG) + assert.deepEqual(input, [ + { type: 'function_call_output', call_id: 'toolu_fixture_0001', output: '# codex-proxy' }, + { role: 'user', content: [{ type: 'input_text', text: 'and now summarise it' }] }, + ]) + }) + + test('a tool_result carrying content blocks is flattened to text', () => { + const input = toResponsesInput([{ + role: 'user', + content: [{ + type: 'tool_result', + tool_use_id: 'toolu_fixture_0003', + content: [{ type: 'text', text: 'line one' }, { type: 'text', text: 'line two' }], + }], + }], CFG) + assert.equal(input[0].output, 'line one\nline two') + }) + + test('an empty error tool_result still sends a non-empty output', () => { + // The Responses API rejects a function_call_output with an empty string, + // so an errored tool with no body has to say something. + const input = toResponsesInput([{ + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'toolu_fixture_0004', content: '', is_error: true }], + }], CFG) + assert.equal(input[0].output, 'error') + }) + + test('a full tool round trip keeps call ids paired across turns', () => { + const input = toResponsesInput([ + { role: 'user', content: 'read the readme' }, + { + role: 'assistant', + content: [{ type: 'tool_use', id: 'toolu_fixture_0005', name: 'Read', input: { file_path: 'README.md' } }], + }, + { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'toolu_fixture_0005', content: '# codex-proxy' }], + }, + ], CFG) + assert.deepEqual(input.map((i: any) => i.type ?? i.role), ['user', 'function_call', 'function_call_output']) + assert.equal(input[1].call_id, input[2].call_id) + }) +}) + +describe('reasoning carried through a thinking block', () => { + const item = { id: 'rs_fixture_0001', type: 'reasoning', encrypted_content: 'ZW5jcnlwdGVkLWJsb2I=' } + + test('encode then decode reproduces the upstream reasoning item', () => { + const block = encodeReasoning(item, CFG) + assert.equal(block?.type, 'thinking') + assert.equal(block?.thinking, REASONING_MARK) + assert.deepEqual(decodeReasoning(block, CFG), { + type: 'reasoning', + id: 'rs_fixture_0001', + encrypted_content: 'ZW5jcnlwdGVkLWJsb2I=', + summary: [], + }) + }) + + test('the marker is invisible: a zero-width space, then a tag', () => { + assert.equal(REASONING_MARK.codePointAt(0), 0x200b) + assert.equal(REASONING_MARK.slice(1), '') + }) + + test('a reasoning item without an encrypted blob encodes to nothing', () => { + assert.equal(encodeReasoning({ id: 'rs_fixture_0002', type: 'reasoning' }, CFG), null) + }) + + test('a real thinking block from another provider is not mistaken for ours', () => { + assert.equal(decodeReasoning({ type: 'thinking', thinking: 'Let me think...', signature: 'abc' }, CFG), null) + assert.equal(decodeReasoning({ type: 'text', text: 'hi' }, CFG), null) + }) + + test('a corrupt signature decodes to nothing instead of throwing', () => { + assert.equal(decodeReasoning({ type: 'thinking', thinking: REASONING_MARK, signature: 'not json' }, CFG), null) + }) + + test('the rebuilt reasoning item is hoisted ahead of the call it produced', () => { + // Responses rejects the pair in the other order. + const input = toResponsesInput([{ + role: 'assistant', + content: [ + encodeReasoning(item, CFG), + { type: 'text', text: 'Looking.' }, + { type: 'tool_use', id: 'toolu_fixture_0006', name: 'Read', input: {} }, + ], + }], CFG) + assert.deepEqual(input.map((i: any) => i.type ?? i.role), ['reasoning', 'assistant', 'function_call']) + }) + + test('with carrying off the marker block is dropped, not sent as text', () => { + assert.equal(encodeReasoning(item, NO_CARRY), null) + const input = toResponsesInput([{ role: 'assistant', content: [encodeReasoning(item, CFG)] }], NO_CARRY) + assert.deepEqual(input, []) + }) +}) + +describe('tool definitions', () => { + const tools = [ + { + name: 'Read', + description: 'Read a file', + input_schema: { type: 'object', properties: { file_path: { type: 'string' } }, required: ['file_path'] }, + }, + ] + + test('an Anthropic tool becomes a non-strict Responses function tool', () => { + assert.deepEqual(toResponsesTools(tools), [{ + type: 'function', + name: 'Read', + description: 'Read a file', + parameters: tools[0].input_schema, + // The client's schemas omit additionalProperties:false, so strict mode + // would reject every one of them. + strict: false, + }]) + }) + + test('a missing description becomes an empty string, never undefined', () => { + assert.equal(toResponsesTools([{ name: 'Bare', input_schema: { type: 'object' } }])?.[0].description, '') + }) + + test('an MCP-style tool nesting its schema under custom is still mapped', () => { + const out = toResponsesTools([{ name: 'mcp__srv__do', custom: { input_schema: { type: 'object' } } }]) + assert.equal(out?.[0].name, 'mcp__srv__do') + assert.deepEqual(out?.[0].parameters, { type: 'object' }) + }) + + test('a tool with no schema at all is filtered out', () => { + const out = toResponsesTools([{ name: 'NoSchema' }, ...tools]) + assert.equal(out?.length, 1) + assert.equal(out?.[0].name, 'Read') + }) + + test('nothing left after filtering means no tools key upstream', () => { + assert.equal(toResponsesTools([{ name: 'NoSchema' }]), undefined) + assert.equal(toResponsesTools([]), undefined) + assert.equal(toResponsesTools(undefined), undefined) + }) +}) + +describe('tool_choice', () => { + test('maps every Anthropic form to its Responses equivalent', () => { + assert.equal(toToolChoice({ type: 'auto' }), 'auto') + assert.equal(toToolChoice({ type: 'any' }), 'required') + assert.equal(toToolChoice({ type: 'none' }), 'none') + assert.deepEqual(toToolChoice({ type: 'tool', name: 'Read' }), { type: 'function', name: 'Read' }) + }) + + test('absent or unknown forms send nothing rather than a guess', () => { + assert.equal(toToolChoice(undefined), undefined) + assert.equal(toToolChoice({ type: 'something_new' }), undefined) + }) +}) + +describe('buildUpstream envelope', () => { + test('never stores the conversation upstream', () => { + assert.equal(buildUpstream({ model: 'codex-sol', messages: [] }, CFG).store, false) + }) + + test('stream is always a boolean, mirroring the request', () => { + assert.equal(buildUpstream({ model: 'codex-sol', messages: [] }, CFG).stream, false) + assert.equal(buildUpstream({ model: 'codex-sol', messages: [], stream: true }, CFG).stream, true) + }) + + test('max_tokens becomes max_output_tokens with a floor of 16', () => { + assert.equal(buildUpstream({ model: 'codex-sol', messages: [], max_tokens: 4096 }, CFG).max_output_tokens, 4096) + // The client probes with max_tokens:1; Responses rejects anything under 16. + assert.equal(buildUpstream({ model: 'codex-sol', messages: [], max_tokens: 1 }, CFG).max_output_tokens, 16) + assert.equal('max_output_tokens' in buildUpstream({ model: 'codex-sol', messages: [] }, CFG), false) + }) + + test('the reasoning knob is sent only to models that accept it', () => { + const reasoning = buildUpstream({ model: 'codex-sol', messages: [] }, CFG) + assert.deepEqual(reasoning.reasoning, { effort: 'medium' }) + assert.deepEqual(reasoning.include, ['reasoning.encrypted_content']) + + // The cheap fallback 400s on reasoning.effort. + const fallback = buildUpstream({ model: 'claude-haiku-4-5-20251001', messages: [] }, CFG) + assert.equal('reasoning' in fallback, false) + assert.equal('include' in fallback, false) + }) + + test('the configured effort reaches the request', () => { + assert.deepEqual( + buildUpstream({ model: 'codex-sol', messages: [] }, { ...CFG, effort: 'high' }).reasoning, + { effort: 'high' }, + ) + }) + + test('with carrying off the encrypted blob is not requested back', () => { + const req = buildUpstream({ model: 'codex-sol', messages: [] }, NO_CARRY) + assert.deepEqual(req.reasoning, { effort: 'medium' }) + assert.equal('include' in req, false) + }) + + test('temperature and other unmapped knobs are not forwarded blind', () => { + // Responses rejects temperature on gpt-5.6-*, so dropping it is the point. + const req = buildUpstream({ model: 'codex-sol', messages: [], temperature: 0.7, top_p: 0.9, metadata: {} }, CFG) + assert.equal('temperature' in req, false) + assert.equal('top_p' in req, false) + assert.equal('metadata' in req, false) + }) + + test('an empty body still produces a well-formed upstream request', () => { + const req = buildUpstream({}, CFG) + assert.equal(req.model, 'gpt-5.4-mini') + assert.deepEqual(req.input, []) + assert.equal(req.stream, false) + }) +}) diff --git a/test/translate-response.test.ts b/test/translate-response.test.ts new file mode 100644 index 0000000..d6943a6 --- /dev/null +++ b/test/translate-response.test.ts @@ -0,0 +1,188 @@ +/** + * Upstream -> downstream: a Responses API response body becomes an Anthropic + * non-streaming message. No network; the response bodies are literals. + */ +import assert from 'node:assert/strict' +import { describe, test } from 'node:test' + +import { + decodeReasoning, + safeParse, + stopReasonFrom, + toAnthropicMessage, + REASONING_MARK, + type Config, +} from '../codex-proxy.ts' + +const CFG: Config = { effort: 'medium', carryReasoning: true, fallbackModel: 'gpt-5.4-mini' } +const NO_CARRY: Config = { ...CFG, carryReasoning: false } +const WHO = { id: 'msg_fixture_0001', model: 'codex-sol' } + +const textResponse = { + id: 'resp_fixture_0001', + status: 'completed', + output: [{ + id: 'msg_upstream_0001', + type: 'message', + role: 'assistant', + content: [{ type: 'output_text', text: 'The readme is 69 lines.' }], + }], + usage: { input_tokens: 1204, output_tokens: 48 }, +} + +describe('safeParse', () => { + test('parses JSON, and never throws on rubbish', () => { + assert.deepEqual(safeParse('{"a":1}'), { a: 1 }) + // A truncated tool call must not take the whole response down. + assert.deepEqual(safeParse('{"a":'), {}) + assert.deepEqual(safeParse(''), {}) + }) +}) + +describe('stopReasonFrom', () => { + test('a completed answer ends the turn', () => { + assert.equal(stopReasonFrom({ status: 'completed', output: [{ type: 'message' }] }), 'end_turn') + }) + + test('an incomplete response reports the token cap', () => { + assert.equal(stopReasonFrom({ status: 'incomplete', output: [] }), 'max_tokens') + }) + + test('a pending tool call outranks everything else', () => { + // The client must run the tool; telling it max_tokens would end the turn. + assert.equal( + stopReasonFrom({ status: 'incomplete', output: [{ type: 'function_call' }] }), + 'tool_use', + ) + }) + + test('a response with no output at all still yields a valid stop reason', () => { + assert.equal(stopReasonFrom({}), 'end_turn') + assert.equal(stopReasonFrom(undefined), 'end_turn') + }) +}) + +describe('toAnthropicMessage', () => { + test('a plain answer becomes one text block with usage carried over', () => { + assert.deepEqual(toAnthropicMessage(textResponse, WHO, CFG), { + id: 'msg_fixture_0001', + type: 'message', + role: 'assistant', + model: 'codex-sol', + content: [{ type: 'text', text: 'The readme is 69 lines.' }], + stop_reason: 'end_turn', + stop_sequence: null, + usage: { input_tokens: 1204, output_tokens: 48 }, + }) + }) + + test('the model echoed back is the alias the client asked for', () => { + // The client validates the echo against what it sent; answering + // "gpt-5.6-sol" to a request for "codex-sol" confuses it. + assert.equal(toAnthropicMessage(textResponse, WHO, CFG).model, 'codex-sol') + }) + + test('a function_call becomes a tool_use block with parsed input', () => { + const msg = toAnthropicMessage({ + status: 'completed', + output: [{ + id: 'fc_upstream_0001', + type: 'function_call', + call_id: 'call_fixture_0001', + name: 'Read', + arguments: '{"file_path":"README.md"}', + }], + }, WHO, CFG) + assert.deepEqual(msg.content, [{ + type: 'tool_use', + // The client sends this id back as tool_use_id, and Responses matches + // function_call_output on call_id -- so it is call_id, not the item id. + id: 'call_fixture_0001', + name: 'Read', + input: { file_path: 'README.md' }, + }]) + assert.equal(msg.stop_reason, 'tool_use') + }) + + test('truncated tool arguments degrade to an empty input, not a crash', () => { + const msg = toAnthropicMessage({ + status: 'incomplete', + output: [{ type: 'function_call', call_id: 'call_fixture_0002', name: 'Read', arguments: '{"file_pa' }], + }, WHO, CFG) + assert.deepEqual(msg.content[0].input, {}) + }) + + test('text and a tool call keep their upstream order', () => { + const msg = toAnthropicMessage({ + status: 'completed', + output: [ + { type: 'message', content: [{ type: 'output_text', text: 'Reading it.' }] }, + { type: 'function_call', call_id: 'call_fixture_0003', name: 'Read', arguments: '{}' }, + ], + }, WHO, CFG) + assert.deepEqual(msg.content.map((b: any) => b.type), ['text', 'tool_use']) + }) + + test('several output_text parts in one message become separate blocks', () => { + const msg = toAnthropicMessage({ + output: [{ + type: 'message', + content: [ + { type: 'output_text', text: 'one' }, + { type: 'refusal', refusal: 'nope' }, + { type: 'output_text', text: 'two' }, + ], + }], + }, WHO, CFG) + assert.deepEqual(msg.content, [{ type: 'text', text: 'one' }, { type: 'text', text: 'two' }]) + }) + + test('a reasoning item survives as a thinking block the client will hand back', () => { + const msg = toAnthropicMessage({ + status: 'completed', + output: [ + { id: 'rs_fixture_0001', type: 'reasoning', encrypted_content: 'ZW5jcnlwdGVkLWJsb2I=' }, + { type: 'message', content: [{ type: 'output_text', text: 'done' }] }, + ], + }, WHO, CFG) + assert.equal(msg.content[0].type, 'thinking') + assert.equal(msg.content[0].thinking, REASONING_MARK) + // The whole point: the next request can rebuild the upstream item from it. + assert.deepEqual(decodeReasoning(msg.content[0], CFG), { + type: 'reasoning', + id: 'rs_fixture_0001', + encrypted_content: 'ZW5jcnlwdGVkLWJsb2I=', + summary: [], + }) + }) + + test('with carrying off the reasoning item is dropped from the answer', () => { + const msg = toAnthropicMessage({ + output: [ + { id: 'rs_fixture_0001', type: 'reasoning', encrypted_content: 'ZW5jcnlwdGVkLWJsb2I=' }, + { type: 'message', content: [{ type: 'output_text', text: 'done' }] }, + ], + }, WHO, NO_CARRY) + assert.deepEqual(msg.content, [{ type: 'text', text: 'done' }]) + }) + + test('an unsummarised reasoning item adds no empty block', () => { + const msg = toAnthropicMessage({ output: [{ id: 'rs_fixture_0002', type: 'reasoning', summary: [] }] }, WHO, CFG) + assert.deepEqual(msg.content, []) + }) + + test('missing usage reports zeros rather than undefined', () => { + const msg = toAnthropicMessage({ status: 'completed', output: [] }, WHO, CFG) + assert.deepEqual(msg.usage, { input_tokens: 0, output_tokens: 0 }) + assert.deepEqual(msg.content, []) + assert.equal(msg.stop_reason, 'end_turn') + }) + + test('an empty upstream body still produces a valid Anthropic message', () => { + const msg = toAnthropicMessage({}, WHO, CFG) + assert.equal(msg.type, 'message') + assert.equal(msg.role, 'assistant') + assert.equal(msg.stop_sequence, null) + assert.deepEqual(msg.content, []) + }) +})