diff --git a/packages/gateway/__tests__/data-plane/chat/messages/errors_test.ts b/packages/gateway/__tests__/data-plane/chat/messages/errors_test.ts index c3ecb6a3df..1ec9368328 100644 --- a/packages/gateway/__tests__/data-plane/chat/messages/errors_test.ts +++ b/packages/gateway/__tests__/data-plane/chat/messages/errors_test.ts @@ -26,7 +26,7 @@ test('translatorInputErrorResult renders an Anthropic 400 invalid_request_error type: 'invalid_request_error', message: "Invalid 'image_url' content part in system or developer message. Only 'text' content parts are supported in system messages on this model.", }); - assert(typeof body.request_id === 'string' && /^req_[A-Za-z0-9]{24}$/.test(body.request_id), `request_id ${String(body.request_id)} must match Anthropic-shape req_<24-base62>`); + assert(typeof body.request_id === 'string' && /^req_[A-Za-z0-9]{24}$/.test(body.request_id), `request_id ${String(body.request_id)} must match Anthropic-shape req_<24 opaque chars>`); }); test('translatorInputErrorResult preserves Anthropic key order: type, error, request_id', () => { diff --git a/packages/gateway/__tests__/data-plane/chat/messages/http_test.ts b/packages/gateway/__tests__/data-plane/chat/messages/http_test.ts index 9c1a639144..d34606f0e2 100644 --- a/packages/gateway/__tests__/data-plane/chat/messages/http_test.ts +++ b/packages/gateway/__tests__/data-plane/chat/messages/http_test.ts @@ -5,6 +5,7 @@ import type { AuthVars } from '../../../../src/middleware/auth.ts'; import { initRepo } from '../../../../src/repo/index.ts'; import type { ApiKey, User } from '../../../../src/repo/types.ts'; import { InMemoryRepo } from '../../../repo/memory.ts'; +import { flushBackground } from '../../../test-utils/background-tracker.ts'; import { doneFrame, eventFrame, type ModelEndpoints, type ProtocolFrame } from '@floway-dev/protocols/common'; import type { MessagesStreamEvent } from '@floway-dev/protocols/messages'; import { type ModelCandidate, directFetcher, type ProviderCallResult, type ProviderStreamResult, type UpstreamCallOptions } from '@floway-dev/provider'; @@ -161,6 +162,43 @@ test('POST /v1/messages returns a single JSON body when stream is omitted', asyn assertEquals(body.role, 'assistant'); }); +test('POST /v1/messages answers the Claude Code model-validation probe without calling the upstream', async () => { + const repo = installRepo(); + const callMessages = vi.fn((): Promise> => { + throw new Error('the probe reached the upstream'); + }); + queueCandidates([makeCandidate({ callMessages })]); + + const response = await makeApp().request('/v1/messages', { + method: 'POST', + headers: new Headers({ 'content-type': 'application/json', 'user-agent': 'claude-cli/2.1.226 (external, cli)' }), + body: JSON.stringify({ + model: 'test-model', + max_tokens: 1, + system: [{ type: 'text', text: "You are Claude Code, Anthropic's official CLI for Claude." }], + messages: [{ role: 'user', content: [{ type: 'text', text: 'Hi', cache_control: { type: 'ephemeral' } }] }], + metadata: { user_id: 'user_0_account__session_0' }, + }), + }); + + assertEquals(response.status, 200); + const body = await response.json() as { model: string; stop_reason: string; usage: { input_tokens: number; output_tokens: number } }; + assertEquals(body.model, 'test-model'); + assertEquals(body.stop_reason, 'max_tokens'); + assertEquals(body.usage.input_tokens, 0); + assertEquals(body.usage.output_tokens, 0); + assertEquals(callMessages.mock.calls.length, 0); + + // The turn is recorded as served, at zero cost, and contributes no latency + // sample — there was no upstream call to measure. + await flushBackground(); + const usage = await repo.usage.listAll(); + assertEquals(usage.length, 1); + assertEquals(usage[0]?.requests, 1); + assertEquals(usage[0]?.metrics, []); + assertEquals(await repo.performance.listAll(), []); +}); + test('POST /v1/messages rejects body anthropic_beta with a 400 before routing', async () => { installRepo(); // No candidates queued — the http entry rejects before reaching the serve. diff --git a/packages/gateway/__tests__/data-plane/chat/messages/interceptors/answer-claude-code-probe_test.ts b/packages/gateway/__tests__/data-plane/chat/messages/interceptors/answer-claude-code-probe_test.ts new file mode 100644 index 0000000000..81bc54ce6b --- /dev/null +++ b/packages/gateway/__tests__/data-plane/chat/messages/interceptors/answer-claude-code-probe_test.ts @@ -0,0 +1,148 @@ +import { test, vi } from 'vitest'; + +import { answerClaudeCodeProbe } from '../../../../../src/data-plane/chat/messages/interceptors/answer-claude-code-probe.ts'; +import type { MessagesInvocation } from '../../../../../src/data-plane/chat/messages/interceptors/types.ts'; +import { mockChatGatewayCtx } from '../../../../test-utils/gateway-ctx.ts'; +import type { ProtocolFrame } from '@floway-dev/protocols/common'; +import { collectMessagesProtocolEventsToResult, type MessagesPayload, type MessagesStreamEvent } from '@floway-dev/protocols/messages'; +import { type ExecuteResult, eventResult } from '@floway-dev/provider'; +import { assert, assertEquals, stubModelCandidate, testTelemetryModelIdentity } from '@floway-dev/test-utils'; + +const stubCtx = mockChatGatewayCtx(); + +// The CLI emits `claude-cli/ (external, …)`, where the +// entrypoint substitutes for `cli` and up to three optional segments append — +// which is why the predicate anchors on the `claude-cli/` prefix only. +// This is the default entrypoint form. +const PROBE_USER_AGENT = 'claude-cli/2.1.226 (external, cli)'; + +// The shape of the 2.1.226 `/model` validation probe: one user turn holding +// one ephemeral text block, `max_tokens: 1`, no tools. The real body also +// carries the CLI's system-prompt array and a `betas` list, neither of which +// the predicate looks at. +const probePayload = (overrides: Partial = {}): MessagesPayload => ({ + model: 'test-model', + max_tokens: 1, + system: [{ type: 'text', text: "You are Claude Code, Anthropic's official CLI for Claude." }], + messages: [{ role: 'user', content: [{ type: 'text', text: 'Hi', cache_control: { type: 'ephemeral' } }] }], + metadata: { user_id: 'user_0_account__session_0' }, + ...overrides, +}); + +const invocation = (payload: MessagesPayload, userAgent: string | null = PROBE_USER_AGENT): MessagesInvocation => ({ + payload, + candidate: stubModelCandidate({ model: { endpoints: { responses: {} } } }), + targetApi: 'responses', + headers: new Headers(userAgent === null ? {} : { 'user-agent': userAgent }), +}); + +const passthrough = async (): Promise>> => + eventResult((async function* (): AsyncGenerator> {})(), testTelemetryModelIdentity); + +const runProbe = async (input: MessagesInvocation) => { + const run = vi.fn(passthrough); + const result = await answerClaudeCodeProbe(input, stubCtx, run); + return { result, run }; +}; + +const assertForwarded = async (input: MessagesInvocation, message: string) => { + const { run } = await runProbe(input); + assertEquals(run.mock.calls.length, 1, message); +}; + +const assertAnswered = async (input: MessagesInvocation, message: string) => { + const { run } = await runProbe(input); + assertEquals(run.mock.calls.length, 0, message); +}; + +test('answers the /model validation probe without dialing the upstream', async () => { + const { result, run } = await runProbe(invocation(probePayload())); + + assertEquals(run.mock.calls.length, 0); + assert(result.type === 'events'); + const message = await collectMessagesProtocolEventsToResult(result.events); + assertEquals(message.type, 'message'); + assertEquals(message.role, 'assistant'); + assertEquals(message.model, 'test-model'); + assertEquals(message.content, []); + assertEquals(message.stop_reason, 'max_tokens'); + assertEquals(message.usage.input_tokens, 0); + assertEquals(message.usage.output_tokens, 0); + assert(message.id.startsWith('msg_')); +}); + +test('reports no performance context so the turn contributes no latency sample', async () => { + const { result } = await runProbe(invocation(probePayload())); + + assert(result.type === 'events'); + assertEquals(result.performance, undefined); + assertEquals(result.finalMetadata, undefined); +}); + +test('answers every fixed probe prompt, in block and bare-string form', async () => { + for (const prompt of ['Hi', 'test']) { + await assertAnswered(invocation(probePayload({ messages: [{ role: 'user', content: prompt }] })), `bare string: ${prompt}`); + await assertAnswered(invocation(probePayload({ messages: [{ role: 'user', content: [{ type: 'text', text: prompt }] }] })), `text block: ${prompt}`); + } +}); + +test('forwards the probe prompts the gateway cannot answer truthfully', async () => { + // `quota` reads rate-limit response headers a synthesized turn cannot carry; + // `.` only ever reaches a Bedrock / Vertex / Mantle base URL; `hello` is a + // third-party report's reconstruction that no observed build sends. + for (const prompt of ['quota', '.', 'hello']) { + await assertForwarded(invocation(probePayload({ messages: [{ role: 'user', content: prompt }] })), `prompt: ${prompt}`); + } +}); + +test('forwards a probe prompt whose casing the CLI has never been observed to send', async () => { + await assertForwarded(invocation(probePayload({ messages: [{ role: 'user', content: 'HI' }] })), 'uppercased'); +}); + +test('forwards a one-token request from a client that is not Claude Code', async () => { + await assertForwarded(invocation(probePayload(), 'python-httpx/0.28.1'), 'other client'); + await assertForwarded(invocation(probePayload(), null), 'no user-agent'); +}); + +test('forwards a Claude Code turn whose output cap is not one token', async () => { + await assertForwarded(invocation(probePayload({ max_tokens: 32_000 })), 'max_tokens 32000'); +}); + +test('forwards a Claude Code turn that carries the session tools', async () => { + await assertForwarded(invocation(probePayload({ + tools: [{ name: 'Bash', input_schema: { type: 'object' } }], + })), 'tools present'); +}); + +test('forwards a conversation whose sole turn is not one of the fixed prompts', async () => { + await assertForwarded(invocation(probePayload({ + messages: [{ role: 'user', content: 'Hi, can you explain this file?' }], + })), 'real question'); +}); + +test('forwards a multi-turn conversation that happens to end on a probe prompt', async () => { + await assertForwarded(invocation(probePayload({ + messages: [ + { role: 'user', content: 'Hi' }, + { role: 'assistant', content: 'Hello!' }, + { role: 'user', content: 'Hi' }, + ], + })), 'three turns'); +}); + +test('forwards a sole turn that carries more than one block', async () => { + await assertForwarded(invocation(probePayload({ + messages: [{ role: 'user', content: [{ type: 'text', text: 'Hi' }, { type: 'text', text: 'and explain this file' }] }], + })), 'two text blocks'); +}); + +test('forwards a sole turn that is not a user turn', async () => { + await assertForwarded(invocation(probePayload({ messages: [{ role: 'assistant', content: 'Hi' }] })), 'assistant role'); + await assertForwarded(invocation(probePayload({ messages: [{ role: 'system', content: 'Hi' }] })), 'system role'); +}); + +test('forwards a turn whose sole block is not text', async () => { + await assertForwarded(invocation(probePayload({ + messages: [{ role: 'user', content: [{ type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'AA==' } }] }], + })), 'image block'); +}); diff --git a/packages/gateway/src/data-plane/chat/messages/errors.ts b/packages/gateway/src/data-plane/chat/messages/errors.ts index 852c3109ad..9aa147e0a3 100644 --- a/packages/gateway/src/data-plane/chat/messages/errors.ts +++ b/packages/gateway/src/data-plane/chat/messages/errors.ts @@ -1,18 +1,10 @@ import { appendFailedUpstreams } from '../../shared/failed-upstreams.ts'; import type { ChatServeFailure } from '../shared/errors.ts'; import type { ProtocolFrame } from '@floway-dev/protocols/common'; -import type { MessagesStreamEvent } from '@floway-dev/protocols/messages'; +import { generateAnthropicId, type MessagesStreamEvent } from '@floway-dev/protocols/messages'; import type { ExecuteResult, PerformanceTelemetryContext } from '@floway-dev/provider'; import type { TranslatorInputError } from '@floway-dev/translate'; -// Mint an Anthropic-shaped synthetic request id (`req_` + 24 base62 chars) -// so a gateway-synthesized 4xx body carries the same top-level `request_id` -// field every real Anthropic response carries. The value is opaque to the -// caller; we never bridge it to an upstream id (these envelopes never -// reached an upstream). 24 chars from crypto.randomUUID yields ~96 bits of -// entropy, plenty for an opaque per-error id. -const mintAnthropicRequestId = (): string => `req_${crypto.randomUUID().replace(/-/g, '').slice(0, 24)}`; - // Anthropic Messages error envelope used to render pre-stream // `ChatServeFailure`s. These are gateway-synthesized rather than received // from any upstream — `source: 'gateway'` so the dump labels them as such. @@ -32,7 +24,7 @@ const anthropicErrorResult = ( body: new TextEncoder().encode(JSON.stringify({ type: 'error', error: { type, message }, - request_id: mintAnthropicRequestId(), + request_id: generateAnthropicId('req'), })), ...(performance ? { performance } : {}), }); diff --git a/packages/gateway/src/data-plane/chat/messages/interceptors/answer-claude-code-probe.ts b/packages/gateway/src/data-plane/chat/messages/interceptors/answer-claude-code-probe.ts new file mode 100644 index 0000000000..928cec19b0 --- /dev/null +++ b/packages/gateway/src/data-plane/chat/messages/interceptors/answer-claude-code-probe.ts @@ -0,0 +1,122 @@ +import type { MessagesInterceptor } from './types.ts'; +import { telemetryModelIdentity } from '../../../shared/telemetry/attribution.ts'; +import { doneFrame, eventFrame, type ProtocolFrame } from '@floway-dev/protocols/common'; +import { generateAnthropicId, type MessagesPayload, type MessagesStreamEvent } from '@floway-dev/protocols/messages'; +import { eventResult, providerModelOf, type ExecuteResult, type ModelCandidate } from '@floway-dev/provider'; + +// Claude Code answers "is this model usable?" by generating one token against +// it. `/model ` runs the CLI's `model_validation` side query — a +// non-streaming `POST /v1/messages?beta=true` with `max_tokens: 1` and a fixed +// throwaway prompt — and reports the model unusable if that call throws. +// Decompiled from the 2.1.226 binary: +// +// await eie({model:r, max_tokens:1, maxRetries:0, querySource:"model_validation", +// messages:[{role:"user",content:[{type:"text",text:"Hi", +// cache_control:{type:"ephemeral"}}]}]}), lra.set(r,!0), {valid:!0} +// +// The verdict is "did not throw". The validation path dereferences +// `usage.input_tokens` / `usage.output_tokens` unguarded to populate the CLI's +// own telemetry event, reads `stop_reason` and the cache counters behind `??`, +// and looks at nothing else. The CLI runs several other one-token probes with +// their own fixed prompts, each from an independent call site rather than +// through `eie` — `quota` for the rate-limit preflight, `test` for credential +// verification, `.` for the Bedrock / Vertex / Mantle reachability checks. +// +// A one-token cap is not portable. OpenAI's Responses API floors +// `max_output_tokens` at 16 and rejects anything lower with a hard 400, so +// every Messages→Responses candidate fails the probe and Claude Code +// concludes the model does not exist — the CLI surfaces the upstream envelope +// verbatim: `API error: 400 {"error":{"message":"Invalid 'max_output_tokens': +// integer below minimum value. Expected a value >= 16, but got 1 instead.", +// "code":"invalid_request_body"}}`. +// +// The gateway answers the probe itself instead. That is honest for what the +// probe actually asks: by the time this interceptor runs, model resolution has +// already picked a real candidate for the requested id, so an id no upstream +// serves still fails at the serve layer with a 404 and the CLI still reports +// it not found. What we suppress is only the pointless one-token generation +// behind it — no upstream call, and no tokens to bill. +const CLAUDE_CODE_USER_AGENT = /^claude-cli\/\d+\.\d+\.\d+/i; + +// The probes we can answer truthfully, written exactly as 2.1.226 sends them: +// `Hi` is the `/model` validation probe, `test` the credential check, which +// reads nothing at all off the response. +// +// Two recorded prompts are deliberately absent. `quota` is the rate-limit +// preflight, and its caller consumes the `anthropic-ratelimit-unified-*` +// response headers rather than the body — a synthesized turn cannot carry +// them, so answering it here would replace a working quota reading, on every +// upstream that serves the probe today, with a silent blank. `.` is issued by +// the Bedrock / Vertex / Mantle SDK clients, which route through their own +// base URLs and never reach an `ANTHROPIC_BASE_URL` gateway. +// +// Matching is exact, and every literal is read off a binary rather than +// inferred: a probe shape we have not observed should reach the upstream +// rather than be answered from a guess. The cost of keying on the prompt at +// all is that these literals are a Claude Code build detail — a report against +// v2.1.220 records the validation prompt as `hello` +// (https://github.com/BerriAI/litellm/issues/35061), a spelling absent from +// 2.1.226 — so a release that renames one re-exposes the 400 until the new +// literal is read off that build and added here. +const PROBE_PROMPTS: ReadonlySet = new Set(['Hi', 'test']); + +// The whole conversation of a probe. `model_validation` sends its prompt as a +// single ephemeral text block and the credential check sends the bare-string +// form; both shapes are current. Anything longer is a real turn. +const soleUserPromptOf = (payload: MessagesPayload): string | null => { + if (payload.messages.length !== 1) return null; + const message = payload.messages[0]!; + if (message.role !== 'user') return null; + const { content } = message; + if (typeof content === 'string') return content; + if (content.length !== 1) return null; + const block = content[0]!; + return block.type === 'text' ? block.text : null; +}; + +const isClaudeCodeProbe = (payload: MessagesPayload, headers: Headers): boolean => { + if (payload.max_tokens !== 1) return false; + // Every real Claude Code turn ships the session's tools; the one-token + // probes never do. + if (payload.tools !== undefined) return false; + const userAgent = headers.get('user-agent'); + if (userAgent === null || !CLAUDE_CODE_USER_AGENT.test(userAgent)) return false; + const prompt = soleUserPromptOf(payload); + return prompt !== null && PROBE_PROMPTS.has(prompt); +}; + +// A turn that stopped at the caller's one-token cap before emitting anything: +// no content blocks, `stop_reason: 'max_tokens'`, and a zero usage block. The +// usage block is load-bearing — the validation path dereferences +// `usage.input_tokens` / `usage.output_tokens` unconditionally, and a missing +// `usage` throws inside the CLI and reads as a failed probe. +const probeFrames = async function* (model: string): AsyncGenerator> { + yield eventFrame({ + type: 'message_start', + message: { + id: generateAnthropicId('msg'), + type: 'message', + role: 'assistant', + model, + content: [], + stop_reason: null, + stop_sequence: null, + usage: { input_tokens: 0, output_tokens: 0 }, + }, + }); + yield eventFrame({ type: 'message_delta', delta: { stop_reason: 'max_tokens', stop_sequence: null }, usage: { output_tokens: 0 } }); + yield eventFrame({ type: 'message_stop' }); + yield doneFrame(); +}; + +// No `performance` context on the result: `settle` reads it to decide whether +// the turn contributes a latency sample, and a turn that never dialed the +// upstream has no latency to report. The usage row still lands, at zero, so +// the request itself stays visible in the dashboard. +const probeResult = (candidate: ModelCandidate, model: string): ExecuteResult> => + eventResult(probeFrames(model), telemetryModelIdentity(candidate, providerModelOf(candidate).id)); + +export const answerClaudeCodeProbe: MessagesInterceptor = async (ctx, _gatewayCtx, run) => + isClaudeCodeProbe(ctx.payload, ctx.headers) + ? probeResult(ctx.candidate, ctx.payload.model) + : await run(); diff --git a/packages/gateway/src/data-plane/chat/messages/interceptors/index.ts b/packages/gateway/src/data-plane/chat/messages/interceptors/index.ts index 3ce6010815..30d754d9b3 100644 --- a/packages/gateway/src/data-plane/chat/messages/interceptors/index.ts +++ b/packages/gateway/src/data-plane/chat/messages/interceptors/index.ts @@ -1,3 +1,4 @@ +import { answerClaudeCodeProbe } from './answer-claude-code-probe.ts'; import { withRoleCompatibilityApplied } from './apply-role-compatibility.ts'; import { withReasoningDisabledOnForcedToolChoice } from './disable-reasoning-on-forced-tool-choice.ts'; import { stripBillingAttribution } from './strip-billing-attribution.ts'; @@ -5,13 +6,19 @@ import type { MessagesCountTokensInterceptor, MessagesInterceptor, MessagesPaylo import { withMessagesWebSearchRequestPrepared, withMessagesWebSearchShim } from './web-search-shim.ts'; // Unified Messages generation chain. All entries are attached to every -// candidate; each interceptor decides whether to act from the candidate's -// model flags and selected target. +// candidate; each interceptor decides whether to act — from the candidate's +// model flags and selected target, or, for the client-compatibility entry, +// from the inbound request itself. // // Translated requests re-enter the selected target protocol's chain. The role // compatibility entry therefore acts only when Messages is the final target. // -// - withMessagesWebSearchShim: registered first so its request preparation, +// - answerClaudeCodeProbe: registered ahead of everything else because it +// replaces the turn rather than shaping it — no payload transform below it +// can matter to a turn that never dials. Unconditional: whether a request +// is one of Claude Code's one-token probes is a property of the request, +// not of the candidate. +// - withMessagesWebSearchShim: registered next so its request preparation, // replay rewrite, and intercept loop wrap the rest of the generation chain. // Unconditional for translated targets (Responses / Chat Completions cannot // carry Anthropic server tools); gated by `messages-web-search-shim` for @@ -41,6 +48,7 @@ const messagesPayloadInterceptors: readonly MessagesPayloadInterceptor[] = [ ]; export const messagesInterceptors: readonly MessagesInterceptor[] = [ + answerClaudeCodeProbe, withMessagesWebSearchShim, ...messagesPayloadInterceptors, ]; diff --git a/packages/protocols/src/messages/id.ts b/packages/protocols/src/messages/id.ts new file mode 100644 index 0000000000..51bb6a7ac4 --- /dev/null +++ b/packages/protocols/src/messages/id.ts @@ -0,0 +1,15 @@ +import { encodeHex } from '../common/base-encoding.ts'; + +// Anthropic identifies a message with `msg_` and a request with `req_`, +// followed by an opaque token. Bodies that no upstream produced — a +// gateway-synthesized error envelope, a turn a gateway answered itself — still +// have to carry those fields, so their ids are generated here rather than +// bridged from an upstream: by construction they name something that never +// reached one. 12 random bytes render as the 24-character token length +// Anthropic emits. +// https://platform.claude.com/docs/en/api/messages +export const generateAnthropicId = (prefix: 'msg' | 'req'): string => { + const bytes = new Uint8Array(12); + crypto.getRandomValues(bytes); + return `${prefix}_${encodeHex(bytes)}`; +}; diff --git a/packages/protocols/src/messages/index.ts b/packages/protocols/src/messages/index.ts index 707cf24383..1eeb2ba1bf 100644 --- a/packages/protocols/src/messages/index.ts +++ b/packages/protocols/src/messages/index.ts @@ -408,6 +408,7 @@ export const parseAnthropicBetaHeader = (raw: string | null | undefined): readon raw ? raw.split(',').map(part => part.trim()).filter(part => part.length > 0) : []; export { MESSAGES_MISSING_TERMINAL_MESSAGE, collectMessagesProtocolEventsToResult } from './to-result.ts'; +export { generateAnthropicId } from './id.ts'; export { reassembleMessagesEvents } from './reassemble.ts'; export { messagesProtocolFrameToSSEFrame } from './to-sse.ts'; export { PROMPT_TOO_LONG_MESSAGE, buildPromptTooLongBody } from './context-window-error.ts';