From 279ecee2e09f05bb022f6acac2d6679cb781b50b Mon Sep 17 00:00:00 2001 From: Roomote Date: Mon, 28 Sep 2026 00:01:55 +0000 Subject: [PATCH] fix(api): sanitize lone UTF-16 surrogates in OpenAI-chat request bodies DeepSeek rejects the entire request body when any string contains a lone UTF-16 surrogate (e.g. after history text was sliced through an astral-plane character), permanently breaking the affected task. Apply the same sanitization pattern PR #1605 added for the VS Code LM provider to the shared OpenAI-chat transforms, covering the DeepSeek and OpenAI providers (and other providers on the same shared path): - Move sanitizeSurrogates/sanitizeIdentifierSurrogates/ sanitizeSurrogatesDeep into a shared transform module; vscode-lm-format re-exports them. - convertToOpenAiMessages/convertToR1Format: sanitize message text, tool_result content, tool_use ids (injective, composed with normalizeToolCallId so call/result pairing survives), tool names, nested tool argument strings, and reasoning_content. - OpenAiHandler: sanitize the system/developer prompt. - BaseProvider.convertToolsForOpenAI: sanitize tool definition name/description/parameters. completePrompt is intentionally out of scope, matching #1605. Fixes #461 --- .../providers/__tests__/base-provider.spec.ts | 44 ++++++ src/api/providers/__tests__/deepseek.spec.ts | 67 +++++++++ src/api/providers/__tests__/openai.spec.ts | 69 +++++++++ src/api/providers/base-provider.ts | 11 +- src/api/providers/openai.ts | 11 +- .../transform/__tests__/openai-format.spec.ts | 128 +++++++++++++++++ src/api/transform/__tests__/r1-format.spec.ts | 135 ++++++++++++++++++ .../__tests__/sanitize-surrogates.spec.ts | 89 ++++++++++++ src/api/transform/openai-format.ts | 35 +++-- src/api/transform/r1-format.ts | 47 +++--- src/api/transform/sanitize-surrogates.ts | 75 ++++++++++ src/api/transform/vscode-lm-format.ts | 78 ++-------- 12 files changed, 686 insertions(+), 103 deletions(-) create mode 100644 src/api/transform/__tests__/sanitize-surrogates.spec.ts create mode 100644 src/api/transform/sanitize-surrogates.ts diff --git a/src/api/providers/__tests__/base-provider.spec.ts b/src/api/providers/__tests__/base-provider.spec.ts index ced452f5a5..0d210ba1e2 100644 --- a/src/api/providers/__tests__/base-provider.spec.ts +++ b/src/api/providers/__tests__/base-provider.spec.ts @@ -267,6 +267,50 @@ describe("BaseProvider", () => { expect(result?.[0].function.parameters.additionalProperties).toBeUndefined() }) + it("should sanitize lone UTF-16 surrogates in name, description, and parameters (#461)", () => { + const tools = [ + { + type: "function", + function: { + name: "read_file", + description: "bad\uD800end", + parameters: { + type: "object", + properties: { + path: { type: "string", description: "bad\uDC00end" }, + }, + }, + }, + }, + ] + + const result = provider.testConvertToolsForOpenAI(tools) + + expect(result?.[0].function.description).toBe("bad\uFFFDend") + expect(result?.[0].function.parameters.properties.path.description).toBe("bad\uFFFDend") + // The serialized body must not contain a lone surrogate anywhere. + expect(JSON.stringify(result)).not.toMatch( + /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + const tools = [ + { + type: "function", + function: { + name: "read\uD800file", + description: "Read a file", + parameters: { type: "object", properties: {} }, + }, + }, + ] + + const result = provider.testConvertToolsForOpenAI(tools) + + expect(result?.[0].function.name).toBe("read\uFFFDfile") + }) + it("should preserve non-function tools unchanged", () => { const tools = [ { diff --git a/src/api/providers/__tests__/deepseek.spec.ts b/src/api/providers/__tests__/deepseek.spec.ts index 4ab247b131..712a84cffa 100644 --- a/src/api/providers/__tests__/deepseek.spec.ts +++ b/src/api/providers/__tests__/deepseek.spec.ts @@ -824,3 +824,70 @@ describe("DeepSeekHandler", () => { }) }) }) + +describe("DeepSeekHandler lone surrogate sanitization (#461)", () => { + // Matches any unpaired UTF-16 code unit; the serialized request body must never contain one. + const LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + mockCreate.mockClear() + }) + + const surrogateOptions: ApiHandlerOptions = { + deepSeekApiKey: "test-api-key", + apiModelId: "deepseek-v4-flash", + deepSeekBaseUrl: "https://api.deepseek.com", + } + + it("sends a request body free of lone surrogates, keeping tool call/result pairing intact", async () => { + const lone = "bad\uD800end" + const sanitized = "bad\uFFFDend" + const handler = new DeepSeekHandler(surrogateOptions) + const messages: Anthropic.Messages.MessageParam[] = [ + { role: "user", content: lone }, + { + role: "assistant", + content: [{ type: "tool_use", id: "call-\uD800", name: "read_file", input: { path: lone } }], + }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "call-\uD800", content: lone }] }, + ] + + await collectStream( + handler.createMessage("system \uD800 prompt", messages, { + taskId: "task-1", + tools: [ + { + type: "function", + function: { + name: "read_file", + description: lone, + parameters: { + type: "object", + properties: { path: { type: "string", description: lone } }, + }, + }, + }, + ], + }), + ) + + expect(mockCreate).toHaveBeenCalledOnce() + const request = mockCreate.mock.calls[0][0] + + // The whole body (messages + tools) must serialize without a lone surrogate. + expect(JSON.stringify(request)).not.toMatch(LONE_SURROGATE) + + // The tool call and its result stay paired after injective id sanitization. + const assistantMessage = request.messages.find( + (message: { role: string }) => message.role === "assistant", + ) as OpenAI.Chat.ChatCompletionAssistantMessageParam + const toolMessage = request.messages.find( + (message: { role: string }) => message.role === "tool", + ) as OpenAI.Chat.ChatCompletionToolMessageParam + const toolCalls = assistantMessage.tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + expect(toolMessage.tool_call_id).toBe(toolCalls[0].id) + expect(toolMessage.content).toBe(sanitized) + expect(JSON.parse(toolCalls[0].function.arguments)).toEqual({ path: sanitized }) + expect(request.tools[0].function.description).toBe(sanitized) + }) +}) diff --git a/src/api/providers/__tests__/openai.spec.ts b/src/api/providers/__tests__/openai.spec.ts index 754d57a6cd..e998ca43fa 100644 --- a/src/api/providers/__tests__/openai.spec.ts +++ b/src/api/providers/__tests__/openai.spec.ts @@ -1877,3 +1877,72 @@ describe("getOpenAiModels", () => { expect(result).toEqual(["gpt-4", "gpt-3.5-turbo"]) }) }) + +describe("OpenAiHandler lone surrogate sanitization (#461)", () => { + // Matches any unpaired UTF-16 code unit; the serialized request body must never contain one. + const LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + mockCreate.mockClear() + }) + + it("sends a request body free of lone surrogates, keeping tool call/result pairing intact", async () => { + const lone = "bad\uD800end" + const sanitized = "bad\uFFFDend" + const handler = new OpenAiHandler( + makeApiHandlerOptions({ + openAiApiKey: "test-api-key", + openAiModelId: "gpt-4", + openAiBaseUrl: "https://api.openai.com/v1", + }), + ) + const messages: Anthropic.Messages.MessageParam[] = [ + { role: "user", content: lone }, + { + role: "assistant", + content: [{ type: "tool_use", id: "call-\uD800", name: "read_file", input: { path: lone } }], + }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "call-\uD800", content: lone }] }, + ] + + await collectStream( + handler.createMessage("system \uD800 prompt", messages, { + taskId: "task-1", + tools: [ + { + type: "function", + function: { + name: "read_file", + description: lone, + parameters: { + type: "object", + properties: { path: { type: "string", description: lone } }, + }, + }, + }, + ], + }), + ) + + expect(mockCreate).toHaveBeenCalledOnce() + const request = mockCreate.mock.calls[0][0] + + // The whole body (system prompt, messages, tools) must serialize without a lone surrogate. + expect(JSON.stringify(request)).not.toMatch(LONE_SURROGATE) + + expect(request.messages[0]).toEqual({ role: "system", content: "system \uFFFD prompt" }) + + // The tool call and its result stay paired after injective id sanitization. + const assistantMessage = request.messages.find( + (message: { role: string }) => message.role === "assistant", + ) as OpenAI.Chat.ChatCompletionAssistantMessageParam + const toolMessage = request.messages.find( + (message: { role: string }) => message.role === "tool", + ) as OpenAI.Chat.ChatCompletionToolMessageParam + const toolCalls = assistantMessage.tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + expect(toolMessage.tool_call_id).toBe(toolCalls[0].id) + expect(toolMessage.content).toBe(sanitized) + expect(JSON.parse(toolCalls[0].function.arguments)).toEqual({ path: sanitized }) + expect(request.tools[0].function.description).toBe(sanitized) + }) +}) diff --git a/src/api/providers/base-provider.ts b/src/api/providers/base-provider.ts index 89366fb619..53a2ac9d3f 100644 --- a/src/api/providers/base-provider.ts +++ b/src/api/providers/base-provider.ts @@ -4,6 +4,7 @@ import type { ModelInfo } from "@roo-code/types" import type { ApiHandler, ApiHandlerCreateMessageMetadata } from "../index" import { ApiStream } from "../transform/stream" +import { sanitizeSurrogates, sanitizeSurrogatesDeep } from "../transform/sanitize-surrogates" import { countTokens } from "../../utils/countTokens" import { isMcpTool } from "../../utils/mcp-name" import { getApiRequestTimeout } from "./utils/timeout-config" @@ -45,10 +46,14 @@ export abstract class BaseProvider implements ApiHandler { ...tool, function: { ...tool.function, + // Sanitize lone UTF-16 surrogates: providers validating the JSON body + // (e.g. DeepSeek) reject the whole request otherwise. See #461. + name: sanitizeSurrogates(tool.function.name), + description: sanitizeSurrogates(tool.function.description), strict: !isMcp, - parameters: isMcp - ? tool.function.parameters - : this.convertToolSchemaForOpenAI(tool.function.parameters), + parameters: sanitizeSurrogatesDeep( + isMcp ? tool.function.parameters : this.convertToolSchemaForOpenAI(tool.function.parameters), + ), }, } }) diff --git a/src/api/providers/openai.ts b/src/api/providers/openai.ts index 619d05d28a..cc00566f88 100644 --- a/src/api/providers/openai.ts +++ b/src/api/providers/openai.ts @@ -18,6 +18,7 @@ import type { ApiHandlerOptions } from "../../shared/api" import { TagMatcher } from "../../utils/tag-matcher" import { convertToOpenAiMessages } from "../transform/openai-format" +import { sanitizeSurrogates } from "../transform/sanitize-surrogates" import { convertToR1Format } from "../transform/r1-format" import { ApiStream, ApiStreamUsageChunk } from "../transform/stream" import { getModelParams } from "../transform/model-params" @@ -101,7 +102,9 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl let systemMessage: OpenAI.Chat.ChatCompletionSystemMessageParam = { role: "system", - content: systemPrompt, + // Sanitize lone UTF-16 surrogates: providers validating the JSON body + // (e.g. DeepSeek) reject the whole request otherwise. See #461. + content: sanitizeSurrogates(systemPrompt), } if (this.options.openAiStreamingEnabled ?? true) { @@ -116,7 +119,7 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl content: [ { type: "text", - text: systemPrompt, + text: sanitizeSurrogates(systemPrompt), // @ts-ignore-next-line cache_control: { type: "ephemeral" }, }, @@ -365,7 +368,7 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl messages: [ { role: "developer", - content: `Formatting re-enabled\n${systemPrompt}`, + content: sanitizeSurrogates(`Formatting re-enabled\n${systemPrompt}`), }, ...convertToOpenAiMessages(messages), ], @@ -402,7 +405,7 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl messages: [ { role: "developer", - content: `Formatting re-enabled\n${systemPrompt}`, + content: sanitizeSurrogates(`Formatting re-enabled\n${systemPrompt}`), }, ...convertToOpenAiMessages(messages), ], diff --git a/src/api/transform/__tests__/openai-format.spec.ts b/src/api/transform/__tests__/openai-format.spec.ts index 214531312c..072c37c6d9 100644 --- a/src/api/transform/__tests__/openai-format.spec.ts +++ b/src/api/transform/__tests__/openai-format.spec.ts @@ -1571,3 +1571,131 @@ describe("sanitizeGeminiMessages", () => { expect(result).toEqual(messages) }) }) + +describe("convertToOpenAiMessages lone surrogate sanitization (#461)", () => { + const lone = "bad\uD800end" + const sanitized = "bad\uFFFDend" + + it("sanitizes a simple string message", () => { + const result = convertToOpenAiMessages([{ role: "user", content: lone }]) + expect(result[0]).toEqual({ role: "user", content: sanitized }) + }) + + it("sanitizes user text blocks", () => { + const result = convertToOpenAiMessages([{ role: "user", content: [{ type: "text", text: lone }] }]) + expect(result[0]).toEqual({ role: "user", content: [{ type: "text", text: sanitized }] }) + }) + + it("sanitizes string tool_result content", () => { + const result = convertToOpenAiMessages([ + { role: "user", content: [{ type: "tool_result", tool_use_id: "tool-1", content: lone }] }, + ]) + expect(result[0]).toEqual({ role: "tool", tool_call_id: "tool-1", content: sanitized }) + }) + + it("sanitizes tool_result text blocks", () => { + const result = convertToOpenAiMessages([ + { + role: "user", + content: [{ type: "tool_result", tool_use_id: "tool-1", content: [{ type: "text", text: lone }] }], + }, + ]) + expect(result[0]).toEqual({ role: "tool", tool_call_id: "tool-1", content: sanitized }) + }) + + it("sanitizes assistant text blocks", () => { + const result = convertToOpenAiMessages([{ role: "assistant", content: [{ type: "text", text: lone }] }]) + expect(result[0]).toMatchObject({ role: "assistant", content: sanitized }) + }) + + it("sanitizes strings nested in tool_use input", () => { + const result = convertToOpenAiMessages([ + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "tool-1", + name: "read_file", + input: { path: lone, nested: { list: [lone] } }, + }, + ], + }, + ]) + const toolCalls = (result[0] as OpenAI.Chat.ChatCompletionAssistantMessageParam) + .tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + expect(JSON.parse(toolCalls[0].function.arguments)).toEqual({ path: sanitized, nested: { list: [sanitized] } }) + }) + + it("sanitizes tool_use name", () => { + const result = convertToOpenAiMessages([ + { role: "assistant", content: [{ type: "tool_use", id: "tool-1", name: `read${lone}`, input: {} }] }, + ]) + const toolCalls = (result[0] as OpenAI.Chat.ChatCompletionAssistantMessageParam) + .tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + expect(toolCalls[0].function.name).toBe(`read${sanitized}`) + }) + + it("sanitizes tool_use ids injectively and keeps them paired with tool_result ids", () => { + const loneIdA = "call-\uD800" + const loneIdB = "call-\uD801" + const result = convertToOpenAiMessages([ + { + role: "assistant", + content: [ + { type: "tool_use", id: loneIdA, name: "read_file", input: {} }, + { type: "tool_use", id: loneIdB, name: "read_file", input: {} }, + ], + }, + { + role: "user", + content: [ + { type: "tool_result", tool_use_id: loneIdA, content: "ok" }, + { type: "tool_result", tool_use_id: loneIdB, content: "ok" }, + ], + }, + ]) + const toolCalls = (result[0] as OpenAI.Chat.ChatCompletionAssistantMessageParam) + .tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + const toolMessages = result.slice(1) as OpenAI.Chat.ChatCompletionToolMessageParam[] + // Injective: the two ids must not collapse onto the same replacement. + expect(toolCalls[0].id).not.toBe(toolCalls[1].id) + // Pairing: each tool result addresses its own call after sanitization. + expect(toolMessages[0].tool_call_id).toBe(toolCalls[0].id) + expect(toolMessages[1].tool_call_id).toBe(toolCalls[1].id) + }) + + it("composes id sanitization with normalizeToolCallId", () => { + const result = convertToOpenAiMessages( + [{ role: "user", content: [{ type: "tool_result", tool_use_id: "call-\uD800", content: "ok" }] }], + { normalizeToolCallId: (id) => id.toUpperCase() }, + ) + expect((result[0] as OpenAI.Chat.ChatCompletionToolMessageParam).tool_call_id).toBe("CALL-\uFFFD" + "D800") + }) + + it("sanitizes reasoning_content pass-through", () => { + const result = convertToOpenAiMessages([ + Object.assign({ role: "assistant" as const, content: lone }, { reasoning_content: lone }), + ]) + expect(result[0]).toMatchObject({ content: sanitized, reasoning_content: sanitized }) + }) + + it("leaves valid surrogate pairs untouched", () => { + const pair = "emoji \uD83D\uDE00 done" + const result = convertToOpenAiMessages([{ role: "user", content: pair }]) + expect(result[0]).toEqual({ role: "user", content: pair }) + }) + + it("produces a JSON body free of lone surrogates", () => { + const result = convertToOpenAiMessages([ + { role: "user", content: lone }, + { + role: "assistant", + content: [{ type: "tool_use", id: "call-\uD800", name: "read_file", input: { path: lone } }], + }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "call-\uD800", content: lone }] }, + ]) + const body = JSON.stringify(result) + expect(body).not.toMatch(/[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { }) }) }) + +describe("convertToR1Format lone surrogate sanitization (#461)", () => { + const lone = "bad\uD800end" + const sanitized = "bad\uFFFDend" + + it("sanitizes a simple string message (including the system prompt user message)", () => { + const result = convertToR1Format([{ role: "user", content: lone }]) + expect(result[0]).toEqual({ role: "user", content: sanitized }) + }) + + it("sanitizes user text blocks", () => { + const result = convertToR1Format([{ role: "user", content: [{ type: "text", text: lone }] }]) + expect(result[0]).toEqual({ role: "user", content: sanitized }) + }) + + it("sanitizes string tool_result content", () => { + const result = convertToR1Format([ + { role: "user", content: [{ type: "tool_result", tool_use_id: "tool-1", content: lone }] }, + ]) + expect(result[0]).toEqual({ role: "tool", tool_call_id: "tool-1", content: sanitized }) + }) + + it("sanitizes tool_result text blocks", () => { + const result = convertToR1Format([ + { + role: "user", + content: [{ type: "tool_result", tool_use_id: "tool-1", content: [{ type: "text", text: lone }] }], + }, + ]) + expect(result[0]).toEqual({ role: "tool", tool_call_id: "tool-1", content: sanitized }) + }) + + it("sanitizes assistant text blocks and reasoning blocks", () => { + const result = convertToR1Format([ + { + role: "assistant", + content: [ + { type: "text", text: lone }, + { type: "reasoning", text: lone } as unknown as Anthropic.Messages.ContentBlockParam, + ], + }, + ]) + expect(result[0]).toMatchObject({ role: "assistant", content: sanitized, reasoning_content: sanitized }) + }) + + it("sanitizes top-level reasoning_content", () => { + const message = Object.assign({ role: "assistant" as const, content: "ok" }, { reasoning_content: lone }) + const result = convertToR1Format([message]) + expect((result[0] as { reasoning_content?: string }).reasoning_content).toBe(sanitized) + }) + + it("sanitizes strings nested in tool_use input", () => { + const result = convertToR1Format([ + { + role: "assistant", + content: [ + { + type: "tool_use", + id: "tool-1", + name: "read_file", + input: { path: lone, nested: { list: [lone] } }, + }, + ], + }, + ]) + const toolCalls = (result[0] as OpenAI.Chat.ChatCompletionAssistantMessageParam) + .tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + expect(JSON.parse(toolCalls[0].function.arguments)).toEqual({ path: sanitized, nested: { list: [sanitized] } }) + }) + + it("sanitizes tool_use name", () => { + const result = convertToR1Format([ + { role: "assistant", content: [{ type: "tool_use", id: "tool-1", name: `read${lone}`, input: {} }] }, + ]) + const toolCalls = (result[0] as OpenAI.Chat.ChatCompletionAssistantMessageParam) + .tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + expect(toolCalls[0].function.name).toBe(`read${sanitized}`) + }) + + it("sanitizes tool_use ids injectively and keeps them paired with tool_result ids", () => { + const loneIdA = "call-\uD800" + const loneIdB = "call-\uD801" + const result = convertToR1Format([ + { + role: "assistant", + content: [ + { type: "tool_use", id: loneIdA, name: "read_file", input: {} }, + { type: "tool_use", id: loneIdB, name: "read_file", input: {} }, + ], + }, + { + role: "user", + content: [ + { type: "tool_result", tool_use_id: loneIdA, content: "ok" }, + { type: "tool_result", tool_use_id: loneIdB, content: "ok" }, + ], + }, + ]) + const toolCalls = (result[0] as OpenAI.Chat.ChatCompletionAssistantMessageParam) + .tool_calls as OpenAI.Chat.ChatCompletionMessageFunctionToolCall[] + const toolMessages = result.slice(1) as OpenAI.Chat.ChatCompletionToolMessageParam[] + // Injective: the two ids must not collapse onto the same replacement. + expect(toolCalls[0].id).not.toBe(toolCalls[1].id) + // Pairing: each tool result addresses its own call after sanitization. + expect(toolMessages[0].tool_call_id).toBe(toolCalls[0].id) + expect(toolMessages[1].tool_call_id).toBe(toolCalls[1].id) + }) + + it("composes id sanitization with the normalizeToolCallId option", () => { + const result = convertToR1Format( + [{ role: "user", content: [{ type: "tool_result", tool_use_id: "call-\uD800", content: "ok" }] }], + { normalizeToolCallId: (id) => id.toUpperCase() }, + ) + expect((result[0] as OpenAI.Chat.ChatCompletionToolMessageParam).tool_call_id).toBe("CALL-\uFFFD" + "D800") + }) + + it("leaves valid surrogate pairs untouched", () => { + const pair = "emoji \uD83D\uDE00 done" + const result = convertToR1Format([{ role: "user", content: pair }]) + expect(result[0]).toEqual({ role: "user", content: pair }) + }) + + it("produces a JSON body free of lone surrogates", () => { + const result = convertToR1Format([ + { role: "user", content: lone }, + { + role: "assistant", + content: [{ type: "tool_use", id: "call-\uD800", name: "read_file", input: { path: lone } }], + }, + { role: "user", content: [{ type: "tool_result", tool_use_id: "call-\uD800", content: lone }] }, + ]) + const body = JSON.stringify(result) + expect(body).not.toMatch(/[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? { + it("leaves plain ASCII unchanged", () => { + expect(sanitizeSurrogates("hello world")).toBe("hello world") + }) + + it("leaves valid surrogate pairs unchanged", () => { + // 😀 U+1F600 and 𐀀 U+10000 are astral-plane code points encoded as surrogate pairs. + expect(sanitizeSurrogates("a\uD83D\uDE00b\uD800\uDC00c")).toBe("a\uD83D\uDE00b\uD800\uDC00c") + }) + + it("replaces a lone high surrogate with U+FFFD", () => { + expect(sanitizeSurrogates("a\uD800b")).toBe("a\uFFFDb") + }) + + it("replaces a lone low surrogate with U+FFFD", () => { + expect(sanitizeSurrogates("a\uDC00b")).toBe("a\uFFFDb") + }) + + it("replaces a trailing lone high surrogate", () => { + expect(sanitizeSurrogates("abc\uD800")).toBe("abc\uFFFD") + }) + + it("replaces a reversed (low-then-high) pair as two lone surrogates", () => { + expect(sanitizeSurrogates("\uDC00\uD800")).toBe("\uFFFD\uFFFD") + }) + + it("returns empty input unchanged", () => { + expect(sanitizeSurrogates("")).toBe("") + }) +}) + +describe("sanitizeIdentifierSurrogates", () => { + it("leaves ordinary ids unchanged", () => { + expect(sanitizeIdentifierSurrogates("toolu_01ABCDEF")).toBe("toolu_01ABCDEF") + }) + + it("leaves valid surrogate pairs unchanged", () => { + expect(sanitizeIdentifierSurrogates("call-\uD83D\uDE00")).toBe("call-\uD83D\uDE00") + }) + + it("keeps ids differing only in a lone surrogate distinct (injective)", () => { + const a = sanitizeIdentifierSurrogates("call-\uD800") + const b = sanitizeIdentifierSurrogates("call-\uD801") + const c = sanitizeIdentifierSurrogates("call-\uDC00") + expect(a).not.toBe(b) + expect(a).not.toBe(c) + expect(b).not.toBe(c) + // No lone surrogates remain, so the id is safe for a validated JSON body. + expect(a).toBe("call-\uFFFD" + "D800") + }) + + it("escapes pre-existing U+FFFD so it cannot be mistaken for an encoded surrogate", () => { + // A literal U+FFFD becomes U+FFFD + "FFFD"; an encoded lone surrogate is U+FFFD + hex. + expect(sanitizeIdentifierSurrogates("a\uFFFDb")).toBe("a\uFFFD" + "FFFDb") + expect(sanitizeIdentifierSurrogates("a\uFFFDb")).not.toBe(sanitizeIdentifierSurrogates("a\uD800b")) + }) + + it("is deterministic, so a tool_use id and its tool_result id stay paired", () => { + const id = "toolu_01\uD800x" + expect(sanitizeIdentifierSurrogates(id)).toBe(sanitizeIdentifierSurrogates(id)) + }) +}) + +describe("sanitizeSurrogatesDeep", () => { + const lone = "bad\uD800end" + const sanitized = "bad\uFFFDend" + + it("sanitizes strings nested in objects and arrays", () => { + expect(sanitizeSurrogatesDeep({ path: lone, nested: { list: [lone, 1, null] } })).toEqual({ + path: sanitized, + nested: { list: [sanitized, 1, null] }, + }) + }) + + it("sanitizes object keys", () => { + expect(sanitizeSurrogatesDeep({ [`a\uD800`]: 1 })).toEqual({ "a\uFFFD": 1 }) + }) + + it("passes through non-string primitives unchanged", () => { + expect(sanitizeSurrogatesDeep(42)).toBe(42) + expect(sanitizeSurrogatesDeep(null)).toBe(null) + expect(sanitizeSurrogatesDeep(undefined)).toBe(undefined) + expect(sanitizeSurrogatesDeep(true)).toBe(true) + }) +}) diff --git a/src/api/transform/openai-format.ts b/src/api/transform/openai-format.ts index c06064f1dd..9b799e5c3d 100644 --- a/src/api/transform/openai-format.ts +++ b/src/api/transform/openai-format.ts @@ -1,6 +1,8 @@ import { Anthropic } from "@anthropic-ai/sdk" import OpenAI from "openai" +import { sanitizeIdentifierSurrogates, sanitizeSurrogates, sanitizeSurrogatesDeep } from "./sanitize-surrogates" + /** * Type for OpenRouter's reasoning detail elements. * @see https://openrouter.ai/docs/use-cases/reasoning-tokens#streaming-response @@ -355,7 +357,9 @@ export function convertToOpenAiMessages( const messageWithDetails = anthropicMessage as AssistantMessageWithReasoning const baseMessage: OpenAI.Chat.ChatCompletionMessageParam & ReasoningPassthroughFields = { role: anthropicMessage.role, - content: anthropicMessage.content, + // Sanitize lone UTF-16 surrogates: providers validating the JSON body + // (e.g. DeepSeek) reject the whole request otherwise. See #461. + content: sanitizeSurrogates(anthropicMessage.content), } if (anthropicMessage.role === "assistant") { @@ -365,7 +369,7 @@ export function convertToOpenAiMessages( } // Pass through reasoning_content for DeepSeek / Z.ai thinking mode. if (typeof messageWithDetails.reasoning_content === "string" && messageWithDetails.reasoning_content) { - baseMessage.reasoning_content = messageWithDetails.reasoning_content + baseMessage.reasoning_content = sanitizeSurrogates(messageWithDetails.reasoning_content) } } @@ -402,7 +406,7 @@ export function convertToOpenAiMessages( let content: string if (typeof toolMessage.content === "string") { - content = toolMessage.content + content = sanitizeSurrogates(toolMessage.content) } else { content = toolMessage.content @@ -415,7 +419,7 @@ export function convertToOpenAiMessages( return "[Image]" } if (part.type === "text") { - return part.text + return sanitizeSurrogates(part.text) } return "" }) @@ -423,7 +427,9 @@ export function convertToOpenAiMessages( } openAiMessages.push({ role: "tool", - tool_call_id: normalizeId(toolMessage.tool_use_id), + // Injective sanitization keeps the id paired with its tool_use after + // lone surrogates are replaced. See #461. + tool_call_id: sanitizeIdentifierSurrogates(normalizeId(toolMessage.tool_use_id)), // Use "(empty)" placeholder for empty content to satisfy providers like Gemini (via OpenRouter) content: content || "(empty)", }) @@ -468,7 +474,7 @@ export function convertToOpenAiMessages( ] as OpenAI.Chat.ChatCompletionToolMessageParam if (lastToolMessage?.role === "tool") { const additionalText = filteredNonToolMessages - .map((part) => (part as Anthropic.TextBlockParam).text) + .map((part) => sanitizeSurrogates((part as Anthropic.TextBlockParam).text)) .join("\n") lastToolMessage.content = `${lastToolMessage.content}\n\n${additionalText}` } @@ -494,7 +500,7 @@ export function convertToOpenAiMessages( } return { type: "text", text: "[Image]" } } - return { type: "text", text: part.text } + return { type: "text", text: sanitizeSurrogates(part.text) } }), }) } @@ -537,19 +543,22 @@ export function convertToOpenAiMessages( if (part.type === "image") { return "" // impossible as the assistant cannot send images } - return part.text + return sanitizeSurrogates(part.text) }) .join("\n") } // Process tool use messages const tool_calls: OpenAI.Chat.ChatCompletionMessageToolCall[] = toolMessages.map((toolMessage) => ({ - id: normalizeId(toolMessage.id), + // Injective sanitization keeps the id paired with its tool_result after + // lone surrogates are replaced. See #461. + id: sanitizeIdentifierSurrogates(normalizeId(toolMessage.id)), type: "function", function: { - name: toolMessage.name, - // json string - arguments: JSON.stringify(toolMessage.input), + name: sanitizeSurrogates(toolMessage.name), + // json string (deep-sanitized: a lone surrogate anywhere in the payload + // makes providers like DeepSeek reject the whole request) + arguments: JSON.stringify(sanitizeSurrogatesDeep(toolMessage.input)), }, })) @@ -579,7 +588,7 @@ export function convertToOpenAiMessages( ? messageWithDetails.reasoning_content : undefined) ?? extractedReasoning if (outgoingReasoningContent) { - baseMessage.reasoning_content = outgoingReasoningContent + baseMessage.reasoning_content = sanitizeSurrogates(outgoingReasoningContent) } // Add tool_calls after reasoning_details diff --git a/src/api/transform/r1-format.ts b/src/api/transform/r1-format.ts index e59b282be2..021ea7d26c 100644 --- a/src/api/transform/r1-format.ts +++ b/src/api/transform/r1-format.ts @@ -1,6 +1,8 @@ import { Anthropic } from "@anthropic-ai/sdk" import OpenAI from "openai" +import { sanitizeIdentifierSurrogates, sanitizeSurrogates, sanitizeSurrogatesDeep } from "./sanitize-surrogates" + type ContentPartText = OpenAI.Chat.ChatCompletionContentPartText type ContentPartImage = OpenAI.Chat.ChatCompletionContentPartImage type UserMessage = OpenAI.Chat.ChatCompletionUserMessageParam @@ -45,10 +47,19 @@ export function convertToR1Format( ): Message[] { const result: Message[] = [] + // Compose any caller-provided id normalization with injective surrogate sanitization: + // a lone UTF-16 surrogate anywhere in the JSON body makes DeepSeek reject the whole + // request ("lone leading surrogate in hex escape"), and the injective variant keeps a + // tool_use id paired with its tool_result. See #461. + const normalizeId = (id: string) => + sanitizeIdentifierSurrogates(options?.normalizeToolCallId ? options.normalizeToolCallId(id) : id) + for (const message of messages) { // Check if the message has reasoning_content (for DeepSeek interleaved thinking) const messageWithReasoning = message as AnthropicMessage & { reasoning_content?: string } const reasoningContent = messageWithReasoning.reasoning_content + ? sanitizeSurrogates(messageWithReasoning.reasoning_content) + : messageWithReasoning.reasoning_content if (message.role === "user") { // Handle user messages - may contain tool_result blocks @@ -59,7 +70,7 @@ export function convertToR1Format( for (const part of message.content) { if (part.type === "text") { - textParts.push(part.text) + textParts.push(sanitizeSurrogates(part.text)) } else if (part.type === "image") { if (part.source.type === "base64") { imageParts.push({ @@ -71,12 +82,12 @@ export function convertToR1Format( // Convert tool_result to OpenAI tool message format let content: string if (typeof part.content === "string") { - content = part.content + content = sanitizeSurrogates(part.content) } else if (Array.isArray(part.content)) { content = part.content ?.map((c) => { - if (c.type === "text") return c.text + if (c.type === "text") return sanitizeSurrogates(c.text) if (c.type === "image") return "(image)" return "" }) @@ -95,9 +106,7 @@ export function convertToR1Format( for (const toolResult of toolResults) { const toolMessage: ToolMessage = { role: "tool", - tool_call_id: options?.normalizeToolCallId - ? options.normalizeToolCallId(toolResult.tool_use_id) - : toolResult.tool_use_id, + tool_call_id: normalizeId(toolResult.tool_use_id), content: toolResult.content, } result.push(toolMessage) @@ -155,18 +164,19 @@ export function convertToR1Format( } } else { // Simple string content + const safeContent = sanitizeSurrogates(message.content) const lastMessage = result[result.length - 1] if (lastMessage?.role === "user") { if (typeof lastMessage.content === "string") { - lastMessage.content += `\n${message.content}` + lastMessage.content += `\n${safeContent}` } else { ;(lastMessage.content as (ContentPartText | ContentPartImage)[]).push({ type: "text", - text: message.content, + text: safeContent, }) } } else { - result.push({ role: "user", content: message.content }) + result.push({ role: "user", content: safeContent }) } } } else if (message.role === "assistant") { @@ -178,19 +188,21 @@ export function convertToR1Format( for (const part of message.content) { if (part.type === "text") { - textParts.push(part.text) + textParts.push(sanitizeSurrogates(part.text)) } else if (part.type === "tool_use") { toolCalls.push({ - id: options?.normalizeToolCallId ? options.normalizeToolCallId(part.id) : part.id, + id: normalizeId(part.id), type: "function", function: { - name: part.name, - arguments: JSON.stringify(part.input), + name: sanitizeSurrogates(part.name), + // Deep-sanitized: a lone surrogate anywhere in the payload makes + // DeepSeek reject the whole request. See #461. + arguments: JSON.stringify(sanitizeSurrogatesDeep(part.input)), }, }) } else if ((part as any).type === "reasoning" && (part as any).text) { // Extract reasoning from content blocks (Task stores it this way) - extractedReasoning = (part as any).text + extractedReasoning = sanitizeSurrogates((part as any).text) } } @@ -224,12 +236,13 @@ export function convertToR1Format( } } else { // Simple string content + const safeContent = sanitizeSurrogates(message.content) const lastMessage = result[result.length - 1] if (lastMessage?.role === "assistant" && !(lastMessage as any).tool_calls) { if (typeof lastMessage.content === "string") { - lastMessage.content += `\n${message.content}` + lastMessage.content += `\n${safeContent}` } else { - lastMessage.content = message.content + lastMessage.content = safeContent } // Preserve reasoning_content from the new message if present if (reasoningContent) { @@ -238,7 +251,7 @@ export function convertToR1Format( } else { const assistantMessage: DeepSeekAssistantMessage = { role: "assistant", - content: message.content, + content: safeContent, ...(reasoningContent && { reasoning_content: reasoningContent }), } result.push(assistantMessage) diff --git a/src/api/transform/sanitize-surrogates.ts b/src/api/transform/sanitize-surrogates.ts new file mode 100644 index 0000000000..e3a421fceb --- /dev/null +++ b/src/api/transform/sanitize-surrogates.ts @@ -0,0 +1,75 @@ +/** + * Shared sanitizers for lone UTF-16 surrogate code units in outbound API request bodies. + * + * A task's history can contain a lone surrogate — e.g. left behind when some upstream step + * slices a string through an astral-plane character (emoji, CJK extension, etc.). Such a code + * unit cannot be encoded as UTF-8, and providers that validate the JSON body (DeepSeek returns + * `400 Failed to parse the request body as JSON: ... lone leading surrogate in hex escape`; + * the VS Code LM backend rejects likewise) refuse the entire request, permanently breaking the + * task. These helpers replace lone surrogates with U+FFFD at the request boundary while leaving + * valid surrogate pairs untouched. + */ + +/** + * Matches unpaired UTF-16 surrogate code units. Valid surrogate pairs are matched by the + * lookahead/lookbehind and left untouched. The regex intentionally omits the `u` flag so it + * operates on UTF-16 code units. + */ +export const LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? `\uFFFD${unit.charCodeAt(0).toString(16).toUpperCase()}`) +} + +/** Non-global twin of {@link LONE_SURROGATE}; `test` on a `/g` regex is stateful via `lastIndex`. */ +export const HAS_LONE_SURROGATE = new RegExp(LONE_SURROGATE.source) + +/** + * Applies {@link sanitizeSurrogates} to every string nested in a tool-call argument object. The + * backend rejects the whole request for a lone surrogate anywhere in the JSON payload, so a tool + * argument carrying a sliced astral character fails the request just as message text would. + * + * LIMITATION: keys are sanitized with the same lossy mapping, so keys differing only in their lone + * surrogate (`"a\uD800"`, `"a\uD801"`) both become `"a\uFFFD"` and the last value wins. This also + * applies to tool schemas, where colliding property definitions collapse and `required` can end up + * with duplicate entries. Accepted deliberately: the alternative is rewriting keys into a form no + * schema reference would match, and a request that reaches the backend beats one rejected outright. + */ +export function sanitizeSurrogatesDeep(value: unknown): unknown { + if (typeof value === "string") { + return sanitizeSurrogates(value) + } + if (Array.isArray(value)) { + return value.map(sanitizeSurrogatesDeep) + } + if (value && typeof value === "object") { + return Object.fromEntries( + Object.entries(value as Record).map(([key, nested]) => [ + sanitizeSurrogates(key), + sanitizeSurrogatesDeep(nested), + ]), + ) + } + return value +} diff --git a/src/api/transform/vscode-lm-format.ts b/src/api/transform/vscode-lm-format.ts index ac90db220e..c20a108788 100644 --- a/src/api/transform/vscode-lm-format.ts +++ b/src/api/transform/vscode-lm-format.ts @@ -1,6 +1,18 @@ import { Anthropic } from "@anthropic-ai/sdk" import * as vscode from "vscode" +import { + HAS_LONE_SURROGATE, + LONE_SURROGATE, + sanitizeIdentifierSurrogates, + sanitizeSurrogates, + sanitizeSurrogatesDeep, +} from "./sanitize-surrogates" + +// Re-exported so existing consumers keep a single import site; the shared definitions +// live in ./sanitize-surrogates so non-VS-Code transforms can use them too. +export { sanitizeIdentifierSurrogates, sanitizeSurrogates, sanitizeSurrogatesDeep } + /** * Safely converts a value into a plain object. */ @@ -28,43 +40,6 @@ function asObjectSafe(value: unknown): object { } } -/** - * Replaces unpaired UTF-16 surrogate code units with the Unicode replacement character (U+FFFD). - * - * The VS Code LM backend forwards requests to model APIs that require valid UTF-8. A lone surrogate - * — e.g. left behind when some upstream step slices a string through an astral-plane character - * (emoji, CJK extension, etc.) — cannot be encoded as UTF-8, so the backend rejects the entire - * request with a 400 ("string contains an unpaired UTF-16 surrogate code point and cannot be - * encoded as valid UTF-8"). Valid surrogate pairs are matched by the lookahead/lookbehind and left - * untouched. The regex intentionally omits the `u` flag so it operates on UTF-16 code units. - */ -const LONE_SURROGATE = /[\uD800-\uDBFF](?![\uDC00-\uDFFF])|(? `\uFFFD${unit.charCodeAt(0).toString(16).toUpperCase()}`) -} - -/** Non-global twin of {@link LONE_SURROGATE}; `test` on a `/g` regex is stateful via `lastIndex`. */ -const HAS_LONE_SURROGATE = new RegExp(LONE_SURROGATE.source) - /** Marks a name whose `_u` sequences would otherwise be read as encoding markers when decoded. */ const TOOL_NAME_MARKER = /_u(?:u|[0-9A-Fa-f]{4})/ @@ -117,35 +92,6 @@ export function decodeToolNameSurrogates(name: string): string { return decoded } -/** - * Applies {@link sanitizeSurrogates} to every string nested in a tool-call argument object. The - * backend rejects the whole request for a lone surrogate anywhere in the JSON payload, so a tool - * argument carrying a sliced astral character fails the request just as message text would. - * - * LIMITATION: keys are sanitized with the same lossy mapping, so keys differing only in their lone - * surrogate (`"a\uD800"`, `"a\uD801"`) both become `"a\uFFFD"` and the last value wins. This also - * applies to tool schemas, where colliding property definitions collapse and `required` can end up - * with duplicate entries. Accepted deliberately: the alternative is rewriting keys into a form no - * schema reference would match, and a request that reaches the backend beats one rejected outright. - */ -export function sanitizeSurrogatesDeep(value: unknown): unknown { - if (typeof value === "string") { - return sanitizeSurrogates(value) - } - if (Array.isArray(value)) { - return value.map(sanitizeSurrogatesDeep) - } - if (value && typeof value === "object") { - return Object.fromEntries( - Object.entries(value as Record).map(([key, nested]) => [ - sanitizeSurrogates(key), - sanitizeSurrogatesDeep(nested), - ]), - ) - } - return value -} - export function convertToVsCodeLmMessages( anthropicMessages: Anthropic.Messages.MessageParam[], ): vscode.LanguageModelChatMessage[] {