diff --git a/apps/server/src/serverSettings.test.ts b/apps/server/src/serverSettings.test.ts index 502ae3430e27..ae134d41d2aa 100644 --- a/apps/server/src/serverSettings.test.ts +++ b/apps/server/src/serverSettings.test.ts @@ -1,6 +1,7 @@ import * as NodeServices from "@effect/platform-node/NodeServices"; import { DEFAULT_SERVER_SETTINGS, + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER, ModelSelection, ProjectId, ProjectScript, @@ -688,6 +689,30 @@ it.layer(NodeServices.layer)("server settings", (it) => { const settings = yield* serverSettings.getSettings; assert.equal(settings.textGenerationModelSelection.instanceId, "claudeAgent"); + assert.equal( + settings.textGenerationModelSelection.model, + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER[ProviderDriverKind.make("claudeAgent")], + ); + }).pipe(Effect.provide(makeServerSettingsLayer())), + ); + + it.effect("keeps the product text-generation model when a custom model is configured", () => + Effect.gen(function* () { + const serverConfig = yield* ServerConfig.ServerConfig; + const fileSystem = yield* FileSystem.FileSystem; + const serverSettings = yield* ServerSettingsModule.ServerSettingsService; + yield* fileSystem.writeFileString( + serverConfig.settingsPath, + '{"providerInstances":{"codex":{"driver":"codex","enabled":false,"config":{}},"claudeAgent":{"driver":"claudeAgent","config":{"customModels":["z-ai/glm-5.3-flash"]}}},"defaultModelSelection":{"instanceId":"claudeAgent","model":"z-ai/glm-5.3-flash"}}', + ); + + const settings = yield* serverSettings.getSettings; + + assert.equal(settings.textGenerationModelSelection.instanceId, "claudeAgent"); + assert.equal( + settings.textGenerationModelSelection.model, + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER[ProviderDriverKind.make("claudeAgent")], + ); }).pipe(Effect.provide(makeServerSettingsLayer())), ); diff --git a/apps/server/src/serverSettings.ts b/apps/server/src/serverSettings.ts index 6949a66d981b..1bb788faabb2 100644 --- a/apps/server/src/serverSettings.ts +++ b/apps/server/src/serverSettings.ts @@ -338,6 +338,12 @@ function resolveTextGenerationProvider(settings: ServerSettings): ServerSettings : fallbackTextGenerationProvider(settings); } +/** + * Pick an enabled provider when the stored text-generation selection cannot + * run. The model stays the product text-generation slug. A configured custom + * model is substituted only after that slug's one-shot attempt reports the + * model itself is unavailable. + */ function fallbackTextGenerationProvider(settings: ServerSettings): ServerSettings { // Same precedence as isModelSelectionProviderEnabled: an explicit provider // instance wins over the legacy providers map, which decodes to defaults diff --git a/apps/server/src/textGeneration/ClaudeTextGeneration.test.ts b/apps/server/src/textGeneration/ClaudeTextGeneration.test.ts index 8fe5152d3450..2b2131e0e108 100644 --- a/apps/server/src/textGeneration/ClaudeTextGeneration.test.ts +++ b/apps/server/src/textGeneration/ClaudeTextGeneration.test.ts @@ -1,6 +1,11 @@ import * as NodeServices from "@effect/platform-node/NodeServices"; import { it } from "@effect/vitest"; -import { ClaudeSettings, ProviderInstanceId } from "@t3tools/contracts"; +import { + ClaudeSettings, + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER, + ProviderDriverKind, + ProviderInstanceId, +} from "@t3tools/contracts"; import { HostProcessPlatform, isHostWindows } from "@t3tools/shared/hostProcess"; import { createModelSelection } from "@t3tools/shared/model"; import * as Effect from "effect/Effect"; @@ -23,6 +28,8 @@ import { sanitizeThreadTitle } from "./TextGenerationUtils.ts"; import { makeClaudeTextGeneration } from "./ClaudeTextGeneration.ts"; import { writeFakeCli } from "../testUtils/fakeCli.ts"; const decodeClaudeSettings = Schema.decodeSync(ClaudeSettings); +/** Encode the fake CLI's per-model response map as a JSON environment value. */ +const encodeUnknownJson = Schema.encodeSync(Schema.fromJsonString(Schema.Unknown)); const ClaudeTextGenerationTestLayer = ServerConfig.ServerConfig.layerTest(process.cwd(), { prefix: "t3code-claude-text-generation-test-", @@ -31,6 +38,10 @@ const ClaudeTextGenerationTestLayer = ServerConfig.ServerConfig.layerTest(proces // The stub behaviour lives in Node so the same implementation runs on Windows, // where a shebang file is not executable and would fall through to the real // Claude CLI on PATH; `writeFakeCli` picks the launcher shape per host. +/** + * Write a fake `claude` binary that records the `--model` argument and can + * answer per model. + */ function makeFakeClaudeBinary(dir: string) { return Effect.gen(function* () { const path = yield* Path.Path; @@ -43,7 +54,7 @@ function makeFakeClaudeBinary(dir: string) { source: [ "const argv = process.argv.slice(2);", 'const args = argv.join(" ");', - 'const { realpathSync } = await import("node:fs");', + 'const { appendFileSync, realpathSync } = await import("node:fs");', "", "function fail(message, code) {", ' process.stderr.write(message + "\\n");', @@ -105,20 +116,42 @@ function makeFakeClaudeBinary(dir: string) { ' fail("CLAUDE_CONFIG_DIR was " + (process.env.CLAUDE_CONFIG_DIR ?? ""), 5);', "}", "", - "const stderrText = process.env.T3_FAKE_CLAUDE_STDERR;", - "if (stderrText) {", - ' process.stderr.write(stderrText + "\\n");', + 'const modelIndex = argv.indexOf("--model");', + 'const model = modelIndex === -1 ? "" : (argv[modelIndex + 1] ?? "");', + "const modelLog = process.env.T3_FAKE_CLAUDE_MODEL_LOG;", + 'if (modelLog) appendFileSync(modelLog, model + "\\n");', + "const modelResponsesRaw = process.env.T3_FAKE_CLAUDE_MODEL_RESPONSES;", + "const modelResponse = modelResponsesRaw ? JSON.parse(modelResponsesRaw)[model] : undefined;", + "if (modelResponse) {", + ' if (modelResponse.stderr) process.stderr.write(modelResponse.stderr + "\\n");', + ' process.stdout.write(modelResponse.stdout ?? "");', + " process.exitCode = Number(modelResponse.exitCode ?? 0);", + "} else {", + " const stderrText = process.env.T3_FAKE_CLAUDE_STDERR;", + " if (stderrText) {", + ' process.stderr.write(stderrText + "\\n");', + " }", + ' process.stdout.write(process.env.T3_FAKE_CLAUDE_OUTPUT ?? "");', + " process.exitCode = Number(process.env.T3_FAKE_CLAUDE_EXIT_CODE ?? 0);", "}", "", - 'process.stdout.write(process.env.T3_FAKE_CLAUDE_OUTPUT ?? "");', - "process.exitCode = Number(process.env.T3_FAKE_CLAUDE_EXIT_CODE ?? 0);", - "", ].join("\n"), }); return binDir; }); } +/** Per-model stdout, stderr, and exit code for the fake Claude CLI. */ +interface FakeClaudeModelResponse { + readonly stdout?: string; + readonly stderr?: string; + readonly exitCode?: number; +} + +/** + * Run `effectFn` against a Claude text-generation service whose `claude` + * binary is the test stub. Restores the process environment afterwards. + */ function withFakeClaudeEnv( input: { output: string; @@ -130,12 +163,18 @@ function withFakeClaudeEnv( configDirMustBe?: string; cwdMustNotBe?: string; claudeConfig?: Partial; + modelResponses?: Readonly>; }, - effectFn: (textGeneration: TextGeneration.TextGeneration["Service"]) => Effect.Effect, + effectFn: ( + textGeneration: TextGeneration.TextGeneration["Service"], + context: { readonly modelLogPath: string }, + ) => Effect.Effect, ) { return Effect.gen(function* () { const fs = yield* FileSystem.FileSystem; + const path = yield* Path.Path; const tempDir = yield* fs.makeTempDirectoryScoped({ prefix: "t3code-claude-text-" }); + const modelLogPath = path.join(tempDir, "claude-models.log"); const binDir = yield* makeFakeClaudeBinary(tempDir); const pathDelimiter = (yield* isHostWindows) ? ";" : ":"; const previousPath = process.env.PATH; @@ -147,6 +186,8 @@ function withFakeClaudeEnv( const previousStdinMustContain = process.env.T3_FAKE_CLAUDE_STDIN_MUST_CONTAIN; const previousConfigDirMustBe = process.env.T3_FAKE_CLAUDE_CONFIG_DIR_MUST_BE; const previousCwdMustNotBe = process.env.T3_FAKE_CLAUDE_CWD_MUST_NOT_BE; + const previousModelLog = process.env.T3_FAKE_CLAUDE_MODEL_LOG; + const previousModelResponses = process.env.T3_FAKE_CLAUDE_MODEL_RESPONSES; yield* Effect.acquireRelease( Effect.sync(() => { @@ -194,6 +235,13 @@ function withFakeClaudeEnv( } else { delete process.env.T3_FAKE_CLAUDE_CONFIG_DIR_MUST_BE; } + + process.env.T3_FAKE_CLAUDE_MODEL_LOG = modelLogPath; + if (input.modelResponses !== undefined) { + process.env.T3_FAKE_CLAUDE_MODEL_RESPONSES = encodeUnknownJson(input.modelResponses); + } else { + delete process.env.T3_FAKE_CLAUDE_MODEL_RESPONSES; + } }), () => Effect.sync(() => { @@ -246,6 +294,18 @@ function withFakeClaudeEnv( } else { process.env.T3_FAKE_CLAUDE_CONFIG_DIR_MUST_BE = previousConfigDirMustBe; } + + if (previousModelLog === undefined) { + delete process.env.T3_FAKE_CLAUDE_MODEL_LOG; + } else { + process.env.T3_FAKE_CLAUDE_MODEL_LOG = previousModelLog; + } + + if (previousModelResponses === undefined) { + delete process.env.T3_FAKE_CLAUDE_MODEL_RESPONSES; + } else { + process.env.T3_FAKE_CLAUDE_MODEL_RESPONSES = previousModelResponses; + } }), ); @@ -255,7 +315,7 @@ function withFakeClaudeEnv( undefined, Effect.succeed(SYNTHETIC_CLAUDE_MODEL_CATALOG), ); - return yield* effectFn(textGeneration); + return yield* effectFn(textGeneration, { modelLogPath }); }).pipe(Effect.scoped); } @@ -579,4 +639,273 @@ it.layer(ClaudeTextGenerationTestLayer)("ClaudeTextGeneration", (it) => { }), ), ); + + const configuredProductModel = + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER[ProviderDriverKind.make("claudeAgent")]; + if (configuredProductModel === undefined) { + throw new Error("Claude product text-generation model is not configured"); + } + const productModel = configuredProductModel; + const customModel = "z-ai/glm-5.3-flash"; + const wrapperStderr = + "Using the OpenRouter credential from the global credential ~/.ori/credentials.json."; + const modelBlockedStdout = JSON.stringify({ + api_error_status: 400, + is_error: true, + result: + "API Error: 400 0 endpoints out of 4 requested are available matching your guardrail restrictions and data policy. Model blocked by guardrail: 4 endpoints excluded", + }); + const contentGuardrailStdout = JSON.stringify({ + api_error_status: 400, + is_error: true, + result: "API Error: 400 Request blocked by a content guardrail. The prompt violates policy.", + }); + + /** + * Read the models the fake Claude CLI was spawned with, in order. + */ + const readSpawnedModels = (modelLogPath: string) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + return (yield* fs.readFileString(modelLogPath)) + .split("\n") + .map((line) => line.trim()) + .filter((line) => line.length > 0); + }); + + /** + * Assert a CLI failure exposes only a bounded category and the process exit. + */ + const expectBoundedCliFailure = ( + error: { readonly detail: string; readonly cause?: unknown }, + expectedDetail: string, + forbidden: ReadonlyArray, + ) => { + expect(error.detail).toBe(expectedDetail); + expect(error.cause).toBeInstanceOf(Error); + if (!(error.cause instanceof Error)) { + throw new Error("expected a process-exit cause"); + } + expect(error.cause.message).toBe("Claude CLI process exited with code 1"); + for (const fragment of [error.detail, error.cause.message]) { + for (const secret of forbidden) { + expect(fragment).not.toContain(secret); + } + } + }; + + it.effect("keeps the product text-generation model when that slug succeeds", () => { + expect(productModel).toBe("claude-haiku-4-5"); + return withFakeClaudeEnv( + { + output: JSON.stringify({ + structured_output: { subject: "Keep the product model", body: "" }, + }), + claudeConfig: { customModels: [customModel] }, + }, + (textGeneration, { modelLogPath }) => + Effect.gen(function* () { + const generated = yield* textGeneration.generateCommitMessage({ + cwd: process.cwd(), + branch: "main", + stagedSummary: "M README.md", + stagedPatch: "diff --git a/README.md b/README.md", + modelSelection: { + instanceId: ProviderInstanceId.make("claudeAgent"), + model: productModel, + }, + }); + + expect(generated.subject).toBe("Keep the product model"); + expect(yield* readSpawnedModels(modelLogPath)).toEqual([productModel]); + }), + ); + }); + + it.effect("uses a configured custom model only after the product slug is unavailable", () => { + expect(productModel).toBe("claude-haiku-4-5"); + return withFakeClaudeEnv( + { + output: "", + claudeConfig: { customModels: [customModel] }, + modelResponses: { + [productModel]: { + exitCode: 1, + stderr: wrapperStderr, + stdout: modelBlockedStdout, + }, + [customModel]: { + exitCode: 0, + stdout: JSON.stringify({ + structured_output: { subject: "Use the configured custom model", body: "" }, + }), + }, + }, + }, + (textGeneration, { modelLogPath }) => + Effect.gen(function* () { + const generated = yield* textGeneration.generateCommitMessage({ + cwd: process.cwd(), + branch: "main", + stagedSummary: "M README.md", + stagedPatch: "diff --git a/README.md b/README.md", + modelSelection: { + instanceId: ProviderInstanceId.make("claudeAgent"), + model: productModel, + }, + }); + + expect(generated.subject).toBe("Use the configured custom model"); + expect(yield* readSpawnedModels(modelLogPath)).toEqual([productModel, customModel]); + }), + ); + }); + + it.effect("does not substitute a custom model for a content guardrail", () => { + expect(productModel).toBe("claude-haiku-4-5"); + return withFakeClaudeEnv( + { + output: contentGuardrailStdout, + exitCode: 1, + stderr: wrapperStderr, + claudeConfig: { customModels: [customModel] }, + }, + (textGeneration, { modelLogPath }) => + Effect.gen(function* () { + const error = yield* Effect.flip( + textGeneration.generateCommitMessage({ + cwd: process.cwd(), + branch: "main", + stagedSummary: "M README.md", + stagedPatch: "diff --git a/README.md b/README.md", + modelSelection: { + instanceId: ProviderInstanceId.make("claudeAgent"), + model: productModel, + }, + }), + ); + + expect(error._tag).toBe("TextGenerationError"); + expectBoundedCliFailure( + error, + "Claude CLI command failed (cli_failed, exit 1, api_status 400).", + [wrapperStderr, "content guardrail", "credentials.json", customModel], + ); + expect(yield* readSpawnedModels(modelLogPath)).toEqual([productModel]); + }), + ); + }); + + it.effect( + "does not substitute a custom model when the product slug fails for another reason", + () => { + expect(productModel).toBe("claude-haiku-4-5"); + return withFakeClaudeEnv( + { + output: "", + exitCode: 1, + stderr: wrapperStderr, + claudeConfig: { customModels: [customModel] }, + }, + (textGeneration, { modelLogPath }) => + Effect.gen(function* () { + const error = yield* Effect.flip( + textGeneration.generateCommitMessage({ + cwd: process.cwd(), + branch: "main", + stagedSummary: "M README.md", + stagedPatch: "diff --git a/README.md b/README.md", + modelSelection: { + instanceId: ProviderInstanceId.make("claudeAgent"), + model: productModel, + }, + }), + ); + + expect(error._tag).toBe("TextGenerationError"); + expectBoundedCliFailure(error, "Claude CLI command failed (cli_failed, exit 1).", [ + wrapperStderr, + "credentials.json", + customModel, + ]); + expect(yield* readSpawnedModels(modelLogPath)).toEqual([productModel]); + }), + ); + }, + ); + + it.effect( + "reports a bounded category when the product model is unavailable and no custom model exists", + () => { + expect(productModel).toBe("claude-haiku-4-5"); + return withFakeClaudeEnv( + { + output: modelBlockedStdout, + exitCode: 1, + stderr: wrapperStderr, + }, + (textGeneration, { modelLogPath }) => + Effect.gen(function* () { + const error = yield* Effect.flip( + textGeneration.generateCommitMessage({ + cwd: process.cwd(), + branch: "main", + stagedSummary: "M README.md", + stagedPatch: "diff --git a/README.md b/README.md", + modelSelection: { + instanceId: ProviderInstanceId.make("claudeAgent"), + model: productModel, + }, + }), + ); + + expect(error._tag).toBe("TextGenerationError"); + expectBoundedCliFailure( + error, + "Claude CLI command failed (model_unavailable, exit 1, api_status 400).", + [wrapperStderr, "guardrail", "endpoints excluded", "credentials.json"], + ); + expect(yield* readSpawnedModels(modelLogPath)).toEqual([productModel]); + }), + ); + }, + ); + + it.effect("does not replace an explicit non-product model when that model is unavailable", () => { + expect(productModel).toBe("claude-haiku-4-5"); + return withFakeClaudeEnv( + { + output: modelBlockedStdout, + exitCode: 1, + stderr: wrapperStderr, + claudeConfig: { customModels: [customModel] }, + }, + (textGeneration, { modelLogPath }) => + Effect.gen(function* () { + const error = yield* Effect.flip( + textGeneration.generateCommitMessage({ + cwd: process.cwd(), + branch: "main", + stagedSummary: "M README.md", + stagedPatch: "diff --git a/README.md b/README.md", + modelSelection: { + instanceId: ProviderInstanceId.make("claudeAgent"), + model: SYNTHETIC_CLAUDE_STANDARD_MODEL, + }, + }), + ); + + expect(error._tag).toBe("TextGenerationError"); + expectBoundedCliFailure( + error, + "Claude CLI command failed (model_unavailable, exit 1, api_status 400).", + [wrapperStderr, customModel], + ); + const models = yield* readSpawnedModels(modelLogPath); + expect(models).toHaveLength(1); + expect(models[0]?.startsWith(SYNTHETIC_CLAUDE_STANDARD_MODEL)).toBe(true); + expect(models.some((model) => model.includes(customModel))).toBe(false); + }), + ); + }); }); diff --git a/apps/server/src/textGeneration/ClaudeTextGeneration.ts b/apps/server/src/textGeneration/ClaudeTextGeneration.ts index 357ecd686e46..9eb32d18ca6b 100644 --- a/apps/server/src/textGeneration/ClaudeTextGeneration.ts +++ b/apps/server/src/textGeneration/ClaudeTextGeneration.ts @@ -14,11 +14,16 @@ import * as Schema from "effect/Schema"; import * as Stream from "effect/Stream"; import { ChildProcess, ChildProcessSpawner } from "effect/unstable/process"; -import { type ClaudeSettings, type ModelSelection } from "@t3tools/contracts"; +import { + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER, + ProviderDriverKind, + TextGenerationError, + type ClaudeSettings, + type ModelSelection, +} from "@t3tools/contracts"; import { sanitizeBranchFragment, sanitizeFeatureBranchName } from "@t3tools/shared/git"; import { resolveSpawnCommand } from "@t3tools/shared/shell"; -import { TextGenerationError } from "@t3tools/contracts"; import * as TextGeneration from "./TextGeneration.ts"; import { buildBranchNamePrompt, @@ -49,8 +54,28 @@ import { scopeClaudeModelCatalog, } from "../provider/ClaudeModelCatalog.ts"; import { makeClaudeEnvironment } from "../provider/Drivers/ClaudeHome.ts"; +import { + boundedCliApiErrorStatus, + customModelForBrokenTextGenerationFallback, + isBrokenProductTextGenerationFallback, + textGenerationCliFailureCategory, +} from "./TextGenerationModelFallback.ts"; const CLAUDE_TIMEOUT_MS = 180_000; +const CLAUDE_PRODUCT_TEXT_GENERATION_MODEL = + DEFAULT_TEXT_GENERATION_MODEL_BY_PROVIDER[ProviderDriverKind.make("claudeAgent")]; + +/** + * Describe a non-zero Claude CLI exit with a bounded category and exit code. + * Stdout and stderr stay out of the string; they can carry credentials or + * arbitrary payloads. Callers attach the process exit as `cause`. + */ +function describeClaudeCliFailure(exitCode: number, stdout: string, stderr: string): string { + const category = textGenerationCliFailureCategory({ stdout, stderr }); + const apiStatus = boundedCliApiErrorStatus(stdout); + const status = apiStatus === undefined ? "" : `, api_status ${apiStatus}`; + return `Claude CLI command failed (${category}, exit ${exitCode}${status}).`; +} /** * Schema for the wrapper JSON returned by `claude -p --output-format json`. @@ -70,6 +95,11 @@ const decodeClaudeOutput = Schema.decodeEffect( Schema.fromJsonString(Schema.Union([ClaudeOutputEnvelope, Schema.Array(ClaudeOutputMessage)])), ); +/** + * Build the Claude text-generation service for one-shot commit, PR, branch, + * and title prompts. A configured custom model is tried only after the product + * slug reports that the model itself is unavailable. + */ export const makeClaudeTextGeneration = Effect.fn("makeClaudeTextGeneration")(function* ( claudeSettings: ClaudeSettings, environment?: NodeJS.ProcessEnv, @@ -119,7 +149,8 @@ export const makeClaudeTextGeneration = Effect.fn("makeClaudeTextGeneration")(fu /** * Spawn the Claude CLI with structured JSON output and return the parsed, - * schema-validated result. + * schema-validated result. Retries once with a configured custom model only + * when the product slug reports that the model itself is unavailable. */ const runClaudeJson = Effect.fn("runClaudeJson")(function* ({ operation, @@ -139,52 +170,61 @@ export const makeClaudeTextGeneration = Effect.fn("makeClaudeTextGeneration")(fu modelSelection: ModelSelection; }): Effect.fn.Return { const catalog = yield* scopedModelCatalog; - const resolvedModelSelection = { - ...modelSelection, - model: resolveClaudeModelSlug(catalog, modelSelection.model), - }; + const requestedModel = resolveClaudeModelSlug(catalog, modelSelection.model); const jsonSchemaStr = yield* encodeJsonForOperation( operation, toJsonSchemaObject(outputSchemaJson), "Failed to encode structured output schema.", ); - const caps = getClaudeCatalogModelCapabilities(catalog, resolvedModelSelection.model); - const descriptors = getProviderOptionDescriptors({ - caps, - selections: resolvedModelSelection.options, - }); - const findDescriptor = (id: string) => descriptors.find((descriptor) => descriptor.id === id); - const rawEffortSelection = getModelSelectionStringOptionValue(resolvedModelSelection, "effort"); - const resolvedEffort = resolveClaudeCatalogEffort( - catalog, - resolvedModelSelection.model, - rawEffortSelection, - ); - const cliEffort = normalizeClaudeCatalogEffort( - catalog, - resolvedEffort, - resolvedModelSelection.model, - ); - const ultracode = isClaudeCatalogUltracodeEffort(resolvedEffort); - const thinkingDescriptor = findDescriptor("thinking"); - const fastModeDescriptor = findDescriptor("fastMode"); - const thinking = - thinkingDescriptor?.type === "boolean" ? thinkingDescriptor.currentValue : undefined; - const fastMode = - fastModeDescriptor?.type === "boolean" ? fastModeDescriptor.currentValue : undefined; - const settings = { - disableAllHooks: true, - ...(typeof thinking === "boolean" ? { alwaysThinkingEnabled: thinking } : {}), - ...(fastMode ? { fastMode: true } : {}), - ...(ultracode ? { ultracode: true } : {}), - }; - const settingsJson = yield* encodeJsonForOperation( - operation, - settings, - "Failed to encode Claude CLI settings.", - ); - const runClaudeCommand = Effect.fn("runClaudeJson.runClaudeCommand")(function* () { + /** + * Spawn one Claude CLI attempt for `selection` and return its output. + * A non-zero exit is returned to the caller so the failure can be classified + * before it becomes an error. + */ + const runClaudeCommand = Effect.fn("runClaudeJson.runClaudeCommand")(function* ( + selection: ModelSelection, + ) { + const resolvedSelection = { + ...selection, + model: resolveClaudeModelSlug(catalog, selection.model), + }; + const caps = getClaudeCatalogModelCapabilities(catalog, resolvedSelection.model); + const descriptors = getProviderOptionDescriptors({ + caps, + selections: resolvedSelection.options, + }); + /** Look up one resolved Claude provider option by id. */ + const findDescriptor = (id: string) => descriptors.find((descriptor) => descriptor.id === id); + const rawEffortSelection = getModelSelectionStringOptionValue(resolvedSelection, "effort"); + const resolvedEffort = resolveClaudeCatalogEffort( + catalog, + resolvedSelection.model, + rawEffortSelection, + ); + const cliEffort = normalizeClaudeCatalogEffort( + catalog, + resolvedEffort, + resolvedSelection.model, + ); + const ultracode = isClaudeCatalogUltracodeEffort(resolvedEffort); + const thinkingDescriptor = findDescriptor("thinking"); + const fastModeDescriptor = findDescriptor("fastMode"); + const thinking = + thinkingDescriptor?.type === "boolean" ? thinkingDescriptor.currentValue : undefined; + const fastMode = + fastModeDescriptor?.type === "boolean" ? fastModeDescriptor.currentValue : undefined; + const settings = { + disableAllHooks: true, + ...(typeof thinking === "boolean" ? { alwaysThinkingEnabled: thinking } : {}), + ...(fastMode ? { fastMode: true } : {}), + ...(ultracode ? { ultracode: true } : {}), + }; + const settingsJson = yield* encodeJsonForOperation( + operation, + settings, + "Failed to encode Claude CLI settings.", + ); // Titles need only the supplied prompt, not configuration from the checkout. const workingDirectory = operation === "generateThreadTitle" @@ -205,7 +245,7 @@ export const makeClaudeTextGeneration = Effect.fn("makeClaudeTextGeneration")(fu "--json-schema", jsonSchemaStr, "--model", - resolveClaudeCatalogApiModelId(catalog, resolvedModelSelection), + resolveClaudeCatalogApiModelId(catalog, resolvedSelection), ...(cliEffort ? ["--effort", cliEffort] : []), "--settings", settingsJson, @@ -249,35 +289,69 @@ export const makeClaudeTextGeneration = Effect.fn("makeClaudeTextGeneration")(fu { concurrency: "unbounded" }, ); - if (exitCode !== 0) { - const stderrDetail = stderr.trim(); - const stdoutDetail = stdout.trim(); - const detail = stderrDetail.length > 0 ? stderrDetail : stdoutDetail; - return yield* new TextGenerationError({ - operation, - detail: - detail.length > 0 - ? `Claude CLI command failed: ${detail}` - : `Claude CLI command failed with code ${exitCode}.`, + return { stdout, stderr, exitCode }; + }); + + /** + * Run one scoped Claude attempt, or none when it exceeds the timeout. + */ + const runOnce = (selection: ModelSelection) => + runClaudeCommand(selection).pipe(Effect.scoped, Effect.timeoutOption(CLAUDE_TIMEOUT_MS)); + + const first = yield* runOnce(modelSelection); + if (Option.isNone(first)) { + return yield* new TextGenerationError({ operation, detail: "Claude CLI request timed out." }); + } + + let outcome = first.value; + if (outcome.exitCode !== 0 && CLAUDE_PRODUCT_TEXT_GENERATION_MODEL !== undefined) { + const customModel = customModelForBrokenTextGenerationFallback( + claudeSettings.customModels, + CLAUDE_PRODUCT_TEXT_GENERATION_MODEL, + ); + // The product slug was just attempted. A custom model is used only when + // that attempt says the model itself is unavailable. + if ( + customModel !== null && + isBrokenProductTextGenerationFallback({ + productModel: CLAUDE_PRODUCT_TEXT_GENERATION_MODEL, + requestedModel, + stdout: outcome.stdout, + stderr: outcome.stderr, + }) + ) { + yield* Effect.logInfo( + "Retrying one-shot text generation with a configured custom model after the product model was unavailable", + { + operation, + productModel: CLAUDE_PRODUCT_TEXT_GENERATION_MODEL, + }, + ); + const second = yield* runOnce({ + instanceId: modelSelection.instanceId, + model: customModel, }); + if (Option.isNone(second)) { + return yield* new TextGenerationError({ + operation, + detail: "Claude CLI request timed out.", + }); + } + outcome = second.value; } + } - return stdout; - }); + if (outcome.exitCode !== 0) { + // Detail is a bounded category. The exit is the process failure; stdout + // and stderr are not copied because they can carry credentials. + return yield* new TextGenerationError({ + operation, + detail: describeClaudeCliFailure(outcome.exitCode, outcome.stdout, outcome.stderr), + cause: new Error(`Claude CLI process exited with code ${outcome.exitCode}`), + }); + } - const rawStdout = yield* runClaudeCommand().pipe( - Effect.scoped, - Effect.timeoutOption(CLAUDE_TIMEOUT_MS), - Effect.flatMap( - Option.match({ - onNone: () => - Effect.fail( - new TextGenerationError({ operation, detail: "Claude CLI request timed out." }), - ), - onSome: (value) => Effect.succeed(value), - }), - ), - ); + const rawStdout = outcome.stdout; const output = yield* decodeClaudeOutput(rawStdout).pipe( Effect.catchTags({ diff --git a/apps/server/src/textGeneration/TextGenerationModelFallback.test.ts b/apps/server/src/textGeneration/TextGenerationModelFallback.test.ts new file mode 100644 index 000000000000..cd0224bf46a9 --- /dev/null +++ b/apps/server/src/textGeneration/TextGenerationModelFallback.test.ts @@ -0,0 +1,227 @@ +import { describe, expect, it } from "vite-plus/test"; + +import { + boundedCliApiErrorStatus, + customModelForBrokenTextGenerationFallback, + isBrokenProductTextGenerationFallback, + textGenerationCliFailureCategory, +} from "./TextGenerationModelFallback.ts"; + +const PRODUCT_MODEL = "claude-haiku-4-5"; +const CUSTOM_MODEL = "z-ai/glm-5.3-flash"; +const WRAPPER_STDERR = + "Using the OpenRouter credential from the global credential ~/.ori/credentials.json."; +const MODEL_BLOCKED_STDOUT = JSON.stringify({ + api_error_status: 400, + is_error: true, + result: + "API Error: 400 0 endpoints out of 4 requested are available matching your guardrail restrictions and data policy. Model blocked by guardrail: 4 endpoints excluded", +}); +const CONTENT_GUARDRAIL_STDOUT = JSON.stringify({ + api_error_status: 400, + is_error: true, + result: "API Error: 400 Request blocked by a content guardrail. The prompt violates policy.", +}); +const PROMPT_GUARDRAIL_STDERR = "guardrail violation: disallowed prompt content"; + +describe("isBrokenProductTextGenerationFallback", () => { + it("keeps a healthy product-model response on the product slug", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: JSON.stringify({ structured_output: { subject: "Keep the product model" } }), + stderr: "", + }), + ).toBe(false); + }); + + it("treats a model-blocked product slug as a broken fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: MODEL_BLOCKED_STDOUT, + stderr: WRAPPER_STDERR, + }), + ).toBe(true); + }); + + it("treats an endpoint exclusion block as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "0 endpoints out of 4 requested are available. 4 endpoints excluded", + stderr: "", + }), + ).toBe(true); + }); + + it("does not treat a content guardrail as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: CONTENT_GUARDRAIL_STDOUT, + stderr: WRAPPER_STDERR, + }), + ).toBe(false); + }); + + it("does not treat a prompt guardrail as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "", + stderr: PROMPT_GUARDRAIL_STDERR, + }), + ).toBe(false); + }); + + it("does not treat the word guardrail alone as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "Your request triggered a guardrail.", + stderr: "", + }), + ).toBe(false); + }); + + it("does not treat a model that blocked prompt content as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "The model blocked this prompt under the content guardrail.", + stderr: "", + }), + ).toBe(false); + }); + + it("does not treat invalid model output as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "invalid model output", + stderr: "", + }), + ).toBe(false); + }); + + it("does not treat one endpoint phrase alone as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "4 endpoints excluded", + stderr: "", + }), + ).toBe(false); + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "0 endpoints out of 4", + stderr: "", + }), + ).toBe(false); + }); + + it("does not treat wrapper stderr alone as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: "", + stderr: WRAPPER_STDERR, + }), + ).toBe(false); + }); + + it("does not treat a rejected non-product model as a broken product fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: "claude-opus-4-6", + stdout: MODEL_BLOCKED_STDOUT, + stderr: WRAPPER_STDERR, + }), + ).toBe(false); + }); + + it("treats an unknown product model as a broken fallback", () => { + expect( + isBrokenProductTextGenerationFallback({ + productModel: PRODUCT_MODEL, + requestedModel: PRODUCT_MODEL, + stdout: `unknown model: ${PRODUCT_MODEL}`, + stderr: "", + }), + ).toBe(true); + }); +}); + +describe("customModelForBrokenTextGenerationFallback", () => { + it("uses the first configured custom model", () => { + expect(customModelForBrokenTextGenerationFallback([CUSTOM_MODEL], PRODUCT_MODEL)).toBe( + CUSTOM_MODEL, + ); + }); + + it("returns null when no custom model is configured", () => { + expect(customModelForBrokenTextGenerationFallback([], PRODUCT_MODEL)).toBeNull(); + }); + + it("returns null when the only custom slug is the product model", () => { + expect(customModelForBrokenTextGenerationFallback([PRODUCT_MODEL], PRODUCT_MODEL)).toBeNull(); + }); + + it("skips the product slug and uses the next custom model", () => { + expect( + customModelForBrokenTextGenerationFallback([PRODUCT_MODEL, CUSTOM_MODEL], PRODUCT_MODEL), + ).toBe(CUSTOM_MODEL); + }); +}); + +describe("textGenerationCliFailureCategory", () => { + it("labels a model block without copying the CLI payload", () => { + expect( + textGenerationCliFailureCategory({ + stdout: MODEL_BLOCKED_STDOUT, + stderr: WRAPPER_STDERR, + }), + ).toBe("model_unavailable"); + }); + + it("labels a content guardrail as a generic CLI failure", () => { + expect( + textGenerationCliFailureCategory({ + stdout: CONTENT_GUARDRAIL_STDOUT, + stderr: PROMPT_GUARDRAIL_STDERR, + }), + ).toBe("cli_failed"); + }); + + it("labels empty output as a generic CLI failure", () => { + expect(textGenerationCliFailureCategory({ stdout: "", stderr: "" })).toBe("cli_failed"); + }); +}); + +describe("boundedCliApiErrorStatus", () => { + it("reads an HTTP status from the JSON envelope", () => { + expect(boundedCliApiErrorStatus(MODEL_BLOCKED_STDOUT)).toBe(400); + }); + + it("drops non-JSON output and statuses outside the HTTP range", () => { + expect(boundedCliApiErrorStatus(WRAPPER_STDERR)).toBeUndefined(); + expect(boundedCliApiErrorStatus(JSON.stringify({ api_error_status: 99 }))).toBeUndefined(); + expect(boundedCliApiErrorStatus(JSON.stringify({ api_error_status: 600 }))).toBeUndefined(); + expect(boundedCliApiErrorStatus(JSON.stringify({ api_error_status: 400.5 }))).toBeUndefined(); + expect(boundedCliApiErrorStatus(JSON.stringify({ api_error_status: "400" }))).toBeUndefined(); + }); +}); diff --git a/apps/server/src/textGeneration/TextGenerationModelFallback.ts b/apps/server/src/textGeneration/TextGenerationModelFallback.ts new file mode 100644 index 000000000000..fc6b8145999f --- /dev/null +++ b/apps/server/src/textGeneration/TextGenerationModelFallback.ts @@ -0,0 +1,126 @@ +/** + * Decide when one-shot text generation may leave the product model. + * + * Commit, PR, branch, and title generation keep the product slug whenever + * that attempt can run. A configured custom model is a substitute only after + * the product slug was the model just tried and the CLI reported that this + * model is unavailable. Prompt and content guardrail failures are not that + * case. + */ +import type { CustomModelSetting } from "@t3tools/contracts"; +import { readCustomModelEntries } from "@t3tools/shared/model"; +import * as Option from "effect/Option"; +import * as Schema from "effect/Schema"; + +/** + * Phrases that name the model as missing or unusable. + * + * A bare `/guardrail/i` match is intentionally absent. Prompt and content + * guardrails use that word without saying the requested model is unavailable. + * `model blocked by guardrail` stays, because that phrase is the model block + * reported for a refused product slug. + */ +const EXPLICIT_MODEL_UNAVAILABLE_PATTERNS: ReadonlyArray = [ + /\bunknown model\b/i, + /\bmodel not found\b/i, + /\bno such model\b/i, + /\bunsupported model\b/i, + /\binvalid model(?:\s+(?:id|name|slug))?\b(?!\s+output)/i, + /\bmodel(?:\s+(?:id|name|slug))? is not available\b/i, + /\bmodel blocked by guardrail\b/i, +]; + +/** + * OpenRouter refuses a model by reporting that none of its endpoints remain. + * Either half alone is not specific enough to retry. + */ +const OPENROUTER_ENDPOINT_COUNT_PATTERN = /\b\d+\s+endpoints?\s+out of\s+\d+\b/i; +const OPENROUTER_ENDPOINTS_EXCLUDED_PATTERN = /\bendpoints? excluded\b/i; + +const CLI_FAILURE_CATEGORIES = ["model_unavailable", "cli_failed"] as const; + +/** Bounded label for a finished CLI attempt. Never includes process output. */ +export type TextGenerationCliFailureCategory = (typeof CLI_FAILURE_CATEGORIES)[number]; + +const ClaudeApiErrorStatusEnvelope = Schema.Struct({ + api_error_status: Schema.optionalKey(Schema.Number), +}); +const decodeClaudeApiErrorStatus = Schema.decodeUnknownOption( + Schema.fromJsonString(ClaudeApiErrorStatusEnvelope), +); + +/** + * Report whether CLI output says the model that was just requested cannot be + * used. Content and prompt guardrail text does not match. + */ +function isModelUnavailableCliOutput(stdout: string, stderr: string): boolean { + const text = `${stdout}\n${stderr}`; + if (EXPLICIT_MODEL_UNAVAILABLE_PATTERNS.some((pattern) => pattern.test(text))) { + return true; + } + return ( + OPENROUTER_ENDPOINT_COUNT_PATTERN.test(text) && OPENROUTER_ENDPOINTS_EXCLUDED_PATTERN.test(text) + ); +} + +/** + * Report whether a failed product-model attempt may be retried with a + * configured custom model. + * + * True only when `requestedModel` is the product slug and the CLI output says + * that model is unavailable. Callers use this to decide a single retry; a + * content or prompt guardrail stays on the original failure. + */ +export function isBrokenProductTextGenerationFallback(input: { + readonly productModel: string; + readonly requestedModel: string; + readonly stdout: string; + readonly stderr: string; +}): boolean { + if (input.requestedModel.trim() !== input.productModel.trim()) { + return false; + } + return isModelUnavailableCliOutput(input.stdout, input.stderr); +} + +/** + * Return the first configured custom slug other than the product model. + * Returns null when there is no distinct custom model to substitute. + */ +export function customModelForBrokenTextGenerationFallback( + customModels: ReadonlyArray, + productModel: string, +): string | null { + return ( + readCustomModelEntries(customModels).find((entry) => entry.slug !== productModel)?.slug ?? null + ); +} + +/** + * Classify CLI output for a caller-facing error. + * The result is a fixed label. Stdout and stderr are not returned. + */ +export function textGenerationCliFailureCategory(input: { + readonly stdout: string; + readonly stderr: string; +}): TextGenerationCliFailureCategory { + return isModelUnavailableCliOutput(input.stdout, input.stderr) + ? "model_unavailable" + : "cli_failed"; +} + +/** + * Read an HTTP status from a Claude JSON error envelope, when one is present. + * Non-integers and values outside 100–599 are dropped so the result stays bounded. + */ +export function boundedCliApiErrorStatus(stdout: string): number | undefined { + const decoded = decodeClaudeApiErrorStatus(stdout); + if (Option.isNone(decoded) || decoded.value.api_error_status === undefined) { + return undefined; + } + const status = decoded.value.api_error_status; + if (!Number.isInteger(status) || status < 100 || status > 599) { + return undefined; + } + return status; +}