From 4cc8b922bf2004d781314e0fcaf4f1511dc3f479 Mon Sep 17 00:00:00 2001 From: tjdoomer Date: Thu, 2 Apr 2026 16:27:33 +1100 Subject: [PATCH 1/2] docs: add roadmap for think block passthrough Surface reasoning traces from DeepSeek-R1, QwQ, Qwen3 etc instead of stripping them. First-class think parts with collapsible UI. --- docs/roadmap/06-think-passthrough.md | 80 ++++++++++++++++++++++++++++ 1 file changed, 80 insertions(+) create mode 100644 docs/roadmap/06-think-passthrough.md diff --git a/docs/roadmap/06-think-passthrough.md b/docs/roadmap/06-think-passthrough.md new file mode 100644 index 00000000000..f80156e3037 --- /dev/null +++ b/docs/roadmap/06-think-passthrough.md @@ -0,0 +1,80 @@ +# Think Block Passthrough + +## Branch: `feature/think-passthrough` + +## Problem + +The OpenAI generator (openaiContentGenerator.ts:537-593) has a `filterThinkTags()` +method that **strips** `...` blocks from streaming output. For +non-streaming, it does `content.replace(/[\s\S]*?<\/think>/g, '')` at +line 1355. + +This means reasoning traces from models like DeepSeek-R1, QwQ, Qwen3-coder, +and others are silently discarded. Users running these models via LM Studio +can't see the chain-of-thought that makes them useful. + +The Anthropic generator doesn't handle think blocks at all. + +## What OpenClaude does + +Wraps `thinking` blocks in `` XML tags and passes them as text content. +Simple, but it at least surfaces the reasoning. + +## Better approach for Delta + +Rather than stripping or flattening, treat think blocks as a first-class part +type. This pairs naturally with the `DeltaPart` types from branch 5, but can +be implemented independently with the current type system. + +## Plan + +### Phase 1: Surface think blocks in streaming (OpenAI generator) + +- Replace `filterThinkTags()` stripping logic with a **think block extractor** +- When a `` tag opens: buffer content into a separate think accumulator +- When `` closes: yield the buffered content as a think part +- Keep the stateful cross-chunk handling (it's well-implemented, just pointed + at the wrong outcome) + +For the current `@google/genai` type system, represent think parts as: +```typescript +{ text: '', thought: true } // extend Part with metadata +``` +Or use the `customMetadata` escape hatch on Part if available. + +### Phase 2: Surface think blocks in non-streaming (OpenAI generator) + +- Replace the regex strip at line 1355 with extraction +- Parse out all `...` blocks, emit as separate parts + +### Phase 3: Handle Anthropic extended thinking + +When/if we adopt the Anthropic SDK (branch 1), Claude's `thinking` content +blocks are returned natively — no XML parsing needed. Map them through. + +### Phase 4: UI rendering + +- `packages/cli/src/ui/` — render think blocks in a collapsible/dimmed style +- Show them by default (reasoning is the point), but respect a config flag + `showThinking: true|false` for users who want clean output +- Consider a `/thinking` toggle command + +### Phase 5: Anthropic generator + +- Handle `thinking` type content blocks in the streaming SSE parser +- Map to the same think part representation + +## Files to modify +- `packages/core/src/core/openaiContentGenerator.ts` — replace filterThinkTags, + modify convertStreamChunkToGenAIFormat and convertToGenAIFormat +- `packages/core/src/core/anthropicContentGenerator.ts` — handle thinking blocks +- `packages/cli/src/ui/` — render think parts (identify exact component) +- `packages/cli/src/config/settingsSchema.ts` — add `showThinking` setting + +## Files to create +- `packages/core/src/utils/thinkBlockParser.ts` — shared parser for XML think tags +- `packages/core/src/utils/thinkBlockParser.test.ts` + +## Dependencies +- Independent of other branches, but aligns naturally with branch 5 (delta types + would include `{ type: 'thinking'; text: string }` as a first-class DeltaPart) From adae23b955f3760c0bad5e00dc427c573f57457a Mon Sep 17 00:00:00 2001 From: tjdoomer Date: Thu, 2 Apr 2026 21:31:09 +1100 Subject: [PATCH 2/2] feat: surface think blocks from reasoning models instead of stripping Previously, ... tags from local reasoning models (DeepSeek-R1, QwQ, Qwen3-coder) were silently stripped. Now they're extracted and emitted as { thought: true, text } parts that the existing Turn system surfaces as DeltaEventType.Thought events. Changes: - New thinkBlockParser.ts with extractThinkBlocks() for non-streaming and StreamingThinkExtractor for cross-chunk streaming - OpenAI generator: replaced filterThinkTags stripping with extraction in both streaming and non-streaming paths - Anthropic generator: handle 'thinking' and 'thinking_delta' content blocks from Claude's extended thinking feature 11 parser tests + all 1873 existing tests pass. --- .../src/core/anthropicContentGenerator.ts | 20 ++- .../src/core/openaiContentGenerator.test.ts | 36 +++-- .../core/src/core/openaiContentGenerator.ts | 98 +++---------- .../core/src/utils/thinkBlockParser.test.ts | 105 ++++++++++++++ packages/core/src/utils/thinkBlockParser.ts | 134 ++++++++++++++++++ 5 files changed, 302 insertions(+), 91 deletions(-) create mode 100644 packages/core/src/utils/thinkBlockParser.test.ts create mode 100644 packages/core/src/utils/thinkBlockParser.ts diff --git a/packages/core/src/core/anthropicContentGenerator.ts b/packages/core/src/core/anthropicContentGenerator.ts index b16a7404d6f..d76dd500ea3 100644 --- a/packages/core/src/core/anthropicContentGenerator.ts +++ b/packages/core/src/core/anthropicContentGenerator.ts @@ -213,6 +213,16 @@ export class AnthropicContentGenerator implements ContentGenerator { FinishReason.FINISH_REASON_UNSPECIFIED, responseId, ); + } else if (delta?.type === 'thinking_delta' && 'thinking' in delta) { + // Extended thinking chunk — surface as thought part + const thinkText = (delta as { thinking?: string }).thinking; + if (thinkText) { + yield this.makeStreamResponse( + [{ thought: true, text: thinkText }], + FinishReason.FINISH_REASON_UNSPECIFIED, + responseId, + ); + } } else if (delta?.type === 'input_json_delta' && 'partial_json' in delta) { // Tool call argument fragment — append to the accumulator. // We don't yield yet because the JSON is incomplete. @@ -539,8 +549,14 @@ export class AnthropicContentGenerator implements ContentGenerator { }, }); } - // thinking blocks are intentionally not surfaced yet — branch 6 - // (feature/think-passthrough) will handle that + else if (block.type === 'thinking' && 'thinking' in block) { + // Anthropic's extended thinking — surface as a thought part so the + // Turn system renders it as DeltaEventType.Thought + const thinkText = (block as { thinking?: string }).thinking; + if (thinkText) { + parts.push({ thought: true, text: thinkText }); + } + } } response.responseId = message.id; diff --git a/packages/core/src/core/openaiContentGenerator.test.ts b/packages/core/src/core/openaiContentGenerator.test.ts index eb23ecd7aae..1983b0e2e34 100644 --- a/packages/core/src/core/openaiContentGenerator.test.ts +++ b/packages/core/src/core/openaiContentGenerator.test.ts @@ -3504,8 +3504,8 @@ describe('OpenAIContentGenerator', () => { }); }); - describe('think tag filtering', () => { - it('should filter complete blocks from non-streaming responses', async () => { + describe('think tag handling', () => { + it('should extract blocks as thought parts from non-streaming responses', async () => { const mockResponse = { id: 'test-id', object: 'chat.completion', @@ -3533,11 +3533,13 @@ describe('OpenAIContentGenerator', () => { }; const response = await generator.generateContent(request, 'test-prompt'); - const text = response.candidates?.[0]?.content?.parts?.[0]; - expect(text).toEqual({ text: 'Here is my answer.' }); + const parts = response.candidates?.[0]?.content?.parts; + // Think content becomes a thought part, visible text is separate + expect(parts?.[0]).toEqual({ thought: true, text: 'I need to think about this...' }); + expect(parts?.[1]).toEqual({ text: 'Here is my answer.' }); }); - it('should filter blocks from streaming responses', async () => { + it('should extract blocks as thought parts from streaming responses', async () => { const chunks = [ { id: 'test-id', @@ -3597,14 +3599,19 @@ describe('OpenAIContentGenerator', () => { responses.push(response); } - // Collect all text parts - const allText = responses - .flatMap((r) => r.candidates?.[0]?.content?.parts || []) - .filter((p) => 'text' in p && p.text) + // Collect all parts + const allParts = responses.flatMap((r) => r.candidates?.[0]?.content?.parts || []); + + // Visible text should not include think content + const visibleText = allParts + .filter((p) => 'text' in p && p.text && !('thought' in p && p.thought)) .map((p) => ('text' in p ? p.text : '')) .join(''); + expect(visibleText).toBe('Hello world'); - expect(allText).toBe('Hello world'); + // Think content should be surfaced as thought parts + const thoughtParts = allParts.filter((p) => 'thought' in p && p.thought); + expect(thoughtParts.length).toBeGreaterThan(0); }); it('should pass through text with no think tags unchanged', async () => { @@ -3638,7 +3645,7 @@ describe('OpenAIContentGenerator', () => { expect(text).toEqual({ text: 'Just a normal response with no tags.' }); }); - it('should filter multiple blocks from non-streaming responses', async () => { + it('should extract multiple blocks as a combined thought part', async () => { const mockResponse = { id: 'test-id', object: 'chat.completion', @@ -3666,8 +3673,11 @@ describe('OpenAIContentGenerator', () => { }; const response = await generator.generateContent(request, 'test-prompt'); - const text = response.candidates?.[0]?.content?.parts?.[0]; - expect(text).toEqual({ text: 'First part. Second part.' }); + const parts = response.candidates?.[0]?.content?.parts; + // Thought part comes first with combined think content + expect(parts?.[0]).toEqual({ thought: true, text: 'thought 1\n\nthought 2' }); + // Visible text follows + expect(parts?.[1]).toEqual({ text: 'First part. Second part.' }); }); }); }); diff --git a/packages/core/src/core/openaiContentGenerator.ts b/packages/core/src/core/openaiContentGenerator.ts index 0328054b43c..94fbba06e93 100644 --- a/packages/core/src/core/openaiContentGenerator.ts +++ b/packages/core/src/core/openaiContentGenerator.ts @@ -32,6 +32,7 @@ import { Config } from '../config/config.js'; import { openaiLogger } from '../utils/openaiLogger.js'; import { safeJsonParse } from '../utils/safeJsonParse.js'; import { normalizeSchemaForProvider, type ProviderType } from '../tools/schemaNormalizer.js'; +import { extractThinkBlocks, StreamingThinkExtractor } from '../utils/thinkBlockParser.js'; // Extended types to support cache_control interface ChatCompletionContentPartTextWithCache @@ -104,8 +105,7 @@ export class OpenAIContentGenerator implements ContentGenerator { arguments: string; } > = new Map(); - private insideThinkBlock: boolean = false; - private thinkTagBuffer: string = ''; + private thinkExtractor: StreamingThinkExtractor = new StreamingThinkExtractor(); constructor( contentGeneratorConfig: ContentGeneratorConfig, @@ -551,72 +551,15 @@ export class OpenAIContentGenerator implements ContentGenerator { } } - /** - * Filter ... tags from streaming text chunks. - * Handles tags that span multiple chunks using stateful buffering. - */ - private filterThinkTags(text: string): string { - let result = ''; - let i = 0; - - while (i < text.length) { - if (this.insideThinkBlock) { - // Look for closing tag - const closeIdx = text.indexOf('', i); - if (closeIdx !== -1) { - // Found closing tag, skip everything up to and including it - this.insideThinkBlock = false; - i = closeIdx + ''.length; - } else { - // No closing tag in this chunk, check if a partial is at the end - // Buffer up to 8 chars (length of "") to handle split tags - const tailLen = Math.min(text.length - i, 8); - const tail = text.substring(text.length - tailLen); - if (''.startsWith(tail) && tail.length < ''.length) { - this.thinkTagBuffer = tail; - } - // Skip all remaining text (we're inside a think block) - break; - } - } else { - // Look for opening tag - const openIdx = text.indexOf('', i); - if (openIdx !== -1) { - // Emit text before the tag - result += text.substring(i, openIdx); - this.insideThinkBlock = true; - i = openIdx + ''.length; - } else { - // Check if a partial tag might be at the end of the chunk - let partialMatch = false; - for (let len = Math.min(7, text.length - i); len >= 1; len--) { - const candidate = text.substring(text.length - len); - if (''.startsWith(candidate)) { - // Potential partial opening tag at end of chunk - result += text.substring(i, text.length - len); - this.thinkTagBuffer = candidate; - partialMatch = true; - break; - } - } - if (!partialMatch) { - result += text.substring(i); - } - break; - } - } - } - - return result; - } + // filterThinkTags removed — replaced by StreamingThinkExtractor in thinkBlockParser.ts + // which extracts think content as thought parts instead of discarding it. private async *streamGenerator( stream: AsyncIterable, ): AsyncGenerator { - // Reset the accumulator for each new stream + // Reset accumulators for each new stream this.streamingToolCalls.clear(); - this.insideThinkBlock = false; - this.thinkTagBuffer = ''; + this.thinkExtractor.reset(); for await (const chunk of stream) { yield this.convertStreamChunkToGenAIFormat(chunk); @@ -1374,11 +1317,15 @@ export class OpenAIContentGenerator implements ContentGenerator { const parts: Part[] = []; - // Handle text content (filter out ... tags from local models) + // Extract blocks from local models as thought parts instead of stripping them. + // The Turn system surfaces these as DeltaEventType.Thought events in the UI. if (choice.message.content) { - const filtered = choice.message.content.replace(/[\s\S]*?<\/think>/g, ''); - if (filtered) { - parts.push({ text: filtered }); + const { text: visibleText, thinkContent } = extractThinkBlocks(choice.message.content); + if (thinkContent) { + parts.push({ thought: true, text: thinkContent }); + } + if (visibleText) { + parts.push({ text: visibleText }); } } @@ -1462,18 +1409,17 @@ export class OpenAIContentGenerator implements ContentGenerator { if (choice) { const parts: Part[] = []; - // Handle text content (filter out ... tags from local models) + // Extract blocks from streaming content instead of stripping them. + // Completed think blocks become { thought: true } parts that the Turn + // system surfaces as DeltaEventType.Thought events. if (choice.delta?.content) { if (typeof choice.delta.content === 'string') { - // Prepend any buffered partial tag from previous chunk - let textToFilter = choice.delta.content; - if (this.thinkTagBuffer) { - textToFilter = this.thinkTagBuffer + textToFilter; - this.thinkTagBuffer = ''; + const { visibleText, completedThink } = this.thinkExtractor.process(choice.delta.content); + if (completedThink) { + parts.push({ thought: true, text: completedThink }); } - const filtered = this.filterThinkTags(textToFilter); - if (filtered) { - parts.push({ text: filtered }); + if (visibleText) { + parts.push({ text: visibleText }); } } } diff --git a/packages/core/src/utils/thinkBlockParser.test.ts b/packages/core/src/utils/thinkBlockParser.test.ts new file mode 100644 index 00000000000..20a1fd692cd --- /dev/null +++ b/packages/core/src/utils/thinkBlockParser.test.ts @@ -0,0 +1,105 @@ +import { describe, it, expect } from 'vitest'; +import { extractThinkBlocks, StreamingThinkExtractor } from './thinkBlockParser.js'; + +describe('extractThinkBlocks', () => { + it('should extract a single think block', () => { + const input = 'Hello reasoning here world'; + const { text, thinkContent } = extractThinkBlocks(input); + expect(text).toBe('Hello world'); + expect(thinkContent).toBe('reasoning here'); + }); + + it('should extract multiple think blocks', () => { + const input = 'first text second'; + const { text, thinkContent } = extractThinkBlocks(input); + expect(text).toBe('text'); + expect(thinkContent).toBe('first\n\nsecond'); + }); + + it('should return null thinkContent when no think blocks', () => { + const { text, thinkContent } = extractThinkBlocks('just plain text'); + expect(text).toBe('just plain text'); + expect(thinkContent).toBeNull(); + }); + + it('should handle multiline think blocks', () => { + const input = '\nline 1\nline 2\nresult'; + const { text, thinkContent } = extractThinkBlocks(input); + expect(text).toBe('result'); + expect(thinkContent).toBe('line 1\nline 2'); + }); + + it('should handle empty think blocks', () => { + const input = 'text'; + const { text, thinkContent } = extractThinkBlocks(input); + expect(text).toBe('text'); + expect(thinkContent).toBeNull(); // empty think = null + }); +}); + +describe('StreamingThinkExtractor', () => { + it('should extract think content from a single chunk', () => { + const ext = new StreamingThinkExtractor(); + const result = ext.process('Hello reason world'); + expect(result.visibleText).toBe('Hello world'); + expect(result.completedThink).toBe('reason'); + }); + + it('should handle think block spanning two chunks', () => { + const ext = new StreamingThinkExtractor(); + + const r1 = ext.process('Hello start of '); + expect(r1.visibleText).toBe('Hello '); + expect(r1.completedThink).toBeNull(); // not complete yet + + const r2 = ext.process('reasoning world'); + expect(r2.visibleText).toBe(' world'); + expect(r2.completedThink).toBe('start of reasoning'); + }); + + it('should handle think block spanning many chunks', () => { + const ext = new StreamingThinkExtractor(); + + ext.process(''); + const r2 = ext.process('chunk 1 '); + expect(r2.completedThink).toBeNull(); + + const r3 = ext.process('chunk 2 '); + expect(r3.completedThink).toBeNull(); + + const r4 = ext.process('chunk 3done'); + expect(r4.completedThink).toBe('chunk 1 chunk 2 chunk 3'); + expect(r4.visibleText).toBe('done'); + }); + + it('should handle partial tag at chunk boundary', () => { + const ext = new StreamingThinkExtractor(); + + // "reasoning'); + expect(r2.completedThink).toBe('reasoning'); + }); + + it('should handle text with no think blocks', () => { + const ext = new StreamingThinkExtractor(); + const result = ext.process('just regular text'); + expect(result.visibleText).toBe('just regular text'); + expect(result.completedThink).toBeNull(); + }); + + it('should reset state between streams', () => { + const ext = new StreamingThinkExtractor(); + + // Start a think block but don't finish + ext.process('partial'); + ext.reset(); + + // After reset, should work cleanly + const result = ext.process('fresh text'); + expect(result.visibleText).toBe('fresh text'); + expect(result.completedThink).toBeNull(); + }); +}); diff --git a/packages/core/src/utils/thinkBlockParser.ts b/packages/core/src/utils/thinkBlockParser.ts new file mode 100644 index 00000000000..12cbd497309 --- /dev/null +++ b/packages/core/src/utils/thinkBlockParser.ts @@ -0,0 +1,134 @@ +/** + * Shared parser for ... XML tags emitted by reasoning models + * (DeepSeek-R1, QwQ, Qwen3-coder, etc). + * + * Instead of stripping think blocks, we extract them so they can be surfaced + * as { thought: true, text: "..." } parts — which the existing Turn system + * already handles via DeltaEventType.Thought. + */ + +export interface ParsedContent { + /** Regular text content (outside think blocks) */ + text: string; + /** Extracted think block content, if any */ + thinkContent: string | null; +} + +/** + * Extract ... blocks from a complete response string. + * Returns the visible text and the think content separately. + * + * Used by the non-streaming path where the full response is available. + */ +export function extractThinkBlocks(content: string): ParsedContent { + const thinkRegex = /([\s\S]*?)<\/think>/g; + const thinkParts: string[] = []; + let match; + + while ((match = thinkRegex.exec(content)) !== null) { + const thinkText = match[1].trim(); + if (thinkText) { + thinkParts.push(thinkText); + } + } + + const text = content.replace(/[\s\S]*?<\/think>/g, '').trim(); + const thinkContent = thinkParts.length > 0 ? thinkParts.join('\n\n') : null; + + return { text, thinkContent }; +} + +// --------------------------------------------------------------------------- +// Streaming think block extractor +// --------------------------------------------------------------------------- + +/** + * Stateful streaming parser that handles tags spanning multiple chunks. + * + * Replaces the old filterThinkTags approach which stripped content. This version + * extracts think content into a separate buffer so it can be yielded as thought + * parts to the Turn system. + */ +export class StreamingThinkExtractor { + private insideThinkBlock = false; + private tagBuffer = ''; + private thinkAccumulator = ''; + + /** Reset state between streams. */ + reset(): void { + this.insideThinkBlock = false; + this.tagBuffer = ''; + this.thinkAccumulator = ''; + } + + /** + * Process a streaming text chunk. Returns the visible text and any + * completed think content. + * + * Think content is only returned when a close tag is found, + * meaning the reasoning block is complete. Partial reasoning stays + * buffered until the block closes. + */ + process(chunk: string): { visibleText: string; completedThink: string | null } { + // Prepend any buffered partial tag from previous chunk + let text = this.tagBuffer + chunk; + this.tagBuffer = ''; + + let visibleText = ''; + let completedThink: string | null = null; + let i = 0; + + while (i < text.length) { + if (this.insideThinkBlock) { + const closeIdx = text.indexOf('', i); + if (closeIdx !== -1) { + // Capture the think content up to the close tag + this.thinkAccumulator += text.substring(i, closeIdx); + completedThink = this.thinkAccumulator.trim() || null; + this.thinkAccumulator = ''; + this.insideThinkBlock = false; + i = closeIdx + ''.length; + } else { + // No close tag yet — check for partial at chunk boundary + const tailLen = Math.min(text.length - i, 8); + const tail = text.substring(text.length - tailLen); + if (''.startsWith(tail) && tail.length < ''.length) { + // Buffer the potential partial tag, accumulate everything before it + this.thinkAccumulator += text.substring(i, text.length - tailLen); + this.tagBuffer = tail; + } else { + // Accumulate all remaining text as think content + this.thinkAccumulator += text.substring(i); + } + break; + } + } else { + const openIdx = text.indexOf('', i); + if (openIdx !== -1) { + // Emit text before the tag as visible + visibleText += text.substring(i, openIdx); + this.insideThinkBlock = true; + i = openIdx + ''.length; + } else { + // Check for partial at chunk boundary + let partialMatch = false; + for (let len = Math.min(7, text.length - i); len >= 1; len--) { + const candidate = text.substring(text.length - len); + if (''.startsWith(candidate)) { + visibleText += text.substring(i, text.length - len); + this.tagBuffer = candidate; + partialMatch = true; + break; + } + } + if (!partialMatch) { + visibleText += text.substring(i); + } + break; + } + } + } + + return { visibleText, completedThink }; + } +}