diff --git a/docs/roadmap/06-think-passthrough.md b/docs/roadmap/06-think-passthrough.md
new file mode 100644
index 00000000000..f80156e3037
--- /dev/null
+++ b/docs/roadmap/06-think-passthrough.md
@@ -0,0 +1,80 @@
+# Think Block Passthrough
+
+## Branch: `feature/think-passthrough`
+
+## Problem
+
+The OpenAI generator (openaiContentGenerator.ts:537-593) has a `filterThinkTags()`
+method that **strips** `...` blocks from streaming output. For
+non-streaming, it does `content.replace(/[\s\S]*?<\/think>/g, '')` at
+line 1355.
+
+This means reasoning traces from models like DeepSeek-R1, QwQ, Qwen3-coder,
+and others are silently discarded. Users running these models via LM Studio
+can't see the chain-of-thought that makes them useful.
+
+The Anthropic generator doesn't handle think blocks at all.
+
+## What OpenClaude does
+
+Wraps `thinking` blocks in `` XML tags and passes them as text content.
+Simple, but it at least surfaces the reasoning.
+
+## Better approach for Delta
+
+Rather than stripping or flattening, treat think blocks as a first-class part
+type. This pairs naturally with the `DeltaPart` types from branch 5, but can
+be implemented independently with the current type system.
+
+## Plan
+
+### Phase 1: Surface think blocks in streaming (OpenAI generator)
+
+- Replace `filterThinkTags()` stripping logic with a **think block extractor**
+- When a `` tag opens: buffer content into a separate think accumulator
+- When `` closes: yield the buffered content as a think part
+- Keep the stateful cross-chunk handling (it's well-implemented, just pointed
+ at the wrong outcome)
+
+For the current `@google/genai` type system, represent think parts as:
+```typescript
+{ text: '', thought: true } // extend Part with metadata
+```
+Or use the `customMetadata` escape hatch on Part if available.
+
+### Phase 2: Surface think blocks in non-streaming (OpenAI generator)
+
+- Replace the regex strip at line 1355 with extraction
+- Parse out all `...` blocks, emit as separate parts
+
+### Phase 3: Handle Anthropic extended thinking
+
+When/if we adopt the Anthropic SDK (branch 1), Claude's `thinking` content
+blocks are returned natively — no XML parsing needed. Map them through.
+
+### Phase 4: UI rendering
+
+- `packages/cli/src/ui/` — render think blocks in a collapsible/dimmed style
+- Show them by default (reasoning is the point), but respect a config flag
+ `showThinking: true|false` for users who want clean output
+- Consider a `/thinking` toggle command
+
+### Phase 5: Anthropic generator
+
+- Handle `thinking` type content blocks in the streaming SSE parser
+- Map to the same think part representation
+
+## Files to modify
+- `packages/core/src/core/openaiContentGenerator.ts` — replace filterThinkTags,
+ modify convertStreamChunkToGenAIFormat and convertToGenAIFormat
+- `packages/core/src/core/anthropicContentGenerator.ts` — handle thinking blocks
+- `packages/cli/src/ui/` — render think parts (identify exact component)
+- `packages/cli/src/config/settingsSchema.ts` — add `showThinking` setting
+
+## Files to create
+- `packages/core/src/utils/thinkBlockParser.ts` — shared parser for XML think tags
+- `packages/core/src/utils/thinkBlockParser.test.ts`
+
+## Dependencies
+- Independent of other branches, but aligns naturally with branch 5 (delta types
+ would include `{ type: 'thinking'; text: string }` as a first-class DeltaPart)
diff --git a/packages/core/src/core/anthropicContentGenerator.ts b/packages/core/src/core/anthropicContentGenerator.ts
index b16a7404d6f..d76dd500ea3 100644
--- a/packages/core/src/core/anthropicContentGenerator.ts
+++ b/packages/core/src/core/anthropicContentGenerator.ts
@@ -213,6 +213,16 @@ export class AnthropicContentGenerator implements ContentGenerator {
FinishReason.FINISH_REASON_UNSPECIFIED,
responseId,
);
+ } else if (delta?.type === 'thinking_delta' && 'thinking' in delta) {
+ // Extended thinking chunk — surface as thought part
+ const thinkText = (delta as { thinking?: string }).thinking;
+ if (thinkText) {
+ yield this.makeStreamResponse(
+ [{ thought: true, text: thinkText }],
+ FinishReason.FINISH_REASON_UNSPECIFIED,
+ responseId,
+ );
+ }
} else if (delta?.type === 'input_json_delta' && 'partial_json' in delta) {
// Tool call argument fragment — append to the accumulator.
// We don't yield yet because the JSON is incomplete.
@@ -539,8 +549,14 @@ export class AnthropicContentGenerator implements ContentGenerator {
},
});
}
- // thinking blocks are intentionally not surfaced yet — branch 6
- // (feature/think-passthrough) will handle that
+ else if (block.type === 'thinking' && 'thinking' in block) {
+ // Anthropic's extended thinking — surface as a thought part so the
+ // Turn system renders it as DeltaEventType.Thought
+ const thinkText = (block as { thinking?: string }).thinking;
+ if (thinkText) {
+ parts.push({ thought: true, text: thinkText });
+ }
+ }
}
response.responseId = message.id;
diff --git a/packages/core/src/core/openaiContentGenerator.test.ts b/packages/core/src/core/openaiContentGenerator.test.ts
index eb23ecd7aae..1983b0e2e34 100644
--- a/packages/core/src/core/openaiContentGenerator.test.ts
+++ b/packages/core/src/core/openaiContentGenerator.test.ts
@@ -3504,8 +3504,8 @@ describe('OpenAIContentGenerator', () => {
});
});
- describe('think tag filtering', () => {
- it('should filter complete blocks from non-streaming responses', async () => {
+ describe('think tag handling', () => {
+ it('should extract blocks as thought parts from non-streaming responses', async () => {
const mockResponse = {
id: 'test-id',
object: 'chat.completion',
@@ -3533,11 +3533,13 @@ describe('OpenAIContentGenerator', () => {
};
const response = await generator.generateContent(request, 'test-prompt');
- const text = response.candidates?.[0]?.content?.parts?.[0];
- expect(text).toEqual({ text: 'Here is my answer.' });
+ const parts = response.candidates?.[0]?.content?.parts;
+ // Think content becomes a thought part, visible text is separate
+ expect(parts?.[0]).toEqual({ thought: true, text: 'I need to think about this...' });
+ expect(parts?.[1]).toEqual({ text: 'Here is my answer.' });
});
- it('should filter blocks from streaming responses', async () => {
+ it('should extract blocks as thought parts from streaming responses', async () => {
const chunks = [
{
id: 'test-id',
@@ -3597,14 +3599,19 @@ describe('OpenAIContentGenerator', () => {
responses.push(response);
}
- // Collect all text parts
- const allText = responses
- .flatMap((r) => r.candidates?.[0]?.content?.parts || [])
- .filter((p) => 'text' in p && p.text)
+ // Collect all parts
+ const allParts = responses.flatMap((r) => r.candidates?.[0]?.content?.parts || []);
+
+ // Visible text should not include think content
+ const visibleText = allParts
+ .filter((p) => 'text' in p && p.text && !('thought' in p && p.thought))
.map((p) => ('text' in p ? p.text : ''))
.join('');
+ expect(visibleText).toBe('Hello world');
- expect(allText).toBe('Hello world');
+ // Think content should be surfaced as thought parts
+ const thoughtParts = allParts.filter((p) => 'thought' in p && p.thought);
+ expect(thoughtParts.length).toBeGreaterThan(0);
});
it('should pass through text with no think tags unchanged', async () => {
@@ -3638,7 +3645,7 @@ describe('OpenAIContentGenerator', () => {
expect(text).toEqual({ text: 'Just a normal response with no tags.' });
});
- it('should filter multiple blocks from non-streaming responses', async () => {
+ it('should extract multiple blocks as a combined thought part', async () => {
const mockResponse = {
id: 'test-id',
object: 'chat.completion',
@@ -3666,8 +3673,11 @@ describe('OpenAIContentGenerator', () => {
};
const response = await generator.generateContent(request, 'test-prompt');
- const text = response.candidates?.[0]?.content?.parts?.[0];
- expect(text).toEqual({ text: 'First part. Second part.' });
+ const parts = response.candidates?.[0]?.content?.parts;
+ // Thought part comes first with combined think content
+ expect(parts?.[0]).toEqual({ thought: true, text: 'thought 1\n\nthought 2' });
+ // Visible text follows
+ expect(parts?.[1]).toEqual({ text: 'First part. Second part.' });
});
});
});
diff --git a/packages/core/src/core/openaiContentGenerator.ts b/packages/core/src/core/openaiContentGenerator.ts
index 0328054b43c..94fbba06e93 100644
--- a/packages/core/src/core/openaiContentGenerator.ts
+++ b/packages/core/src/core/openaiContentGenerator.ts
@@ -32,6 +32,7 @@ import { Config } from '../config/config.js';
import { openaiLogger } from '../utils/openaiLogger.js';
import { safeJsonParse } from '../utils/safeJsonParse.js';
import { normalizeSchemaForProvider, type ProviderType } from '../tools/schemaNormalizer.js';
+import { extractThinkBlocks, StreamingThinkExtractor } from '../utils/thinkBlockParser.js';
// Extended types to support cache_control
interface ChatCompletionContentPartTextWithCache
@@ -104,8 +105,7 @@ export class OpenAIContentGenerator implements ContentGenerator {
arguments: string;
}
> = new Map();
- private insideThinkBlock: boolean = false;
- private thinkTagBuffer: string = '';
+ private thinkExtractor: StreamingThinkExtractor = new StreamingThinkExtractor();
constructor(
contentGeneratorConfig: ContentGeneratorConfig,
@@ -551,72 +551,15 @@ export class OpenAIContentGenerator implements ContentGenerator {
}
}
- /**
- * Filter ... tags from streaming text chunks.
- * Handles tags that span multiple chunks using stateful buffering.
- */
- private filterThinkTags(text: string): string {
- let result = '';
- let i = 0;
-
- while (i < text.length) {
- if (this.insideThinkBlock) {
- // Look for closing tag
- const closeIdx = text.indexOf('', i);
- if (closeIdx !== -1) {
- // Found closing tag, skip everything up to and including it
- this.insideThinkBlock = false;
- i = closeIdx + ''.length;
- } else {
- // No closing tag in this chunk, check if a partial is at the end
- // Buffer up to 8 chars (length of "") to handle split tags
- const tailLen = Math.min(text.length - i, 8);
- const tail = text.substring(text.length - tailLen);
- if (''.startsWith(tail) && tail.length < ''.length) {
- this.thinkTagBuffer = tail;
- }
- // Skip all remaining text (we're inside a think block)
- break;
- }
- } else {
- // Look for opening tag
- const openIdx = text.indexOf('', i);
- if (openIdx !== -1) {
- // Emit text before the tag
- result += text.substring(i, openIdx);
- this.insideThinkBlock = true;
- i = openIdx + ''.length;
- } else {
- // Check if a partial tag might be at the end of the chunk
- let partialMatch = false;
- for (let len = Math.min(7, text.length - i); len >= 1; len--) {
- const candidate = text.substring(text.length - len);
- if (''.startsWith(candidate)) {
- // Potential partial opening tag at end of chunk
- result += text.substring(i, text.length - len);
- this.thinkTagBuffer = candidate;
- partialMatch = true;
- break;
- }
- }
- if (!partialMatch) {
- result += text.substring(i);
- }
- break;
- }
- }
- }
-
- return result;
- }
+ // filterThinkTags removed — replaced by StreamingThinkExtractor in thinkBlockParser.ts
+ // which extracts think content as thought parts instead of discarding it.
private async *streamGenerator(
stream: AsyncIterable,
): AsyncGenerator {
- // Reset the accumulator for each new stream
+ // Reset accumulators for each new stream
this.streamingToolCalls.clear();
- this.insideThinkBlock = false;
- this.thinkTagBuffer = '';
+ this.thinkExtractor.reset();
for await (const chunk of stream) {
yield this.convertStreamChunkToGenAIFormat(chunk);
@@ -1374,11 +1317,15 @@ export class OpenAIContentGenerator implements ContentGenerator {
const parts: Part[] = [];
- // Handle text content (filter out ... tags from local models)
+ // Extract blocks from local models as thought parts instead of stripping them.
+ // The Turn system surfaces these as DeltaEventType.Thought events in the UI.
if (choice.message.content) {
- const filtered = choice.message.content.replace(/[\s\S]*?<\/think>/g, '');
- if (filtered) {
- parts.push({ text: filtered });
+ const { text: visibleText, thinkContent } = extractThinkBlocks(choice.message.content);
+ if (thinkContent) {
+ parts.push({ thought: true, text: thinkContent });
+ }
+ if (visibleText) {
+ parts.push({ text: visibleText });
}
}
@@ -1462,18 +1409,17 @@ export class OpenAIContentGenerator implements ContentGenerator {
if (choice) {
const parts: Part[] = [];
- // Handle text content (filter out ... tags from local models)
+ // Extract blocks from streaming content instead of stripping them.
+ // Completed think blocks become { thought: true } parts that the Turn
+ // system surfaces as DeltaEventType.Thought events.
if (choice.delta?.content) {
if (typeof choice.delta.content === 'string') {
- // Prepend any buffered partial tag from previous chunk
- let textToFilter = choice.delta.content;
- if (this.thinkTagBuffer) {
- textToFilter = this.thinkTagBuffer + textToFilter;
- this.thinkTagBuffer = '';
+ const { visibleText, completedThink } = this.thinkExtractor.process(choice.delta.content);
+ if (completedThink) {
+ parts.push({ thought: true, text: completedThink });
}
- const filtered = this.filterThinkTags(textToFilter);
- if (filtered) {
- parts.push({ text: filtered });
+ if (visibleText) {
+ parts.push({ text: visibleText });
}
}
}
diff --git a/packages/core/src/utils/thinkBlockParser.test.ts b/packages/core/src/utils/thinkBlockParser.test.ts
new file mode 100644
index 00000000000..20a1fd692cd
--- /dev/null
+++ b/packages/core/src/utils/thinkBlockParser.test.ts
@@ -0,0 +1,105 @@
+import { describe, it, expect } from 'vitest';
+import { extractThinkBlocks, StreamingThinkExtractor } from './thinkBlockParser.js';
+
+describe('extractThinkBlocks', () => {
+ it('should extract a single think block', () => {
+ const input = 'Hello reasoning here world';
+ const { text, thinkContent } = extractThinkBlocks(input);
+ expect(text).toBe('Hello world');
+ expect(thinkContent).toBe('reasoning here');
+ });
+
+ it('should extract multiple think blocks', () => {
+ const input = 'first text second';
+ const { text, thinkContent } = extractThinkBlocks(input);
+ expect(text).toBe('text');
+ expect(thinkContent).toBe('first\n\nsecond');
+ });
+
+ it('should return null thinkContent when no think blocks', () => {
+ const { text, thinkContent } = extractThinkBlocks('just plain text');
+ expect(text).toBe('just plain text');
+ expect(thinkContent).toBeNull();
+ });
+
+ it('should handle multiline think blocks', () => {
+ const input = '\nline 1\nline 2\nresult';
+ const { text, thinkContent } = extractThinkBlocks(input);
+ expect(text).toBe('result');
+ expect(thinkContent).toBe('line 1\nline 2');
+ });
+
+ it('should handle empty think blocks', () => {
+ const input = 'text';
+ const { text, thinkContent } = extractThinkBlocks(input);
+ expect(text).toBe('text');
+ expect(thinkContent).toBeNull(); // empty think = null
+ });
+});
+
+describe('StreamingThinkExtractor', () => {
+ it('should extract think content from a single chunk', () => {
+ const ext = new StreamingThinkExtractor();
+ const result = ext.process('Hello reason world');
+ expect(result.visibleText).toBe('Hello world');
+ expect(result.completedThink).toBe('reason');
+ });
+
+ it('should handle think block spanning two chunks', () => {
+ const ext = new StreamingThinkExtractor();
+
+ const r1 = ext.process('Hello start of ');
+ expect(r1.visibleText).toBe('Hello ');
+ expect(r1.completedThink).toBeNull(); // not complete yet
+
+ const r2 = ext.process('reasoning world');
+ expect(r2.visibleText).toBe(' world');
+ expect(r2.completedThink).toBe('start of reasoning');
+ });
+
+ it('should handle think block spanning many chunks', () => {
+ const ext = new StreamingThinkExtractor();
+
+ ext.process('');
+ const r2 = ext.process('chunk 1 ');
+ expect(r2.completedThink).toBeNull();
+
+ const r3 = ext.process('chunk 2 ');
+ expect(r3.completedThink).toBeNull();
+
+ const r4 = ext.process('chunk 3done');
+ expect(r4.completedThink).toBe('chunk 1 chunk 2 chunk 3');
+ expect(r4.visibleText).toBe('done');
+ });
+
+ it('should handle partial tag at chunk boundary', () => {
+ const ext = new StreamingThinkExtractor();
+
+ // "reasoning');
+ expect(r2.completedThink).toBe('reasoning');
+ });
+
+ it('should handle text with no think blocks', () => {
+ const ext = new StreamingThinkExtractor();
+ const result = ext.process('just regular text');
+ expect(result.visibleText).toBe('just regular text');
+ expect(result.completedThink).toBeNull();
+ });
+
+ it('should reset state between streams', () => {
+ const ext = new StreamingThinkExtractor();
+
+ // Start a think block but don't finish
+ ext.process('partial');
+ ext.reset();
+
+ // After reset, should work cleanly
+ const result = ext.process('fresh text');
+ expect(result.visibleText).toBe('fresh text');
+ expect(result.completedThink).toBeNull();
+ });
+});
diff --git a/packages/core/src/utils/thinkBlockParser.ts b/packages/core/src/utils/thinkBlockParser.ts
new file mode 100644
index 00000000000..12cbd497309
--- /dev/null
+++ b/packages/core/src/utils/thinkBlockParser.ts
@@ -0,0 +1,134 @@
+/**
+ * Shared parser for ... XML tags emitted by reasoning models
+ * (DeepSeek-R1, QwQ, Qwen3-coder, etc).
+ *
+ * Instead of stripping think blocks, we extract them so they can be surfaced
+ * as { thought: true, text: "..." } parts — which the existing Turn system
+ * already handles via DeltaEventType.Thought.
+ */
+
+export interface ParsedContent {
+ /** Regular text content (outside think blocks) */
+ text: string;
+ /** Extracted think block content, if any */
+ thinkContent: string | null;
+}
+
+/**
+ * Extract ... blocks from a complete response string.
+ * Returns the visible text and the think content separately.
+ *
+ * Used by the non-streaming path where the full response is available.
+ */
+export function extractThinkBlocks(content: string): ParsedContent {
+ const thinkRegex = /([\s\S]*?)<\/think>/g;
+ const thinkParts: string[] = [];
+ let match;
+
+ while ((match = thinkRegex.exec(content)) !== null) {
+ const thinkText = match[1].trim();
+ if (thinkText) {
+ thinkParts.push(thinkText);
+ }
+ }
+
+ const text = content.replace(/[\s\S]*?<\/think>/g, '').trim();
+ const thinkContent = thinkParts.length > 0 ? thinkParts.join('\n\n') : null;
+
+ return { text, thinkContent };
+}
+
+// ---------------------------------------------------------------------------
+// Streaming think block extractor
+// ---------------------------------------------------------------------------
+
+/**
+ * Stateful streaming parser that handles tags spanning multiple chunks.
+ *
+ * Replaces the old filterThinkTags approach which stripped content. This version
+ * extracts think content into a separate buffer so it can be yielded as thought
+ * parts to the Turn system.
+ */
+export class StreamingThinkExtractor {
+ private insideThinkBlock = false;
+ private tagBuffer = '';
+ private thinkAccumulator = '';
+
+ /** Reset state between streams. */
+ reset(): void {
+ this.insideThinkBlock = false;
+ this.tagBuffer = '';
+ this.thinkAccumulator = '';
+ }
+
+ /**
+ * Process a streaming text chunk. Returns the visible text and any
+ * completed think content.
+ *
+ * Think content is only returned when a close tag is found,
+ * meaning the reasoning block is complete. Partial reasoning stays
+ * buffered until the block closes.
+ */
+ process(chunk: string): { visibleText: string; completedThink: string | null } {
+ // Prepend any buffered partial tag from previous chunk
+ let text = this.tagBuffer + chunk;
+ this.tagBuffer = '';
+
+ let visibleText = '';
+ let completedThink: string | null = null;
+ let i = 0;
+
+ while (i < text.length) {
+ if (this.insideThinkBlock) {
+ const closeIdx = text.indexOf('', i);
+ if (closeIdx !== -1) {
+ // Capture the think content up to the close tag
+ this.thinkAccumulator += text.substring(i, closeIdx);
+ completedThink = this.thinkAccumulator.trim() || null;
+ this.thinkAccumulator = '';
+ this.insideThinkBlock = false;
+ i = closeIdx + ''.length;
+ } else {
+ // No close tag yet — check for partial at chunk boundary
+ const tailLen = Math.min(text.length - i, 8);
+ const tail = text.substring(text.length - tailLen);
+ if (''.startsWith(tail) && tail.length < ''.length) {
+ // Buffer the potential partial tag, accumulate everything before it
+ this.thinkAccumulator += text.substring(i, text.length - tailLen);
+ this.tagBuffer = tail;
+ } else {
+ // Accumulate all remaining text as think content
+ this.thinkAccumulator += text.substring(i);
+ }
+ break;
+ }
+ } else {
+ const openIdx = text.indexOf('', i);
+ if (openIdx !== -1) {
+ // Emit text before the tag as visible
+ visibleText += text.substring(i, openIdx);
+ this.insideThinkBlock = true;
+ i = openIdx + ''.length;
+ } else {
+ // Check for partial at chunk boundary
+ let partialMatch = false;
+ for (let len = Math.min(7, text.length - i); len >= 1; len--) {
+ const candidate = text.substring(text.length - len);
+ if (''.startsWith(candidate)) {
+ visibleText += text.substring(i, text.length - len);
+ this.tagBuffer = candidate;
+ partialMatch = true;
+ break;
+ }
+ }
+ if (!partialMatch) {
+ visibleText += text.substring(i);
+ }
+ break;
+ }
+ }
+ }
+
+ return { visibleText, completedThink };
+ }
+}