From cde35303d65c41e59410b33f777b87d2661c12a5 Mon Sep 17 00:00:00 2001 From: elhoim Date: Fri, 18 Sep 2026 09:32:30 +0000 Subject: [PATCH] fix(metrics): count one API response once when logged per content block Claude Code writes an assistant response to the transcript once per content block - a `thinking` line, a `tool_use` line, and so on - and every one of those lines repeats the same `usage` object, the same `message.id` and an increasing `apiBlockIndex`. The transcript fallback counted each line, so a two-block response was billed twice: Tokens Input / Output / Cached / Total and the speed widgets' request count all read roughly 2x on current transcripts. Continuation blocks are now skipped - `apiBlockIndex` when the transcript carries it, a repeated `message.id` otherwise - and in the speed path the later block only stretches the request's interval and takes its final usage, so a streamed response (`stop_reason: null` lines followed by the real count) is still counted exactly once. The live `context_window` path in the status JSON is untouched, and so is `contextLength`: repeats carry identical usage, so dropping them cannot move the most-recent-usage figures. Co-Authored-By: Claude Opus 5 (1M context) --- src/types/TokenMetrics.ts | 6 +- src/utils/__tests__/jsonl-metrics.test.ts | 163 ++++++++++++++++++++++ src/utils/jsonl-metrics.ts | 54 ++++++- 3 files changed, 219 insertions(+), 4 deletions(-) diff --git a/src/types/TokenMetrics.ts b/src/types/TokenMetrics.ts index 042385d92..a39b8df50 100644 --- a/src/types/TokenMetrics.ts +++ b/src/types/TokenMetrics.ts @@ -6,8 +6,12 @@ export interface TokenUsage { } export interface TranscriptLine { - message?: { usage?: TokenUsage; stop_reason?: string | null }; + message?: { id?: string; usage?: TokenUsage; stop_reason?: string | null }; isSidechain?: boolean; + // Index of the content block this line carries within its API response. + // Claude Code writes one line per block (thinking, text, tool_use, ...) and + // repeats the same `usage` on each of them, so only block 0 is billable. + apiBlockIndex?: number; timestamp?: string; isApiErrorMessage?: boolean; type?: 'user' | 'assistant' | 'system' | 'progress' | 'file-history-snapshot'; diff --git a/src/utils/__tests__/jsonl-metrics.test.ts b/src/utils/__tests__/jsonl-metrics.test.ts index 348ea432a..dad860ba5 100644 --- a/src/utils/__tests__/jsonl-metrics.test.ts +++ b/src/utils/__tests__/jsonl-metrics.test.ts @@ -69,12 +69,16 @@ function makeUsageLine(params: { isSidechain?: boolean; isApiErrorMessage?: boolean; stopReason?: string | null; + messageId?: string; + apiBlockIndex?: number; }): string { return JSON.stringify({ timestamp: params.timestamp, isSidechain: params.isSidechain, isApiErrorMessage: params.isApiErrorMessage, + apiBlockIndex: params.apiBlockIndex, message: { + id: params.messageId, stop_reason: params.stopReason, usage: { input_tokens: params.input, @@ -93,14 +97,18 @@ function makeTranscriptLine(params: { output?: number; isSidechain?: boolean; isApiErrorMessage?: boolean; + messageId?: string; + apiBlockIndex?: number; }): string { return JSON.stringify({ timestamp: params.timestamp, type: params.type, isSidechain: params.isSidechain, isApiErrorMessage: params.isApiErrorMessage, + apiBlockIndex: params.apiBlockIndex, message: typeof params.input === 'number' || typeof params.output === 'number' ? { + id: params.messageId, usage: { input_tokens: params.input ?? 0, output_tokens: params.output ?? 0 @@ -468,6 +476,161 @@ describe('jsonl transcript metrics', () => { }); }); + it('counts one API response once when it is logged per content block', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ccstatusline-jsonl-metrics-')); + tempRoots.push(root); + const transcriptPath = path.join(root, 'api-blocks.jsonl'); + + // Claude Code writes one line per content block of a single API + // response - here a thinking block and a tool_use block - and repeats + // the same usage on both. + const blocks = { + timestamp: '2026-01-01T10:00:00.000Z', + input: 100, + output: 50, + cacheRead: 20, + cacheCreate: 10, + stopReason: 'tool_use', + messageId: 'msg_1' + }; + const lines = [ + makeUsageLine({ ...blocks, apiBlockIndex: 0 }), + makeUsageLine({ ...blocks, apiBlockIndex: 1 }), + makeUsageLine({ + timestamp: '2026-01-01T10:01:00.000Z', + input: 5, + output: 7, + stopReason: 'end_turn', + messageId: 'msg_2', + apiBlockIndex: 0 + }) + ]; + fs.writeFileSync(transcriptPath, `${lines.join('\n')}\n`); + + const metrics = await getTokenMetrics(transcriptPath); + + expect(metrics).toEqual({ + inputTokens: 105, + outputTokens: 57, + cachedTokens: 30, + cacheReadTokens: 20, + cacheCreationTokens: 10, + totalTokens: 192, + contextLength: 5 + }); + }); + + it('falls back to the repeated message id when the transcript has no apiBlockIndex', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ccstatusline-jsonl-metrics-')); + tempRoots.push(root); + const transcriptPath = path.join(root, 'repeated-message-id.jsonl'); + + const blocks = { + timestamp: '2026-01-01T10:00:00.000Z', + input: 10, + output: 20, + stopReason: 'end_turn', + messageId: 'msg_1' + }; + fs.writeFileSync(transcriptPath, `${[ + makeUsageLine(blocks), + makeUsageLine(blocks) + ].join('\n')}\n`); + + const metrics = await getTokenMetrics(transcriptPath); + + expect(metrics.inputTokens).toBe(10); + expect(metrics.outputTokens).toBe(20); + }); + + it('keeps counting distinct responses that carry no message id', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ccstatusline-jsonl-metrics-')); + tempRoots.push(root); + const transcriptPath = path.join(root, 'no-message-id.jsonl'); + + // Without an id or a block index there is nothing to tell a repeat from + // a separate response, so both are counted as before. + fs.writeFileSync(transcriptPath, `${[ + makeUsageLine({ timestamp: '2026-01-01T10:00:00.000Z', input: 10, output: 20, stopReason: 'end_turn' }), + makeUsageLine({ timestamp: '2026-01-01T10:01:00.000Z', input: 10, output: 20, stopReason: 'end_turn' }) + ].join('\n')}\n`); + + const metrics = await getTokenMetrics(transcriptPath); + + expect(metrics.inputTokens).toBe(20); + expect(metrics.outputTokens).toBe(40); + }); + + it('still counts the final line of a streamed response that repeats its message id', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ccstatusline-jsonl-metrics-')); + tempRoots.push(root); + const transcriptPath = path.join(root, 'streamed-response.jsonl'); + + // In-flight lines carry `stop_reason: null` and a partial usage; the + // final line of the same message holds the real count. Skipping repeats + // must not swallow that final line. + fs.writeFileSync(transcriptPath, `${[ + makeUsageLine({ + timestamp: '2026-01-01T10:00:00.000Z', + input: 100, + output: 5, + stopReason: null, + messageId: 'msg_1' + }), + makeUsageLine({ + timestamp: '2026-01-01T10:00:05.000Z', + input: 100, + output: 50, + stopReason: 'end_turn', + messageId: 'msg_1' + }) + ].join('\n')}\n`); + + const metrics = await getTokenMetrics(transcriptPath); + + expect(metrics.inputTokens).toBe(100); + expect(metrics.outputTokens).toBe(50); + }); + + it('treats a per-block logged response as one request in speed metrics', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ccstatusline-jsonl-speed-')); + tempRoots.push(root); + const transcriptPath = path.join(root, 'api-blocks-speed.jsonl'); + + const lines = [ + makeTranscriptLine({ timestamp: '2026-01-01T10:00:00.000Z', type: 'user' }), + makeTranscriptLine({ + timestamp: '2026-01-01T10:00:10.000Z', + type: 'assistant', + input: 100, + output: 50, + messageId: 'msg_1', + apiBlockIndex: 0 + }), + makeTranscriptLine({ + timestamp: '2026-01-01T10:00:20.000Z', + type: 'assistant', + input: 100, + output: 50, + messageId: 'msg_1', + apiBlockIndex: 1 + }) + ]; + fs.writeFileSync(transcriptPath, `${lines.join('\n')}\n`); + + const metrics = await getSpeedMetrics(transcriptPath); + + // One request, counted once, over the span from the prompt to the last + // block of the response. + expect(metrics).toEqual({ + totalDurationMs: 20000, + inputTokens: 100, + outputTokens: 50, + totalTokens: 150, + requestCount: 1 + }); + }); + it('reports post-compaction context length from the boundary marker before the next turn', async () => { const root = fs.mkdtempSync(path.join(os.tmpdir(), 'ccstatusline-jsonl-metrics-')); tempRoots.push(root); diff --git a/src/utils/jsonl-metrics.ts b/src/utils/jsonl-metrics.ts index 88f4c47d8..aa72f9cf8 100644 --- a/src/utils/jsonl-metrics.ts +++ b/src/utils/jsonl-metrics.ts @@ -103,6 +103,7 @@ interface TokenMetricState { metrics: TokenMetricAccumulator; hasStopReasonField: boolean; lastUsageEntry: TokenMetricEntry | null; + lastCountedMessageId: string | undefined; sawCompactBoundary: boolean; boundaryAfterLastUsage: boolean; lastCompactBoundaryPostTokens: number | null; @@ -170,12 +171,31 @@ function createTokenMetricState(): TokenMetricState { metrics: createTokenMetricAccumulator(), hasStopReasonField: false, lastUsageEntry: null, + lastCountedMessageId: undefined, sawCompactBoundary: false, boundaryAfterLastUsage: false, lastCompactBoundaryPostTokens: null }; } +/** + * One API response is written to the transcript once per content block - a + * `thinking` line, a `tool_use` line, and so on - and every one of those lines + * repeats the same `usage` object, the same `message.id` and an increasing + * `apiBlockIndex`. Counting each line would multiply a single response's cost + * by its block count, so continuation blocks are skipped: `apiBlockIndex` when + * the transcript carries it, the repeated message id otherwise. + */ +function isUsageContinuationBlock(data: TranscriptLine | null, lastCountedMessageId: string | undefined): boolean { + const blockIndex = data?.apiBlockIndex; + if (typeof blockIndex === 'number' && blockIndex > 0) { + return true; + } + + const messageId = data?.message?.id; + return messageId !== undefined && messageId === lastCountedMessageId; +} + function collectTokenMetricRecord(state: TokenMetricState, data: TranscriptLine | null, timestampMs: number | null): void { const compactBoundary = isCompactBoundary(data); if (compactBoundary) { @@ -187,7 +207,7 @@ function collectTokenMetricRecord(state: TokenMetricState, data: TranscriptLine const message = data?.message; const usage = message?.usage; - if (usage) { + if (usage && !isUsageContinuationBlock(data, state.lastCountedMessageId)) { const entry: TokenMetricEntry = { usage: parseUsageTokens(usage), stopReason: message.stop_reason, @@ -202,6 +222,7 @@ function collectTokenMetricRecord(state: TokenMetricState, data: TranscriptLine } if (!state.hasStopReasonField || entry.stopReason) { accumulateTokenMetricEntry(state.metrics, entry, !compactBoundary); + state.lastCountedMessageId = message.id; } state.lastUsageEntry = entry; state.boundaryAfterLastUsage = compactBoundary; @@ -317,13 +338,17 @@ function normalizeWindowSeconds(value: number | undefined): number | null { return normalized > 0 ? normalized : null; } -interface SpeedMetricCollectorState extends CollectedSpeedMetrics { lastUserTimestampMs: number | null } +interface SpeedMetricCollectorState extends CollectedSpeedMetrics { + lastUserTimestampMs: number | null; + lastCountedMessageId: string | undefined; +} function createSpeedMetricCollector(): SpeedMetricCollectorState { return { requests: [], latestTimestampMs: null, - lastUserTimestampMs: null + lastUserTimestampMs: null, + lastCountedMessageId: undefined }; } @@ -355,6 +380,28 @@ function collectSpeedMetricRecord( interval = { startMs: state.lastUserTimestampMs, endMs: timestampMs }; } + // A continuation block is part of the request already recorded, so it only + // stretches that request's interval to when the later block landed. + if (isUsageContinuationBlock(data, state.lastCountedMessageId)) { + const pending = state.requests[state.requests.length - 1]; + if (pending) { + // The later line carries the response's final usage: identical for a + // block repeat, complete for a streamed one. + const latest = parseUsageTokens(data.message.usage); + pending.inputTokens = latest.input; + pending.outputTokens = latest.output; + + if (timestampMs !== null) { + pending.assistantTimestampMs = timestampMs; + if (pending.interval && timestampMs > pending.interval.endMs) { + pending.interval = { startMs: pending.interval.startMs, endMs: timestampMs }; + } + } + } + + return; + } + const usage = parseUsageTokens(data.message.usage); state.requests.push({ inputTokens: usage.input, @@ -362,6 +409,7 @@ function collectSpeedMetricRecord( assistantTimestampMs: timestampMs, interval }); + state.lastCountedMessageId = data.message.id; } async function collectSpeedMetricsFromFile(filePath: string, ignoreSidechain: boolean): Promise {