From 8c5bd5201a7d9ef0f77b3b51a18ae0703f1e174a Mon Sep 17 00:00:00 2001 From: coryslater <25396141+fivestarspicy@users.noreply.github.com> Date: Fri, 7 Aug 2026 10:39:55 -0700 Subject: [PATCH] fix(aio): stop over-billing cache-heavy gemini generations The exclusive-reporting fallback treated any `input < cache_read + cache_write` as proof that a provider reports cache tokens as a separate pool. Inclusive providers can break that invariant by a few percent without changing accounting model: with Gemini explicit context caching, `cached_content_token_count` is counted when the cache is created while `prompt_token_count` is counted per request, so the two land slightly apart. Flipping those events to exclusive bills the full input at the prompt rate and the cache again at the cache rate, so a nearly fully cached prompt bills several times over. Require the cache pool to exceed input by more than 1.5x before resolving exclusive, and clamp the residual text pool to zero on every path so a small overshoot cannot bill a negative residual. Generated-By: PostHog Code Task-Id: 06160e48-feb9-4d39-8b7b-3dcfd1d9ca24 --- .../pipelines/ai/costs/input-costs.test.ts | 30 ++++++++++-- .../pipelines/ai/costs/input-costs.ts | 47 +++++++++++++------ 2 files changed, 59 insertions(+), 18 deletions(-) diff --git a/nodejs/src/ingestion/pipelines/ai/costs/input-costs.test.ts b/nodejs/src/ingestion/pipelines/ai/costs/input-costs.test.ts index 8455ae306b23..8a40c38bb570 100644 --- a/nodejs/src/ingestion/pipelines/ai/costs/input-costs.test.ts +++ b/nodejs/src/ingestion/pipelines/ai/costs/input-costs.test.ts @@ -170,6 +170,19 @@ describe('resolveCacheReportingExclusive()', () => { }, expected: true, }, + { + // Gemini explicit context caching counts `cached_content_token_count` when + // the cache is created and `prompt_token_count` per request, so an inclusive + // event can report a cache pool a few percent over its input total. + name: 'stays inclusive when cache reads exceed input tokens only slightly', + properties: { + $ai_provider: 'gemini', + $ai_model: 'models/gemini-3-flash-preview', + $ai_input_tokens: 23000, + $ai_cache_read_input_tokens: 25000, + }, + expected: false, + }, { name: 'stays inclusive for a non-Anthropic provider when cache reads fit within input tokens', properties: { @@ -628,6 +641,18 @@ describe('calculateInputCost()', () => { // Regular: 10000 * 0.00000125 = 0.0125 expectCostToBeCloseTo(result, 0.0125) }) + + it('clamps the residual to zero when cache reads slightly exceed input tokens', () => { + const event = createGeminiTestEvent(23000, 25000) + const result = calculateInputCost(event, GEMINI_MODEL) + + // Stays inclusive, so the whole prompt is cached and the residual text pool + // clamps to zero instead of going negative. + // Regular: max(23000 - 25000, 0) = 0 + // Read: 25000 * 3.1e-7 = 0.00775 + expectCostToBeCloseTo(result, 0.00775) + expect(event.properties!['$ai_cache_reporting_exclusive']).toBe(false) + }) }) describe('default provider - cache handling', () => { @@ -710,12 +735,11 @@ describe('calculateInputCost()', () => { expect(parseFloat(result)).toBeGreaterThan(0) }) - it('handles negative token counts gracefully', () => { + it('clamps negative token counts to zero rather than billing a negative cost', () => { const event = createOpenAITestEvent(-1000, undefined, { $ai_model: 'gpt-4' }) const result = calculateInputCost(event, testModel) - // Should calculate even with negative (though invalid in practice) - expect(parseFloat(result)).toBeLessThan(0) + expect(parseFloat(result)).toBe(0) }) it('handles null cache token values', () => { diff --git a/nodejs/src/ingestion/pipelines/ai/costs/input-costs.ts b/nodejs/src/ingestion/pipelines/ai/costs/input-costs.ts index 2526a9bd427a..a9e5eca9cafe 100644 --- a/nodejs/src/ingestion/pipelines/ai/costs/input-costs.ts +++ b/nodejs/src/ingestion/pipelines/ai/costs/input-costs.ts @@ -48,6 +48,15 @@ const hasNumericProperty = (event: PluginEvent, key: string): boolean => { ) } +/** + * How far the cache pool has to exceed the reported input total before an undeclared + * event is treated as exclusive-reporting. This sits in the empty band between the two + * populations that break the inclusive invariant: reporting slop on inclusive providers + * clusters a few percent over the input, while genuine exclusive reporting runs orders + * of magnitude over it. + */ +const EXCLUSIVE_DETECTION_RATIO = 1.5 + export const resolveCacheReportingExclusive = (event: PluginEvent): boolean => { if (!event.properties) { return false @@ -66,7 +75,21 @@ export const resolveCacheReportingExclusive = (event: PluginEvent): boolean => { const inputTokens = numericProperty(event, '$ai_input_tokens') const cacheReadTokens = numericProperty(event, '$ai_cache_read_input_tokens') const cacheWriteTokens = numericProperty(event, '$ai_cache_creation_input_tokens') - const provablyExclusive = inputTokens < cacheReadTokens + cacheWriteTokens + + // An inclusive provider can still report a cache pool slightly above its input + // total, because the two counts do not have to come from the same measurement. + // Gemini explicit context caching is the clearest case: `cached_content_token_count` + // is a property of the cache object and is counted when the cache is created, while + // `prompt_token_count` is counted per request. They describe the same tokens but + // land a few percent apart, so a bare `input < cache` break is not by itself proof + // of exclusive accounting. + // + // Only treat the break as proof when the cache pool exceeds the input by more than + // EXCLUSIVE_DETECTION_RATIO, because genuine exclusive reporting sends only the + // incremental turn as input and so overshoots by orders of magnitude rather than by + // a few percent. Inclusive events that trip the smaller break are handled by + // clamping the residual text pool to zero instead, in `clampTextTokens`. + const provablyExclusive = cacheReadTokens + cacheWriteTokens > inputTokens * EXCLUSIVE_DETECTION_RATIO if (provablyExclusive) { aiCacheExclusiveFallbackCounter.labels({ prior: anthropicStyle ? 'anthropic_inclusive' : 'inclusive' }).inc() @@ -76,16 +99,13 @@ export const resolveCacheReportingExclusive = (event: PluginEvent): boolean => { } /** - * Clamp the residual text-token pool to zero when modality tokens are present - * and the subtraction would push it negative. This guards against modality - * counts that overlap with cache tokens or exceed the reported input total — - * either case would otherwise produce a negative text contribution that - * silently offsets the modality bill. + * Clamp the residual text-token pool to zero whenever the subtraction would push it + * negative. A negative residual is never meaningful: it only ever means we + * over-subtracted, either because modality counts overlap the cache pool or because an + * inclusive provider reported a cache pool slightly above its input total. Billing it + * would silently credit back part of the cache and modality cost. */ -const clampTextTokens = (value: string | number, hasModalityTokens: boolean): string | number => { - if (!hasModalityTokens) { - return value - } +const clampTextTokens = (value: string | number): string | number => { const num = Number(value) return Number.isFinite(num) && num < 0 ? '0' : value } @@ -182,7 +202,6 @@ export const calculateInputCost = (event: PluginEvent, cost: ResolvedModelCost): const cachedAudioInputCost = computeCachedAudioInputCost(event, cost, cachedAudioInputTokens) const imageInputCost = computeImageInputCost(event, cost, imageInputTokens) const modalityInputCost = bigDecimal.add(bigDecimal.add(audioInputCost, cachedAudioInputCost), imageInputCost) - const hasModalityTokens = audioInputTokens > 0 || imageInputTokens > 0 // Text-only portion of the cache pool. Subtracting cached_audio gives us // the cached-text count, which we bill at the standard cache_read rate. @@ -222,8 +241,7 @@ export const calculateInputCost = (event: PluginEvent, cost: ResolvedModelCost): ? inputTokens : bigDecimal.subtract(bigDecimal.subtract(inputTokens, cachedTextTokens), cacheWriteTokens) const uncachedTextTokens = clampTextTokens( - bigDecimal.subtract(bigDecimal.subtract(baseUncachedTokens, audioInputTokens), imageInputTokens), - hasModalityTokens + bigDecimal.subtract(bigDecimal.subtract(baseUncachedTokens, audioInputTokens), imageInputTokens) ) const uncachedCost = bigDecimal.multiply(cost.cost.prompt_token, uncachedTextTokens) @@ -232,8 +250,7 @@ export const calculateInputCost = (event: PluginEvent, cost: ResolvedModelCost): const baseRegularTokens = exclusive ? inputTokens : bigDecimal.subtract(inputTokens, cachedTextTokens) const regularTextTokens = clampTextTokens( - bigDecimal.subtract(bigDecimal.subtract(baseRegularTokens, audioInputTokens), imageInputTokens), - hasModalityTokens + bigDecimal.subtract(bigDecimal.subtract(baseRegularTokens, audioInputTokens), imageInputTokens) ) let cacheReadCost: string