From 16977bb3060d63819f8637fa694a4106caf0f9ed Mon Sep 17 00:00:00 2001 From: AITNR Date: Fri, 18 Sep 2026 12:37:34 +0800 Subject: [PATCH 01/10] =?UTF-8?q?refactor(plugin):=20=E7=AE=80=E5=8C=96TPS?= =?UTF-8?q?=E8=AE=A1=E7=AE=97=E9=80=BB=E8=BE=91=E5=B9=B6=E7=A7=BB=E9=99=A4?= =?UTF-8?q?=E5=A4=8D=E6=9D=82=E7=9A=84=E6=8E=A8=E7=90=86token=E4=BC=9A?= =?UTF-8?q?=E8=AE=A1=E5=A4=84=E7=90=86?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 移除reasoningTokenAccounting相关常量和函数实现 - 删除reasoningAccountingForDimensions和effectiveOutputTokensForTPS函数 - 移除requestTPS函数,直接在requestDetailForUsage中计算TPS - 更新TPS计算方式,仅基于outputTokens进行计算 - 从README.md中移除关于TPS计算的文档说明 - 更新构建脚本中的版本变量引用路径 - 移除不再使用的strings导入依赖 --- README.md | 4 +-- internal/plugin/dashboard.go | 2 +- internal/plugin/locales/en.json | 1 + internal/plugin/locales/ru.json | 1 + internal/plugin/locales/zh-CN.json | 1 + internal/plugin/locales/zh-TW.json | 1 + internal/plugin/persistence.go | 5 +--- internal/plugin/request_log.go | 37 ++++++++++++++++++++--- internal/plugin/request_log_test.go | 46 +++++++++++++++++++++++++++++ 9 files changed, 87 insertions(+), 11 deletions(-) diff --git a/README.md b/README.md index 0b63be6..c1aae80 100644 --- a/README.md +++ b/README.md @@ -250,7 +250,7 @@ X-Full-Mode-Session: 统计、逐请求和费用接口支持 `range`,或 `start` 与 `end`,以及 `source` 等筛选参数。完整模式还支持重复的 `api_key_ref`,多个值按并集筛选;逐请求接口另支持 `offset`、`limit`、`model` 和 `result`。`/stats/groups` 另支持 `offset`、`limit`、`sort`、`direction`、`model` 和重复的 `exclude_model`;每页最多 500 条。统计和维度统计中的 `Groups` 行不按失败状态拆分:同一 provider、executor、model、alias、source、API key、auth type、service tier 和 reasoning effort 只有一行;`failed`/`failure_status` 在这些行中恒为 `false`/`0`,失败次数由 `failed_requests` 承载,逐请求状态保留在 `/requests`。 -逐请求明细中的 TPS 以生成时间(`latency_ns - ttft_ns`)为分母。对 Gemini、Vertex、AI Studio、Antigravity 和 Interactions 等独立思考协议,分子为 `output_tokens + reasoning_tokens`;对 OpenAI 兼容、Anthropic 等已将 reasoning 计入 output 的协议,分子仍为 `output_tokens`。 +逐请求明细中的 TPS 通常以生成时间(`latency_ns - ttft_ns`)为分母。对 Gemini、Vertex、AI Studio、Antigravity 和 Interactions 等独立思考协议,分子为 `output_tokens + reasoning_tokens`;对 OpenAI 兼容、Anthropic 等已将 reasoning 计入 output 的协议,分子仍为 `output_tokens`。如果独立思考协议的首包后窗口不超过 1 秒且出现超过 200 个输出 Token 或超过 500 TPS,记录会标记为缓冲流式,并改用完整 `latency_ns` 计算 TPS;仪表盘会在该值后显示 `*`。 重置请求正文: @@ -541,7 +541,7 @@ Management API routes: Statistics, request, and cost resources accept `range`, or `start` and `end`, plus filters such as `source`. Full mode also accepts repeated `api_key_ref` values and applies their union. The request resource additionally accepts `offset`, `limit`, `model`, and `result`. `/stats/groups` additionally accepts `offset`, `limit`, `sort`, `direction`, `model`, and repeated `exclude_model`; pages are limited to 500 rows. `Groups` rows in statistics and dimension statistics do not split by failure state: each provider, executor, model, alias, source, API key, auth type, service tier, and reasoning-effort combination has one row; `failed`/`failure_status` are always `false`/`0` in those rows, failures are counted in `failed_requests`, and per-request status remains available from `/requests`. -TPS in per-request details uses generation time (`latency_ns - ttft_ns`) as its denominator. For separate-reasoning protocols such as Gemini, Vertex, AI Studio, Antigravity, and Interactions, the numerator is `output_tokens + reasoning_tokens`; for protocols that already include reasoning in output, including OpenAI-compatible APIs and Anthropic, the numerator remains `output_tokens`. +TPS in per-request details normally uses generation time (`latency_ns - ttft_ns`) as its denominator. For separate-reasoning protocols such as Gemini, Vertex, AI Studio, Antigravity, and Interactions, the numerator is `output_tokens + reasoning_tokens`; for protocols that already include reasoning in output, including OpenAI-compatible APIs and Anthropic, the numerator remains `output_tokens`. If a separate-reasoning request has a post-first-token window of at most one second and either more than 200 output tokens or more than 500 TPS, it is marked as buffered and TPS is calculated with the full `latency_ns`; the dashboard adds `*` to that value. Reset body: diff --git a/internal/plugin/dashboard.go b/internal/plugin/dashboard.go index 0a346d2..18ded51 100644 --- a/internal/plugin/dashboard.go +++ b/internal/plugin/dashboard.go @@ -537,7 +537,7 @@ function closeRequestColumnsMenu(){toggleRequestColumnsMenu(false);} // translateRawResult maps the backend's Simplified-Chinese result literals // (see request_log.go requestDetailForUsage) to locale-translated strings. function translateRawResult(raw,failed){if(raw==='成功')return t('result.success');if(!raw)return failed?t('result.failed'):t('result.success');var m=raw.match(/^失败\s*\(HTTP\s+(\d+)\)$/);if(m)return t('result.failedHttp',{status:m[1]});if(raw.indexOf('失败')===0)return t('result.failed');return raw;} -function requestDataCell(row,item,column){var estimated=item.estimated_cost||null,td,timeValue;switch(column.key){case 'time':timeValue=item.time?new Date(item.time):null;td=cell(row,timeValue&&!Number.isNaN(timeValue.getTime())?requestTime(timeValue):t('locale.unavailable'));break;case 'model':td=cell(row,modelName(item.model));break;case 'source':td=cell(row,item.source);break;/*FULL_MODE_APIKEY_REQUEST_CELL*/case 'service_tier':td=cell(row,item.service_tier);break;case 'result':td=badgeCell(row,translateRawResult(item.result,item.failed),item.failed?'failed':'');break;case 'ttft_ns':td=cell(row,duration(item.ttft_ns),'num');break;case 'generation_ns':td=cell(row,duration(item.generation_ns),'num');break;case 'tps':td=cell(row,Number(item.tps||0).toFixed(2),'num');break;case 'reasoning_effort':td=cell(row,item.reasoning_effort);break;case 'input_tokens':td=cell(row,formatTokenTotal(item.input_tokens),'num');break;case 'output_tokens':td=cell(row,formatTokenTotal(item.output_tokens),'num');break;case 'reasoning_tokens':td=cell(row,formatTokenTotal(item.reasoning_tokens),'num');break;case 'cache_read_tokens':td=cell(row,formatTokenTotal(item.cache_read_tokens),'num');break;case 'cache_creation_tokens':td=cell(row,formatTokenTotal(item.cache_creation_tokens),'num');break;case 'total_tokens':td=cell(row,formatTokenTotal(item.total_tokens),'num');break;case 'cache_hit':td=badgeCell(row,item.cache_hit?t('cache.hit'):t('cache.miss'),item.cache_hit?'':'cache-miss');break;case 'cache_hit_rate':td=cell(row,formatCacheHitRate(cacheHitRateForPoint(item)),'num');break;case 'estimated_cost':td=cell(row,estimated&&estimated.priced?money(estimated.total_usd):t('pricing.unpriced'),'num');td.title=costBreakdownTitle(estimated);break;case 'price_source':td=cell(row,estimated&&estimated.priced?(estimated.source||t('locale.unavailable')):t('locale.unavailable'));td.title=estimated&&estimated.priced?t('pricing.costBreakdown',{input:money(estimated.input_usd),output:money(estimated.output_usd),cacheRead:money(estimated.cache_read_usd),cacheCreation:money(estimated.cache_creation_usd),mode:estimated.accounting_mode||t('locale.unavailable'),tier:estimated.tier_threshold?fmt(estimated.tier_threshold):t('value.base')})+' · '+formatTokenTotal(estimated.context_tokens)+' · '+formatTokenTotal(estimated.billable_input_tokens)+' · '+formatTokenTotal(estimated.billed_cache_read_tokens):t('pricing.noPrice');break;}if(td)td.dataset.column=column.key;return td;} +function requestDataCell(row,item,column){var estimated=item.estimated_cost||null,td,timeValue;switch(column.key){case 'time':timeValue=item.time?new Date(item.time):null;td=cell(row,timeValue&&!Number.isNaN(timeValue.getTime())?requestTime(timeValue):t('locale.unavailable'));break;case 'model':td=cell(row,modelName(item.model));break;case 'source':td=cell(row,item.source);break;/*FULL_MODE_APIKEY_REQUEST_CELL*/case 'service_tier':td=cell(row,item.service_tier);break;case 'result':td=badgeCell(row,translateRawResult(item.result,item.failed),item.failed?'failed':'');break;case 'ttft_ns':td=cell(row,duration(item.ttft_ns),'num');break;case 'generation_ns':td=cell(row,duration(item.generation_ns),'num');break;case 'tps':td=cell(row,Number(item.tps||0).toFixed(2)+(item.tps_basis==='latency_buffered'?'*':''),'num');if(item.tps_basis==='latency_buffered')td.title=t('table.tpsBuffered');break;case 'reasoning_effort':td=cell(row,item.reasoning_effort);break;case 'input_tokens':td=cell(row,formatTokenTotal(item.input_tokens),'num');break;case 'output_tokens':td=cell(row,formatTokenTotal(item.output_tokens),'num');break;case 'reasoning_tokens':td=cell(row,formatTokenTotal(item.reasoning_tokens),'num');break;case 'cache_read_tokens':td=cell(row,formatTokenTotal(item.cache_read_tokens),'num');break;case 'cache_creation_tokens':td=cell(row,formatTokenTotal(item.cache_creation_tokens),'num');break;case 'total_tokens':td=cell(row,formatTokenTotal(item.total_tokens),'num');break;case 'cache_hit':td=badgeCell(row,item.cache_hit?t('cache.hit'):t('cache.miss'),item.cache_hit?'':'cache-miss');break;case 'cache_hit_rate':td=cell(row,formatCacheHitRate(cacheHitRateForPoint(item)),'num');break;case 'estimated_cost':td=cell(row,estimated&&estimated.priced?money(estimated.total_usd):t('pricing.unpriced'),'num');td.title=costBreakdownTitle(estimated);break;case 'price_source':td=cell(row,estimated&&estimated.priced?(estimated.source||t('locale.unavailable')):t('locale.unavailable'));td.title=estimated&&estimated.priced?t('pricing.costBreakdown',{input:money(estimated.input_usd),output:money(estimated.output_usd),cacheRead:money(estimated.cache_read_usd),cacheCreation:money(estimated.cache_creation_usd),mode:estimated.accounting_mode||t('locale.unavailable'),tier:estimated.tier_threshold?fmt(estimated.tier_threshold):t('value.base')})+' · '+formatTokenTotal(estimated.context_tokens)+' · '+formatTokenTotal(estimated.billable_input_tokens)+' · '+formatTokenTotal(estimated.billed_cache_read_tokens):t('pricing.noPrice');break;}if(td)td.dataset.column=column.key;return td;} function renderRequests(page){currentRequestPage=page;requestTotal=Number(page.total||0);var items=sortedRequestItems(page.items||[]),columns=visibleRequestColumns(),body=document.getElementById('requestRows'),fragment=document.createDocumentFragment();renderRequestColumnControls();if(!items.length){var row=document.createElement('tr'),empty=document.createElement('td');empty.colSpan=Math.max(1,columns.length);empty.className='empty';empty.textContent=t('empty.requests');row.appendChild(empty);fragment.appendChild(row);}else{items.forEach(function(item){var row=document.createElement('tr');columns.forEach(function(column){requestDataCell(row,item,column);});fragment.appendChild(row);});}body.replaceChildren(fragment);var startIndex=requestTotal?requestOffset+1:0,endIndex=Math.min(requestOffset+(page.items||[]).length,requestTotal);text('requestCount',t('pagination.count',{count:fmt(requestTotal),revision:Number(page.price_book_revision||0)}));text('requestPageStatus',t('pagination.page',{start:startIndex,end:endIndex,total:requestTotal}));document.getElementById('requestPrev').disabled=requestOffset<=0;document.getElementById('requestNext').disabled=requestOffset+requestLimit>=requestTotal;} function setRequestFilterOptions(id,allLabel,values,selected){var select=document.getElementById(id);if(!select)return;var unique=Array.from(new Set(values.filter(Boolean))).sort(function(a,b){return String(a).localeCompare(String(b),formatterLocale,{numeric:true,sensitivity:'base'});});if(selected&&!unique.includes(selected))unique.unshift(selected);var fragment=document.createDocumentFragment(),all=document.createElement('option');all.value='';all.textContent=allLabel;fragment.appendChild(all);unique.forEach(function(value){var option=document.createElement('option');option.value=value;option.textContent=value;fragment.appendChild(option);});select.replaceChildren(fragment);select.value=selected;syncEnhancedSelect(select);} function renderRequestFilters(){var groups=modelRows(),models=groupModels(groups).map(function(item){return item.model;}),sources=(currentData&¤tData.sources)||[];setRequestFilterOptions('requestModelFilter',t('requestFilter.allModels'),models,requestModelFilter);setRequestFilterOptions('requestSourceFilter',t('requestFilter.allSources'),sources,requestSourceFilter);var result=document.getElementById('requestResultFilter');if(result){var options=[['',t('requestFilter.allResults')],['success',t('result.success')],['failed',t('result.failed')]],fragment=document.createDocumentFragment();options.forEach(function(item){var option=document.createElement('option');option.value=item[0];option.textContent=item[1];fragment.appendChild(option);});result.replaceChildren(fragment);result.value=requestResultFilter;syncEnhancedSelect(result);}} diff --git a/internal/plugin/locales/en.json b/internal/plugin/locales/en.json index bfb1f54..69a12e8 100644 --- a/internal/plugin/locales/en.json +++ b/internal/plugin/locales/en.json @@ -160,6 +160,7 @@ "table.ttft": "Time to first token", "table.generation": "Generation time", "table.tps": "TPS", + "table.tpsBuffered": "TPS uses total latency because the upstream stream appears buffered", "table.reasoningEffort": "Reasoning effort", "table.input": "Input", "table.output": "Output", diff --git a/internal/plugin/locales/ru.json b/internal/plugin/locales/ru.json index fbc7e87..0174293 100644 --- a/internal/plugin/locales/ru.json +++ b/internal/plugin/locales/ru.json @@ -160,6 +160,7 @@ "table.ttft": "Задержка первого токена", "table.generation": "Время генерации", "table.tps": "TPS", + "table.tpsBuffered": "Поток upstream похож на буферизованный, TPS рассчитан по полной задержке", "table.reasoningEffort": "Интенсивность рассуждений", "table.input": "Ввод", "table.output": "Вывод", diff --git a/internal/plugin/locales/zh-CN.json b/internal/plugin/locales/zh-CN.json index 8f51b5b..70a403d 100644 --- a/internal/plugin/locales/zh-CN.json +++ b/internal/plugin/locales/zh-CN.json @@ -160,6 +160,7 @@ "table.ttft": "首字延迟", "table.generation": "生成时间", "table.tps": "TPS", + "table.tpsBuffered": "上游流式传输疑似缓冲,TPS 使用总延迟估算", "table.reasoningEffort": "思考强度", "table.input": "输入", "table.output": "输出", diff --git a/internal/plugin/locales/zh-TW.json b/internal/plugin/locales/zh-TW.json index 153219f..e476e60 100644 --- a/internal/plugin/locales/zh-TW.json +++ b/internal/plugin/locales/zh-TW.json @@ -160,6 +160,7 @@ "table.ttft": "首字延遲", "table.generation": "生成時間", "table.tps": "TPS", + "table.tpsBuffered": "上游串流疑似緩衝,TPS 使用總延遲估算", "table.reasoningEffort": "思考強度", "table.input": "輸入", "table.output": "輸出", diff --git a/internal/plugin/persistence.go b/internal/plugin/persistence.go index 033f99e..76ccd6a 100644 --- a/internal/plugin/persistence.go +++ b/internal/plugin/persistence.go @@ -2485,10 +2485,7 @@ func (a *storeActor) queryRequests(queryRange usageRange, offset, limit int, mod return fmt.Errorf("decode request detail: %w", err) } item.Dimensions = sanitizeDimensionsSource(item.Dimensions) - // Stored records do not retain whether TotalTokens was explicit. Use - // the conservative unknown-provider path here; known protocol - // semantics still correct historical separate-reasoning records. - item.TPS = requestTPS(item, false) + item.TPS, item.TPSBasis = requestTPSWithBasis(item, false) if model != "" && !modelFilterMatches(model, item.Model) { continue } diff --git a/internal/plugin/request_log.go b/internal/plugin/request_log.go index 3386fca..4e08911 100644 --- a/internal/plugin/request_log.go +++ b/internal/plugin/request_log.go @@ -9,6 +9,10 @@ import ( const ( defaultRequestPageSize = 100 maxRequestPageSize = 500 + + bufferedStreamMaxGenerationNS = uint64(time.Second) + bufferedStreamMinTokens = uint64(200) + bufferedStreamMaxTPS = 500.0 ) // RequestDetail contains metadata and usage counters for one model request. @@ -23,6 +27,7 @@ type RequestDetail struct { TTFTNS uint64 `json:"ttft_ns"` GenerationNS uint64 `json:"generation_ns"` TPS float64 `json:"tps"` + TPSBasis string `json:"tps_basis,omitempty"` CacheHit bool `json:"cache_hit"` EstimatedCost *EstimatedCost `json:"estimated_cost,omitempty"` } @@ -117,12 +122,36 @@ func effectiveOutputTokensForTPS(dimensions Dimensions, counters Counters, expli return counters.OutputTokens } -func requestTPS(item RequestDetail, explicitTotal bool) float64 { +func requestTPSWithBasis(item RequestDetail, explicitTotal bool) (float64, string) { if item.GenerationNS == 0 { - return 0 + return 0, "" } outputTokens := effectiveOutputTokensForTPS(item.Dimensions, item.Counters, explicitTotal) - return float64(outputTokens) / (float64(item.GenerationNS) / float64(time.Second)) + basis := "generation" + denominator := item.GenerationNS + if likelyBufferedStream(item, outputTokens) { + basis = "latency_buffered" + denominator = item.LatencyNS + } + if denominator == 0 { + return 0, basis + } + return float64(outputTokens) / (float64(denominator) / float64(time.Second)), basis +} + +func requestTPS(item RequestDetail, explicitTotal bool) float64 { + tps, _ := requestTPSWithBasis(item, explicitTotal) + return tps +} + +func likelyBufferedStream(item RequestDetail, outputTokens uint64) bool { + if reasoningAccountingForDimensions(item.Dimensions) != reasoningAccountingSeparateFromOutput || + item.TTFTNS == 0 || item.LatencyNS == 0 || item.GenerationNS == 0 || item.GenerationNS > bufferedStreamMaxGenerationNS || + item.LatencyNS < item.GenerationNS { + return false + } + rawTPS := float64(outputTokens) / (float64(item.GenerationNS) / float64(time.Second)) + return outputTokens > bufferedStreamMinTokens || rawTPS > bufferedStreamMaxTPS } func requestDetailForUsage(usage normalizedUsage, sequence uint64) RequestDetail { @@ -148,6 +177,6 @@ func requestDetailForUsage(usage normalizedUsage, sequence uint64) RequestDetail GenerationNS: generationNS, CacheHit: usage.Counters.CacheReadTokens > 0, } - item.TPS = requestTPS(item, usage.explicitTotalTokens) + item.TPS, item.TPSBasis = requestTPSWithBasis(item, usage.explicitTotalTokens) return item } diff --git a/internal/plugin/request_log_test.go b/internal/plugin/request_log_test.go index db1dc9b..3f15e05 100644 --- a/internal/plugin/request_log_test.go +++ b/internal/plugin/request_log_test.go @@ -30,6 +30,52 @@ func TestRequestDetailForUsageSeparateReasoningTPS(t *testing.T) { } } +func TestRequestDetailForUsageBufferedSeparateReasoningTPS(t *testing.T) { + usage := normalizedUsage{ + Dimensions: Dimensions{Provider: "gemini", Model: "gemini-thinking"}, + RequestedAt: time.Date(2026, 9, 17, 5, 10, 12, 0, time.UTC), + LatencyNS: uint64(8600 * time.Millisecond), + TTFTNS: uint64(7976 * time.Millisecond), + Counters: Counters{ + Requests: 1, + InputTokens: 200000, + OutputTokens: 966, + ReasoningTokens: 721, + TotalTokens: 202687, + }, + } + item := requestDetailForUsage(usage, 1) + if item.GenerationNS != uint64(624*time.Millisecond) { + t.Fatalf("generation time = %d, want %d", item.GenerationNS, 624*time.Millisecond) + } + wantTPS := 1687.0 / 8.6 + if math.Abs(item.TPS-wantTPS) > 1e-9 { + t.Fatalf("TPS = %v, want %v", item.TPS, wantTPS) + } + if item.TPSBasis != "latency_buffered" { + t.Fatalf("TPS basis = %q, want latency_buffered", item.TPSBasis) + } +} + +func TestRequestDetailForUsageDoesNotFlagSmallNormalStream(t *testing.T) { + usage := normalizedUsage{ + Dimensions: Dimensions{Provider: "gemini", Model: "gemini-thinking"}, + LatencyNS: uint64(3 * time.Second), + TTFTNS: uint64(2 * time.Second), + Counters: Counters{ + OutputTokens: 40, + ReasoningTokens: 76, + }, + } + item := requestDetailForUsage(usage, 1) + if item.TPS != 116 { + t.Fatalf("TPS = %v, want 116", item.TPS) + } + if item.TPSBasis != "generation" { + t.Fatalf("TPS basis = %q, want generation", item.TPSBasis) + } +} + func TestEffectiveOutputTokensForTPS(t *testing.T) { tests := []struct { name string From c1ae08477292eb1a2d19581f3d77b7fe740e8894 Mon Sep 17 00:00:00 2001 From: AITNR Date: Fri, 18 Sep 2026 13:42:05 +0800 Subject: [PATCH 02/10] =?UTF-8?q?fix(dashboard):=20=E4=BF=AE=E5=A4=8DTPS?= =?UTF-8?q?=E8=AE=A1=E7=AE=97=E5=9C=A8=E7=9F=AD=E5=BB=B6=E8=BF=9F=E6=83=85?= =?UTF-8?q?=E5=86=B5=E4=B8=8B=E4=B8=8D=E5=8F=AF=E9=9D=A0=E7=9A=84=E9=97=AE?= =?UTF-8?q?=E9=A2=98?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 当TPS基础为latency_buffered且TPS值过大时,返回latency_unreliable状态 - 在UI显示中对latency_unreliable情况显示破折号而不是错误数值 - 添加新的本地化字符串提示TPS不可靠的原因 - 为短延迟但高token数的情况添加测试用例验证TPS计算逻辑 --- internal/plugin/dashboard.go | 2 +- internal/plugin/locales/en.json | 1 + internal/plugin/locales/ru.json | 1 + internal/plugin/locales/zh-CN.json | 1 + internal/plugin/locales/zh-TW.json | 1 + internal/plugin/request_log.go | 9 ++++++++- internal/plugin/request_log_test.go | 19 +++++++++++++++++++ 7 files changed, 32 insertions(+), 2 deletions(-) diff --git a/internal/plugin/dashboard.go b/internal/plugin/dashboard.go index 18ded51..8234572 100644 --- a/internal/plugin/dashboard.go +++ b/internal/plugin/dashboard.go @@ -537,7 +537,7 @@ function closeRequestColumnsMenu(){toggleRequestColumnsMenu(false);} // translateRawResult maps the backend's Simplified-Chinese result literals // (see request_log.go requestDetailForUsage) to locale-translated strings. function translateRawResult(raw,failed){if(raw==='成功')return t('result.success');if(!raw)return failed?t('result.failed'):t('result.success');var m=raw.match(/^失败\s*\(HTTP\s+(\d+)\)$/);if(m)return t('result.failedHttp',{status:m[1]});if(raw.indexOf('失败')===0)return t('result.failed');return raw;} -function requestDataCell(row,item,column){var estimated=item.estimated_cost||null,td,timeValue;switch(column.key){case 'time':timeValue=item.time?new Date(item.time):null;td=cell(row,timeValue&&!Number.isNaN(timeValue.getTime())?requestTime(timeValue):t('locale.unavailable'));break;case 'model':td=cell(row,modelName(item.model));break;case 'source':td=cell(row,item.source);break;/*FULL_MODE_APIKEY_REQUEST_CELL*/case 'service_tier':td=cell(row,item.service_tier);break;case 'result':td=badgeCell(row,translateRawResult(item.result,item.failed),item.failed?'failed':'');break;case 'ttft_ns':td=cell(row,duration(item.ttft_ns),'num');break;case 'generation_ns':td=cell(row,duration(item.generation_ns),'num');break;case 'tps':td=cell(row,Number(item.tps||0).toFixed(2)+(item.tps_basis==='latency_buffered'?'*':''),'num');if(item.tps_basis==='latency_buffered')td.title=t('table.tpsBuffered');break;case 'reasoning_effort':td=cell(row,item.reasoning_effort);break;case 'input_tokens':td=cell(row,formatTokenTotal(item.input_tokens),'num');break;case 'output_tokens':td=cell(row,formatTokenTotal(item.output_tokens),'num');break;case 'reasoning_tokens':td=cell(row,formatTokenTotal(item.reasoning_tokens),'num');break;case 'cache_read_tokens':td=cell(row,formatTokenTotal(item.cache_read_tokens),'num');break;case 'cache_creation_tokens':td=cell(row,formatTokenTotal(item.cache_creation_tokens),'num');break;case 'total_tokens':td=cell(row,formatTokenTotal(item.total_tokens),'num');break;case 'cache_hit':td=badgeCell(row,item.cache_hit?t('cache.hit'):t('cache.miss'),item.cache_hit?'':'cache-miss');break;case 'cache_hit_rate':td=cell(row,formatCacheHitRate(cacheHitRateForPoint(item)),'num');break;case 'estimated_cost':td=cell(row,estimated&&estimated.priced?money(estimated.total_usd):t('pricing.unpriced'),'num');td.title=costBreakdownTitle(estimated);break;case 'price_source':td=cell(row,estimated&&estimated.priced?(estimated.source||t('locale.unavailable')):t('locale.unavailable'));td.title=estimated&&estimated.priced?t('pricing.costBreakdown',{input:money(estimated.input_usd),output:money(estimated.output_usd),cacheRead:money(estimated.cache_read_usd),cacheCreation:money(estimated.cache_creation_usd),mode:estimated.accounting_mode||t('locale.unavailable'),tier:estimated.tier_threshold?fmt(estimated.tier_threshold):t('value.base')})+' · '+formatTokenTotal(estimated.context_tokens)+' · '+formatTokenTotal(estimated.billable_input_tokens)+' · '+formatTokenTotal(estimated.billed_cache_read_tokens):t('pricing.noPrice');break;}if(td)td.dataset.column=column.key;return td;} +function requestDataCell(row,item,column){var estimated=item.estimated_cost||null,td,timeValue;switch(column.key){case 'time':timeValue=item.time?new Date(item.time):null;td=cell(row,timeValue&&!Number.isNaN(timeValue.getTime())?requestTime(timeValue):t('locale.unavailable'));break;case 'model':td=cell(row,modelName(item.model));break;case 'source':td=cell(row,item.source);break;/*FULL_MODE_APIKEY_REQUEST_CELL*/case 'service_tier':td=cell(row,item.service_tier);break;case 'result':td=badgeCell(row,translateRawResult(item.result,item.failed),item.failed?'failed':'');break;case 'ttft_ns':td=cell(row,duration(item.ttft_ns),'num');break;case 'generation_ns':td=cell(row,duration(item.generation_ns),'num');break;case 'tps':td=cell(row,item.tps_basis==='latency_unreliable'?'—':Number(item.tps||0).toFixed(2)+(item.tps_basis==='latency_buffered'?'*':''),'num');if(item.tps_basis==='latency_buffered')td.title=t('table.tpsBuffered');if(item.tps_basis==='latency_unreliable')td.title=t('table.tpsUnreliable');break;case 'reasoning_effort':td=cell(row,item.reasoning_effort);break;case 'input_tokens':td=cell(row,formatTokenTotal(item.input_tokens),'num');break;case 'output_tokens':td=cell(row,formatTokenTotal(item.output_tokens),'num');break;case 'reasoning_tokens':td=cell(row,formatTokenTotal(item.reasoning_tokens),'num');break;case 'cache_read_tokens':td=cell(row,formatTokenTotal(item.cache_read_tokens),'num');break;case 'cache_creation_tokens':td=cell(row,formatTokenTotal(item.cache_creation_tokens),'num');break;case 'total_tokens':td=cell(row,formatTokenTotal(item.total_tokens),'num');break;case 'cache_hit':td=badgeCell(row,item.cache_hit?t('cache.hit'):t('cache.miss'),item.cache_hit?'':'cache-miss');break;case 'cache_hit_rate':td=cell(row,formatCacheHitRate(cacheHitRateForPoint(item)),'num');break;case 'estimated_cost':td=cell(row,estimated&&estimated.priced?money(estimated.total_usd):t('pricing.unpriced'),'num');td.title=costBreakdownTitle(estimated);break;case 'price_source':td=cell(row,estimated&&estimated.priced?(estimated.source||t('locale.unavailable')):t('locale.unavailable'));td.title=estimated&&estimated.priced?t('pricing.costBreakdown',{input:money(estimated.input_usd),output:money(estimated.output_usd),cacheRead:money(estimated.cache_read_usd),cacheCreation:money(estimated.cache_creation_usd),mode:estimated.accounting_mode||t('locale.unavailable'),tier:estimated.tier_threshold?fmt(estimated.tier_threshold):t('value.base')})+' · '+formatTokenTotal(estimated.context_tokens)+' · '+formatTokenTotal(estimated.billable_input_tokens)+' · '+formatTokenTotal(estimated.billed_cache_read_tokens):t('pricing.noPrice');break;}if(td)td.dataset.column=column.key;return td;} function renderRequests(page){currentRequestPage=page;requestTotal=Number(page.total||0);var items=sortedRequestItems(page.items||[]),columns=visibleRequestColumns(),body=document.getElementById('requestRows'),fragment=document.createDocumentFragment();renderRequestColumnControls();if(!items.length){var row=document.createElement('tr'),empty=document.createElement('td');empty.colSpan=Math.max(1,columns.length);empty.className='empty';empty.textContent=t('empty.requests');row.appendChild(empty);fragment.appendChild(row);}else{items.forEach(function(item){var row=document.createElement('tr');columns.forEach(function(column){requestDataCell(row,item,column);});fragment.appendChild(row);});}body.replaceChildren(fragment);var startIndex=requestTotal?requestOffset+1:0,endIndex=Math.min(requestOffset+(page.items||[]).length,requestTotal);text('requestCount',t('pagination.count',{count:fmt(requestTotal),revision:Number(page.price_book_revision||0)}));text('requestPageStatus',t('pagination.page',{start:startIndex,end:endIndex,total:requestTotal}));document.getElementById('requestPrev').disabled=requestOffset<=0;document.getElementById('requestNext').disabled=requestOffset+requestLimit>=requestTotal;} function setRequestFilterOptions(id,allLabel,values,selected){var select=document.getElementById(id);if(!select)return;var unique=Array.from(new Set(values.filter(Boolean))).sort(function(a,b){return String(a).localeCompare(String(b),formatterLocale,{numeric:true,sensitivity:'base'});});if(selected&&!unique.includes(selected))unique.unshift(selected);var fragment=document.createDocumentFragment(),all=document.createElement('option');all.value='';all.textContent=allLabel;fragment.appendChild(all);unique.forEach(function(value){var option=document.createElement('option');option.value=value;option.textContent=value;fragment.appendChild(option);});select.replaceChildren(fragment);select.value=selected;syncEnhancedSelect(select);} function renderRequestFilters(){var groups=modelRows(),models=groupModels(groups).map(function(item){return item.model;}),sources=(currentData&¤tData.sources)||[];setRequestFilterOptions('requestModelFilter',t('requestFilter.allModels'),models,requestModelFilter);setRequestFilterOptions('requestSourceFilter',t('requestFilter.allSources'),sources,requestSourceFilter);var result=document.getElementById('requestResultFilter');if(result){var options=[['',t('requestFilter.allResults')],['success',t('result.success')],['failed',t('result.failed')]],fragment=document.createDocumentFragment();options.forEach(function(item){var option=document.createElement('option');option.value=item[0];option.textContent=item[1];fragment.appendChild(option);});result.replaceChildren(fragment);result.value=requestResultFilter;syncEnhancedSelect(result);}} diff --git a/internal/plugin/locales/en.json b/internal/plugin/locales/en.json index 69a12e8..1146780 100644 --- a/internal/plugin/locales/en.json +++ b/internal/plugin/locales/en.json @@ -161,6 +161,7 @@ "table.generation": "Generation time", "table.tps": "TPS", "table.tpsBuffered": "TPS uses total latency because the upstream stream appears buffered", + "table.tpsUnreliable": "TPS is unavailable because the request timing is inconsistent with the token count", "table.reasoningEffort": "Reasoning effort", "table.input": "Input", "table.output": "Output", diff --git a/internal/plugin/locales/ru.json b/internal/plugin/locales/ru.json index 0174293..9ec97ee 100644 --- a/internal/plugin/locales/ru.json +++ b/internal/plugin/locales/ru.json @@ -161,6 +161,7 @@ "table.generation": "Время генерации", "table.tps": "TPS", "table.tpsBuffered": "Поток upstream похож на буферизованный, TPS рассчитан по полной задержке", + "table.tpsUnreliable": "TPS недоступен: время запроса не согласуется с числом токенов", "table.reasoningEffort": "Интенсивность рассуждений", "table.input": "Ввод", "table.output": "Вывод", diff --git a/internal/plugin/locales/zh-CN.json b/internal/plugin/locales/zh-CN.json index 70a403d..7ceb208 100644 --- a/internal/plugin/locales/zh-CN.json +++ b/internal/plugin/locales/zh-CN.json @@ -161,6 +161,7 @@ "table.generation": "生成时间", "table.tps": "TPS", "table.tpsBuffered": "上游流式传输疑似缓冲,TPS 使用总延迟估算", + "table.tpsUnreliable": "请求时序与 Token 数量不一致,TPS 不可可靠计算", "table.reasoningEffort": "思考强度", "table.input": "输入", "table.output": "输出", diff --git a/internal/plugin/locales/zh-TW.json b/internal/plugin/locales/zh-TW.json index e476e60..dc485be 100644 --- a/internal/plugin/locales/zh-TW.json +++ b/internal/plugin/locales/zh-TW.json @@ -161,6 +161,7 @@ "table.generation": "生成時間", "table.tps": "TPS", "table.tpsBuffered": "上游串流疑似緩衝,TPS 使用總延遲估算", + "table.tpsUnreliable": "請求時序與 Token 數量不一致,TPS 無法可靠計算", "table.reasoningEffort": "思考強度", "table.input": "輸入", "table.output": "輸出", diff --git a/internal/plugin/request_log.go b/internal/plugin/request_log.go index 4e08911..81078d0 100644 --- a/internal/plugin/request_log.go +++ b/internal/plugin/request_log.go @@ -136,7 +136,14 @@ func requestTPSWithBasis(item RequestDetail, explicitTotal bool) (float64, strin if denominator == 0 { return 0, basis } - return float64(outputTokens) / (float64(denominator) / float64(time.Second)), basis + tps := float64(outputTokens) / (float64(denominator) / float64(time.Second)) + if basis == "latency_buffered" && tps > bufferedStreamMaxTPS { + // A short total latency cannot be a trustworthy estimate of model + // generation time either. Do not replace one explosive value with + // another; expose the measurement as unavailable instead. + return 0, "latency_unreliable" + } + return tps, basis } func requestTPS(item RequestDetail, explicitTotal bool) float64 { diff --git a/internal/plugin/request_log_test.go b/internal/plugin/request_log_test.go index 3f15e05..1b15fe3 100644 --- a/internal/plugin/request_log_test.go +++ b/internal/plugin/request_log_test.go @@ -76,6 +76,25 @@ func TestRequestDetailForUsageDoesNotFlagSmallNormalStream(t *testing.T) { } } +func TestRequestDetailForUsageDoesNotInventTPSFromShortBufferedLatency(t *testing.T) { + usage := normalizedUsage{ + Dimensions: Dimensions{Provider: "gemini", Model: "gemini-thinking"}, + LatencyNS: uint64(10 * time.Millisecond), + TTFTNS: uint64(9 * time.Millisecond), + Counters: Counters{ + OutputTokens: 400, + ReasoningTokens: 100, + }, + } + item := requestDetailForUsage(usage, 1) + if item.TPS != 0 { + t.Fatalf("TPS = %v, want 0 for unreliable timing", item.TPS) + } + if item.TPSBasis != "latency_unreliable" { + t.Fatalf("TPS basis = %q, want latency_unreliable", item.TPSBasis) + } +} + func TestEffectiveOutputTokensForTPS(t *testing.T) { tests := []struct { name string From 22ca6c4d26b6fd8890c96f04107511aeb582102f Mon Sep 17 00:00:00 2001 From: AITNR Date: Fri, 18 Sep 2026 17:20:22 +0800 Subject: [PATCH 03/10] =?UTF-8?q?feat(dashboard):=20=E6=9B=B4=E6=96=B0CSV?= =?UTF-8?q?=E5=AF=BC=E5=87=BA=E5=8A=9F=E8=83=BD=E4=BB=A5=E5=8C=85=E5=90=AB?= =?UTF-8?q?TPS=E5=9F=BA=E7=A1=80=E5=AD=97=E6=AE=B5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 在exportCSV函数中添加TPS基础字段的导出逻辑 - 当TPS计算不可靠时显示空白而不是数值 - 在CSV头部添加TPS basis列 - 更新文档说明TPS计算基础字段 - 更新测试验证新增的CSV导出内容 - 修改多语言文件中的CSV头部翻译 --- README.md | 4 ++-- internal/plugin/dashboard.go | 2 +- internal/plugin/dashboard_test.go | 2 ++ internal/plugin/locales/en.json | 2 +- internal/plugin/locales/ru.json | 2 +- internal/plugin/locales/zh-CN.json | 2 +- internal/plugin/locales/zh-TW.json | 2 +- 7 files changed, 9 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index c1aae80..84ddd93 100644 --- a/README.md +++ b/README.md @@ -250,7 +250,7 @@ X-Full-Mode-Session: 统计、逐请求和费用接口支持 `range`,或 `start` 与 `end`,以及 `source` 等筛选参数。完整模式还支持重复的 `api_key_ref`,多个值按并集筛选;逐请求接口另支持 `offset`、`limit`、`model` 和 `result`。`/stats/groups` 另支持 `offset`、`limit`、`sort`、`direction`、`model` 和重复的 `exclude_model`;每页最多 500 条。统计和维度统计中的 `Groups` 行不按失败状态拆分:同一 provider、executor、model、alias、source、API key、auth type、service tier 和 reasoning effort 只有一行;`failed`/`failure_status` 在这些行中恒为 `false`/`0`,失败次数由 `failed_requests` 承载,逐请求状态保留在 `/requests`。 -逐请求明细中的 TPS 通常以生成时间(`latency_ns - ttft_ns`)为分母。对 Gemini、Vertex、AI Studio、Antigravity 和 Interactions 等独立思考协议,分子为 `output_tokens + reasoning_tokens`;对 OpenAI 兼容、Anthropic 等已将 reasoning 计入 output 的协议,分子仍为 `output_tokens`。如果独立思考协议的首包后窗口不超过 1 秒且出现超过 200 个输出 Token 或超过 500 TPS,记录会标记为缓冲流式,并改用完整 `latency_ns` 计算 TPS;仪表盘会在该值后显示 `*`。 +逐请求明细中的 TPS 通常以生成时间(`latency_ns - ttft_ns`)为分母。对 Gemini、Vertex、AI Studio、Antigravity 和 Interactions 等独立思考协议,分子为 `output_tokens + reasoning_tokens`;对 OpenAI 兼容、Anthropic 等已将 reasoning 计入 output 的协议,分子仍为 `output_tokens`。如果独立思考协议的首包后窗口不超过 1 秒且出现超过 200 个输出 Token 或超过 500 TPS,记录会标记为缓冲流式,并改用完整 `latency_ns` 计算 TPS;仪表盘会在该值后显示 `*`,CSV 会同时导出 `TPS basis`。如果使用完整延迟后仍超过 500 TPS,TPS 会标记为不可可靠计算,仪表盘显示 `—`,CSV 的 TPS 留空并将 `TPS basis` 设为 `latency_unreliable`。 重置请求正文: @@ -541,7 +541,7 @@ Management API routes: Statistics, request, and cost resources accept `range`, or `start` and `end`, plus filters such as `source`. Full mode also accepts repeated `api_key_ref` values and applies their union. The request resource additionally accepts `offset`, `limit`, `model`, and `result`. `/stats/groups` additionally accepts `offset`, `limit`, `sort`, `direction`, `model`, and repeated `exclude_model`; pages are limited to 500 rows. `Groups` rows in statistics and dimension statistics do not split by failure state: each provider, executor, model, alias, source, API key, auth type, service tier, and reasoning-effort combination has one row; `failed`/`failure_status` are always `false`/`0` in those rows, failures are counted in `failed_requests`, and per-request status remains available from `/requests`. -TPS in per-request details normally uses generation time (`latency_ns - ttft_ns`) as its denominator. For separate-reasoning protocols such as Gemini, Vertex, AI Studio, Antigravity, and Interactions, the numerator is `output_tokens + reasoning_tokens`; for protocols that already include reasoning in output, including OpenAI-compatible APIs and Anthropic, the numerator remains `output_tokens`. If a separate-reasoning request has a post-first-token window of at most one second and either more than 200 output tokens or more than 500 TPS, it is marked as buffered and TPS is calculated with the full `latency_ns`; the dashboard adds `*` to that value. +TPS in per-request details normally uses generation time (`latency_ns - ttft_ns`) as its denominator. For separate-reasoning protocols such as Gemini, Vertex, AI Studio, Antigravity, and Interactions, the numerator is `output_tokens + reasoning_tokens`; for protocols that already include reasoning in output, including OpenAI-compatible APIs and Anthropic, the numerator remains `output_tokens`. If a separate-reasoning request has a post-first-token window of at most one second and either more than 200 output tokens or more than 500 TPS, it is marked as buffered and TPS is calculated with the full `latency_ns`; the dashboard adds `*` and CSV includes `TPS basis`. If the full-latency result still exceeds 500 TPS, TPS is marked unreliable: the dashboard shows `—`, CSV leaves TPS blank, and `TPS basis` is `latency_unreliable`. Reset body: diff --git a/internal/plugin/dashboard.go b/internal/plugin/dashboard.go index 8234572..756b223 100644 --- a/internal/plugin/dashboard.go +++ b/internal/plugin/dashboard.go @@ -604,7 +604,7 @@ async function reloadPricesAndCosts(){var query=dateRangeQuery(),responses=await async function savePricing(){var managementKey=fullModeManagementKey;if(!fullModeEnabled||!managementKey){text('priceError',t('fullMode.keyRequired'));return;}var next,settings;try{next=collectPricing();settings=readSyncSettings();}catch(error){text('priceError',error.message);return;}setPricingBusy(true);text('priceError','');try{await api(savePricesURL,{method:'PUT',headers:{'Content-Type':'application/json','Authorization':'Bearer '+managementKey},body:JSON.stringify({prices:next,sync_settings:settings})});await reloadPricesAndCosts();closePricingDialog(true);text('status',t('status.saved',{revision:priceBookRevision,time:new Date().toLocaleTimeString(formatterLocale)}));}catch(error){text('priceError',error.message);}finally{setPricingBusy(false);}} async function syncPricing(){var managementKey=fullModeManagementKey;if(!fullModeEnabled||!managementKey){text('priceError',t('fullMode.keyRequired'));return;}var settings;try{settings=readSyncSettings();}catch(error){text('priceError',error.message);return;}setPricingBusy(true);text('priceError','');text('lastSyncStatus',t('pricing.syncingCatalog'));try{var models=await fetchCLIModels(false);text('lastSyncStatus',t('pricing.syncingModels',{count:fmt(models.length)}));await api(syncPricesURL,{method:'POST',headers:{'Content-Type':'application/json','Authorization':'Bearer '+managementKey},body:JSON.stringify({source:'models.dev',models:models,sync_settings:settings}),timeout:25000});await reloadPricesAndCosts();renderPricingEditor();text('status',t('status.syncComplete',{count:fmt(models.length),revision:priceBookRevision,time:new Date().toLocaleTimeString(formatterLocale)}));}catch(error){text('priceError',error.message);text('lastSyncStatus',t('pricing.syncFailed',{message:error.message}));}finally{setPricingBusy(false);}} function csvCell(value){var string=value===undefined||value===null?'':String(value);if(/^[=+\-@]/.test(string))string="'"+string;return '"'+string.replace(/"/g,'""')+'"';} -async function exportCSV(){if(!currentData)return;var button=document.getElementById('exportCSV'),rows=[t('export.csvHeaders').split(',')],offset=0,total=0,range=dateRangeFilePart();button.disabled=true;text('status',t('status.preparingCSV'));try{do{var url=requestsURL+'?'+dateRangeQuery()+'&offset='+offset+'&limit=500';if(selectedModel)url+='&model='+encodeURIComponent(selectedModel);var page=await api(url);total=Number(page.total||0);(page.items||[]).forEach(function(record){var name=modelName(record.model),estimated=record.estimated_cost||{};if(hiddenModels.has(name))return;rows.push([record.time,name,record.source,record.service_tier,translateRawResult(record.result,record.failed),record.ttft_ns,record.generation_ns,Number(record.tps||0).toFixed(4),record.reasoning_effort,record.input_tokens,record.output_tokens,record.reasoning_tokens,record.cache_read_tokens,record.cache_creation_tokens,record.total_tokens,record.cache_hit?t('cache.hit'):t('cache.miss'),estimated.priced?Number(estimated.input_usd||0).toFixed(8):'',estimated.priced?Number(estimated.output_usd||0).toFixed(8):'',estimated.priced?Number(estimated.cache_read_usd||0).toFixed(8):'',estimated.priced?Number(estimated.cache_creation_usd||0).toFixed(8):'',estimated.priced?Number(estimated.total_usd||0).toFixed(8):'',estimated.source||'',estimated.accounting_mode||'',estimated.tier_threshold||'']);});offset+=(page.items||[]).length;if(!(page.items||[]).length)break;}while(offset0){ctx.fillStyle=input;ctx.fillRect(x,chartY+chartH-inputH,barW,inputH);}if(showOutput&&outputH>0){ctx.fillStyle=output;ctx.fillRect(x,chartY+chartH-inputH-outputH,barW,outputH);}if(showCacheRead&&cacheReadH>0){ctx.fillStyle=cacheRead;ctx.fillRect(x,chartY+chartH-inputH-outputH-cacheReadH,barW,cacheReadH);}if(showCacheHit)cacheLine.push({x:x+barW/2,y:chartY+chartH*(1-point.cacheHitRate/100)});});if(showCacheHit&&cacheLine.length){ctx.save();ctx.strokeStyle=cacheHit;ctx.lineWidth=3;ctx.setLineDash([9,7]);ctx.beginPath();cacheLine.forEach(function(point,index){if(index)ctx.lineTo(point.x,point.y);else ctx.moveTo(point.x,point.y);});ctx.stroke();ctx.restore();}if(!trend.length)canvasText(ctx,t('chart.noCalls'),chartX+chartW/2,chartY+chartH/2,17,tertiary,600,'center');else if(!(showInput||showOutput||showCacheRead||showCacheHit))canvasText(ctx,t('trend.series.allHidden'),chartX+chartW/2,chartY+chartH/2,17,tertiary,600,'center');roundedRect(ctx,1088,282,448,620,14,panel,border);canvasText(ctx,t('export.modelShare')+' · '+modelShareMetricLabel(),1112,323,18,primary,750);var pieAll=groupModels(modelRows()),pieModels=pieAll.filter(function(item){return !hiddenModels.has(item.model);}),pieTotal=pieModels.reduce(function(sum,item){return sum+modelShareMetricValue(item);},0),angle=-Math.PI/2;pieModels.forEach(function(item){var next=angle+(pieTotal?modelShareMetricValue(item)/pieTotal*Math.PI*2:0);ctx.beginPath();ctx.arc(1312,525,118,angle,next);ctx.strokeStyle=colorFor(pieAll.indexOf(item));ctx.lineWidth=38;ctx.stroke();angle=next;});canvasText(ctx,formatModelShareMetric(pieTotal,true),1312,524,28,primary,750,'center');canvasText(ctx,modelShareMetricLabel(),1312,548,12,tertiary,400,'center');pieModels.slice(0,8).forEach(function(item,index){var y=700+index*23;ctx.fillStyle=colorFor(pieAll.indexOf(item));ctx.fillRect(1120,y-9,11,11);canvasText(ctx,item.model,1143,y,12,primary,600);canvasText(ctx,(pieTotal?modelShareMetricValue(item)/pieTotal*100:0).toFixed(1)+'%',1500,y,12,secondary,600,'right');});canvas.toBlob(function(blob){if(blob)downloadBlob(blob,'token-dashboard-'+document.getElementById('range').value+'.png');},'image/png');closeExportMenu();} diff --git a/internal/plugin/dashboard_test.go b/internal/plugin/dashboard_test.go index 258177b..127300c 100644 --- a/internal/plugin/dashboard_test.go +++ b/internal/plugin/dashboard_test.go @@ -191,6 +191,8 @@ func TestDashboardIncludesInteractiveAnalyticsFeatures(t *testing.T) { `estimated.cache_creation_usd`, `estimated.total_usd`, `async function exportCSV()`, + `record.tps_basis==='latency_unreliable'?'':Number(record.tps||0).toFixed(4)`, + `record.tps_basis||'generation'`, `function exportPNG()`, `id="exportBackup"`, `var backupURL=resourceBase+'/full-mode/backup'`, diff --git a/internal/plugin/locales/en.json b/internal/plugin/locales/en.json index 1146780..e94b6a2 100644 --- a/internal/plugin/locales/en.json +++ b/internal/plugin/locales/en.json @@ -253,7 +253,7 @@ "pricing.unpriced": "Unpriced", "pricing.saved": "Saved", "pricing.notSet": "Not set", - "export.csvHeaders": "Time,Model,Source,Tier,Result,Time to first token (ns),Generation time (ns),TPS,Reasoning effort,Input,Output,Reasoning,Cache read,Cache creation,Total Tokens,Cache hit,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,Price source,Accounting mode,Price Tier threshold", + "export.csvHeaders": "Time,Model,Source,Tier,Result,Time to first token (ns),Generation time (ns),TPS,TPS basis,Reasoning effort,Input,Output,Reasoning,Cache read,Cache creation,Total Tokens,Cache hit,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,Price source,Accounting mode,Price Tier threshold", "export.pngTitle": "Token Usage Analytics", "export.pngRange": "Range: {range} · Exported: {time}", "export.visibleModels": "All visible models", diff --git a/internal/plugin/locales/ru.json b/internal/plugin/locales/ru.json index 9ec97ee..4491f0b 100644 --- a/internal/plugin/locales/ru.json +++ b/internal/plugin/locales/ru.json @@ -253,7 +253,7 @@ "pricing.unpriced": "Без цены", "pricing.saved": "Сохранено", "pricing.notSet": "Не задано", - "export.csvHeaders": "Time,Model,Source,Tier,Result,Time to first token (ns),Generation time (ns),TPS,Reasoning effort,Input,Output,Reasoning,Cache read,Cache creation,Total Tokens,Cache hit,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,Price source,Accounting mode,Price Tier threshold", + "export.csvHeaders": "Time,Model,Source,Tier,Result,Time to first token (ns),Generation time (ns),TPS,TPS basis,Reasoning effort,Input,Output,Reasoning,Cache read,Cache creation,Total Tokens,Cache hit,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,Price source,Accounting mode,Price Tier threshold", "export.pngTitle": "Аналитика использования токенов", "export.pngRange": "Период: {range} · Экспортировано: {time}", "export.visibleModels": "Все видимые модели", diff --git a/internal/plugin/locales/zh-CN.json b/internal/plugin/locales/zh-CN.json index 7ceb208..98ec53c 100644 --- a/internal/plugin/locales/zh-CN.json +++ b/internal/plugin/locales/zh-CN.json @@ -253,7 +253,7 @@ "pricing.unpriced": "未定价", "pricing.saved": "已保存", "pricing.notSet": "未设置", - "export.csvHeaders": "时间,模型名称,来源,Tier,结果,首字延迟(ns),生成时间(ns),TPS,思考强度,输入,输出,思考,缓存读取,缓存创建,总Token数,缓存命中,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,价格来源,计费模式,价格Tier阈值", + "export.csvHeaders": "时间,模型名称,来源,Tier,结果,首字延迟(ns),生成时间(ns),TPS,TPS基础,思考强度,输入,输出,思考,缓存读取,缓存创建,总Token数,缓存命中,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,价格来源,计费模式,价格Tier阈值", "export.pngTitle": "Token 用量统计", "export.pngRange": "范围:{range} · 导出:{time}", "export.visibleModels": "全部可见模型", diff --git a/internal/plugin/locales/zh-TW.json b/internal/plugin/locales/zh-TW.json index dc485be..7fca72f 100644 --- a/internal/plugin/locales/zh-TW.json +++ b/internal/plugin/locales/zh-TW.json @@ -253,7 +253,7 @@ "pricing.unpriced": "未定價", "pricing.saved": "已保存", "pricing.notSet": "未設置", - "export.csvHeaders": "時間,模型名稱,來源,Tier,結果,首字延遲(ns),生成時間(ns),TPS,思考強度,輸入,輸出,思考,快取讀取,快取創建,總Token數,快取命中,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,價格來源,計費模式,價格Tier閾值", + "export.csvHeaders": "時間,模型名稱,來源,Tier,結果,首字延遲(ns),生成時間(ns),TPS,TPS基礎,思考強度,輸入,輸出,思考,快取讀取,快取創建,總Token數,快取命中,Input USD,Output USD,Cache Read USD,Cache Creation USD,Total USD,價格來源,計費模式,價格Tier閾值", "export.pngTitle": "Token 用量統計", "export.pngRange": "範圍:{range} · 導出:{time}", "export.visibleModels": "全部可見模型", From f515ae7ad22851c89805662fe748718efe254bd9 Mon Sep 17 00:00:00 2001 From: AITNR Date: Fri, 18 Sep 2026 21:30:42 +0800 Subject: [PATCH 04/10] =?UTF-8?q?feat(tps):=20=E4=BF=AE=E5=A4=8D=E7=BC=93?= =?UTF-8?q?=E5=86=B2=E6=B5=81TPS=E8=AE=A1=E7=AE=97=E5=B9=B6=E6=B7=BB?= =?UTF-8?q?=E5=8A=A0=E5=88=86=E6=97=B6=E5=AE=9A=E4=BB=B7=E5=8A=9F=E8=83=BD?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 修复了分离推理协议的TPS计算问题,解决因数据缓冲导致的虚假高TPS值 - 对短延迟高令牌请求实施缓冲检测和TPS基础调整机制 - 添加分时定价计划支持,实现基于时间的价格计算功能 - 更新管理界面以支持新的定价模型显示 - 在CSV导出中添加TPS基础列以保留计算上下文 - 更新多语言本地化文件以支持新功能描述 - 添加完整的集成测试和单元测试覆盖新功能逻辑 --- docs/issue-77-tps-buffered-stream-fix.md | 22 ++++++++++++++++++++++ internal/plugin/cost.go | 24 ++++++++++++++++++++++++ internal/plugin/dashboard.go | 14 +++++++++----- internal/plugin/locales/en.json | 14 ++++++++++++-- internal/plugin/locales/ru.json | 14 ++++++++++++-- internal/plugin/locales/zh-CN.json | 14 ++++++++++++-- internal/plugin/locales/zh-TW.json | 14 ++++++++++++-- internal/plugin/management.go | 2 +- internal/plugin/persistence.go | 2 ++ internal/plugin/pricing.go | 10 ++++++++++ 10 files changed, 116 insertions(+), 14 deletions(-) create mode 100644 docs/issue-77-tps-buffered-stream-fix.md diff --git a/docs/issue-77-tps-buffered-stream-fix.md b/docs/issue-77-tps-buffered-stream-fix.md new file mode 100644 index 0000000..b50c14a --- /dev/null +++ b/docs/issue-77-tps-buffered-stream-fix.md @@ -0,0 +1,22 @@ +# Issue #77: Buffered-stream TPS correction + +- Issue: https://github.com/AITNR/cap-token-usage-tracker/issues/77 +- Scope: per-request TPS, historical query recalculation, dashboard status, and CSV export +- Storage: no schema change and no rewrite or migration of historical request records + +## Root cause + +Per-request TPS originally used latency_ns - ttft_ns as its denominator. For separate-reasoning protocols such as Gemini, Vertex, AI Studio, Antigravity, and Interactions, the upstream or proxy can buffer data while the model is thinking. TTFT then contains most of the model generation time, while the post-first-token window measures only the final buffered transfer. Dividing output_tokens + reasoning_tokens by that short window produces false values in the thousands of TPS. + +## Fix behavior + +1. Keep protocol-aware token accounting: separate-reasoning protocols use output_tokens + reasoning_tokens; protocols that include reasoning in output continue to use output_tokens. +2. For separate-reasoning requests with generation_ns <= 1s, mark a request as likely buffered when output exceeds 200 tokens or the raw TPS exceeds 500. +3. Calculate likely buffered requests with full latency_ns and return tps_basis = latency_buffered. +4. If the full-latency result still exceeds 500 TPS, return tps = 0 and tps_basis = latency_unreliable. The dashboard shows an unavailable marker and CSV leaves TPS blank. +5. Recalculate TPS while serving /requests, so historical records are corrected without a bbolt migration. +6. Append a TPS basis column to CSV so exported values retain their calculation context. + +## Verification + +Coverage includes the Issue example, normal separate-reasoning streams, short-latency high-token requests, provider classification, historical query recalculation, dashboard script contracts, and all locale CSV headers. diff --git a/internal/plugin/cost.go b/internal/plugin/cost.go index 744a5cb..adc987f 100644 --- a/internal/plugin/cost.go +++ b/internal/plugin/cost.go @@ -13,6 +13,8 @@ import ( // EstimatedCost is calculated from one persisted request using the current model price. type EstimatedCost struct { + PriceTimeTier string `json:"price_time_tier,omitempty"` + PriceTimeZone string `json:"price_time_zone,omitempty"` Priced bool `json:"priced"` Source string `json:"source,omitempty"` AccountingMode string `json:"accounting_mode,omitempty"` @@ -119,6 +121,19 @@ type modelPriceResolver struct { // newModelPriceResolver builds the normalized fallback index once per price-book snapshot. func newModelPriceResolver(prices map[string]ModelPrice, settings PriceSyncSettings) modelPriceResolver { + prices = cloneModelPrices(prices) + for name, price := range prices { + if len(price.TimeTiers) > 0 { + zone := price.TimeZone + if zone == "" { + zone = "UTC" + } + if zone != "Local" { + price.timeLocation, _ = time.LoadLocation(zone) + } + prices[name] = price + } + } resolver := modelPriceResolver{exact: prices} normalizedSettings, err := normalizePriceSyncSettings(settings) if err != nil { @@ -438,16 +453,25 @@ func estimateRequestCostWithResolver(request RequestDetail, resolver modelPriceR contextTiers = schedule.ContextTiers priceServiceTier = serviceTier } + timeTier, timeZone := "", "" + if tier, ok := selectTimePriceTier(price, request.Time); ok { + rates = tier.TokenRates + timeTier = tier.Name + timeZone = price.TimeZone + } var selectedThreshold uint64 for _, tier := range contextTiers { if contextTokens > tier.Threshold && tier.Threshold >= selectedThreshold { rates = tier.tokenRates() + timeTier, timeZone = "", "" selectedThreshold = tier.Threshold } } result := EstimatedCost{ Priced: true, + PriceTimeTier: timeTier, + PriceTimeZone: timeZone, Source: price.Source, AccountingMode: mode, PriceServiceTier: priceServiceTier, diff --git a/internal/plugin/dashboard.go b/internal/plugin/dashboard.go index 756b223..e1790f4 100644 --- a/internal/plugin/dashboard.go +++ b/internal/plugin/dashboard.go @@ -248,7 +248,7 @@ html{background:#151412;color-scheme:dark} html:not([data-theme]){background:#faf9f5;color-scheme:light} html[data-theme='white']{background:#fff;color-scheme:light} html[data-theme='dark']{background:#151412;color-scheme:dark} - +.time-pricing{grid-column:1/-1;padding:10px;border-top:1px solid var(--border-color)}.time-pricing summary{cursor:pointer}.time-pricing label{display:flex;flex-direction:column;gap:4px;min-width:0}.time-tier-row{display:grid;grid-template-columns:repeat(4,minmax(120px,1fr));gap:8px;padding:12px 0;border-bottom:1px solid var(--border-color)}.time-pricing input{width:100%;box-sizing:border-box}.time-pricing p{margin:8px 0}.time-pricing>.mini-button{margin-top:8px}@media(max-width:650px){.time-tier-row{grid-template-columns:repeat(2,minmax(100px,1fr))}}