From 6930f64a54b41cafb3aacf258f5ac10a47394094 Mon Sep 17 00:00:00 2001 From: Bai Li Date: Wed, 23 Sep 2026 10:16:07 -0700 Subject: [PATCH] fix(pricing): add Opus 5.5 and GPT-6, correct rates that moved publicly New rows for claude-opus-5-5, gpt-6-astra, gpt-6-sol and gpt-6-luna. GPT-5.6 cache writes move to 1.25x input, the Pro tiers lose a cached-input discount they never had, and the OpenRouter GLM 5.2 / DeepSeek V4 Pro fallback headlines follow the current listing. Checked against the providers' pricing pages on 2026-09-23; every other row still matches. Co-Authored-By: Claude Opus 5.5 --- evalboard/lib/pricing.generated.ts | 10 +++++++--- src/coder_eval/pricing.py | 28 +++++++++++++++++----------- 2 files changed, 24 insertions(+), 14 deletions(-) diff --git a/evalboard/lib/pricing.generated.ts b/evalboard/lib/pricing.generated.ts index b11cd688a..977c4af0d 100644 --- a/evalboard/lib/pricing.generated.ts +++ b/evalboard/lib/pricing.generated.ts @@ -30,6 +30,7 @@ export const PRICING: Record = { "claude-opus-4-7": { inputPerMTok: 5.0, outputPerMTok: 25.0, cacheWritePerMTok: 6.25, cacheReadPerMTok: 0.5 }, "claude-opus-4-8": { inputPerMTok: 5.0, outputPerMTok: 25.0, cacheWritePerMTok: 6.25, cacheReadPerMTok: 0.5 }, "claude-opus-5": { inputPerMTok: 5.0, outputPerMTok: 25.0, cacheWritePerMTok: 6.25, cacheReadPerMTok: 0.5 }, + "claude-opus-5-5": { inputPerMTok: 4.0, outputPerMTok: 20.0, cacheWritePerMTok: 5.0, cacheReadPerMTok: 0.2 }, "claude-sonnet-4-20250514": { inputPerMTok: 3.0, outputPerMTok: 15.0, cacheWritePerMTok: 3.75, cacheReadPerMTok: 0.3 }, "claude-sonnet-4-5": { inputPerMTok: 3.0, outputPerMTok: 15.0, cacheWritePerMTok: 3.75, cacheReadPerMTok: 0.3 }, "claude-sonnet-4-5-20250929": { inputPerMTok: 3.0, outputPerMTok: 15.0, cacheWritePerMTok: 3.75, cacheReadPerMTok: 0.3 }, @@ -66,9 +67,12 @@ export const PRICING: Record = { "gpt-5.3-codex": { inputPerMTok: 1.75, outputPerMTok: 14.0, cacheWritePerMTok: 1.75, cacheReadPerMTok: 0.175 }, "gpt-5.4": { inputPerMTok: 2.5, outputPerMTok: 15.0, cacheWritePerMTok: 2.5, cacheReadPerMTok: 0.25 }, "gpt-5.5": { inputPerMTok: 5.0, outputPerMTok: 30.0, cacheWritePerMTok: 5.0, cacheReadPerMTok: 0.5 }, - "gpt-5.6-luna": { inputPerMTok: 0.2, outputPerMTok: 1.2, cacheWritePerMTok: 0.2, cacheReadPerMTok: 0.02 }, - "gpt-5.6-sol": { inputPerMTok: 4.0, outputPerMTok: 20.0, cacheWritePerMTok: 4.0, cacheReadPerMTok: 0.4 }, - "gpt-5.6-terra": { inputPerMTok: 2.0, outputPerMTok: 12.0, cacheWritePerMTok: 2.0, cacheReadPerMTok: 0.2 }, + "gpt-5.6-luna": { inputPerMTok: 0.2, outputPerMTok: 1.2, cacheWritePerMTok: 0.25, cacheReadPerMTok: 0.02 }, + "gpt-5.6-sol": { inputPerMTok: 4.0, outputPerMTok: 20.0, cacheWritePerMTok: 5.0, cacheReadPerMTok: 0.4 }, + "gpt-5.6-terra": { inputPerMTok: 2.0, outputPerMTok: 12.0, cacheWritePerMTok: 2.5, cacheReadPerMTok: 0.2 }, + "gpt-6-astra": { inputPerMTok: 10.0, outputPerMTok: 50.0, cacheWritePerMTok: 12.5, cacheReadPerMTok: 1.0 }, + "gpt-6-luna": { inputPerMTok: 0.1, outputPerMTok: 0.5, cacheWritePerMTok: 0.125, cacheReadPerMTok: 0.01 }, + "gpt-6-sol": { inputPerMTok: 2.0, outputPerMTok: 10.0, cacheWritePerMTok: 2.5, cacheReadPerMTok: 0.2 }, "jev-1.13.0": { inputPerMTok: 0.042, outputPerMTok: 0.0, cacheWritePerMTok: 0.042, cacheReadPerMTok: 0.0 }, "jev-latest": { inputPerMTok: 0.042, outputPerMTok: 0.0, cacheWritePerMTok: 0.042, cacheReadPerMTok: 0.0 }, "kimi-k2-7-code": { inputPerMTok: 0.95, outputPerMTok: 4.0, cacheWritePerMTok: 0.0, cacheReadPerMTok: 0.19 }, diff --git a/src/coder_eval/pricing.py b/src/coder_eval/pricing.py index ff2782517..2a52afba1 100644 --- a/src/coder_eval/pricing.py +++ b/src/coder_eval/pricing.py @@ -64,6 +64,8 @@ class ModelPricing: # every other Claude model uses. Fable 5 pays $1 on the identical $10 base. "claude-fable-5-1": ModelPricing(10.0, 50.0, 12.50, 0.25), "claude-fable-5": ModelPricing(10.0, 50.0, 12.50, 1.0), + # Opus 5.5 prices cache hits at 0.05x input, not the usual 0.1x. + "claude-opus-5-5": ModelPricing(4.0, 20.0, 5.0, 0.20), # Opus 4.5 and later dropped to $5/$25; 4.1 and 4 keep the old $15/$75. The # version boundary is the price boundary: a newer Opus is not the dearer one. "claude-opus-5": ModelPricing(5.0, 25.0, 6.25, 0.50), @@ -91,7 +93,7 @@ class ModelPricing: "claude-3-opus-20240229": ModelPricing(15.0, 75.0, 18.75, 1.50), "claude-3-sonnet-20240229": ModelPricing(3.0, 15.0, 3.75, 0.30), "claude-3-haiku-20240307": ModelPricing(0.25, 1.25, 0.30, 0.03), - # OpenAI GPT-5 / Codex (direct or Azure). No cache-write fee: cache_write == input below. + # OpenAI (direct or Azure). Cache writes are free before GPT-5.6 (== input), 1.25x input from 5.6 on. "gpt-5-codex": ModelPricing(1.25, 10.0, 1.25, 0.125), "gpt-5": ModelPricing(1.25, 10.0, 1.25, 0.125), "gpt-5.1-codex-max": ModelPricing(1.25, 10.0, 1.25, 0.125), @@ -105,21 +107,25 @@ class ModelPricing: # CAVEAT: flat rate, so gpt-5.5's long-context surcharge (2x input / 1.5x # output past 272K input tokens) is not modelled and reads low. "gpt-5.5": ModelPricing(5.0, 30.0, 5.0, 0.50), - # Pro tiers offer no prompt caching, so cache_read is nominal. - "gpt-5.5-pro": ModelPricing(30.0, 180.0, 30.0, 3.0), - "gpt-5.4-pro": ModelPricing(30.0, 180.0, 30.0, 3.0), + # Pro tiers have no cached-input discount, so cache_read == input. + "gpt-5.5-pro": ModelPricing(30.0, 180.0, 30.0, 30.0), + "gpt-5.4-pro": ModelPricing(30.0, 180.0, 30.0, 30.0), "gpt-5.4-mini": ModelPricing(0.75, 4.5, 0.75, 0.075), "gpt-5.4-nano": ModelPricing(0.20, 1.25, 0.20, 0.02), # HAZARD: a single CURRENT-rate card with no effective date, so a repriced # model makes historical runs re-price at today's rate. Sol's rate is # promotional through at least 2026-11-21; re-check then. - "gpt-5.6-sol": ModelPricing(4.0, 20.0, 4.0, 0.40), - "gpt-5.6-terra": ModelPricing(2.0, 12.0, 2.0, 0.20), - "gpt-5.6-luna": ModelPricing(0.20, 1.20, 0.20, 0.02), + "gpt-5.6-sol": ModelPricing(4.0, 20.0, 5.0, 0.40), + "gpt-5.6-terra": ModelPricing(2.0, 12.0, 2.50, 0.20), + "gpt-5.6-luna": ModelPricing(0.20, 1.20, 0.25, 0.02), + # CAVEAT: flat rate; GPT-6's >272K-input tier (2x input / 1.5x output) is not modelled and reads low. + "gpt-6-astra": ModelPricing(10.0, 50.0, 12.50, 1.00), + "gpt-6-sol": ModelPricing(2.0, 10.0, 2.50, 0.20), + "gpt-6-luna": ModelPricing(0.10, 0.50, 0.125, 0.01), # Keyed on the literal ids ListModels returns. No cache-write fee, so # cache_write == input (unused). CAVEAT: Pro's >200K-token tier costs more, so - # a very-large-context run reads LOW. List rates; discounted by half through - # 2026-12-31. + # a very-large-context run reads LOW. List rates; 3.6-3.8 Flash are discounted + # by half through 2026-12-31. "gemini-3.8-flash": ModelPricing(1.5, 7.5, 1.5, 0.15), "gemini-3.7-flash": ModelPricing(1.5, 7.5, 1.5, 0.15), "gemini-3.6-flash": ModelPricing(1.5, 7.5, 1.5, 0.15), @@ -148,8 +154,8 @@ class ModelPricing: # why these three carry per_request_billing (the mirror omits them). # Rationale: .claude/notes/reporting.md ยง Cost joining "moonshotai/kimi-k3": ModelPricing(3.0, 15.0, 3.0, 0.30, per_request_billing=True), - "z-ai/glm-5.2": ModelPricing(0.966, 3.036, 0.966, 0.1932, per_request_billing=True), - "deepseek/deepseek-v4-pro": ModelPricing(1.030776, 2.061552, 1.030776, 0.085898, per_request_billing=True), + "z-ai/glm-5.2": ModelPricing(0.6496, 2.0416, 0.6496, 0.12064, per_request_billing=True), + "deepseek/deepseek-v4-pro": ModelPricing(0.951432, 1.902864, 0.951432, 0.079286, per_request_billing=True), **_DELEGATE_PRICING, }