diff --git a/packages/cli/src/__tests__/hook-handlers.test.ts b/packages/cli/src/__tests__/hook-handlers.test.ts index bfd9006..e76af23 100644 --- a/packages/cli/src/__tests__/hook-handlers.test.ts +++ b/packages/cli/src/__tests__/hook-handlers.test.ts @@ -635,13 +635,13 @@ describe("handleStop", () => { const cwd = "/tmp/test-cwd"; await _handleSessionStart({ session_id: sid, cwd }, testDbPath); - // Ceiling chosen so 5000 input tokens on opus 4.7 (~$0.075) lands - // squarely in the 80–99% warning band: maxUsd ≈ $0.085. + // Ceiling chosen so 5000 input tokens on opus 4.7 (~$0.025 at $5/MTok) + // lands squarely in the 80–99% warning band: maxUsd ≈ $0.028. insertPolicy(getDb(testDbPath), { id: createPolicyId("p_warn_band"), name: "Soft ceiling", type: PolicyType.CostCeiling, - config: { type: PolicyType.CostCeiling, maxUsd: 0.085 }, + config: { type: PolicyType.CostCeiling, maxUsd: 0.028 }, severity: PolicySeverity.Error, enabled: true, createdAt: new Date().toISOString(), @@ -758,11 +758,11 @@ describe("handleSessionEnd", () => { await runHook(() => _handleSessionEnd({ session_id: sid, cwd }, testDbPath)); const run = getRun(getDb(testDbPath), createRunId(state.runId))!; - expect(run.metrics.costUsd).toBeCloseTo(15, 4); + expect(run.metrics.costUsd).toBeCloseTo(5, 4); expect(run.metrics.tokenUsage.input).toBe(1_000_000); // Backend + per-model cost are captured (no env => direct Anthropic). expect(run.metrics.backend).toBe("anthropic"); - expect(run.metrics.byModel?.["claude-opus-4-7"]).toBeCloseTo(15, 4); + expect(run.metrics.byModel?.["claude-opus-4-7"]).toBeCloseTo(5, 4); }); it("tags the run 'bedrock' when CLAUDE_CODE_USE_BEDROCK is set", async () => { @@ -784,10 +784,10 @@ describe("handleSessionEnd", () => { const run = getRun(getDb(testDbPath), createRunId(state.runId))!; expect(run.metrics.backend).toBe("bedrock"); - expect(run.metrics.costUsd).toBeCloseTo(15, 4); + expect(run.metrics.costUsd).toBeCloseTo(5, 4); expect( run.metrics.byModel?.["us.anthropic.claude-opus-4-7-v1:0"], - ).toBeCloseTo(15, 4); + ).toBeCloseTo(5, 4); } finally { if (prev === undefined) delete process.env["CLAUDE_CODE_USE_BEDROCK"]; else process.env["CLAUDE_CODE_USE_BEDROCK"] = prev; diff --git a/packages/cli/src/__tests__/transcript.test.ts b/packages/cli/src/__tests__/transcript.test.ts index 1ee4fc3..529da3d 100644 --- a/packages/cli/src/__tests__/transcript.test.ts +++ b/packages/cli/src/__tests__/transcript.test.ts @@ -16,6 +16,21 @@ describe("transcriptPath", () => { join(homedir(), ".claude", "projects", "-Users-x-code-repo", "abc-123.jsonl"), ); }); + + it("encodes dots, underscores, and spaces like Claude Code does", () => { + // Claude Code replaces every non-alphanumeric character, not just "/". + // A dot-only replacement resolves to a nonexistent dir and reads $0 usage. + const path = transcriptPath("/Users/x/my.app_v2/sub dir", "abc-123"); + expect(path).toBe( + join( + homedir(), + ".claude", + "projects", + "-Users-x-my-app-v2-sub-dir", + "abc-123.jsonl", + ), + ); + }); }); describe("readSessionUsage", () => { @@ -135,8 +150,8 @@ describe("readSessionUsage", () => { writeFileSync(path, lines.join("\n"), "utf-8"); const usage = readSessionUsage(path, "bedrock"); - expect(usage.totalCostUsd).toBeCloseTo(15, 4); - expect(usage.byModel["us.anthropic.claude-opus-4-7-v1:0"]).toBeCloseTo(15, 4); + expect(usage.totalCostUsd).toBeCloseTo(5, 4); + expect(usage.byModel["us.anthropic.claude-opus-4-7-v1:0"]).toBeCloseTo(5, 4); }); it("aggregates costs per model across mixed-model sessions", () => { @@ -160,9 +175,90 @@ describe("readSessionUsage", () => { writeFileSync(path, lines.join("\n"), "utf-8"); const usage = readSessionUsage(path); - expect(usage.byModel["claude-opus-4-7"]).toBeCloseTo(15, 4); + expect(usage.byModel["claude-opus-4-7"]).toBeCloseTo(5, 4); expect(usage.byModel["claude-haiku-4-5"]).toBeCloseTo(1, 4); - expect(usage.totalCostUsd).toBeCloseTo(16, 4); + expect(usage.totalCostUsd).toBeCloseTo(6, 4); + }); + + it("counts each message.id once even when repeated across content-block lines", () => { + // Claude Code writes one transcript line per content block; every line + // for the same API response carries the same message.id and identical + // usage. This is the real transcript shape — summing per-line inflates + // cost by blocks-per-message. + const path = join(tmpDir, "dupes.jsonl"); + const usageBlock = { + input_tokens: 1_000_000, + output_tokens: 1_000_000, + }; + const lines = [ + // Same message, three content blocks + ...Array.from({ length: 3 }, () => + JSON.stringify({ + type: "assistant", + message: { id: "msg_aaa", model: "claude-opus-4-8", usage: usageBlock }, + }), + ), + // A different message + JSON.stringify({ + type: "assistant", + message: { id: "msg_bbb", model: "claude-opus-4-8", usage: usageBlock }, + }), + ]; + writeFileSync(path, lines.join("\n"), "utf-8"); + + const usage = readSessionUsage(path); + // 2 unique messages x (1M in @ $5 + 1M out @ $25) = $60, not $120 + expect(usage.totalCostUsd).toBeCloseTo(60, 4); + expect(usage.inputTokens).toBe(2_000_000); + expect(usage.outputTokens).toBe(2_000_000); + }); + + it("still counts every line when message ids are absent", () => { + const path = join(tmpDir, "no-ids.jsonl"); + const lines = Array.from({ length: 2 }, () => + JSON.stringify({ + type: "assistant", + message: { + model: "claude-haiku-4-5", + usage: { input_tokens: 1_000_000, output_tokens: 0 }, + }, + }), + ); + writeFileSync(path, lines.join("\n"), "utf-8"); + + const usage = readSessionUsage(path); + expect(usage.inputTokens).toBe(2_000_000); + expect(usage.totalCostUsd).toBeCloseTo(2, 4); + }); + + it("reports models with no pricing entry instead of silently pricing $0", () => { + const path = join(tmpDir, "unknown-model.jsonl"); + const lines = [ + JSON.stringify({ + type: "assistant", + message: { + id: "msg_x", + model: "claude-nova-9", + usage: { input_tokens: 1_000_000, output_tokens: 0 }, + }, + }), + JSON.stringify({ + type: "assistant", + message: { + id: "msg_y", + model: "claude-haiku-4-5", + usage: { input_tokens: 1_000_000, output_tokens: 0 }, + }, + }), + ]; + writeFileSync(path, lines.join("\n"), "utf-8"); + + const usage = readSessionUsage(path); + expect(usage.unknownModels).toEqual(["claude-nova-9"]); + // Tokens still counted, cost only from the known model + expect(usage.inputTokens).toBe(2_000_000); + expect(usage.totalCostUsd).toBeCloseTo(1, 4); + expect(usage.byModel["claude-nova-9"]).toBe(0); }); }); diff --git a/packages/cli/src/commands/hook.ts b/packages/cli/src/commands/hook.ts index 3b737d2..dcd4d23 100644 --- a/packages/cli/src/commands/hook.ts +++ b/packages/cli/src/commands/hook.ts @@ -109,6 +109,7 @@ function cleanupState(claudeSessionId: string): void { interface HookInput { session_id: string; cwd?: string; + transcript_path?: string; hook_event_name?: string; tool_name?: string; tool_input?: Record; @@ -119,6 +120,34 @@ interface HookInput { tool_use_id?: string; } +// ─── Transcript usage for a hook event ─────────────────────────────────────── +// +// Prefer the transcript_path Claude Code sends in every hook payload — it is +// authoritative. Reconstructing the path from cwd is the fallback, and it can +// miss (the encoding must match Claude Code's exactly), which silently reads +// as $0 spend. Unknown-model pricing is the same failure mode, so both warn +// on stderr instead of failing silent: a cost pipeline that quietly reports +// $0 defeats every budget policy downstream. + +function readHookUsage( + input: HookInput, + state: HookState, +): { usage: SessionUsage; backend: Backend } { + const backend = state.backend ?? detectBackend(); + const cwd = state.cwd ?? input.cwd; + const path = + input.transcript_path ?? (cwd ? transcriptPath(cwd, input.session_id) : null); + const usage = path ? readSessionUsage(path, backend) : ZERO_USAGE; + if (usage.unknownModels.length > 0) { + process.stderr.write( + `[agentops] ⚠ No pricing entry for model(s): ${usage.unknownModels.join(", ")}. ` + + `Their spend is recorded as $0, so cost ceilings and budgets cannot see it. ` + + `Update ANTHROPIC_PRICING in @agentops/core.\n`, + ); + } + return { usage, backend }; +} + // ─── Stale state detection ─────────────────────────────────────────────────── function handleStaleState(sessionId: string, state: HookState): void { @@ -334,11 +363,7 @@ async function handlePreToolUse(input: HookInput, dbPath?: string): Promise { process.exit(0); } - const cwd = state.cwd ?? input.cwd; - const backend = state.backend ?? detectBackend(); - const usage: SessionUsage = cwd - ? readSessionUsage(transcriptPath(cwd, input.session_id), backend) - : ZERO_USAGE; + const { usage, backend } = readHookUsage(input, state); const ops = createOps(opsConfigFromState(state, dbPath), input.session_id); diff --git a/packages/cli/src/transcript.ts b/packages/cli/src/transcript.ts index 1852f8d..7e48f64 100644 --- a/packages/cli/src/transcript.ts +++ b/packages/cli/src/transcript.ts @@ -1,7 +1,7 @@ import { existsSync, readFileSync } from "node:fs"; import { homedir } from "node:os"; import { join } from "node:path"; -import { computeCost, type Backend, type TokenUsageBlock } from "@agentops/core"; +import { computeCost, resolvePricing, type Backend, type TokenUsageBlock } from "@agentops/core"; export interface SessionUsage { readonly totalCostUsd: number; @@ -10,6 +10,10 @@ export interface SessionUsage { readonly cacheReadTokens: number; readonly cacheWriteTokens: number; readonly byModel: Record; + // Models seen in the transcript that have no pricing entry. Their tokens + // are counted above but contribute $0 to totalCostUsd — callers must + // surface this loudly, or cost ceilings silently fail open. + readonly unknownModels: readonly string[]; } export const ZERO_USAGE: SessionUsage = { @@ -19,6 +23,7 @@ export const ZERO_USAGE: SessionUsage = { cacheReadTokens: 0, cacheWriteTokens: 0, byModel: {}, + unknownModels: [], }; // Reads CLAUDE_CODE_USE_BEDROCK / CLAUDE_CODE_USE_VERTEX style env vars. @@ -32,14 +37,19 @@ export function detectBackend(env: NodeJS.ProcessEnv = process.env): Backend { return "anthropic"; } +// Claude Code encodes the project cwd by replacing every non-alphanumeric +// character with "-" (not just "/" — dots, underscores, and spaces too). +// Prefer the transcript_path field from the hook payload when available; +// this reconstruction is the fallback for older payloads. export function transcriptPath(cwd: string, sessionId: string): string { - const encoded = cwd.replace(/\//g, "-"); + const encoded = cwd.replace(/[^a-zA-Z0-9]/g, "-"); return join(homedir(), ".claude", "projects", encoded, `${sessionId}.jsonl`); } interface TranscriptLine { readonly type?: string; readonly message?: { + readonly id?: string; readonly model?: string; readonly usage?: TokenUsageBlock; }; @@ -61,12 +71,14 @@ export function readSessionUsage( } const lines = raw.split("\n"); - let totalCostUsd = 0; - let inputTokens = 0; - let outputTokens = 0; - let cacheReadTokens = 0; - let cacheWriteTokens = 0; - const byModel: Record = {}; + + // Claude Code writes one transcript line per content block, and every line + // for the same API response repeats the same message.id with identical + // usage. Summing per-line multiplies real cost by blocks-per-message + // (2-7x in practice) — count each message.id once. Lines without an id + // (older transcript formats) are counted individually. + const byMessage = new Map(); + let anonKey = 0; for (const line of lines) { if (line.trim().length === 0) continue; @@ -81,6 +93,20 @@ export function readSessionUsage( const usage = entry.message?.usage ?? entry.usage; if (!model || !usage) continue; + const key = entry.message?.id ?? `__no_id_${anonKey++}`; + byMessage.set(key, { model, usage }); + } + + let totalCostUsd = 0; + let inputTokens = 0; + let outputTokens = 0; + let cacheReadTokens = 0; + let cacheWriteTokens = 0; + const byModel: Record = {}; + const unknownModels = new Set(); + + for (const { model, usage } of byMessage.values()) { + if (!resolvePricing(model, backend)) unknownModels.add(model); const cost = computeCost(model, usage, backend); totalCostUsd += cost; inputTokens += usage.input_tokens ?? 0; @@ -97,5 +123,6 @@ export function readSessionUsage( cacheReadTokens, cacheWriteTokens, byModel, + unknownModels: [...unknownModels], }; } diff --git a/packages/core/src/__tests__/pricing.test.ts b/packages/core/src/__tests__/pricing.test.ts index a0821da..c0eef59 100644 --- a/packages/core/src/__tests__/pricing.test.ts +++ b/packages/core/src/__tests__/pricing.test.ts @@ -14,6 +14,24 @@ describe("resolvePricing", () => { ); }); + it("resolves the current model generation", () => { + expect(resolvePricing("claude-opus-4-8")).toBe(ANTHROPIC_PRICING["claude-opus-4-8"]); + expect(resolvePricing("claude-sonnet-5")).toBe(ANTHROPIC_PRICING["claude-sonnet-5"]); + expect(resolvePricing("claude-fable-5")).toBe(ANTHROPIC_PRICING["claude-fable-5"]); + }); + + it("resolves dated variants of both claude-sonnet-4-5 and claude-sonnet-5", () => { + // "claude-sonnet-5-..." must not prefix-match "claude-sonnet-4-5" or vice + // versa; both should land on the Sonnet tier via their own keys. + expect(resolvePricing("claude-sonnet-4-5-20250929")?.inputPerMTok).toBe(3); + expect(resolvePricing("claude-sonnet-5-20260201")?.inputPerMTok).toBe(3); + }); + + it("prices pre-4.5 Opus at the legacy $15/$75 tier", () => { + expect(resolvePricing("claude-opus-4-1")!.inputPerMTok).toBe(15); + expect(resolvePricing("claude-opus-4-8")!.inputPerMTok).toBe(5); + }); + it("matches by prefix for dated suffixes", () => { expect(resolvePricing("claude-sonnet-4-5-20250929")).toBe( ANTHROPIC_PRICING["claude-sonnet-4-5"], @@ -83,28 +101,37 @@ describe("Bedrock/Anthropic parity (as of 2026-05-13)", () => { describe("computeCost", () => { it("computes base input + output for Opus 4.7", () => { - // 1M input @ $15 + 1M output @ $75 = $90 + // 1M input @ $5 + 1M output @ $25 = $30 const cost = computeCost("claude-opus-4-7", { input_tokens: 1_000_000, output_tokens: 1_000_000, }); - expect(cost).toBeCloseTo(90, 6); + expect(cost).toBeCloseTo(30, 6); }); it("applies 0.10x rate to cache reads", () => { - // 1M cache reads on Opus = $1.50 + // 1M cache reads on Opus = $0.50 const cost = computeCost("claude-opus-4-7", { cache_read_input_tokens: 1_000_000, }); - expect(cost).toBeCloseTo(1.5, 6); + expect(cost).toBeCloseTo(0.5, 6); }); it("applies 1.25x rate to cache writes", () => { - // 1M cache writes on Opus = $18.75 + // 1M cache writes on Opus = $6.25 const cost = computeCost("claude-opus-4-7", { cache_creation_input_tokens: 1_000_000, }); - expect(cost).toBeCloseTo(18.75, 6); + expect(cost).toBeCloseTo(6.25, 6); + }); + + it("computes Fable 5 at the Mythos-class tier", () => { + // 1M input @ $10 + 1M output @ $50 = $60 + const cost = computeCost("claude-fable-5", { + input_tokens: 1_000_000, + output_tokens: 1_000_000, + }); + expect(cost).toBeCloseTo(60, 6); }); it("computes mixed usage for Sonnet 4.6", () => { @@ -136,10 +163,10 @@ describe("computeCost", () => { { input_tokens: 1_000_000, output_tokens: 0 }, "bedrock", ); - expect(cost).toBeCloseTo(15, 4); + expect(cost).toBeCloseTo(5, 4); }); - it("returns 0 when model is unknown to selected backend", () => { + it("resolves a Bedrock-formatted id even against the anthropic table", () => { // Forcing anthropic backend on a Bedrock-formatted id (no normalization // path that matches an Anthropic key) const cost = computeCost( @@ -150,23 +177,23 @@ describe("computeCost", () => { // Anthropic table doesn't normalize Bedrock prefixes the same way, // but normalizeModelId is shared — so it WILL resolve. Document the // current behavior rather than over-constraining. - expect(cost).toBeCloseTo(15, 4); + expect(cost).toBeCloseTo(5, 4); }); it("matches real transcript line for Opus 4.7", () => { // Realistic line from a session: // input=6, cache_creation=14483, cache_read=16963, output=472 - // 6*15/1e6 = 0.00009 - // 14483*18.75/1e6 = 0.27155625 - // 16963*1.5/1e6 = 0.0254445 - // 472*75/1e6 = 0.0354 - // sum ~= 0.3324907 + // 6*5/1e6 = 0.00003 + // 14483*6.25/1e6 = 0.09051875 + // 16963*0.5/1e6 = 0.0084815 + // 472*25/1e6 = 0.0118 + // sum ~= 0.11083025 const cost = computeCost("claude-opus-4-7", { input_tokens: 6, cache_creation_input_tokens: 14483, cache_read_input_tokens: 16963, output_tokens: 472, }); - expect(cost).toBeCloseTo(0.3324907, 4); + expect(cost).toBeCloseTo(0.11083025, 4); }); }); diff --git a/packages/core/src/pricing.ts b/packages/core/src/pricing.ts index e91fa64..2163dd2 100644 --- a/packages/core/src/pricing.ts +++ b/packages/core/src/pricing.ts @@ -23,48 +23,76 @@ export interface ModelPricing { readonly cacheReadPerMTok: number; } +// Rates verified against platform.claude.com pricing on 2026-07-11. +// Opus 4.5 dropped the Opus tier to $5/$25; every Opus since (4.6/4.7/4.8) +// stays there. Only the pre-4.5 Opus models (4.0/4.1) bill at $15/$75. +export const PRICING_VERIFIED_DATE = "2026-07-11"; + +const OPUS_TIER: ModelPricing = { + inputPerMTok: 5, + outputPerMTok: 25, + cacheWritePerMTok: 6.25, + cacheReadPerMTok: 0.5, +}; + +const SONNET_TIER: ModelPricing = { + inputPerMTok: 3, + outputPerMTok: 15, + cacheWritePerMTok: 3.75, + cacheReadPerMTok: 0.3, +}; + +const LEGACY_OPUS_TIER: ModelPricing = { + inputPerMTok: 15, + outputPerMTok: 75, + cacheWritePerMTok: 18.75, + cacheReadPerMTok: 1.5, +}; + export const ANTHROPIC_PRICING: Record = { - "claude-opus-4-7": { - inputPerMTok: 15, - outputPerMTok: 75, - cacheWritePerMTok: 18.75, - cacheReadPerMTok: 1.5, - }, - "claude-opus-4-6": { - inputPerMTok: 15, - outputPerMTok: 75, - cacheWritePerMTok: 18.75, - cacheReadPerMTok: 1.5, - }, - "claude-sonnet-4-6": { - inputPerMTok: 3, - outputPerMTok: 15, - cacheWritePerMTok: 3.75, - cacheReadPerMTok: 0.3, + // Mythos-class tier ($10/$50) + "claude-fable-5": { + inputPerMTok: 10, + outputPerMTok: 50, + cacheWritePerMTok: 12.5, + cacheReadPerMTok: 1, }, - "claude-sonnet-4-5": { - inputPerMTok: 3, - outputPerMTok: 15, - cacheWritePerMTok: 3.75, - cacheReadPerMTok: 0.3, + "claude-mythos-5": { + inputPerMTok: 10, + outputPerMTok: 50, + cacheWritePerMTok: 12.5, + cacheReadPerMTok: 1, }, + // Opus tier ($5/$25 since Opus 4.5) + "claude-opus-4-8": OPUS_TIER, + "claude-opus-4-7": OPUS_TIER, + "claude-opus-4-6": OPUS_TIER, + "claude-opus-4-5": OPUS_TIER, + // Sonnet tier ($3/$15). Sonnet 5 has an intro rate of $2/$10 through + // 2026-08-31; we bill at the sticker rate, which slightly overestimates + // until then — the safe direction for budget enforcement. + "claude-sonnet-5": SONNET_TIER, + "claude-sonnet-4-6": SONNET_TIER, + "claude-sonnet-4-5": SONNET_TIER, + // Haiku tier "claude-haiku-4-5": { inputPerMTok: 1, outputPerMTok: 5, cacheWritePerMTok: 1.25, cacheReadPerMTok: 0.1, }, + // Pre-4.5 Opus (deprecated but still billable on old transcripts) + "claude-opus-4-1": LEGACY_OPUS_TIER, + "claude-opus-4-0": LEGACY_OPUS_TIER, }; // Verified parity with Anthropic-direct rates for us-east-1 / us-west-2 on // 2026-05-13. If you ship into a non-US region or AWS adjusts published -// rates, override these entries. +// rates, override individual entries here. Spreading the whole table keeps +// Bedrock in lockstep as models are added, so new models never silently +// price at $0 on one backend only. export const BEDROCK_PRICING: Record = { - "claude-opus-4-7": ANTHROPIC_PRICING["claude-opus-4-7"]!, - "claude-opus-4-6": ANTHROPIC_PRICING["claude-opus-4-6"]!, - "claude-sonnet-4-6": ANTHROPIC_PRICING["claude-sonnet-4-6"]!, - "claude-sonnet-4-5": ANTHROPIC_PRICING["claude-sonnet-4-5"]!, - "claude-haiku-4-5": ANTHROPIC_PRICING["claude-haiku-4-5"]!, + ...ANTHROPIC_PRICING, }; // BEDROCK_PRICING above is a parity copy of the Anthropic-direct rates, not