From 66a3d282596d970700071f79d93957524a85e4e9 Mon Sep 17 00:00:00 2001 From: Jai Khurana <64320051+jaikhuranna@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:35:49 +0530 Subject: [PATCH 1/3] fix(usage): price Claude 1-hour cache writes at the 1-hour rate Claude Code writes most prompt-cache entries with a 1-hour TTL, which Anthropic bills at 2x input, but every cache write was priced at the 5-minute rate (1.25x input). Read the ephemeral_1h_input_tokens split from Claude transcripts and LiteLLM's cache_creation_input_token_cost_above_1hr rate, falling back to the 5-minute rate when either is missing. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../server/src/usage/usageAggregation.test.ts | 1 + apps/server/src/usage/usagePricing.test.ts | 50 +++++++++++++++++++ apps/server/src/usage/usagePricing.ts | 47 +++++++++++------ apps/server/src/usage/usageScanCache.test.ts | 14 ++++-- apps/server/src/usage/usageScanCache.ts | 12 +++-- .../server/src/usage/usageTranscripts.test.ts | 21 ++++++++ apps/server/src/usage/usageTranscripts.ts | 15 +++++- docs/user/usage.md | 2 +- 8 files changed, 139 insertions(+), 23 deletions(-) diff --git a/apps/server/src/usage/usageAggregation.test.ts b/apps/server/src/usage/usageAggregation.test.ts index 75435de08ff7..a6b609fecfd3 100644 --- a/apps/server/src/usage/usageAggregation.test.ts +++ b/apps/server/src/usage/usageAggregation.test.ts @@ -12,6 +12,7 @@ const rates: RateTable = new Map([ outputCostPerToken: 5e-5, cacheReadCostPerToken: 1e-6, cacheCreationCostPerToken: 1.25e-5, + cacheCreation1hCostPerToken: 2e-5, fastMultiplier: 1, }, ], diff --git a/apps/server/src/usage/usagePricing.test.ts b/apps/server/src/usage/usagePricing.test.ts index ca340a44a64e..7727c9310519 100644 --- a/apps/server/src/usage/usagePricing.test.ts +++ b/apps/server/src/usage/usagePricing.test.ts @@ -6,6 +6,7 @@ import { lookupRate, parseRateTable, priceUsage, + type RateTable, } from "./usagePricing.ts"; const rate = (input: number, cacheRead?: number) => ({ @@ -110,6 +111,55 @@ describe("usage pricing", () => { ); }); + describe("cache writes", () => { + const table = parseRateTable({ + "claude-sonnet-5": { + ...rate(2e-6), + cache_creation_input_token_cost: 2.5e-6, + cache_creation_input_token_cost_above_1hr: 4e-6, + }, + "claude-no-1h-rate": { ...rate(2e-6), cache_creation_input_token_cost: 2.5e-6 }, + }); + const writes = { + uncachedInputTokens: 0, + cachedInputTokens: 0, + cacheCreationTokens: 1_000_000, + outputTokens: 0, + reasoningTokens: 0, + }; + const cost = (model: string, cacheCreation1hTokens?: number, overrides?: RateTable) => + priceUsage( + table, + { + model, + totals: writes, + reportedCostUsd: null, + fast: false, + ...(cacheCreation1hTokens === undefined ? {} : { cacheCreation1hTokens }), + }, + overrides, + ).costUsd; + + it("prices 5-minute, 1-hour, and mixed writes at their own rates", () => { + expect(cost("claude-sonnet-5")).toBeCloseTo(2.5); + expect(cost("claude-sonnet-5", 0)).toBeCloseTo(2.5); + expect(cost("claude-sonnet-5", 1_000_000)).toBeCloseTo(4); + expect(cost("claude-sonnet-5", 250_000)).toBeCloseTo(0.75 * 2.5 + 0.25 * 4); + }); + + it("falls back to the 5-minute rate when no 1-hour rate is published or set", () => { + expect(cost("claude-no-1h-rate", 1_000_000)).toBeCloseTo(2.5); + const overrides = createOverrideRateTable({ + "claude-sonnet-5": { + inputCostPerMillionTokens: 2, + outputCostPerMillionTokens: 10, + cacheWriteCostPerMillionTokens: 3, + }, + }); + expect(cost("claude-sonnet-5", 1_000_000, overrides)).toBeCloseTo(3); + }); + }); + it("keeps the canonical Fable rate separate from DeepInfra in either order", () => { const canonical = ["claude-fable-5", rate(1e-5, 1e-6)] as const; const deepInfra = ["deepinfra/anthropic/claude-fable-5", rate(1e-5)] as const; diff --git a/apps/server/src/usage/usagePricing.ts b/apps/server/src/usage/usagePricing.ts index 60bf31b93e31..09fab5891fb2 100644 --- a/apps/server/src/usage/usagePricing.ts +++ b/apps/server/src/usage/usagePricing.ts @@ -23,7 +23,10 @@ export interface ModelRate { readonly inputCostPerToken: number; readonly outputCostPerToken: number; readonly cacheReadCostPerToken: number; + /** Cache writes at the default 5-minute TTL. */ readonly cacheCreationCostPerToken: number; + /** Cache writes at the 1-hour TTL, which Claude Code uses by default. */ + readonly cacheCreation1hCostPerToken: number; /** * Multiple of the rates above billed for a fast-mode request, from LiteLLM's * `provider_specific_entry.fast`. `1` when the model publishes no fast tier. @@ -41,27 +44,32 @@ export function createOverrideRateTable( overrides: Readonly>, ): RateTable { return new Map( - Object.entries(overrides).map(([model, prices]) => [ - model.trim(), - { - inputCostPerToken: prices.inputCostPerMillionTokens / 1_000_000, - outputCostPerToken: prices.outputCostPerMillionTokens / 1_000_000, - cacheReadCostPerToken: - (prices.cacheReadCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, - cacheCreationCostPerToken: - (prices.cacheWriteCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, - fastMultiplier: 1, - }, - ]), + Object.entries(overrides).map(([model, prices]) => [model.trim(), overrideRate(prices)]), ); } +function overrideRate(prices: UsageModelPriceOverride): ModelRate { + const cacheWrite = + (prices.cacheWriteCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000; + return { + inputCostPerToken: prices.inputCostPerMillionTokens / 1_000_000, + outputCostPerToken: prices.outputCostPerMillionTokens / 1_000_000, + cacheReadCostPerToken: + (prices.cacheReadCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, + // A custom price has one cache-write rate; it applies to both TTLs. + cacheCreationCostPerToken: cacheWrite, + cacheCreation1hCostPerToken: cacheWrite, + fastMultiplier: 1, + }; +} + /** Raw shape of one LiteLLM entry, narrowed to the fields we read. */ interface LiteLlmEntry { readonly input_cost_per_token?: unknown; readonly output_cost_per_token?: unknown; readonly cache_read_input_token_cost?: unknown; readonly cache_creation_input_token_cost?: unknown; + readonly cache_creation_input_token_cost_above_1hr?: unknown; readonly provider_specific_entry?: unknown; } @@ -100,6 +108,7 @@ export function parseRateTable(document: unknown): RateTable { const key = normalizeRateKey(name); if (key.length === 0) continue; + const cacheCreation = finiteNumber(entry.cache_creation_input_token_cost) ?? input; table.set(key, { inputCostPerToken: input, outputCostPerToken: output, @@ -107,7 +116,9 @@ export function parseRateTable(document: unknown): RateTable { // premium. When a model omits them, cached input is priced as plain // input rather than as free. cacheReadCostPerToken: finiteNumber(entry.cache_read_input_token_cost) ?? input, - cacheCreationCostPerToken: finiteNumber(entry.cache_creation_input_token_cost) ?? input, + cacheCreationCostPerToken: cacheCreation, + cacheCreation1hCostPerToken: + finiteNumber(entry.cache_creation_input_token_cost_above_1hr) ?? cacheCreation, fastMultiplier: fastMultiplier(entry), }); } @@ -137,6 +148,7 @@ function sameRate(a: ModelRate, b: ModelRate): boolean { a.outputCostPerToken === b.outputCostPerToken && a.cacheReadCostPerToken === b.cacheReadCostPerToken && a.cacheCreationCostPerToken === b.cacheCreationCostPerToken && + a.cacheCreation1hCostPerToken === b.cacheCreation1hCostPerToken && a.fastMultiplier === b.fastMultiplier ); } @@ -184,7 +196,10 @@ export function lookupRate(table: RateTable, model: string): ModelRate | null { } /** The parts of a transcript record that decide its price. */ -export type PricedRecord = Pick; +export type PricedRecord = Pick< + UsageRecord, + "model" | "totals" | "fast" | "cacheCreation1hTokens" | "reportedCostUsd" +>; export interface PricedUsage { readonly costUsd: number; @@ -211,10 +226,12 @@ export function priceUsage( const rate = override ?? lookupRate(table, model); if (rate === null) return { costUsd: 0, costSource: "unpriced" }; + const cacheCreation1hTokens = record.cacheCreation1hTokens ?? 0; const standardCostUsd = totals.uncachedInputTokens * rate.inputCostPerToken + totals.cachedInputTokens * rate.cacheReadCostPerToken + - totals.cacheCreationTokens * rate.cacheCreationCostPerToken + + (totals.cacheCreationTokens - cacheCreation1hTokens) * rate.cacheCreationCostPerToken + + cacheCreation1hTokens * rate.cacheCreation1hCostPerToken + totals.outputTokens * rate.outputCostPerToken; return { diff --git a/apps/server/src/usage/usageScanCache.test.ts b/apps/server/src/usage/usageScanCache.test.ts index 6455b1eb7770..04eb421d7659 100644 --- a/apps/server/src/usage/usageScanCache.test.ts +++ b/apps/server/src/usage/usageScanCache.test.ts @@ -61,7 +61,15 @@ describe("scan cache round trip", () => { [ "/a.jsonl", 100, - [record(), record({ dedupeKey: "msg_2:", model: "claude-opus-5-5", fast: true })], + [ + record(), + record({ + dedupeKey: "msg_2:", + model: "claude-opus-5-5", + fast: true, + cacheCreation1hTokens: 8, + }), + ], ], ["/b.jsonl", 200, [record({ sessionId: "session-b", reportedCostUsd: 1.5 })]], ]); @@ -133,7 +141,7 @@ describe("scan cache round trip", () => { const row = encoded.files["/a.jsonl"]!.r[0]!; const poisoned = { ...encoded, - files: { "/a.jsonl": { ...encoded.files["/a.jsonl"]!, r: [[...row.slice(0, 10), true]] } }, + files: { "/a.jsonl": { ...encoded.files["/a.jsonl"]!, r: [[...row.slice(0, 10), true, 0]] } }, }; expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); @@ -141,7 +149,7 @@ describe("scan cache round trip", () => { it("rejects a document from the previous cache version", () => { const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record()]]])); - const previous = { ...encoded, version: 3 }; + const previous = { ...encoded, version: 4 }; expect(decodeScanCache(JSON.parse(JSON.stringify(previous))).size).toBe(0); }); diff --git a/apps/server/src/usage/usageScanCache.ts b/apps/server/src/usage/usageScanCache.ts index 79cca5cff8e2..94601af61097 100644 --- a/apps/server/src/usage/usageScanCache.ts +++ b/apps/server/src/usage/usageScanCache.ts @@ -24,7 +24,8 @@ import type { CodexScanState, UsageRecord } from "./usageTranscripts.ts"; // v3: entries carry the parse position and reducer state so a grown file // re-parses only its appended bytes instead of starting over. // v4: records carry Claude fast mode, which v3 rows never captured. -const USAGE_SCAN_CACHE_VERSION = 4 as const; +// v5: records carry Claude 1-hour cache writes, which v4 rows never captured. +const USAGE_SCAN_CACHE_VERSION = 5 as const; export interface CachedFile { readonly size: number; @@ -60,6 +61,7 @@ type SerializedRecord = readonly [ dedupeKey: string | null, reportedCostUsd: number | null, fast: 0 | 1, + cacheCreation1hTokens: number, ]; interface SerializedFile { @@ -112,6 +114,7 @@ export function encodeScanCache(cache: ScanCache): SerializedCache { record.dedupeKey, record.reportedCostUsd, record.fast ? 1 : 0, + record.cacheCreation1hTokens ?? 0, ]; const files: Record = {}; @@ -168,7 +171,7 @@ export function decodeScanCache(document: unknown): ScanCache { ): UsageRecord[] | null => { const records: UsageRecord[] = []; for (const row of rows) { - if (!isRecordArray(row) || row.length < 11) return null; + if (!isRecordArray(row) || row.length < 12) return null; const [ timestampMs, modelIndex, @@ -181,6 +184,7 @@ export function decodeScanCache(document: unknown): ScanCache { dedupeKey, reportedCostUsd, fast, + cacheCreation1h, ] = row as SerializedRecord; const model = typeof modelIndex === "number" ? models[modelIndex] : undefined; @@ -193,7 +197,8 @@ export function decodeScanCache(document: unknown): ScanCache { !Number.isFinite(cacheCreation) || !Number.isFinite(output) || !Number.isFinite(reasoning) || - (fast !== 0 && fast !== 1) + (fast !== 0 && fast !== 1) || + !Number.isFinite(cacheCreation1h) ) { return null; } @@ -212,6 +217,7 @@ export function decodeScanCache(document: unknown): ScanCache { }, reportedCostUsd: typeof reportedCostUsd === "number" ? reportedCostUsd : null, fast: fast === 1, + ...(cacheCreation1h > 0 ? { cacheCreation1hTokens: cacheCreation1h } : {}), dedupeKey: typeof dedupeKey === "string" ? dedupeKey : null, }); } diff --git a/apps/server/src/usage/usageTranscripts.test.ts b/apps/server/src/usage/usageTranscripts.test.ts index ace3b7d18cf7..a514da8e8f94 100644 --- a/apps/server/src/usage/usageTranscripts.test.ts +++ b/apps/server/src/usage/usageTranscripts.test.ts @@ -16,6 +16,7 @@ function claudeLine(overrides: { model?: string; outputTokens?: number; speed?: string; + cacheCreation?: Record; }): string { return JSON.stringify({ type: "assistant", @@ -33,6 +34,9 @@ function claudeLine(overrides: { cache_read_input_tokens: 1000, output_tokens: overrides.outputTokens ?? 286, ...(overrides.speed === undefined ? {} : { speed: overrides.speed }), + ...(overrides.cacheCreation === undefined + ? {} + : { cache_creation: overrides.cacheCreation }), }, }, }); @@ -64,6 +68,23 @@ describe("parseClaudeLine", () => { expect(line("standard")?.fast).toBe(false); }); + it("reads the 1-hour share of cache writes", () => { + const line = (cacheCreation?: Record) => + parseClaudeLine( + claudeLine({ + messageId: "msg_1", + contentType: "text", + ...(cacheCreation === undefined ? {} : { cacheCreation }), + }), + )?.cacheCreation1hTokens; + + expect(line({ ephemeral_5m_input_tokens: 818, ephemeral_1h_input_tokens: 66000 })).toBe(66000); + expect(line({ ephemeral_5m_input_tokens: 66818, ephemeral_1h_input_tokens: 0 })).toBe(0); + expect(line()).toBe(0); + // Never more than the total, so the 5-minute remainder cannot go negative. + expect(line({ ephemeral_1h_input_tokens: 99_999_999 })).toBe(66818); + }); + it("gives every content block of one message the same dedupe key", () => { // T3 Code writes one record per content block, each repeating the parent // message's full usage. Summing them would overcount ~2.4x on real data. diff --git a/apps/server/src/usage/usageTranscripts.ts b/apps/server/src/usage/usageTranscripts.ts index 5da13168a1af..1b17dd7afd93 100644 --- a/apps/server/src/usage/usageTranscripts.ts +++ b/apps/server/src/usage/usageTranscripts.ts @@ -20,6 +20,12 @@ export interface UsageRecord { * multiple of the standard rate. Only Claude Code records this. */ readonly fast: boolean; + /** + * The part of `totals.cacheCreationTokens` written to the 1-hour cache, + * which bills at a higher rate than the 5-minute default. Only Claude Code + * records this. + */ + readonly cacheCreation1hTokens?: number; /** * Key for cross-file de-duplication, or `null` when the record is inherently * unique and needs no dedup. @@ -135,6 +141,12 @@ export function parseClaudeLine(line: string): UsageRecord | null { messageId === null && requestId === null ? null : `${messageId ?? ""}:${requestId ?? ""}`; const cost = record["costUSD"]; + const cacheCreationTokens = int(usageRecord["cache_creation_input_tokens"]); + const cacheCreation = usageRecord["cache_creation"]; + const cacheCreation1hTokens = + typeof cacheCreation === "object" && cacheCreation !== null + ? int((cacheCreation as Record)["ephemeral_1h_input_tokens"]) + : 0; return { provider: "claude", @@ -144,13 +156,14 @@ export function parseClaudeLine(line: string): UsageRecord | null { totals: { uncachedInputTokens: int(usageRecord["input_tokens"]), cachedInputTokens: int(usageRecord["cache_read_input_tokens"]), - cacheCreationTokens: int(usageRecord["cache_creation_input_tokens"]), + cacheCreationTokens, outputTokens: int(usageRecord["output_tokens"]), // Anthropic folds thinking tokens into output and does not break them out. reasoningTokens: 0, }, reportedCostUsd: typeof cost === "number" && Number.isFinite(cost) ? cost : null, fast: usageRecord["speed"] === "fast", + cacheCreation1hTokens: Math.min(cacheCreation1hTokens, cacheCreationTokens), dedupeKey, }; } diff --git a/docs/user/usage.md b/docs/user/usage.md index ab77f70a6b3a..de9e45940b57 100644 --- a/docs/user/usage.md +++ b/docs/user/usage.md @@ -47,7 +47,7 @@ rates per million input and output tokens. You can enter any model ID, including without public pricing. Cache read and cache write rates are optional and use the input rate when blank. Enter `0` for -tokens that are free. Saved prices replace automatic pricing for all of that environment's +tokens that are free. The cache write rate applies to both 5-minute and 1-hour cache writes. Saved prices replace automatic pricing for all of that environment's history and are shared with clients connected to it. When environments have different prices, cells show **Mixed**. Edit rates directly in the table, then choose **Save changes** to apply all edited rows. Untouched cells keep each environment's rate. Select one environment to inspect its From 5609c53a091b91d518351e62a62cde227361e7db Mon Sep 17 00:00:00 2001 From: Jai Khurana <64320051+jaikhuranna@users.noreply.github.com> Date: Sat, 26 Sep 2026 10:52:59 +0530 Subject: [PATCH 2/3] fix(usage): re-parse cached rows with an out-of-range 1-hour cache count A cached 1-hour cache-write count above the row's cache-write total would price the 5-minute remainder as negative on a warm scan. Treat it as a corrupt row so the file is parsed again. Co-Authored-By: Claude Opus 5.5 (1M context) --- apps/server/src/usage/usageScanCache.test.ts | 18 ++++++++++++++++++ apps/server/src/usage/usageScanCache.ts | 4 +++- 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/apps/server/src/usage/usageScanCache.test.ts b/apps/server/src/usage/usageScanCache.test.ts index 04eb421d7659..a5f573ff6abf 100644 --- a/apps/server/src/usage/usageScanCache.test.ts +++ b/apps/server/src/usage/usageScanCache.test.ts @@ -147,6 +147,24 @@ describe("scan cache round trip", () => { expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); }); + it("drops an entry whose 1-hour cache writes fall outside the cache-write total", () => { + const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record()]]])); + const row = encoded.files["/a.jsonl"]!.r[0]!; + for (const cacheCreation1h of [-1, 11]) { + const poisoned = { + ...encoded, + files: { + "/a.jsonl": { + ...encoded.files["/a.jsonl"]!, + r: [[...row.slice(0, 11), cacheCreation1h]], + }, + }, + }; + + expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); + } + }); + it("rejects a document from the previous cache version", () => { const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record()]]])); const previous = { ...encoded, version: 4 }; diff --git a/apps/server/src/usage/usageScanCache.ts b/apps/server/src/usage/usageScanCache.ts index 94601af61097..fe656eae1b76 100644 --- a/apps/server/src/usage/usageScanCache.ts +++ b/apps/server/src/usage/usageScanCache.ts @@ -198,7 +198,9 @@ export function decodeScanCache(document: unknown): ScanCache { !Number.isFinite(output) || !Number.isFinite(reasoning) || (fast !== 0 && fast !== 1) || - !Number.isFinite(cacheCreation1h) + !Number.isFinite(cacheCreation1h) || + cacheCreation1h < 0 || + cacheCreation1h > cacheCreation ) { return null; } From 0e414d0fa145e97a4a513a02732fb12b0513369f Mon Sep 17 00:00:00 2001 From: Jai Khurana <64320051+jaikhuranna@users.noreply.github.com> Date: Sun, 27 Sep 2026 09:19:46 +0530 Subject: [PATCH 3/3] docs(usage): rewrap the cache write rate note Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/user/usage.md | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/docs/user/usage.md b/docs/user/usage.md index de9e45940b57..9e67fd93ea2e 100644 --- a/docs/user/usage.md +++ b/docs/user/usage.md @@ -47,9 +47,9 @@ rates per million input and output tokens. You can enter any model ID, including without public pricing. Cache read and cache write rates are optional and use the input rate when blank. Enter `0` for -tokens that are free. The cache write rate applies to both 5-minute and 1-hour cache writes. Saved prices replace automatic pricing for all of that environment's -history and are shared with clients connected to it. When environments have different prices, -cells show **Mixed**. Edit rates directly in the table, then choose **Save changes** to apply all +tokens that are free. The cache write rate applies to both 5-minute and 1-hour cache writes. +Saved prices replace automatic pricing for all of that environment's history and are shared with +clients connected to it. When environments have different prices, cells show **Mixed**. Edit rates directly in the table, then choose **Save changes** to apply all edited rows. Untouched cells keep each environment's rate. Select one environment to inspect its prices. **Reset to automatic** marks a model's override for removal when you save; you can undo it before saving.