diff --git a/apps/server/src/usage/usageAggregation.test.ts b/apps/server/src/usage/usageAggregation.test.ts index 75435de08ff7..a6b609fecfd3 100644 --- a/apps/server/src/usage/usageAggregation.test.ts +++ b/apps/server/src/usage/usageAggregation.test.ts @@ -12,6 +12,7 @@ const rates: RateTable = new Map([ outputCostPerToken: 5e-5, cacheReadCostPerToken: 1e-6, cacheCreationCostPerToken: 1.25e-5, + cacheCreation1hCostPerToken: 2e-5, fastMultiplier: 1, }, ], diff --git a/apps/server/src/usage/usagePricing.test.ts b/apps/server/src/usage/usagePricing.test.ts index db4f68c2cb42..1d72326056ac 100644 --- a/apps/server/src/usage/usagePricing.test.ts +++ b/apps/server/src/usage/usagePricing.test.ts @@ -7,6 +7,7 @@ import { lookupRate, parseRateTable, priceUsage, + type RateTable, } from "./usagePricing.ts"; const rate = (input: number, cacheRead?: number) => ({ @@ -131,6 +132,55 @@ describe("usage pricing", () => { ); }); + describe("cache writes", () => { + const table = parseRateTable({ + "claude-sonnet-5": { + ...rate(2e-6), + cache_creation_input_token_cost: 2.5e-6, + cache_creation_input_token_cost_above_1hr: 4e-6, + }, + "claude-no-1h-rate": { ...rate(2e-6), cache_creation_input_token_cost: 2.5e-6 }, + }); + const writes = { + uncachedInputTokens: 0, + cachedInputTokens: 0, + cacheCreationTokens: 1_000_000, + outputTokens: 0, + reasoningTokens: 0, + }; + const cost = (model: string, cacheCreation1hTokens?: number, overrides?: RateTable) => + priceUsage( + table, + { + model, + totals: writes, + reportedCostUsd: null, + fast: false, + ...(cacheCreation1hTokens === undefined ? {} : { cacheCreation1hTokens }), + }, + overrides, + ).costUsd; + + it("prices 5-minute, 1-hour, and mixed writes at their own rates", () => { + expect(cost("claude-sonnet-5")).toBeCloseTo(2.5); + expect(cost("claude-sonnet-5", 0)).toBeCloseTo(2.5); + expect(cost("claude-sonnet-5", 1_000_000)).toBeCloseTo(4); + expect(cost("claude-sonnet-5", 250_000)).toBeCloseTo(0.75 * 2.5 + 0.25 * 4); + }); + + it("falls back to the 5-minute rate when no 1-hour rate is published or set", () => { + expect(cost("claude-no-1h-rate", 1_000_000)).toBeCloseTo(2.5); + const overrides = createOverrideRateTable({ + "claude-sonnet-5": { + inputCostPerMillionTokens: 2, + outputCostPerMillionTokens: 10, + cacheWriteCostPerMillionTokens: 3, + }, + }); + expect(cost("claude-sonnet-5", 1_000_000, overrides)).toBeCloseTo(3); + }); + }); + it("keeps the canonical Fable rate separate from DeepInfra in either order", () => { const canonical = ["claude-fable-5", rate(1e-5, 1e-6)] as const; const deepInfra = ["deepinfra/anthropic/claude-fable-5", rate(1e-5)] as const; diff --git a/apps/server/src/usage/usagePricing.ts b/apps/server/src/usage/usagePricing.ts index 78f6cf2c5cd9..961a36304826 100644 --- a/apps/server/src/usage/usagePricing.ts +++ b/apps/server/src/usage/usagePricing.ts @@ -23,7 +23,10 @@ export interface ModelRate { readonly inputCostPerToken: number; readonly outputCostPerToken: number; readonly cacheReadCostPerToken: number; + /** Cache writes at the default 5-minute TTL. */ readonly cacheCreationCostPerToken: number; + /** Cache writes at the 1-hour TTL, which Claude Code uses by default. */ + readonly cacheCreation1hCostPerToken: number; /** * Multiple of the rates above billed for a fast-mode request, from LiteLLM's * `provider_specific_entry.fast`. `1` when the model publishes no fast tier. @@ -41,27 +44,32 @@ export function createOverrideRateTable( overrides: Readonly>, ): RateTable { return new Map( - Object.entries(overrides).map(([model, prices]) => [ - model.trim(), - { - inputCostPerToken: prices.inputCostPerMillionTokens / 1_000_000, - outputCostPerToken: prices.outputCostPerMillionTokens / 1_000_000, - cacheReadCostPerToken: - (prices.cacheReadCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, - cacheCreationCostPerToken: - (prices.cacheWriteCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, - fastMultiplier: 1, - }, - ]), + Object.entries(overrides).map(([model, prices]) => [model.trim(), overrideRate(prices)]), ); } +function overrideRate(prices: UsageModelPriceOverride): ModelRate { + const cacheWrite = + (prices.cacheWriteCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000; + return { + inputCostPerToken: prices.inputCostPerMillionTokens / 1_000_000, + outputCostPerToken: prices.outputCostPerMillionTokens / 1_000_000, + cacheReadCostPerToken: + (prices.cacheReadCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, + // A custom price has one cache-write rate; it applies to both TTLs. + cacheCreationCostPerToken: cacheWrite, + cacheCreation1hCostPerToken: cacheWrite, + fastMultiplier: 1, + }; +} + /** Raw shape of one LiteLLM entry, narrowed to the fields we read. */ interface LiteLlmEntry { readonly input_cost_per_token?: unknown; readonly output_cost_per_token?: unknown; readonly cache_read_input_token_cost?: unknown; readonly cache_creation_input_token_cost?: unknown; + readonly cache_creation_input_token_cost_above_1hr?: unknown; readonly provider_specific_entry?: unknown; } @@ -100,6 +108,7 @@ export function parseRateTable(document: unknown): RateTable { const key = normalizeRateKey(name); if (key.length === 0) continue; + const cacheCreation = finiteNumber(entry.cache_creation_input_token_cost) ?? input; table.set(key, { inputCostPerToken: input, outputCostPerToken: output, @@ -107,7 +116,9 @@ export function parseRateTable(document: unknown): RateTable { // premium. When a model omits them, cached input is priced as plain // input rather than as free. cacheReadCostPerToken: finiteNumber(entry.cache_read_input_token_cost) ?? input, - cacheCreationCostPerToken: finiteNumber(entry.cache_creation_input_token_cost) ?? input, + cacheCreationCostPerToken: cacheCreation, + cacheCreation1hCostPerToken: + finiteNumber(entry.cache_creation_input_token_cost_above_1hr) ?? cacheCreation, fastMultiplier: fastMultiplier(entry), }); } @@ -137,6 +148,7 @@ function sameRate(a: ModelRate, b: ModelRate): boolean { a.outputCostPerToken === b.outputCostPerToken && a.cacheReadCostPerToken === b.cacheReadCostPerToken && a.cacheCreationCostPerToken === b.cacheCreationCostPerToken && + a.cacheCreation1hCostPerToken === b.cacheCreation1hCostPerToken && a.fastMultiplier === b.fastMultiplier ); } @@ -186,7 +198,7 @@ export function lookupRate(table: RateTable, model: string): ModelRate | null { /** The parts of a transcript record that decide its price. */ export type PricedRecord = Pick< UsageRecord, - "model" | "rateModel" | "totals" | "fast" | "reportedCostUsd" + "model" | "rateModel" | "totals" | "fast" | "cacheCreation1hTokens" | "reportedCostUsd" >; export interface PricedUsage { @@ -214,10 +226,12 @@ export function priceUsage( const rate = override ?? lookupRate(table, record.rateModel ?? model); if (rate === null) return { costUsd: 0, costSource: "unpriced" }; + const cacheCreation1hTokens = record.cacheCreation1hTokens ?? 0; const standardCostUsd = totals.uncachedInputTokens * rate.inputCostPerToken + totals.cachedInputTokens * rate.cacheReadCostPerToken + - totals.cacheCreationTokens * rate.cacheCreationCostPerToken + + (totals.cacheCreationTokens - cacheCreation1hTokens) * rate.cacheCreationCostPerToken + + cacheCreation1hTokens * rate.cacheCreation1hCostPerToken + totals.outputTokens * rate.outputCostPerToken; return { diff --git a/apps/server/src/usage/usageScanCache.test.ts b/apps/server/src/usage/usageScanCache.test.ts index 6455b1eb7770..a5f573ff6abf 100644 --- a/apps/server/src/usage/usageScanCache.test.ts +++ b/apps/server/src/usage/usageScanCache.test.ts @@ -61,7 +61,15 @@ describe("scan cache round trip", () => { [ "/a.jsonl", 100, - [record(), record({ dedupeKey: "msg_2:", model: "claude-opus-5-5", fast: true })], + [ + record(), + record({ + dedupeKey: "msg_2:", + model: "claude-opus-5-5", + fast: true, + cacheCreation1hTokens: 8, + }), + ], ], ["/b.jsonl", 200, [record({ sessionId: "session-b", reportedCostUsd: 1.5 })]], ]); @@ -133,15 +141,33 @@ describe("scan cache round trip", () => { const row = encoded.files["/a.jsonl"]!.r[0]!; const poisoned = { ...encoded, - files: { "/a.jsonl": { ...encoded.files["/a.jsonl"]!, r: [[...row.slice(0, 10), true]] } }, + files: { "/a.jsonl": { ...encoded.files["/a.jsonl"]!, r: [[...row.slice(0, 10), true, 0]] } }, }; expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); }); + it("drops an entry whose 1-hour cache writes fall outside the cache-write total", () => { + const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record()]]])); + const row = encoded.files["/a.jsonl"]!.r[0]!; + for (const cacheCreation1h of [-1, 11]) { + const poisoned = { + ...encoded, + files: { + "/a.jsonl": { + ...encoded.files["/a.jsonl"]!, + r: [[...row.slice(0, 11), cacheCreation1h]], + }, + }, + }; + + expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); + } + }); + it("rejects a document from the previous cache version", () => { const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record()]]])); - const previous = { ...encoded, version: 3 }; + const previous = { ...encoded, version: 4 }; expect(decodeScanCache(JSON.parse(JSON.stringify(previous))).size).toBe(0); }); diff --git a/apps/server/src/usage/usageScanCache.ts b/apps/server/src/usage/usageScanCache.ts index 79cca5cff8e2..fe656eae1b76 100644 --- a/apps/server/src/usage/usageScanCache.ts +++ b/apps/server/src/usage/usageScanCache.ts @@ -24,7 +24,8 @@ import type { CodexScanState, UsageRecord } from "./usageTranscripts.ts"; // v3: entries carry the parse position and reducer state so a grown file // re-parses only its appended bytes instead of starting over. // v4: records carry Claude fast mode, which v3 rows never captured. -const USAGE_SCAN_CACHE_VERSION = 4 as const; +// v5: records carry Claude 1-hour cache writes, which v4 rows never captured. +const USAGE_SCAN_CACHE_VERSION = 5 as const; export interface CachedFile { readonly size: number; @@ -60,6 +61,7 @@ type SerializedRecord = readonly [ dedupeKey: string | null, reportedCostUsd: number | null, fast: 0 | 1, + cacheCreation1hTokens: number, ]; interface SerializedFile { @@ -112,6 +114,7 @@ export function encodeScanCache(cache: ScanCache): SerializedCache { record.dedupeKey, record.reportedCostUsd, record.fast ? 1 : 0, + record.cacheCreation1hTokens ?? 0, ]; const files: Record = {}; @@ -168,7 +171,7 @@ export function decodeScanCache(document: unknown): ScanCache { ): UsageRecord[] | null => { const records: UsageRecord[] = []; for (const row of rows) { - if (!isRecordArray(row) || row.length < 11) return null; + if (!isRecordArray(row) || row.length < 12) return null; const [ timestampMs, modelIndex, @@ -181,6 +184,7 @@ export function decodeScanCache(document: unknown): ScanCache { dedupeKey, reportedCostUsd, fast, + cacheCreation1h, ] = row as SerializedRecord; const model = typeof modelIndex === "number" ? models[modelIndex] : undefined; @@ -193,7 +197,10 @@ export function decodeScanCache(document: unknown): ScanCache { !Number.isFinite(cacheCreation) || !Number.isFinite(output) || !Number.isFinite(reasoning) || - (fast !== 0 && fast !== 1) + (fast !== 0 && fast !== 1) || + !Number.isFinite(cacheCreation1h) || + cacheCreation1h < 0 || + cacheCreation1h > cacheCreation ) { return null; } @@ -212,6 +219,7 @@ export function decodeScanCache(document: unknown): ScanCache { }, reportedCostUsd: typeof reportedCostUsd === "number" ? reportedCostUsd : null, fast: fast === 1, + ...(cacheCreation1h > 0 ? { cacheCreation1hTokens: cacheCreation1h } : {}), dedupeKey: typeof dedupeKey === "string" ? dedupeKey : null, }); } diff --git a/apps/server/src/usage/usageTranscripts.test.ts b/apps/server/src/usage/usageTranscripts.test.ts index ace3b7d18cf7..a514da8e8f94 100644 --- a/apps/server/src/usage/usageTranscripts.test.ts +++ b/apps/server/src/usage/usageTranscripts.test.ts @@ -16,6 +16,7 @@ function claudeLine(overrides: { model?: string; outputTokens?: number; speed?: string; + cacheCreation?: Record; }): string { return JSON.stringify({ type: "assistant", @@ -33,6 +34,9 @@ function claudeLine(overrides: { cache_read_input_tokens: 1000, output_tokens: overrides.outputTokens ?? 286, ...(overrides.speed === undefined ? {} : { speed: overrides.speed }), + ...(overrides.cacheCreation === undefined + ? {} + : { cache_creation: overrides.cacheCreation }), }, }, }); @@ -64,6 +68,23 @@ describe("parseClaudeLine", () => { expect(line("standard")?.fast).toBe(false); }); + it("reads the 1-hour share of cache writes", () => { + const line = (cacheCreation?: Record) => + parseClaudeLine( + claudeLine({ + messageId: "msg_1", + contentType: "text", + ...(cacheCreation === undefined ? {} : { cacheCreation }), + }), + )?.cacheCreation1hTokens; + + expect(line({ ephemeral_5m_input_tokens: 818, ephemeral_1h_input_tokens: 66000 })).toBe(66000); + expect(line({ ephemeral_5m_input_tokens: 66818, ephemeral_1h_input_tokens: 0 })).toBe(0); + expect(line()).toBe(0); + // Never more than the total, so the 5-minute remainder cannot go negative. + expect(line({ ephemeral_1h_input_tokens: 99_999_999 })).toBe(66818); + }); + it("gives every content block of one message the same dedupe key", () => { // T3 Code writes one record per content block, each repeating the parent // message's full usage. Summing them would overcount ~2.4x on real data. diff --git a/apps/server/src/usage/usageTranscripts.ts b/apps/server/src/usage/usageTranscripts.ts index 6e01c2c5a8ed..9d75d5470429 100644 --- a/apps/server/src/usage/usageTranscripts.ts +++ b/apps/server/src/usage/usageTranscripts.ts @@ -25,6 +25,12 @@ export interface UsageRecord { * multiple of the standard rate. Only Claude Code records this. */ readonly fast: boolean; + /** + * The part of `totals.cacheCreationTokens` written to the 1-hour cache, + * which bills at a higher rate than the 5-minute default. Only Claude Code + * records this. + */ + readonly cacheCreation1hTokens?: number; /** * Key for cross-file de-duplication, or `null` when the record is inherently * unique and needs no dedup. @@ -140,6 +146,12 @@ export function parseClaudeLine(line: string): UsageRecord | null { messageId === null && requestId === null ? null : `${messageId ?? ""}:${requestId ?? ""}`; const cost = record["costUSD"]; + const cacheCreationTokens = int(usageRecord["cache_creation_input_tokens"]); + const cacheCreation = usageRecord["cache_creation"]; + const cacheCreation1hTokens = + typeof cacheCreation === "object" && cacheCreation !== null + ? int((cacheCreation as Record)["ephemeral_1h_input_tokens"]) + : 0; return { provider: "claude", @@ -149,13 +161,14 @@ export function parseClaudeLine(line: string): UsageRecord | null { totals: { uncachedInputTokens: int(usageRecord["input_tokens"]), cachedInputTokens: int(usageRecord["cache_read_input_tokens"]), - cacheCreationTokens: int(usageRecord["cache_creation_input_tokens"]), + cacheCreationTokens, outputTokens: int(usageRecord["output_tokens"]), // Anthropic folds thinking tokens into output and does not break them out. reasoningTokens: 0, }, reportedCostUsd: typeof cost === "number" && Number.isFinite(cost) ? cost : null, fast: usageRecord["speed"] === "fast", + cacheCreation1hTokens: Math.min(cacheCreation1hTokens, cacheCreationTokens), dedupeKey, }; } diff --git a/docs/user/usage.md b/docs/user/usage.md index ab77f70a6b3a..9e67fd93ea2e 100644 --- a/docs/user/usage.md +++ b/docs/user/usage.md @@ -47,9 +47,9 @@ rates per million input and output tokens. You can enter any model ID, including without public pricing. Cache read and cache write rates are optional and use the input rate when blank. Enter `0` for -tokens that are free. Saved prices replace automatic pricing for all of that environment's -history and are shared with clients connected to it. When environments have different prices, -cells show **Mixed**. Edit rates directly in the table, then choose **Save changes** to apply all +tokens that are free. The cache write rate applies to both 5-minute and 1-hour cache writes. +Saved prices replace automatic pricing for all of that environment's history and are shared with +clients connected to it. When environments have different prices, cells show **Mixed**. Edit rates directly in the table, then choose **Save changes** to apply all edited rows. Untouched cells keep each environment's rate. Select one environment to inspect its prices. **Reset to automatic** marks a model's override for removal when you save; you can undo it before saving.