diff --git a/apps/server/src/usage/usageAggregation.test.ts b/apps/server/src/usage/usageAggregation.test.ts index 8da4e920ac06..75435de08ff7 100644 --- a/apps/server/src/usage/usageAggregation.test.ts +++ b/apps/server/src/usage/usageAggregation.test.ts @@ -12,6 +12,7 @@ const rates: RateTable = new Map([ outputCostPerToken: 5e-5, cacheReadCostPerToken: 1e-6, cacheCreationCostPerToken: 1.25e-5, + fastMultiplier: 1, }, ], ]); @@ -31,6 +32,7 @@ function record(overrides: Partial = {}): UsageRecord { reasoningTokens: 0, }, reportedCostUsd: null, + fast: false, dedupeKey: null, ...overrides, }; diff --git a/apps/server/src/usage/usageAggregation.ts b/apps/server/src/usage/usageAggregation.ts index 2ad3893ad4e6..684f0a520614 100644 --- a/apps/server/src/usage/usageAggregation.ts +++ b/apps/server/src/usage/usageAggregation.ts @@ -161,20 +161,13 @@ export class UsageAggregator { this.#buckets.set(key, bucket); } - const priced = priceUsage( - this.#options.rates, - record.model, - record.totals, - record.reportedCostUsd, - this.#options.priceOverrides, - ); + const priced = priceUsage(this.#options.rates, record, this.#options.priceOverrides); bucket.totals = addTotals(bucket.totals, record.totals); bucket.costUsd += priced.costUsd; bucket.cacheSavingsUsd += cacheSavingsUsd( this.#options.rates, - record.model, - record.totals, + record, this.#options.priceOverrides, ); bucket.records += 1; diff --git a/apps/server/src/usage/usagePricing.test.ts b/apps/server/src/usage/usagePricing.test.ts index 713d860999cb..ca340a44a64e 100644 --- a/apps/server/src/usage/usagePricing.test.ts +++ b/apps/server/src/usage/usagePricing.test.ts @@ -22,6 +22,12 @@ describe("usage pricing", () => { outputTokens: 1_000_000, reasoningTokens: 500_000, }; + const record = (model: string, reportedCostUsd: number | null = null, fast = false) => ({ + model, + totals, + reportedCostUsd, + fast, + }); it("uses custom token rates ahead of public and provider-reported costs", () => { const table = parseRateTable({ "example-model": rate(1) }); @@ -35,12 +41,12 @@ describe("usage pricing", () => { }); for (const reportedCostUsd of [null, 99]) { - expect(priceUsage(table, "example-model", totals, reportedCostUsd, overrides)).toEqual({ + expect(priceUsage(table, record("example-model", reportedCostUsd), overrides)).toEqual({ costUsd: 13.5, costSource: "modelPriced", }); } - expect(cacheSavingsUsd(table, "example-model", totals, overrides)).toBe(1.5); + expect(cacheSavingsUsd(table, record("example-model"), overrides)).toBe(1.5); }); it("prices unknown models offline and uses input prices for omitted cache rates", () => { @@ -49,11 +55,11 @@ describe("usage pricing", () => { "example-model": { inputCostPerMillionTokens: 2, outputCostPerMillionTokens: 8 }, }); - expect(priceUsage(table, "example-model", totals, null, overrides)).toEqual({ + expect(priceUsage(table, record("example-model"), overrides)).toEqual({ costUsd: 14, costSource: "modelPriced", }); - expect(cacheSavingsUsd(table, "example-model", totals, overrides)).toBe(0); + expect(cacheSavingsUsd(table, record("example-model"), overrides)).toBe(0); }); it("preserves explicit zero rates and matches only the exact trimmed model ID", () => { @@ -64,7 +70,7 @@ describe("usage pricing", () => { outputCostPerMillionTokens: 0, }, }); - expect(priceUsage(table, " vendor/example-model[1m] ", totals, 99, overrides)).toEqual({ + expect(priceUsage(table, record(" vendor/example-model[1m] ", 99), overrides)).toEqual({ costUsd: 0, costSource: "modelPriced", }); @@ -74,14 +80,36 @@ describe("usage pricing", () => { "vendor/Example-model[1m]", "other/example-model[1m]", ]) { - expect(priceUsage(table, model, totals, null, overrides).costSource).toBe("unpriced"); - expect(priceUsage(table, model, totals, 99, overrides)).toEqual({ + expect(priceUsage(table, record(model), overrides).costSource).toBe("unpriced"); + expect(priceUsage(table, record(model, 99), overrides)).toEqual({ costUsd: 99, costSource: "providerReported", }); } }); + it("prices fast-mode requests at the model's published fast multiple", () => { + const table = parseRateTable({ + "claude-opus-5-5": { ...rate(4e-6, 2e-7), provider_specific_entry: { fast: 2, us: 1.1 } }, + "claude-fable-5-1": { ...rate(1e-5, 2.5e-7), provider_specific_entry: { us: 1.1 } }, + }); + const overrides = createOverrideRateTable({ + "claude-opus-5-5": { inputCostPerMillionTokens: 4, outputCostPerMillionTokens: 20 }, + }); + const cost = (model: string, fast: boolean, custom?: typeof overrides) => + priceUsage(table, record(model, null, fast), custom).costUsd; + + expect(cost("claude-opus-5-5", true)).toBeCloseTo(2 * cost("claude-opus-5-5", false)); + expect(cacheSavingsUsd(table, record("claude-opus-5-5", null, true))).toBeCloseTo( + 2 * cacheSavingsUsd(table, record("claude-opus-5-5")), + ); + // No published fast tier, and custom prices, both stay at the standard rate. + expect(cost("claude-fable-5-1", true)).toBe(cost("claude-fable-5-1", false)); + expect(cost("claude-opus-5-5", true, overrides)).toBe( + cost("claude-opus-5-5", false, overrides), + ); + }); + it("keeps the canonical Fable rate separate from DeepInfra in either order", () => { const canonical = ["claude-fable-5", rate(1e-5, 1e-6)] as const; const deepInfra = ["deepinfra/anthropic/claude-fable-5", rate(1e-5)] as const; diff --git a/apps/server/src/usage/usagePricing.ts b/apps/server/src/usage/usagePricing.ts index 6c94be424827..60bf31b93e31 100644 --- a/apps/server/src/usage/usagePricing.ts +++ b/apps/server/src/usage/usagePricing.ts @@ -7,11 +7,9 @@ * * @module usagePricing */ -import type { - UsageCostSource, - UsageModelPriceOverride, - UsageTokenTotals, -} from "@t3tools/contracts"; +import type { UsageCostSource, UsageModelPriceOverride } from "@t3tools/contracts"; + +import type { UsageRecord } from "./usageTranscripts.ts"; /** * The subset of a LiteLLM entry we price against. All values are USD per token. @@ -26,11 +24,19 @@ export interface ModelRate { readonly outputCostPerToken: number; readonly cacheReadCostPerToken: number; readonly cacheCreationCostPerToken: number; + /** + * Multiple of the rates above billed for a fast-mode request, from LiteLLM's + * `provider_specific_entry.fast`. `1` when the model publishes no fast tier. + */ + readonly fastMultiplier: number; } export type RateTable = ReadonlyMap; -/** Custom IDs keep their case, provider prefix, and variant suffix. */ +/** + * Custom IDs keep their case, provider prefix, and variant suffix. Custom rates + * apply as entered, fast-mode requests included. + */ export function createOverrideRateTable( overrides: Readonly>, ): RateTable { @@ -44,6 +50,7 @@ export function createOverrideRateTable( (prices.cacheReadCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, cacheCreationCostPerToken: (prices.cacheWriteCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, + fastMultiplier: 1, }, ]), ); @@ -55,12 +62,21 @@ interface LiteLlmEntry { readonly output_cost_per_token?: unknown; readonly cache_read_input_token_cost?: unknown; readonly cache_creation_input_token_cost?: unknown; + readonly provider_specific_entry?: unknown; } function finiteNumber(value: unknown): number | null { return typeof value === "number" && Number.isFinite(value) ? value : null; } +/** Reads `provider_specific_entry.fast`, e.g. `2` for Claude Opus 5.5. */ +function fastMultiplier(entry: LiteLlmEntry): number { + const specific = entry.provider_specific_entry; + if (typeof specific !== "object" || specific === null) return 1; + const fast = finiteNumber((specific as Record)["fast"]); + return fast !== null && fast > 0 ? fast : 1; +} + /** * Projects the LiteLLM document into a rate table. * @@ -92,6 +108,7 @@ export function parseRateTable(document: unknown): RateTable { // input rather than as free. cacheReadCostPerToken: finiteNumber(entry.cache_read_input_token_cost) ?? input, cacheCreationCostPerToken: finiteNumber(entry.cache_creation_input_token_cost) ?? input, + fastMultiplier: fastMultiplier(entry), }); } @@ -119,7 +136,8 @@ function sameRate(a: ModelRate, b: ModelRate): boolean { a.inputCostPerToken === b.inputCostPerToken && a.outputCostPerToken === b.outputCostPerToken && a.cacheReadCostPerToken === b.cacheReadCostPerToken && - a.cacheCreationCostPerToken === b.cacheCreationCostPerToken + a.cacheCreationCostPerToken === b.cacheCreationCostPerToken && + a.fastMultiplier === b.fastMultiplier ); } @@ -165,24 +183,26 @@ export function lookupRate(table: RateTable, model: string): ModelRate | null { return table.get(key) ?? null; } +/** The parts of a transcript record that decide its price. */ +export type PricedRecord = Pick; + export interface PricedUsage { readonly costUsd: number; readonly costSource: UsageCostSource; } /** - * Prices a bucket's tokens. + * Prices one record's tokens. * * `reasoningTokens` is intentionally not charged separately: it is already * counted inside `outputTokens`. */ export function priceUsage( table: RateTable, - model: string, - totals: UsageTokenTotals, - reportedCostUsd: number | null, + record: PricedRecord, overrides?: RateTable, ): PricedUsage { + const { model, totals, reportedCostUsd } = record; const override = overrides?.get(model.trim()); if (override === undefined && reportedCostUsd !== null && Number.isFinite(reportedCostUsd)) { return { costUsd: reportedCostUsd, costSource: "providerReported" }; @@ -191,13 +211,16 @@ export function priceUsage( const rate = override ?? lookupRate(table, model); if (rate === null) return { costUsd: 0, costSource: "unpriced" }; - const costUsd = + const standardCostUsd = totals.uncachedInputTokens * rate.inputCostPerToken + totals.cachedInputTokens * rate.cacheReadCostPerToken + totals.cacheCreationTokens * rate.cacheCreationCostPerToken + totals.outputTokens * rate.outputCostPerToken; - return { costUsd, costSource: "modelPriced" }; + return { + costUsd: standardCostUsd * (record.fast ? rate.fastMultiplier : 1), + costSource: "modelPriced", + }; } /** @@ -206,11 +229,14 @@ export function priceUsage( */ export function cacheSavingsUsd( table: RateTable, - model: string, - totals: UsageTokenTotals, + record: PricedRecord, overrides?: RateTable, ): number { - const rate = overrides?.get(model.trim()) ?? lookupRate(table, model); + const rate = overrides?.get(record.model.trim()) ?? lookupRate(table, record.model); if (rate === null) return 0; - return totals.cachedInputTokens * (rate.inputCostPerToken - rate.cacheReadCostPerToken); + return ( + record.totals.cachedInputTokens * + (rate.inputCostPerToken - rate.cacheReadCostPerToken) * + (record.fast ? rate.fastMultiplier : 1) + ); } diff --git a/apps/server/src/usage/usageScanCache.test.ts b/apps/server/src/usage/usageScanCache.test.ts index cc1bdbcc1626..6455b1eb7770 100644 --- a/apps/server/src/usage/usageScanCache.test.ts +++ b/apps/server/src/usage/usageScanCache.test.ts @@ -24,6 +24,7 @@ function record(overrides: Partial = {}): UsageRecord { reasoningTokens: 0, }, reportedCostUsd: null, + fast: false, dedupeKey: "msg_1:", ...overrides, }; @@ -57,7 +58,11 @@ function cacheWith(entries: readonly [string, number, readonly UsageRecord[]][]) describe("scan cache round trip", () => { it("restores records unchanged", () => { const original = cacheWith([ - ["/a.jsonl", 100, [record(), record({ dedupeKey: "msg_2:", model: "claude-opus-5" })]], + [ + "/a.jsonl", + 100, + [record(), record({ dedupeKey: "msg_2:", model: "claude-opus-5-5", fast: true })], + ], ["/b.jsonl", 200, [record({ sessionId: "session-b", reportedCostUsd: 1.5 })]], ]); original.set("/grok.jsonl", { @@ -123,9 +128,20 @@ describe("scan cache round trip", () => { expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); }); + it("drops an entry whose fast flag is not 0 or 1", () => { + const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record({ fast: true })]]])); + const row = encoded.files["/a.jsonl"]!.r[0]!; + const poisoned = { + ...encoded, + files: { "/a.jsonl": { ...encoded.files["/a.jsonl"]!, r: [[...row.slice(0, 10), true]] } }, + }; + + expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); + }); + it("rejects a document from the previous cache version", () => { const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record()]]])); - const previous = { ...encoded, version: 2 }; + const previous = { ...encoded, version: 3 }; expect(decodeScanCache(JSON.parse(JSON.stringify(previous))).size).toBe(0); }); diff --git a/apps/server/src/usage/usageScanCache.ts b/apps/server/src/usage/usageScanCache.ts index 71ef25051eb6..79cca5cff8e2 100644 --- a/apps/server/src/usage/usageScanCache.ts +++ b/apps/server/src/usage/usageScanCache.ts @@ -23,7 +23,8 @@ import type { CodexScanState, UsageRecord } from "./usageTranscripts.ts"; // entries would keep serving double-counted records forever. // v3: entries carry the parse position and reducer state so a grown file // re-parses only its appended bytes instead of starting over. -const USAGE_SCAN_CACHE_VERSION = 3 as const; +// v4: records carry Claude fast mode, which v3 rows never captured. +const USAGE_SCAN_CACHE_VERSION = 4 as const; export interface CachedFile { readonly size: number; @@ -58,6 +59,7 @@ type SerializedRecord = readonly [ reasoningTokens: number, dedupeKey: string | null, reportedCostUsd: number | null, + fast: 0 | 1, ]; interface SerializedFile { @@ -109,6 +111,7 @@ export function encodeScanCache(cache: ScanCache): SerializedCache { record.totals.reasoningTokens, record.dedupeKey, record.reportedCostUsd, + record.fast ? 1 : 0, ]; const files: Record = {}; @@ -165,7 +168,7 @@ export function decodeScanCache(document: unknown): ScanCache { ): UsageRecord[] | null => { const records: UsageRecord[] = []; for (const row of rows) { - if (!isRecordArray(row) || row.length < 10) return null; + if (!isRecordArray(row) || row.length < 11) return null; const [ timestampMs, modelIndex, @@ -177,6 +180,7 @@ export function decodeScanCache(document: unknown): ScanCache { reasoning, dedupeKey, reportedCostUsd, + fast, ] = row as SerializedRecord; const model = typeof modelIndex === "number" ? models[modelIndex] : undefined; @@ -188,7 +192,8 @@ export function decodeScanCache(document: unknown): ScanCache { !Number.isFinite(cached) || !Number.isFinite(cacheCreation) || !Number.isFinite(output) || - !Number.isFinite(reasoning) + !Number.isFinite(reasoning) || + (fast !== 0 && fast !== 1) ) { return null; } @@ -206,6 +211,7 @@ export function decodeScanCache(document: unknown): ScanCache { reasoningTokens: reasoning, }, reportedCostUsd: typeof reportedCostUsd === "number" ? reportedCostUsd : null, + fast: fast === 1, dedupeKey: typeof dedupeKey === "string" ? dedupeKey : null, }); } diff --git a/apps/server/src/usage/usageTranscripts.test.ts b/apps/server/src/usage/usageTranscripts.test.ts index b09db613ed85..ace3b7d18cf7 100644 --- a/apps/server/src/usage/usageTranscripts.test.ts +++ b/apps/server/src/usage/usageTranscripts.test.ts @@ -15,6 +15,7 @@ function claudeLine(overrides: { contentType: string; model?: string; outputTokens?: number; + speed?: string; }): string { return JSON.stringify({ type: "assistant", @@ -31,6 +32,7 @@ function claudeLine(overrides: { cache_creation_input_tokens: 66818, cache_read_input_tokens: 1000, output_tokens: overrides.outputTokens ?? 286, + ...(overrides.speed === undefined ? {} : { speed: overrides.speed }), }, }, }); @@ -51,6 +53,15 @@ describe("parseClaudeLine", () => { reasoningTokens: 0, }); expect(record?.dedupeKey).toBe("msg_1:"); + expect(record?.fast).toBe(false); + }); + + it("marks fast-mode requests", () => { + const line = (speed: string) => + parseClaudeLine(claudeLine({ messageId: "msg_1", contentType: "text", speed })); + + expect(line("fast")?.fast).toBe(true); + expect(line("standard")?.fast).toBe(false); }); it("gives every content block of one message the same dedupe key", () => { diff --git a/apps/server/src/usage/usageTranscripts.ts b/apps/server/src/usage/usageTranscripts.ts index 5d909379eb10..5da13168a1af 100644 --- a/apps/server/src/usage/usageTranscripts.ts +++ b/apps/server/src/usage/usageTranscripts.ts @@ -15,6 +15,11 @@ export interface UsageRecord { readonly sessionId: string; readonly totals: UsageTokenTotals; readonly reportedCostUsd: number | null; + /** + * Whether the request ran in fast mode, which bills at a model-specific + * multiple of the standard rate. Only Claude Code records this. + */ + readonly fast: boolean; /** * Key for cross-file de-duplication, or `null` when the record is inherently * unique and needs no dedup. @@ -145,6 +150,7 @@ export function parseClaudeLine(line: string): UsageRecord | null { reasoningTokens: 0, }, reportedCostUsd: typeof cost === "number" && Number.isFinite(cost) ? cost : null, + fast: usageRecord["speed"] === "fast", dedupeKey, }; } @@ -304,6 +310,7 @@ export function parseCodexLine(line: string, state: CodexScanState): UsageRecord totals, // Codex does not report cost in the rollout. reportedCostUsd: null, + fast: false, // Events surviving the fork-copy suppression above are unique to this // rollout, so they need no global dedup. dedupeKey: null, @@ -433,6 +440,7 @@ export function parseGrokLine(line: string): readonly UsageRecord[] { sessionId, totals: grokTotalsToUsage(topLevel), reportedCostUsd: grokCostTicksToUsd(topLevel.costUsdTicks), + fast: false, // No prompt id means we cannot tell two same-second updates apart. dedupeKey: promptId === null ? null : `${sessionId}:${promptId}:grok`, }, @@ -479,6 +487,7 @@ export function parseGrokLine(line: string): readonly UsageRecord[] { sessionId, totals, reportedCostUsd, + fast: false, dedupeKey: promptId === null ? null : `${sessionId}:${promptId}:${entry.model}`, }); }