diff --git a/apps/mobile/src/features/usage/UsageRouteScreen.tsx b/apps/mobile/src/features/usage/UsageRouteScreen.tsx index 18cfe0ace9..cf5ddc0fe4 100644 --- a/apps/mobile/src/features/usage/UsageRouteScreen.tsx +++ b/apps/mobile/src/features/usage/UsageRouteScreen.tsx @@ -290,8 +290,8 @@ export function UsageRouteScreen() { ) : null} {merged.approximateEnvironments.length > 0 ? ( - Totals may count shared usage more than once because these environments run older - servers or have directories whose filesystem identity could not be read:{" "} + Totals for these environments are approximate because they run older servers, read + overlapping history folders, or have folders whose identity could not be read:{" "} {selectedEnvironments .filter((environment) => merged.approximateEnvironments.includes(environment.environmentId), diff --git a/apps/mobile/src/state/usage.test.tsx b/apps/mobile/src/state/usage.test.tsx new file mode 100644 index 0000000000..b29d9c4eb9 --- /dev/null +++ b/apps/mobile/src/state/usage.test.tsx @@ -0,0 +1,112 @@ +import { EnvironmentId, UsageDay, USAGE_CONTRACT_VERSION } from "@t3tools/contracts"; +import { act, useLayoutEffect } from "react"; +import { create, type ReactTestRenderer } from "react-test-renderer"; +import { describe, expect, it, vi } from "vite-plus/test"; + +import { useUsage, type EnvironmentUsageStatus, type UsageView } from "./usage"; + +const state = vi.hoisted(() => ({ environments: [] as EnvironmentUsageStatus[] })); +vi.mock("@effect/atom-react", () => ({ useAtomValue: () => state.environments })); +vi.mock("./atom-registry", () => ({ appAtomRegistry: {} })); +vi.mock("./presentation", () => ({ environmentPresentations: {} })); +vi.mock("./server", () => ({ serverEnvironment: {} })); + +const input = { + sinceDay: UsageDay.make("2026-09-04"), + untilDay: UsageDay.make("2026-09-05"), + timeZone: "UTC", +}; +function reading(id: string, partial: boolean): EnvironmentUsageStatus { + const cell = (day: UsageDay, costUsd: number) => ({ + day, + provider: "codex" as const, + model: "gpt-6", + totals: { + uncachedInputTokens: 10, + cachedInputTokens: 0, + cacheCreationTokens: 0, + outputTokens: 5, + reasoningTokens: 0, + }, + costUsd, + cacheSavingsUsd: 0, + costSource: "modelPriced" as const, + records: 1, + unpricedRecords: 0, + sessions: 1, + }); + const buckets = partial + ? [cell(input.sinceDay, 4), cell(input.untilDay, 3)] + : [cell(input.sinceDay, 10)]; + return { + environmentId: EnvironmentId.make(id), + label: id, + isPending: false, + isConnected: true, + error: null, + summary: { + ...input, + contractVersion: USAGE_CONTRACT_VERSION, + readAt: partial ? "2026-09-05T12:00:00Z" : "2026-09-04T12:00:00Z", + buckets, + sources: [ + { + fingerprint: { + hostId: "host", + provider: "codex", + resolvedHomePath: "/sessions", + volumeId: "physical-volume", + }, + status: partial ? "partial" : "ok", + scannedFiles: 1, + skippedFiles: partial ? 1 : 0, + malformedRecords: 0, + distinctSessions: partial ? 2 : 1, + message: partial ? "One source file could not be read." : null, + buckets, + }, + ], + pricing: { status: "fresh", source: "test", fetchedAt: null, knownModels: 1 }, + scanDurationMs: 1, + }, + }; +} + +Object.assign(globalThis, { IS_REACT_ACT_ENVIRONMENT: true }); +describe("mobile usage scan selection", () => { + it("retains complete cells and new partial cells, then recomputes ownership when the selection changes", async () => { + state.environments = [reading("complete", false), reading("partial", true)]; + let latest: UsageView | undefined; + let renderer: ReactTestRenderer | undefined; + function Probe({ selected }: { selected: ReadonlySet | null }) { + const usage = useUsage(input, selected); + useLayoutEffect(() => { + latest = usage; + }, [usage]); + return null; + } + try { + await act(async () => { + renderer = create(); + }); + expect(latest?.merged.costUsd).toBe(13); + expect(latest?.merged.sessions).toBe(2); + expect(latest?.merged.duplicateSources).toEqual(["partial: /sessions"]); + await act(async () => + renderer?.update(), + ); + expect(latest?.merged.costUsd).toBe(7); + expect(latest?.merged.duplicateSources).toEqual([]); + await act(async () => + renderer?.update(), + ); + expect(latest?.merged.costUsd).toBe(10); + state.environments = state.environments.toReversed(); + await act(async () => renderer?.update()); + expect(latest?.merged.costUsd).toBe(13); + expect(latest?.merged.sessions).toBe(2); + } finally { + await act(async () => renderer?.unmount()); + } + }); +}); diff --git a/apps/web/src/components/usage/UsagePage.tsx b/apps/web/src/components/usage/UsagePage.tsx index a8e26e1b8b..2847166c79 100644 --- a/apps/web/src/components/usage/UsagePage.tsx +++ b/apps/web/src/components/usage/UsagePage.tsx @@ -797,8 +797,8 @@ function UsageCoverageNotice({ ) : null} {approximate.length > 0 ? ( - Totals may include shared usage more than once because these environments run older - servers or have directories whose filesystem identity could not be read:{" "} + Totals for these environments are approximate because they run older servers, read + overlapping history folders, or have folders whose identity could not be read:{" "} {approximate.map((entry) => entry.label).join(", ")}. ) : null} diff --git a/apps/web/src/state/usage.test.tsx b/apps/web/src/state/usage.test.tsx index cb0f17081a..20eb78cdfe 100644 --- a/apps/web/src/state/usage.test.tsx +++ b/apps/web/src/state/usage.test.tsx @@ -117,6 +117,59 @@ afterEach(async () => { }); describe("usage environment selection", () => { + it("keeps complete and supplemental partial cells when selection changes", async () => { + const complete = environment("complete", 10, "shared"); + const partial = environment("partial", 4, "shared"); + if (complete.summary === null || partial.summary === null) + throw new Error("Missing fixture summary"); + const completeBuckets = complete.summary.buckets.map((cell) => ({ + ...cell, + model: "shared-model", + })); + const partialBuckets = [ + ...partial.summary.buckets.map((cell) => ({ ...cell, model: "shared-model" })), + { ...partial.summary.buckets[0]!, model: "new-model", costUsd: 3 }, + ]; + testState.environments = [ + { + ...complete, + summary: { + ...complete.summary, + buckets: completeBuckets, + sources: complete.summary.sources.map((source) => ({ + ...source, + buckets: completeBuckets, + })), + }, + }, + { + ...partial, + summary: { + ...partial.summary, + readAt: "2026-09-04T13:00:00Z", + buckets: partialBuckets, + sources: partial.summary.sources.map((source) => ({ + ...source, + status: "partial" as const, + distinctSessions: 2, + buckets: partialBuckets, + })), + }, + }, + ]; + await act(() => renderer?.update()); + expect(latest.merged.costUsd).toBe(13); + expect(latest.merged.sessions).toBe(2); + await select("partial"); + expect(latest.merged.costUsd).toBe(7); + expect(latest.merged.duplicateSources).toEqual([]); + await select("complete"); + expect(latest.merged.costUsd).toBe(10); + testState.environments = testState.environments.toReversed(); + await act(() => renderer?.update()); + expect(latest.merged.costUsd).toBe(13); + }); + it("starts with all environments and adds results as they arrive", async () => { expect(latest.merged.costUsd).toBe(30); expect(latest.isPending).toBe(false); diff --git a/packages/shared/src/usageMerge.test.ts b/packages/shared/src/usageMerge.test.ts index 173d6cbc75..c46435956e 100644 --- a/packages/shared/src/usageMerge.test.ts +++ b/packages/shared/src/usageMerge.test.ts @@ -183,7 +183,8 @@ describe("mergeUsage", () => { ); expect(merged.costUsd).toBe(22); - expect(merged.approximateEnvironments).toEqual([]); + // env-a's unique home is kept while its shared home goes to env-b. + expect(merged.approximateEnvironments).toEqual(["env-a"]); }); it("counts overlapping Claude homes once while retaining both unique homes", () => { @@ -223,7 +224,8 @@ describe("mergeUsage", () => { expect(merged.records).toBe(3); expect(merged.sessions).toBe(3); expect(merged.duplicateSources).toHaveLength(1); - expect(merged.approximateEnvironments).toEqual([]); + // env-b's unique home is kept while its shared home goes to env-a. + expect(merged.approximateEnvironments).toEqual(["env-b"]); }); it("does not collapse identical paths on different hosts or filesystems", () => { @@ -289,7 +291,14 @@ describe("mergeUsage", () => { [ environment( "env-a", - summary([bucket({ costUsd: 4 })], [{ ...shared, buckets: [bucket({ costUsd: 4 })] }]), + // A unique home keeps env-b's homes from containing env-a's. + summary( + [bucket({ costUsd: 6 })], + [ + { ...shared, buckets: [bucket({ costUsd: 4 })] }, + { ...shared, homePath: "/unique-a", buckets: [bucket({ costUsd: 2 })] }, + ], + ), ), environment( "env-b", @@ -303,7 +312,7 @@ describe("mergeUsage", () => { USAGE_CONTRACT_VERSION, ); - expect(merged.costUsd).toBe(19); + expect(merged.costUsd).toBe(21); expect(merged.approximateEnvironments).toEqual(["env-b"]); expect(merged.contractMismatches).toEqual([]); }); @@ -314,7 +323,14 @@ describe("mergeUsage", () => { [ environment( "env-a", - summary([bucket({ costUsd: 4 })], [{ ...shared, buckets: [bucket({ costUsd: 4 })] }]), + // A unique home keeps env-b's homes from containing env-a's. + summary( + [bucket({ costUsd: 6 })], + [ + { ...shared, buckets: [bucket({ costUsd: 4 })] }, + { ...shared, homePath: "/unique-a", buckets: [bucket({ costUsd: 2 })] }, + ], + ), ), environment( "env-b", @@ -330,7 +346,7 @@ describe("mergeUsage", () => { USAGE_CONTRACT_VERSION, ); - expect(merged.costUsd).toBe(19); + expect(merged.costUsd).toBe(21); expect(merged.providers.map((provider) => provider.provider)).toEqual(["claude"]); expect(merged.approximateEnvironments).toEqual(["env-b"]); }); @@ -715,3 +731,435 @@ describe("mergeUsage", () => { expect(merged.contributingEnvironments).toEqual(["env-v4", "env-v5", "env-v6"]); }); }); + +describe("complete and partial usage scans", () => { + const shared = { provider: "claude" as const, hostId: "mac", homePath: "/shared" }; + function scan( + id: string, + buckets: readonly UsageBucket[], + status: "ok" | "partial" | "failed", + readAt: string, + attributed = true, + distinctSessions = 1, + ) { + const reading = summary(buckets, [ + { ...shared, distinctSessions, ...(attributed ? { buckets } : {}) }, + ]); + return environment(id, { + ...reading, + readAt, + sources: reading.sources.map((source) => ({ ...source, status })), + }); + } + const oldTime = "2026-08-07T00:00:00.000Z"; + const newTime = "2026-08-08T00:00:00.000Z"; + function orders(...readings: EnvironmentUsage[]) { + return [readings, readings.toReversed()]; + } + it.each([false, true])( + "retains complete cells over a newer partial or failed scan (attributed=%s)", + (attributed) => { + const complete = scan("old", [bucket()], "ok", oldTime, attributed); + for (const status of ["partial", "failed"] as const) { + const incomplete = scan( + "new", + [bucket({ costUsd: 4, records: 6 })], + status, + newTime, + attributed, + 2, + ); + for (const readings of orders(complete, incomplete)) { + const merged = mergeUsage(readings, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(10); + expect(merged.records).toBe(5); + expect(merged.totalTokens).toBe(1160); + expect(merged.sessions).toBe(1); + expect(merged.contributingEnvironments).toEqual(["old"]); + expect(merged.duplicateSources).toEqual(["new: /shared"]); + } + } + }, + ); + it.each([false, true])( + "adds only new day, hour and model cells from newer partial scans (attributed=%s)", + (attributed) => { + const original = bucket({ hourStart: "2026-08-07T00:00:00.000Z" }); + const day = bucket({ + day: "2026-08-08" as UsageDay, + hourStart: "2026-08-08T00:00:00.000Z", + costUsd: 3, + records: 1, + }); + const hour = bucket({ hourStart: "2026-08-07T01:00:00.000Z", costUsd: 2, records: 1 }); + const model = bucket({ + hourStart: original.hourStart, + model: "claude-other", + costUsd: 1, + records: 1, + }); + const complete = scan("old", [original], "ok", oldTime, attributed); + const partial = scan( + "new", + [{ ...original, costUsd: 4 }, day, hour, model], + "partial", + newTime, + attributed, + 3, + ); + for (const readings of orders(complete, partial)) { + const merged = mergeUsage(readings, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(16); + expect(merged.records).toBe(8); + expect(merged.sessions).toBe(3); + expect(merged.providers[0]?.sessions).toBe(3); + expect(merged.daily.map(({ day, costUsd }) => [day, costUsd])).toEqual([ + ["2026-08-07", 13], + ["2026-08-08", 3], + ]); + expect(merged.hourly.map(({ hourStart, costUsd }) => [hourStart, costUsd])).toEqual([ + [original.hourStart, 11], + [hour.hourStart, 2], + [day.hourStart, 3], + ]); + expect(merged.contributingEnvironments).toEqual( + readings.map(({ environmentId }) => environmentId), + ); + expect(merged.duplicateSources).toEqual(["new: /shared"]); + } + }, + ); + it("retains complete priority without supplementing when its scan time is unknown", () => { + const complete = scan("complete", [bucket()], "ok", "invalid"); + const partial = scan( + "partial", + [bucket({ day: "2026-08-08" as UsageDay, costUsd: 3 })], + "partial", + newTime, + ); + for (const readings of orders(complete, partial)) { + expect(mergeUsage(readings, USAGE_CONTRACT_VERSION).costUsd).toBe(10); + } + }); + it("takes each supplemental cell from the newest partial scan once", () => { + const old = scan("old", [bucket()], "ok", oldTime); + const cell = bucket({ day: "2026-08-08" as UsageDay, costUsd: 3, records: 1 }); + const middle = scan("middle", [cell], "partial", "2026-08-08T00:00:00.000Z", true, 2); + const newest = scan( + "newest", + [{ ...cell, costUsd: 5 }], + "partial", + "2026-08-08T01:00:00.000Z", + true, + 3, + ); + for (const readings of orders(old, middle, newest)) { + const merged = mergeUsage(readings, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(15); + expect(merged.sessions).toBe(3); + expect(merged.contributingEnvironments).toEqual( + readings.filter((entry) => entry !== middle).map(({ environmentId }) => environmentId), + ); + } + }); + it.each([ + { status: "partial" as const, time: oldTime }, + { status: "partial" as const, time: "2026-08-06T00:00:00.000Z" }, + { status: "partial" as const, time: "invalid" }, + { status: "failed" as const, time: newTime }, + ])("does not supplement unproven new usage from $status at $time", ({ status, time }) => { + const complete = scan("complete", [bucket()], "ok", oldTime); + const other = scan( + "other", + [bucket({ day: "2026-08-08" as UsageDay, costUsd: 3 })], + status, + time, + true, + 2, + ); + for (const readings of orders(complete, other)) { + expect(mergeUsage(readings, USAGE_CONTRACT_VERSION).costUsd).toBe(10); + expect(mergeUsage(readings, USAGE_CONTRACT_VERSION).sessions).toBe(1); + } + }); + it("retains a partial scan when no complete scan is available and restores complete priority later", () => { + const partial = scan("partial", [bucket({ costUsd: 4 })], "partial", oldTime); + const failed = scan("failed", [], "failed", newTime); + const complete = scan("complete", [bucket()], "ok", newTime); + expect(mergeUsage([partial], USAGE_CONTRACT_VERSION).costUsd).toBe(4); + expect(mergeUsage([failed, partial], USAGE_CONTRACT_VERSION).costUsd).toBe(4); + expect(mergeUsage([failed, partial, complete], USAGE_CONTRACT_VERSION).costUsd).toBe(10); + expect(mergeUsage([failed], USAGE_CONTRACT_VERSION).costUsd).toBe(0); + }); + it("supplements the shared home while retaining each host's unique attributed home", () => { + const original = bucket(); + const extra = bucket({ day: "2026-08-08" as UsageDay, costUsd: 3, records: 1 }); + const first = summary( + [bucket({ costUsd: 12 })], + [ + { ...shared, buckets: [original] }, + { ...shared, homePath: "/unique-a", buckets: [bucket({ costUsd: 2 })] }, + ], + ); + const second = summary( + [bucket({ costUsd: 9 })], + [ + { ...shared, distinctSessions: 2, buckets: [{ ...original, costUsd: 1 }, extra] }, + { ...shared, homePath: "/unique-b", buckets: [bucket({ costUsd: 5 })] }, + ], + ); + const complete = environment("a", { ...first, readAt: oldTime }); + const partial = environment("b", { + ...second, + readAt: newTime, + sources: second.sources.map((source) => ({ ...source, status: "partial" as const })), + }); + for (const readings of orders(complete, partial)) { + const merged = mergeUsage(readings, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(20); + expect(merged.sessions).toBe(4); + // b's unique home is kept while its shared home goes to a. + expect(merged.approximateEnvironments).toEqual(["b"]); + } + }); + it("does not invent source attribution for legacy multi-home totals", () => { + const complete = scan("old", [bucket()], "ok", oldTime); + const reading = summary( + [bucket({ day: "2026-08-08" as UsageDay, costUsd: 3 })], + [shared, { ...shared, homePath: "/unique-b" }], + USAGE_MERGE_COMPATIBLE_SINCE, + ); + const legacy = environment("legacy", { + ...reading, + readAt: newTime, + sources: reading.sources.map((source) => ({ ...source, status: "partial" as const })), + }); + const merged = mergeUsage([complete, legacy], USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(13); + expect(merged.approximateEnvironments).toEqual(["legacy"]); + expect(merged.sessions).toBe(2); + }); + it("keeps one owner for a provider's homes when a newer scan attributes a record elsewhere", () => { + // Both servers dedupe one record found in D1 and D2. The complete scan + // attributes it to D1; the newer scan could not read D1 and attributes it + // to D2. Splitting D1 and D2 between them would count it twice. + const record = bucket(); + const later = bucket({ day: "2026-08-08" as UsageDay, costUsd: 3, records: 1 }); + const complete = summary( + [record], + [ + { ...shared, homePath: "/d1", buckets: [record] }, + { ...shared, homePath: "/d2", buckets: [] }, + ], + ); + const newer = summary( + [record, later], + [ + { ...shared, homePath: "/d1", buckets: [] }, + { ...shared, homePath: "/d2", distinctSessions: 2, buckets: [record, later] }, + ], + ); + const readings = orders( + environment("complete", { ...complete, readAt: oldTime }), + environment("newer", { + ...newer, + readAt: newTime, + sources: newer.sources.map((source) => + source.fingerprint.resolvedHomePath === "/d1" + ? { ...source, status: "partial" as const } + : source, + ), + }), + ); + for (const environments of readings) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(13); + expect(merged.records).toBe(6); + expect(merged.daily.map(({ day, costUsd }) => [day, costUsd])).toEqual([ + ["2026-08-07", 10], + ["2026-08-08", 3], + ]); + expect(merged.approximateEnvironments).toEqual([]); + expect(merged.duplicateSources).toEqual(["newer: /d1", "newer: /d2"]); + } + }); + it("keeps a legacy complete multi-home total exact against a newer partial scan", () => { + const legacy = summary( + [bucket({ costUsd: 15 })], + [ + { ...shared, homePath: "/d1" }, + { ...shared, homePath: "/d2" }, + ], + USAGE_MERGE_COMPATIBLE_SINCE, + ); + const current = summary( + [bucket({ costUsd: 15 })], + [ + { ...shared, homePath: "/d1", buckets: [bucket({ costUsd: 10 })] }, + { ...shared, homePath: "/d2", buckets: [bucket({ costUsd: 5 })] }, + ], + ); + for (const environments of orders( + environment("legacy", { ...legacy, readAt: oldTime }), + environment("current", { + ...current, + readAt: newTime, + sources: current.sources.map((source) => + source.fingerprint.resolvedHomePath === "/d1" + ? { ...source, status: "partial" as const } + : source, + ), + }), + )) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(15); + expect(merged.approximateEnvironments).toEqual([]); + expect(merged.contributingEnvironments).toEqual(["legacy"]); + } + }); + it("does not let an unshared failed home demote a scan's shared homes", () => { + const complete = summary( + [bucket({ costUsd: 15 })], + [ + { ...shared, homePath: "/d1", buckets: [bucket({ costUsd: 10 })] }, + { ...shared, homePath: "/d2", buckets: [bucket({ costUsd: 5 })] }, + { ...shared, homePath: "/d3", buckets: [] }, + ], + ); + const partial = summary( + [bucket({ costUsd: 4 })], + [{ ...shared, homePath: "/d1", buckets: [bucket({ costUsd: 4 })] }], + ); + for (const [completeAt, partialAt] of [ + [newTime, oldTime], + [oldTime, newTime], + ] as const) { + for (const environments of orders( + environment("a", { + ...complete, + readAt: completeAt, + sources: complete.sources.map((source) => + source.fingerprint.resolvedHomePath === "/d3" + ? { ...source, status: "failed" as const } + : source, + ), + }), + environment("b", { + ...partial, + readAt: partialAt, + sources: partial.sources.map((source) => ({ ...source, status: "partial" as const })), + }), + )) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(15); + expect(merged.contributingEnvironments).toEqual(["a"]); + expect(merged.approximateEnvironments).toEqual([]); + } + } + }); + it("flags scans whose homes are split between owners", () => { + const record = bucket(); + const d1 = { ...shared, homePath: "/d1" }; + const d2 = { ...shared, homePath: "/d2" }; + const d3 = { ...shared, homePath: "/d3" }; + // Record R exists in D1 and D3; each server credits it to a different home. + const a = environment("a", { + ...summary( + [record], + [ + { ...d1, buckets: [record] }, + { ...d2, buckets: [] }, + ], + ), + readAt: "2026-08-09T00:00:00.000Z", + }); + const b = environment("b", { + ...summary( + [record], + [ + { ...d2, buckets: [] }, + { ...d3, buckets: [record] }, + ], + ), + readAt: newTime, + }); + const c = environment("c", { + ...summary( + [record], + [ + { ...d1, buckets: [record] }, + { ...d3, buckets: [] }, + ], + ), + readAt: oldTime, + }); + for (const environments of orders(a, b, c)) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(20); + expect([...merged.approximateEnvironments].sort()).toEqual(["b", "c"]); + } + for (const environments of orders(a, b)) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(20); + expect(merged.approximateEnvironments).toEqual(["b"]); + } + }); + it("lets a scan of a superset of homes own them all regardless of scan order", () => { + const desktop = summary([bucket()], [{ ...shared, homePath: "/.claude", buckets: [bucket()] }]); + const dev = summary( + [bucket({ costUsd: 13 })], + [ + { ...shared, homePath: "/.claude", buckets: [bucket()] }, + { ...shared, homePath: "/.claude-alt", buckets: [bucket({ costUsd: 3 })] }, + ], + ); + for (const [desktopAt, devAt] of [ + [newTime, oldTime], + [oldTime, newTime], + ] as const) { + for (const environments of orders( + environment("desktop", { ...desktop, readAt: desktopAt }), + environment("dev", { ...dev, readAt: devAt }), + )) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(13); + expect(merged.contributingEnvironments).toEqual(["dev"]); + expect(merged.duplicateSources).toEqual(["desktop: /.claude"]); + expect(merged.approximateEnvironments).toEqual([]); + } + } + // A legacy provider-wide total over a superset of homes stays exact. + const legacy = summary( + [bucket({ costUsd: 13 })], + [ + { ...shared, homePath: "/.claude" }, + { ...shared, homePath: "/.claude-alt" }, + ], + USAGE_MERGE_COMPATIBLE_SINCE, + ); + for (const environments of orders( + environment("desktop", { ...desktop, readAt: newTime }), + environment("legacy", { ...legacy, readAt: oldTime }), + )) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(13); + expect(merged.approximateEnvironments).toEqual([]); + } + // Neither home set contains the other, so the split stays flagged. + const other = summary( + [bucket({ costUsd: 12 })], + [ + { ...shared, homePath: "/.claude", buckets: [bucket()] }, + { ...shared, homePath: "/.claude-other", buckets: [bucket({ costUsd: 2 })] }, + ], + ); + for (const environments of orders( + environment("other", { ...other, readAt: newTime }), + environment("dev", { ...dev, readAt: oldTime }), + )) { + const merged = mergeUsage(environments, USAGE_CONTRACT_VERSION); + expect(merged.costUsd).toBe(15); + expect(merged.approximateEnvironments).toEqual(["dev"]); + } + }); +}); diff --git a/packages/shared/src/usageMerge.ts b/packages/shared/src/usageMerge.ts index 926f85f204..14e85e0592 100644 --- a/packages/shared/src/usageMerge.ts +++ b/packages/shared/src/usageMerge.ts @@ -11,6 +11,7 @@ import { type EnvironmentId, type UsageBucket, type UsageProviderKind, + type UsageSource, type UsageSourceFingerprint, type UsageSummary, } from "@t3tools/contracts"; @@ -124,32 +125,190 @@ function fingerprintKey(fingerprint: UsageSourceFingerprint, environmentId: Envi ]); } +/** One environment's scan of every non-missing source for one provider. */ +interface ProviderScan { + readonly environment: EnvironmentUsage; + readonly provider: UsageProviderKind; + readonly sources: readonly UsageSource[]; + /** + * The least complete status among sources another scan also reads: 0 ok, + * 1 partial, 2 failed. A home no other scan reads cannot be split between + * owners, so its status does not demote the scan's shared homes. + */ + readonly rank: number; + readonly readAt: number; +} + +const STATUS_RANK = { ok: 0, partial: 1, failed: 2 } as const; + +function hasSourceBuckets(sources: readonly UsageSource[]): boolean { + return sources.every( + (source) => + source.buckets !== undefined && + source.buckets.every((bucket) => bucket.provider === source.fingerprint.provider), + ); +} + +function providerScans(environments: readonly EnvironmentUsage[]): ProviderScan[] { + const readers = new Map(); + for (const environment of environments) { + const keys = new Set( + environment.summary.sources + .filter((source) => source.status !== "missing") + .map((source) => fingerprintKey(source.fingerprint, environment.environmentId)), + ); + for (const key of keys) readers.set(key, (readers.get(key) ?? 0) + 1); + } + const scans: ProviderScan[] = []; + for (const environment of environments) { + const sourcesByProvider = new Map(); + for (const source of environment.summary.sources) { + if (source.status === "missing") continue; + const sources = sourcesByProvider.get(source.fingerprint.provider) ?? []; + sources.push(source); + sourcesByProvider.set(source.fingerprint.provider, sources); + } + for (const [provider, sources] of sourcesByProvider) { + scans.push({ + environment, + provider, + sources, + rank: Math.max( + 0, + ...sources.map((source) => + source.status === "missing" || + (readers.get(fingerprintKey(source.fingerprint, environment.environmentId)) ?? 0) < 2 + ? 0 + : STATUS_RANK[source.status], + ), + ), + readAt: Date.parse(environment.summary.readAt), + }); + } + } + return scans; +} + +function newestFirst(a: ProviderScan, b: ProviderScan): number { + return ( + (b.readAt || 0) - (a.readAt || 0) || + a.environment.environmentId.localeCompare(b.environment.environmentId) + ); +} + +/** + * Cells reported for `sources` of one provider scan, or undefined when the + * scan cannot attribute them. A legacy provider-wide list covers all of the + * provider's sources at once, so it is attributable only as a whole. + */ +function cellsFor( + scan: ProviderScan, + sources: readonly UsageSource[], +): + | readonly { + readonly sources: readonly UsageSource[]; + readonly buckets: readonly UsageBucket[]; + }[] + | undefined { + if (hasSourceBuckets(scan.sources)) { + return sources.map((source) => ({ sources: [source], buckets: source.buckets ?? [] })); + } + if (sources.length !== scan.sources.length) return undefined; + return [ + { + sources, + buckets: scan.environment.summary.buckets.filter( + (bucket) => bucket.provider === scan.provider, + ), + }, + ]; +} + +/** + * Identifies a usage cell. Model ids are compared verbatim, so two scans that + * label one model differently produce distinct cells. + */ +function bucketKey(bucket: UsageBucket) { + return JSON.stringify([bucket.day, bucket.hourStart ?? null, bucket.provider, bucket.model]); +} + +/** + * Orders ranked scans for claiming. Within a rank, a scan whose homes strictly + * contain another's claims first, so the nested scan is a whole duplicate + * rather than a split. Otherwise the ranked order holds. Strict containment + * is acyclic, so some remaining scan is always uncontained. + */ +function claimOrder( + scans: readonly ProviderScan[], + keyOf: (scan: ProviderScan, source: UsageSource) => string, +): ProviderScan[] { + const keys = new Map( + scans.map((scan) => [scan, new Set(scan.sources.map((source) => keyOf(scan, source)))]), + ); + const contains = (outer: ProviderScan, inner: ProviderScan) => { + const outerKeys = keys.get(outer) ?? new Set(); + const innerKeys = keys.get(inner) ?? new Set(); + return ( + outer.rank === inner.rank && + outerKeys.size > innerKeys.size && + [...innerKeys].every((key) => outerKeys.has(key)) + ); + }; + const remaining = [...scans]; + const ordered: ProviderScan[] = []; + while (remaining.length > 0) { + const index = remaining.findIndex( + (scan) => !remaining.some((other) => other !== scan && contains(other, scan)), + ); + ordered.push(...remaining.splice(Math.max(index, 0), 1)); + } + return ordered; +} + /** * Decides which environment owns each physical transcript directory. * * Several environments on one machine (worktree servers, for instance) resolve - * the same provider home and would otherwise double count every token. The - * most recently read summary claims a fingerprint; environment ids break ties - * so ownership stays stable when two summaries have the same read time. + * the same provider home and would otherwise double count every token. + * + * A server attributes a record found in several of its homes to the first one + * it could read, so two scans of the same homes can attribute one record to + * different homes. Ownership is therefore ranked per provider scan rather than + * per home: a scan whose every shared home was read completely claims its + * homes before any scan with a partial or failed shared home. Within a rank the + * most recent scan wins, with environment ids breaking ties. Homes already + * claimed are duplicates; unclaimed ones still go to the next scan. Within a + * rank, a scan whose homes strictly contain another scan's homes claims first. + * + * Scans of identical or nested homes therefore have one owner. Scans of overlapping but + * unequal homes cannot: keeping each scan's unique homes splits its shared + * homes from them, and a record in both may be credited twice. Such scans are + * reported as approximate rather than dropping their unique usage. + * + * A newer scan can then add cells absent from a complete scan that owns all of + * its homes, provided every home of the newer scan is owned by that complete + * scan or by the newer scan itself. */ function claimSources(environments: readonly EnvironmentUsage[]): { readonly ownerByFingerprint: ReadonlyMap; + readonly supplementalBucketsByEnvironment: ReadonlyMap; + readonly sessionsByFingerprint: ReadonlyMap; readonly duplicates: readonly string[]; readonly uncertainEnvironments: readonly EnvironmentId[]; } { const ownerByFingerprint = new Map(); + const ownerScanByFingerprint = new Map(); + const supplementalBucketsByEnvironment = new Map(); + const sessionsByFingerprint = new Map(); const duplicates: string[] = []; const weakMatches = new Map(); - const ordered = [...environments].sort( - (a, b) => - (Date.parse(b.summary.readAt) || 0) - (Date.parse(a.summary.readAt) || 0) || - a.environmentId.localeCompare(b.environmentId), - ); + const scans = providerScans(environments).sort((a, b) => a.rank - b.rank || newestFirst(a, b)); + const keyOf = (scan: ProviderScan, source: UsageSource) => + fingerprintKey(source.fingerprint, scan.environment.environmentId); - for (const environment of ordered) { - for (const source of environment.summary.sources) { - if (source.status === "missing") continue; + for (const scan of claimOrder(scans, keyOf)) { + for (const source of scan.sources) { const weakKey = JSON.stringify([ source.fingerprint.hostId, source.fingerprint.provider, @@ -157,20 +316,88 @@ function claimSources(environments: readonly EnvironmentUsage[]): { ]); const matches = weakMatches.get(weakKey) ?? []; matches.push({ - environmentId: environment.environmentId, + environmentId: scan.environment.environmentId, volumeId: source.fingerprint.volumeId, }); weakMatches.set(weakKey, matches); - const key = fingerprintKey(source.fingerprint, environment.environmentId); + const key = keyOf(scan, source); if (ownerByFingerprint.has(key)) { - duplicates.push(`${environment.label}: ${source.fingerprint.resolvedHomePath}`); + duplicates.push(`${scan.environment.label}: ${source.fingerprint.resolvedHomePath}`); continue; } - ownerByFingerprint.set(key, environment.environmentId); + ownerByFingerprint.set(key, scan.environment.environmentId); + ownerScanByFingerprint.set(key, scan); + sessionsByFingerprint.set(key, source.distinctSessions); } } + // Aggregates cannot reconcile overlapping records. Retain the complete + // cells, and add only cells the complete scan never saw in any of its homes, + // taking each from the newest scan that reports it. + const seenByOwner = new Map>(); + for (const scan of [...scans].sort(newestFirst)) { + const owners = new Set( + scan.sources.map((source) => ownerScanByFingerprint.get(keyOf(scan, source))), + ); + owners.delete(scan); + const [owner] = owners; + if (owner === undefined || owners.size !== 1) continue; + if ( + owner.rank !== STATUS_RANK.ok || + !Number.isFinite(scan.readAt) || + !Number.isFinite(owner.readAt) || + scan.readAt <= owner.readAt || + owner.sources.some((source) => ownerScanByFingerprint.get(keyOf(owner, source)) !== owner) + ) { + continue; + } + const ownerCells = cellsFor(owner, owner.sources); + const cells = cellsFor( + scan, + scan.sources.filter( + (source) => + source.status !== "failed" && ownerScanByFingerprint.get(keyOf(scan, source)) === owner, + ), + ); + if (ownerCells === undefined || cells === undefined) continue; + let seen = seenByOwner.get(owner); + if (seen === undefined) { + seen = new Set(ownerCells.flatMap(({ buckets }) => buckets.map(bucketKey))); + seenByOwner.set(owner, seen); + } + // One scan never reports a record twice, so its cells are compared only + // against earlier scans, not against each other. + const added: UsageBucket[] = []; + for (const { sources, buckets } of cells) { + const fresh = buckets.filter((bucket) => !seen.has(bucketKey(bucket))); + if (fresh.length === 0) continue; + added.push(...fresh); + // Per-source session counts do not carry IDs to union. Keep the largest + // observed count instead of recounting sessions spanning multiple cells. + for (const source of sources) { + const key = keyOf(scan, source); + sessionsByFingerprint.set( + key, + Math.max(sessionsByFingerprint.get(key) ?? 0, source.distinctSessions), + ); + } + } + if (added.length === 0) continue; + for (const bucket of added) seen.add(bucketKey(bucket)); + const environmentId = scan.environment.environmentId; + supplementalBucketsByEnvironment.set(environmentId, [ + ...(supplementalBucketsByEnvironment.get(environmentId) ?? []), + ...added, + ]); + } + const uncertainEnvironments = new Set(); + for (const scan of scans) { + const owners = new Set( + scan.sources.map((source) => ownerScanByFingerprint.get(keyOf(scan, source))), + ); + if (owners.size > 1) uncertainEnvironments.add(scan.environment.environmentId); + } for (const matches of weakMatches.values()) { if ( matches.some((match) => match.volumeId === "") && @@ -180,13 +407,21 @@ function claimSources(environments: readonly EnvironmentUsage[]): { } } - return { ownerByFingerprint, duplicates, uncertainEnvironments: [...uncertainEnvironments] }; + return { + ownerByFingerprint, + supplementalBucketsByEnvironment, + sessionsByFingerprint, + duplicates, + uncertainEnvironments: [...uncertainEnvironments], + }; } /** Sources this environment owns after fingerprint claims, plus their buckets. */ function ownedContribution( environment: EnvironmentUsage, ownerByFingerprint: ReadonlyMap, + supplementalBuckets: readonly UsageBucket[], + sessionsByFingerprint: ReadonlyMap, ): { readonly buckets: readonly UsageBucket[]; readonly sessionsByProvider: ReadonlyMap; @@ -207,7 +442,8 @@ function ownedContribution( // would count a session once per day and model it spans. sessionsByProvider.set( provider, - (sessionsByProvider.get(provider) ?? 0) + source.distinctSessions, + (sessionsByProvider.get(provider) ?? 0) + + (sessionsByFingerprint.get(key) ?? source.distinctSessions), ); } } @@ -218,12 +454,7 @@ function ownedContribution( // Mixed-version peers still have provider-wide buckets. Keep those totals // visible, but explicitly report their attribution as approximate when a // provider spans both owned and duplicated directories. - const granular = sources.every( - (source) => - source.buckets !== undefined && - source.buckets.every((bucket) => bucket.provider === source.fingerprint.provider), - ); - if (granular) { + if (hasSourceBuckets(sources)) { for (const source of sources) { if ( ownerByFingerprint.get(fingerprintKey(source.fingerprint, environment.environmentId)) === @@ -247,7 +478,7 @@ function ownedContribution( } } return { - buckets, + buckets: [...buckets, ...supplementalBuckets], sessionsByProvider, approximate, }; @@ -327,7 +558,13 @@ export function mergeUsage( } } - const { ownerByFingerprint, duplicates, uncertainEnvironments } = claimSources(current); + const { + ownerByFingerprint, + supplementalBucketsByEnvironment, + sessionsByFingerprint, + duplicates, + uncertainEnvironments, + } = claimSources(current); let costUsd = 0; let uncachedInputTokens = 0; @@ -380,6 +617,8 @@ export function mergeUsage( const { buckets, sessionsByProvider, approximate } = ownedContribution( environment, ownerByFingerprint, + supplementalBucketsByEnvironment.get(environment.environmentId) ?? [], + sessionsByFingerprint, ); if (approximate) approximateEnvironments.add(environment.environmentId); if (buckets.length > 0) contributingEnvironments.push(environment.environmentId);