diff --git a/apps/mobile/src/features/usage/UsageRouteScreen.test.tsx b/apps/mobile/src/features/usage/UsageRouteScreen.test.tsx index bf951c620..cceafaaec 100644 --- a/apps/mobile/src/features/usage/UsageRouteScreen.test.tsx +++ b/apps/mobile/src/features/usage/UsageRouteScreen.test.tsx @@ -73,6 +73,16 @@ vi.mock("./UsageLimitsPooled", () => ({ UsageLimitsSection: () => null })); vi.mock("./usageProviders", () => ({ PROVIDER_LABEL: { opencode: "OpenCode", codex: "Codex" }, useProviderColors: () => ({ opencode: "#000", codex: "#fff" }), + useUsageMixColors: () => ({ + input: "#111", + cacheRead: "#222", + cacheWrite: "#333", + output: "#444", + other: "#555", + standard: "#666", + fast: "#777", + ultrafast: "#888", + }), })); import { UsageRouteScreen } from "./UsageRouteScreen"; diff --git a/apps/mobile/src/features/usage/UsageRouteScreen.tsx b/apps/mobile/src/features/usage/UsageRouteScreen.tsx index bf03827c0..d532071bf 100644 --- a/apps/mobile/src/features/usage/UsageRouteScreen.tsx +++ b/apps/mobile/src/features/usage/UsageRouteScreen.tsx @@ -37,7 +37,15 @@ import { UsageLimitsSection } from "./UsageLimitsPooled"; import { ControlPillMenu } from "../../components/ControlPill"; import { SymbolView } from "../../components/AppSymbol"; import type { UsageChartMetric } from "./usageChartData"; -import { PROVIDER_LABEL, useProviderColors } from "./usageProviders"; +import { + costTypeSegments, + hasFasterSpeedCost, + SPEED_COST_FOOTNOTE, + speedCostSegments, + visibleCostSegments, + type CostMixSegment, +} from "./usageCostMix"; +import { PROVIDER_LABEL, useProviderColors, useUsageMixColors } from "./usageProviders"; type UsageTab = "usage" | "limits"; const TAB_OPTIONS = [ @@ -336,6 +344,7 @@ export function UsageRouteScreen() { /> + )} @@ -597,6 +606,71 @@ function TotalsSection(props: { readonly merged: MergedUsage; readonly isPast24H ); } +function CostSection(props: { readonly merged: MergedUsage }) { + const { categoryCost, speedCost } = props.merged; + const colors = useUsageMixColors(); + if (props.merged.costUsd <= 0) return null; + + return ( + + + {hasFasterSpeedCost(speedCost) ? ( + + + + ) : null} + + ); +} + +/** One part-to-whole cost bar with its legend. Empty segments are left out. */ +function ShareBar(props: { + readonly label: string; + readonly segments: readonly CostMixSegment[]; + readonly aside?: string; + readonly footnote?: string; +}) { + const visible = visibleCostSegments(props.segments); + if (visible.length === 0) return null; + + return ( + + + {props.label} + {props.aside ? ( + {props.aside} + ) : null} + + + {visible.map((segment) => ( + + ))} + + + {visible.map((segment) => ( + + + {segment.label} + {formatUsd(segment.value)} + + ))} + + {props.footnote ? ( + {props.footnote} + ) : null} + + ); +} + function MetricCell(props: { readonly label: string; readonly value: string; diff --git a/apps/mobile/src/features/usage/usageCostMix.test.ts b/apps/mobile/src/features/usage/usageCostMix.test.ts new file mode 100644 index 000000000..a6b180988 --- /dev/null +++ b/apps/mobile/src/features/usage/usageCostMix.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from "vite-plus/test"; + +import { + costTypeSegments, + hasFasterSpeedCost, + speedCostSegments, + visibleCostSegments, + type CostMixColors, +} from "./usageCostMix"; + +const colors: CostMixColors = { + input: "input", + cacheRead: "cacheRead", + cacheWrite: "cacheWrite", + output: "output", + other: "other", + standard: "standard", + fast: "fast", + ultrafast: "ultrafast", +}; + +describe("usage cost mix", () => { + it("splits cost by type and hides sub-cent unsplit rounding", () => { + const cost = { input: 1, cacheRead: 2, cacheWrite: 0, output: 4, unsplit: 0.004 }; + + expect( + visibleCostSegments(costTypeSegments(cost, colors)).map(({ label, value }) => [label, value]), + ).toEqual([ + ["Input", 1], + ["Cache read", 2], + ["Output", 4], + ]); + expect(visibleCostSegments(costTypeSegments({ ...cost, unsplit: 5 }, colors)).at(-1)).toEqual({ + label: "Other", + value: 5, + color: "other", + }); + }); + + it("shows the speed bar only when some cost ran faster than standard", () => { + const standardOnly = { standard: 9, fast: 0, ultrafast: 0, premium: 0 }; + const mixed = { standard: 9, fast: 6, ultrafast: 2, premium: 4 }; + + expect(hasFasterSpeedCost(standardOnly)).toBe(false); + expect(hasFasterSpeedCost(mixed)).toBe(true); + expect(speedCostSegments(mixed, colors)).toEqual([ + { label: "Standard", value: 9, color: "standard" }, + { label: "Fast", value: 6, color: "fast" }, + { label: "Ultrafast", value: 2, color: "ultrafast" }, + ]); + }); +}); diff --git a/apps/mobile/src/features/usage/usageCostMix.ts b/apps/mobile/src/features/usage/usageCostMix.ts new file mode 100644 index 000000000..2fe9e7b8f --- /dev/null +++ b/apps/mobile/src/features/usage/usageCostMix.ts @@ -0,0 +1,66 @@ +/** + * Shapes merged cost into the part-to-whole segments the Usage cost section + * draws: by token type, and by request speed. + * + * @module usageCostMix + */ +import type { CategoryCost, SpeedCost } from "@t3tools/shared/usageMerge"; + +export interface CostMixSegment { + readonly label: string; + readonly value: number; + readonly color: string; +} + +export interface CostMixColors { + readonly input: string; + readonly cacheRead: string; + readonly cacheWrite: string; + readonly output: string; + readonly other: string; + readonly standard: string; + readonly fast: string; + readonly ultrafast: string; +} + +export function costTypeSegments( + cost: CategoryCost, + colors: CostMixColors, +): readonly CostMixSegment[] { + return [ + { label: "Input", value: cost.input, color: colors.input }, + { label: "Cache read", value: cost.cacheRead, color: colors.cacheRead }, + { label: "Cache write", value: cost.cacheWrite, color: colors.cacheWrite }, + { label: "Output", value: cost.output, color: colors.output }, + // Reported cost with no rates to split it, or from older servers. Below a + // cent it is rounding, not usage. + { label: "Other", value: cost.unsplit >= 0.005 ? cost.unsplit : 0, color: colors.other }, + ]; +} + +export function speedCostSegments( + cost: SpeedCost, + colors: CostMixColors, +): readonly CostMixSegment[] { + return [ + { label: "Standard", value: cost.standard, color: colors.standard }, + { label: "Fast", value: cost.fast, color: colors.fast }, + { label: "Ultrafast", value: cost.ultrafast, color: colors.ultrafast }, + ]; +} + +/** Servers from before the speed split count all of their cost as standard. */ +export const SPEED_COST_FOOTNOTE = + "Servers that predate speed tracking count all cost as Standard."; + +/** The speed bar only appears once some cost ran faster than standard. */ +export function hasFasterSpeedCost(cost: SpeedCost): boolean { + return cost.fast + cost.ultrafast > 0; +} + +/** Empty segments are left out of a bar and its legend. */ +export function visibleCostSegments( + segments: readonly CostMixSegment[], +): readonly CostMixSegment[] { + return segments.filter((segment) => segment.value > 0); +} diff --git a/apps/mobile/src/features/usage/usageProviders.ts b/apps/mobile/src/features/usage/usageProviders.ts index c969bbc5a..4bb4cc536 100644 --- a/apps/mobile/src/features/usage/usageProviders.ts +++ b/apps/mobile/src/features/usage/usageProviders.ts @@ -1,5 +1,6 @@ import type { UsageProviderKind } from "@t3tools/contracts"; import { useAppearancePreferences } from "../settings/appearance/AppearancePreferencesProvider"; +import type { CostMixColors } from "./usageCostMix"; /** * Series and table order. The chart stacks providers from the bottom in this @@ -35,3 +36,23 @@ export function useProviderColors(): Record { opencode: "#8b5cf6", }; } + +/** + * Neutral steps for cost and token mixes, so they never borrow a provider's + * color. Matches the web steps: oklab mixes of the codex ink into the + * background, above the 15 ΔE separation floor for adjacent segments. + */ +export function useUsageMixColors(): CostMixColors { + const { themeAppearance: scheme } = useAppearancePreferences(); + const dark = scheme === "dark"; + return { + input: dark ? "#737373" : "#848484", + cacheRead: dark ? "#282828" : "#c0c0c0", + cacheWrite: dark ? "#949494" : "#6d6d6d", + output: dark ? "#e6e6e6" : "#3c3c43", + other: dark ? "#494949" : "#a3a3a3", + standard: dark ? "#313131" : "#b8b8b8", + fast: dark ? "#838383" : "#797979", + ultrafast: dark ? "#e6e6e6" : "#3c3c43", + }; +} diff --git a/apps/server/src/usage/UsageService.test.ts b/apps/server/src/usage/UsageService.test.ts index 3480e4629..d782f559c 100644 --- a/apps/server/src/usage/UsageService.test.ts +++ b/apps/server/src/usage/UsageService.test.ts @@ -567,6 +567,99 @@ describe("UsageService", () => { }).pipe(Effect.scoped), ); + it.live( + "upgrades a v4 cache: reprices live Codex tiers, keeps deleted rollouts, leaves v4 intact", + () => + Effect.gen(function* () { + const { home, settings } = yield* setup; + const sessions = NodePath.join(home, "codex", "sessions"); + const rollout = (sessionId: string, outputTokens: number) => + [ + { type: "session_meta", payload: { id: sessionId } }, + { type: "turn_context", payload: { model: "gpt-6-astra" } }, + { + type: "event_msg", + payload: { + type: "thread_settings_applied", + thread_settings: { service_tier: "ultrafast" }, + }, + }, + { + type: "event_msg", + timestamp: "2026-08-01T10:00:00Z", + payload: { + type: "token_count", + info: { last_token_usage: { input_tokens: 0, output_tokens: outputTokens } }, + }, + }, + ] + .map((line) => encodeUnknownJsonString(line)) + .join("\n") + "\n"; + const live = NodePath.join(sessions, "live.jsonl"); + const deleted = NodePath.join(sessions, "deleted.jsonl"); + yield* Effect.promise(async () => { + await NodeFSP.mkdir(sessions, { recursive: true }); + await NodeFSP.writeFile(live, rollout("live", 10)); + await NodeFSP.writeFile(deleted, rollout("deleted", 20)); + }); + + yield* Effect.gen(function* () { + const { stateDir } = yield* ServerConfig.ServerConfig; + const cachePath = NodePath.join(stateDir, "usage-scan-cache-v5.json"); + const legacyPath = NodePath.join(stateDir, "usage-scan-cache.json"); + yield* (yield* UsageService.make).readSummary(WINDOW); + + // Rewrite the cache as a v4 server left it: every Codex record at + // speed 0 (standard), and no tier in the reducer state. + const legacy = yield* Effect.promise(async () => { + const document = decodeUnknownJsonString(await NodeFSP.readFile(cachePath, "utf8")) as { + files: Record; + }; + for (const file of Object.values(document.files)) { + file.r = file.r.map((row) => [...row.slice(0, 10), 0]); + // Claude transcripts in the shared fixture home carry no reducer state. + if (file.cs !== null) delete file.cs.speed; + } + const text = encodeUnknownJsonString({ ...document, version: 4 }); + await NodeFSP.writeFile(legacyPath, text); + await NodeFSP.rm(cachePath); + await NodeFSP.rm(deleted); + return text; + }); + + const summary = yield* (yield* UsageService.make).readSummary(WINDOW); + // The live rollout re-parses at the ultrafast rate (10 x 6); the + // deleted one keeps its saved v4 usage at the standard rate (20 x 1). + assert.strictEqual(totalOutputTokens(summary), 30); + assert.strictEqual( + summary.buckets.reduce((sum, bucket) => sum + bucket.costUsd, 0), + 80, + ); + // A v4 server sharing this state directory still finds its own cache. + assert.strictEqual( + yield* Effect.promise(() => NodeFSP.readFile(legacyPath, "utf8")), + legacy, + ); + }).pipe( + Effect.provide( + serviceLayers({ + prefix: "usage-service-v4-upgrade-test", + home, + settings, + ratesDocument: { + "gpt-6-astra": { + input_cost_per_token: 0, + output_cost_per_token: 1, + input_cost_per_token_ultrafast: 0, + output_cost_per_token_ultrafast: 6, + }, + }, + }), + ), + ); + }).pipe(Effect.scoped), + ); + it.live("preserves saved tokens, costs and sessions after transcript cleanup and restart", () => Effect.gen(function* () { const { transcript, settings, home } = yield* setup; @@ -671,7 +764,8 @@ describe("UsageService", () => { yield* Effect.gen(function* () { const config = yield* ServerConfig.ServerConfig; - const cachePath = NodePath.join(config.stateDir, "usage-scan-cache.json"); + const cachePath = NodePath.join(config.stateDir, "usage-scan-cache-v5.json"); + const legacyPath = NodePath.join(config.stateDir, "usage-scan-cache.json"); const first = yield* UsageService.make; const before = yield* first.readSummary(WINDOW); assert.strictEqual(totalOutputTokens(before), 10); @@ -689,7 +783,9 @@ describe("UsageService", () => { entry.r = entry.r.map((row) => row.slice(0, 10)); entry.t = entry.t.map((row) => row.slice(0, 10)); } - await NodeFSP.writeFile(cachePath, encodeUnknownJsonString(cache)); + // A v3 server only ever wrote the legacy file name. + await NodeFSP.writeFile(legacyPath, encodeUnknownJsonString(cache)); + await NodeFSP.rm(cachePath); await NodeFSP.rm(deletedTranscript); }); @@ -702,7 +798,7 @@ describe("UsageService", () => { const persisted = decodeUnknownJsonString( yield* Effect.promise(() => NodeFSP.readFile(cachePath, "utf8")), ) as { version: number; files: Record }; - assert.strictEqual(persisted.version, 4); + assert.strictEqual(persisted.version, 5); assert.isDefined(persisted.files[liveCachePath]); assert.isUndefined(persisted.files[liveCachePath]?.fr); assert.strictEqual(persisted.files[deletedCachePath]?.fr, 1); @@ -737,7 +833,7 @@ describe("UsageService", () => { const settingsService = yield* ServerSettings.ServerSettingsService; const service = yield* UsageService.make; yield* service.readSummary(WINDOW); - const cachePath = NodePath.join(config.stateDir, "usage-scan-cache.json"); + const cachePath = NodePath.join(config.stateDir, "usage-scan-cache-v5.json"); const oldDir = NodePath.join(home, "claude", "projects"); assert.include(yield* Effect.promise(() => NodeFSP.readFile(cachePath, "utf8")), oldDir); diff --git a/apps/server/src/usage/UsageService.ts b/apps/server/src/usage/UsageService.ts index ad6ee883a..f71bdd091 100644 --- a/apps/server/src/usage/UsageService.ts +++ b/apps/server/src/usage/UsageService.ts @@ -71,7 +71,9 @@ import { decodeScanCache, dedupeWithinFile, encodeScanCache, + LEGACY_SCAN_CACHE_FILE_NAME, pruneScanCache, + SCAN_CACHE_FILE_NAME, type ScanCache, } from "./usageScanCache.ts"; import type { UsageRecord } from "./usageTranscripts.ts"; @@ -186,7 +188,8 @@ export const make = Effect.gen(function* () { }; const ratesCachePath = path.join(config.stateDir, "usage-model-rates.json"); - const scanCachePath = path.join(config.stateDir, "usage-scan-cache.json"); + const scanCachePath = path.join(config.stateDir, SCAN_CACHE_FILE_NAME); + const legacyScanCachePath = path.join(config.stateDir, LEGACY_SCAN_CACHE_FILE_NAME); let rates: RateTable = new Map(); let ratesFetchedAtMs: number | null = null; let ratesStatus: UsagePricing["status"] = "unavailable"; @@ -522,10 +525,17 @@ export const make = Effect.gen(function* () { */ const ensureScanCacheLoaded = yield* Effect.cached( Effect.gen(function* () { - const document = yield* fileSystem.readFileString(scanCachePath).pipe( - Effect.flatMap((raw) => decodeScanCacheFile(raw)), - Effect.catchCause(() => Effect.succeed(null)), - ); + const readDocument = (filePath: string) => + fileSystem.readFileString(filePath).pipe( + Effect.flatMap((raw) => decodeScanCacheFile(raw)), + Effect.catchCause(() => Effect.succeed(null)), + ); + let document = yield* readDocument(scanCachePath); + if (document === null) { + document = yield* readDocument(legacyScanCachePath); + // Write the migrated cache to its own file on the next scan. + cacheDirty = document !== null; + } if (document === null) return; for (const [path, entry] of decodeScanCache(document)) fileCache.set(path, entry); if ( diff --git a/apps/server/src/usage/antigravityUsageReader.test.ts b/apps/server/src/usage/antigravityUsageReader.test.ts index 56a73addd..bbf156d53 100644 --- a/apps/server/src/usage/antigravityUsageReader.test.ts +++ b/apps/server/src/usage/antigravityUsageReader.test.ts @@ -120,7 +120,7 @@ describe("antigravityUsageReader", () => { reasoningTokens: 150, }); expect(outcome.record?.reportedCostUsd).toBeNull(); - expect(outcome.record?.fast).toBe(false); + expect(outcome.record?.speed).toBe("standard"); expect(outcome.contextSnapshot).toEqual({ timestampMs: 1785578400500, diff --git a/apps/server/src/usage/antigravityUsageReader.ts b/apps/server/src/usage/antigravityUsageReader.ts index 464cdb6df..5018bc0d5 100644 --- a/apps/server/src/usage/antigravityUsageReader.ts +++ b/apps/server/src/usage/antigravityUsageReader.ts @@ -88,7 +88,7 @@ export interface AntigravityUsageRecord { readonly model: string; readonly totals: AntigravityTokenTotals; readonly reportedCostUsd: null; - readonly fast: false; + readonly speed: "standard"; readonly contextSnapshot?: AntigravityContextSnapshot | undefined; } @@ -712,7 +712,7 @@ export function parseAntigravityGenMetadataBlob( reasoningTokens: thinkingOutputTokens, }, reportedCostUsd: null, - fast: false, + speed: "standard", ...(contextSnapshot ? { contextSnapshot } : {}), }; diff --git a/apps/server/src/usage/opencodeUsageReader.ts b/apps/server/src/usage/opencodeUsageReader.ts index c42b850a2..742cdc6a1 100644 --- a/apps/server/src/usage/opencodeUsageReader.ts +++ b/apps/server/src/usage/opencodeUsageReader.ts @@ -77,7 +77,7 @@ function parseOpenCodeMessage( // OpenCode writes zero for models without a known rate, including paid // subscription models. Let the shared price table estimate those records. reportedCostUsd: typeof cost === "number" && Number.isFinite(cost) && cost > 0 ? cost : null, - fast: false, + speed: "standard", dedupeKey: id ? `opencode:${id}` : null, }, malformed: false, diff --git a/apps/server/src/usage/usageAggregation.test.ts b/apps/server/src/usage/usageAggregation.test.ts index 75435de08..ab260d339 100644 --- a/apps/server/src/usage/usageAggregation.test.ts +++ b/apps/server/src/usage/usageAggregation.test.ts @@ -12,7 +12,13 @@ const rates: RateTable = new Map([ outputCostPerToken: 5e-5, cacheReadCostPerToken: 1e-6, cacheCreationCostPerToken: 1.25e-5, - fastMultiplier: 1, + fast: { + inputCostPerToken: 2e-5, + outputCostPerToken: 1e-4, + cacheReadCostPerToken: 2e-6, + cacheCreationCostPerToken: 2.5e-5, + }, + ultrafast: null, }, ], ]); @@ -32,7 +38,7 @@ function record(overrides: Partial = {}): UsageRecord { reasoningTokens: 0, }, reportedCostUsd: null, - fast: false, + speed: "standard", dedupeKey: null, ...overrides, }; @@ -76,6 +82,24 @@ describe("UsageAggregator", () => { ).toThrow("requires exact time bounds"); }); + it("splits a bucket's cost by category and speed", () => { + const [bucket] = aggregate([record(), record({ speed: "fast" })]).buckets; + + // Standard costs $0.005625 and fast twice that. + expect(bucket).toMatchObject({ + costUsd: expect.closeTo(0.016875), + categoryCostUsd: { + input: expect.closeTo(0.003), + cacheRead: expect.closeTo(0.003), + cacheWrite: expect.closeTo(0.000375), + output: expect.closeTo(0.0075), + }, + fastCostUsd: expect.closeTo(0.01125), + speedPremiumUsd: expect.closeTo(0.005625), + }); + expect(bucket).not.toHaveProperty("ultrafastCostUsd"); + }); + it("keeps only the first record for a repeated dedupe key", () => { const result = aggregate([ record({ dedupeKey: "msg_1:" }), diff --git a/apps/server/src/usage/usageAggregation.ts b/apps/server/src/usage/usageAggregation.ts index 684f0a520..2c28ee6d5 100644 --- a/apps/server/src/usage/usageAggregation.ts +++ b/apps/server/src/usage/usageAggregation.ts @@ -12,7 +12,13 @@ * * @module usageAggregation */ -import type { UsageBucket, UsageDay, UsageResolution, UsageTokenTotals } from "@t3tools/contracts"; +import type { + UsageBucket, + UsageCategoryCost, + UsageDay, + UsageResolution, + UsageTokenTotals, +} from "@t3tools/contracts"; import { addTotals, EMPTY_TOTALS, type UsageRecord } from "./usageTranscripts.ts"; import { cacheSavingsUsd, priceUsage, type RateTable } from "./usagePricing.ts"; @@ -50,6 +56,10 @@ interface MutableBucket { totals: UsageTokenTotals; costUsd: number; cacheSavingsUsd: number; + categoryCostUsd: UsageCategoryCost | null; + fastCostUsd: number; + ultrafastCostUsd: number; + speedPremiumUsd: number; records: number; unpricedRecords: number; providerReportedRecords: number; @@ -153,6 +163,10 @@ export class UsageAggregator { totals: EMPTY_TOTALS, costUsd: 0, cacheSavingsUsd: 0, + categoryCostUsd: null, + fastCostUsd: 0, + ultrafastCostUsd: 0, + speedPremiumUsd: 0, records: 0, unpricedRecords: 0, providerReportedRecords: 0, @@ -165,6 +179,22 @@ export class UsageAggregator { bucket.totals = addTotals(bucket.totals, record.totals); bucket.costUsd += priced.costUsd; + if (priced.categoryCostUsd !== null) { + const sum = bucket.categoryCostUsd; + const add = priced.categoryCostUsd; + bucket.categoryCostUsd = + sum === null + ? add + : { + input: sum.input + add.input, + cacheRead: sum.cacheRead + add.cacheRead, + cacheWrite: sum.cacheWrite + add.cacheWrite, + output: sum.output + add.output, + }; + } + if (record.speed === "fast") bucket.fastCostUsd += priced.costUsd; + if (record.speed === "ultrafast") bucket.ultrafastCostUsd += priced.costUsd; + bucket.speedPremiumUsd += priced.speedPremiumUsd; bucket.cacheSavingsUsd += cacheSavingsUsd( this.#options.rates, record, @@ -181,6 +211,10 @@ export class UsageAggregator { const buckets: UsageBucket[] = []; for (const [key, bucket] of this.#buckets) { const [day = "", hourStart = "", provider = "", model = ""] = key.split("\u0000"); + const category = bucket.categoryCostUsd; + const fastCostUsd = roundUsd(bucket.fastCostUsd); + const ultrafastCostUsd = roundUsd(bucket.ultrafastCostUsd); + const speedPremiumUsd = roundUsd(bucket.speedPremiumUsd); buckets.push({ day: day as UsageDay, ...(hourStart === "" ? {} : { hourStart }), @@ -189,6 +223,20 @@ export class UsageAggregator { totals: bucket.totals, costUsd: bucket.costUsd, cacheSavingsUsd: bucket.cacheSavingsUsd, + // Zero and unknown figures are omitted to keep payloads small. + ...(category === null + ? {} + : { + categoryCostUsd: { + input: roundUsd(category.input), + cacheRead: roundUsd(category.cacheRead), + cacheWrite: roundUsd(category.cacheWrite), + output: roundUsd(category.output), + }, + }), + ...(fastCostUsd === 0 ? {} : { fastCostUsd }), + ...(ultrafastCostUsd === 0 ? {} : { ultrafastCostUsd }), + ...(speedPremiumUsd === 0 ? {} : { speedPremiumUsd }), costSource: resolveCostSource(bucket), records: bucket.records, unpricedRecords: bucket.unpricedRecords, @@ -212,6 +260,14 @@ export class UsageAggregator { } } +/** + * Rounds to micro-dollars. The split and speed figures need no more precision, + * and shorter numbers keep them cheap on the wire. + */ +function roundUsd(value: number): number { + return Math.round(value * 1e6) / 1e6; +} + /** * A bucket mixes records from one model, but their cost provenance can differ * when only some records carried a reported cost. The weakest provenance in the diff --git a/apps/server/src/usage/usagePricing.test.ts b/apps/server/src/usage/usagePricing.test.ts index 8f03af031..7a1a51451 100644 --- a/apps/server/src/usage/usagePricing.test.ts +++ b/apps/server/src/usage/usagePricing.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from "@effect/vitest"; +import type { UsageSpeed } from "./usageTranscripts.ts"; import { cacheSavingsUsd, createOverrideRateTable, @@ -22,11 +23,15 @@ describe("usage pricing", () => { outputTokens: 1_000_000, reasoningTokens: 500_000, }; - const record = (model: string, reportedCostUsd: number | null = null, fast = false) => ({ + const record = ( + model: string, + reportedCostUsd: number | null = null, + speed: UsageSpeed = "standard", + ) => ({ model, totals, reportedCostUsd, - fast, + speed, }); it("uses custom token rates ahead of public and provider-reported costs", () => { @@ -41,7 +46,7 @@ describe("usage pricing", () => { }); for (const reportedCostUsd of [null, 99]) { - expect(priceUsage(table, record("example-model", reportedCostUsd), overrides)).toEqual({ + expect(priceUsage(table, record("example-model", reportedCostUsd), overrides)).toMatchObject({ costUsd: 13.5, costSource: "modelPriced", }); @@ -55,7 +60,7 @@ describe("usage pricing", () => { "example-model": { inputCostPerMillionTokens: 2, outputCostPerMillionTokens: 8 }, }); - expect(priceUsage(table, record("example-model"), overrides)).toEqual({ + expect(priceUsage(table, record("example-model"), overrides)).toMatchObject({ costUsd: 14, costSource: "modelPriced", }); @@ -70,7 +75,7 @@ describe("usage pricing", () => { outputCostPerMillionTokens: 0, }, }); - expect(priceUsage(table, record(" vendor/example-model[1m] ", 99), overrides)).toEqual({ + expect(priceUsage(table, record(" vendor/example-model[1m] ", 99), overrides)).toMatchObject({ costUsd: 0, costSource: "modelPriced", }); @@ -84,6 +89,8 @@ describe("usage pricing", () => { expect(priceUsage(table, record(model, 99), overrides)).toEqual({ costUsd: 99, costSource: "providerReported", + categoryCostUsd: null, + speedPremiumUsd: 0, }); } }); @@ -96,20 +103,82 @@ describe("usage pricing", () => { const overrides = createOverrideRateTable({ "claude-opus-5-5": { inputCostPerMillionTokens: 4, outputCostPerMillionTokens: 20 }, }); - const cost = (model: string, fast: boolean, custom?: typeof overrides) => - priceUsage(table, record(model, null, fast), custom).costUsd; + const cost = (model: string, speed: UsageSpeed, custom?: typeof overrides) => + priceUsage(table, record(model, null, speed), custom).costUsd; - expect(cost("claude-opus-5-5", true)).toBeCloseTo(2 * cost("claude-opus-5-5", false)); - expect(cacheSavingsUsd(table, record("claude-opus-5-5", null, true))).toBeCloseTo( + expect(cost("claude-opus-5-5", "fast")).toBeCloseTo(2 * cost("claude-opus-5-5", "standard")); + expect(cacheSavingsUsd(table, record("claude-opus-5-5", null, "fast"))).toBeCloseTo( 2 * cacheSavingsUsd(table, record("claude-opus-5-5")), ); // No published fast tier, and custom prices, both stay at the standard rate. - expect(cost("claude-fable-5-1", true)).toBe(cost("claude-fable-5-1", false)); - expect(cost("claude-opus-5-5", true, overrides)).toBe( - cost("claude-opus-5-5", false, overrides), + expect(cost("claude-fable-5-1", "fast")).toBe(cost("claude-fable-5-1", "standard")); + expect(cost("claude-opus-5-5", "fast", overrides)).toBe( + cost("claude-opus-5-5", "standard", overrides), ); }); + it("splits cost by category and prices the speed premium", () => { + const table = parseRateTable({ + "claude-opus-5-5": { + ...rate(4e-6, 4e-7), + cache_creation_input_token_cost: 5e-6, + provider_specific_entry: { fast: 2 }, + }, + }); + const split = (input: number, cacheRead: number, cacheWrite: number, output: number) => ({ + input: expect.closeTo(input), + cacheRead: expect.closeTo(cacheRead), + cacheWrite: expect.closeTo(cacheWrite), + output: expect.closeTo(output), + }); + + expect(priceUsage(table, record("claude-opus-5-5", null, "fast"))).toEqual({ + costUsd: expect.closeTo(58.8), + costSource: "modelPriced", + categoryCostUsd: split(8, 0.8, 10, 40), + speedPremiumUsd: expect.closeTo(29.4), + }); + // A reported cost keeps its total and splits in proportion to list rates. + expect(priceUsage(table, record("claude-opus-5-5", 29.4, "fast"))).toEqual({ + costUsd: 29.4, + costSource: "providerReported", + categoryCostUsd: split(4, 0.4, 5, 20), + speedPremiumUsd: expect.closeTo(14.7), + }); + // Without rates there is nothing to split it by. + expect(priceUsage(table, record("unknown-model", 29.4, "fast"))).toMatchObject({ + costUsd: 29.4, + categoryCostUsd: null, + speedPremiumUsd: 0, + }); + }); + + it("prices Codex priority and ultrafast requests at their published tier rates", () => { + const table = parseRateTable({ + "gpt-6-astra": { + ...rate(1e-5, 1e-6), + input_cost_per_token_priority: 2e-5, + output_cost_per_token_priority: 1e-4, + cache_read_input_token_cost_priority: 2e-6, + input_cost_per_token_ultrafast: 6e-5, + output_cost_per_token_ultrafast: 3e-4, + // No ultrafast cache rate: keeps the standard 10:1 input-to-cache ratio. + }, + "gpt-6-sol": rate(2e-6, 2e-7), + }); + const cost = (model: string, speed: UsageSpeed) => + priceUsage(table, record(model, null, speed)).costUsd; + const standard = cost("gpt-6-astra", "standard"); + + expect(cost("gpt-6-astra", "fast")).toBeCloseTo(2 * standard); + expect(cost("gpt-6-astra", "ultrafast")).toBeCloseTo(6 * standard); + expect(cacheSavingsUsd(table, record("gpt-6-astra", null, "ultrafast"))).toBeCloseTo( + 6 * cacheSavingsUsd(table, record("gpt-6-astra")), + ); + // A tier the model does not publish bills at the standard rate. + expect(cost("gpt-6-sol", "ultrafast")).toBe(cost("gpt-6-sol", "standard")); + }); + it("keeps the canonical Fable rate separate from DeepInfra in either order", () => { const canonical = ["claude-fable-5", rate(1e-5, 1e-6)] as const; const deepInfra = ["deepinfra/anthropic/claude-fable-5", rate(1e-5)] as const; diff --git a/apps/server/src/usage/usagePricing.ts b/apps/server/src/usage/usagePricing.ts index e767c7093..f9b46774d 100644 --- a/apps/server/src/usage/usagePricing.ts +++ b/apps/server/src/usage/usagePricing.ts @@ -7,35 +7,46 @@ * * @module usagePricing */ -import type { UsageCostSource, UsageModelPriceOverride } from "@t3tools/contracts"; +import type { + UsageCategoryCost, + UsageCostSource, + UsageModelPriceOverride, + UsageTokenTotals, +} from "@t3tools/contracts"; -import type { UsageRecord } from "./usageTranscripts.ts"; +import type { UsageRecord, UsageSpeed } from "./usageTranscripts.ts"; -/** - * The subset of a LiteLLM entry we price against. All values are USD per token. - * - * LiteLLM also publishes tiered variants (`*_above_272k_tokens`, `*_flex`, - * `*_priority`, `*_batches`). We deliberately price at the base tier: the - * transcripts don't record which tier served a request, so anything else would - * be a guess dressed up as precision. - */ -export interface ModelRate { +/** Token rates for one billing speed. All values are USD per token. */ +export interface TokenRates { readonly inputCostPerToken: number; readonly outputCostPerToken: number; readonly cacheReadCostPerToken: number; readonly cacheCreationCostPerToken: number; +} + +/** + * The subset of a LiteLLM entry we price against: standard rates, plus rates + * for each faster speed the model publishes. A request at a speed with no + * published rates bills at the standard rates. + * + * LiteLLM also publishes `*_above_272k_tokens`, `*_flex`, and `*_batches` + * variants. Transcripts don't record those, so we don't price them. + */ +export interface ModelRate extends TokenRates { /** - * Multiple of the rates above billed for a fast-mode request, from LiteLLM's - * `provider_specific_entry.fast`. `1` when the model publishes no fast tier. + * From LiteLLM's `provider_specific_entry.fast` multiple (Claude fast mode), + * or else its `*_priority` rates (Codex `priority`). */ - readonly fastMultiplier: number; + readonly fast: TokenRates | null; + /** From LiteLLM's `*_ultrafast` rates (Codex `ultrafast`). */ + readonly ultrafast: TokenRates | null; } export type RateTable = ReadonlyMap; /** * Custom IDs keep their case, provider prefix, and variant suffix. Custom rates - * apply as entered, fast-mode requests included. + * apply as entered, at every speed. */ export function createOverrideRateTable( overrides: Readonly>, @@ -50,31 +61,68 @@ export function createOverrideRateTable( (prices.cacheReadCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, cacheCreationCostPerToken: (prices.cacheWriteCostPerMillionTokens ?? prices.inputCostPerMillionTokens) / 1_000_000, - fastMultiplier: 1, + fast: null, + ultrafast: null, }, ]), ); } -/** Raw shape of one LiteLLM entry, narrowed to the fields we read. */ -interface LiteLlmEntry { - readonly input_cost_per_token?: unknown; - readonly output_cost_per_token?: unknown; - readonly cache_read_input_token_cost?: unknown; - readonly cache_creation_input_token_cost?: unknown; - readonly provider_specific_entry?: unknown; -} +/** One raw LiteLLM entry. Field names carry a tier suffix, e.g. `_priority`. */ +type LiteLlmEntry = Readonly>; function finiteNumber(value: unknown): number | null { return typeof value === "number" && Number.isFinite(value) ? value : null; } +/** + * Reads one rate set, `suffix` selecting a tier such as `_priority`. Returns + * `null` without both an input and an output rate. + * + * Anthropic bills cache reads at a discount and cache writes at a premium. + * When the standard tier omits them, cached input is priced as plain input + * rather than as free. A faster tier that omits them keeps the standard tier's + * cache-to-input ratio. + */ +function readTokenRates( + entry: LiteLlmEntry, + suffix: string, + standard?: TokenRates, +): TokenRates | null { + const input = finiteNumber(entry[`input_cost_per_token${suffix}`]); + const output = finiteNumber(entry[`output_cost_per_token${suffix}`]); + if (input === null || output === null) return null; + const cacheRate = (name: string, field: "cacheReadCostPerToken" | "cacheCreationCostPerToken") => + finiteNumber(entry[`${name}${suffix}`]) ?? + (standard !== undefined && standard.inputCostPerToken > 0 + ? (standard[field] / standard.inputCostPerToken) * input + : input); + return { + inputCostPerToken: input, + outputCostPerToken: output, + cacheReadCostPerToken: cacheRate("cache_read_input_token_cost", "cacheReadCostPerToken"), + cacheCreationCostPerToken: cacheRate( + "cache_creation_input_token_cost", + "cacheCreationCostPerToken", + ), + }; +} + +function scaleTokenRates(rates: TokenRates, multiple: number): TokenRates { + return { + inputCostPerToken: rates.inputCostPerToken * multiple, + outputCostPerToken: rates.outputCostPerToken * multiple, + cacheReadCostPerToken: rates.cacheReadCostPerToken * multiple, + cacheCreationCostPerToken: rates.cacheCreationCostPerToken * multiple, + }; +} + /** Reads `provider_specific_entry.fast`, e.g. `2` for Claude Opus 5.5. */ -function fastMultiplier(entry: LiteLlmEntry): number { - const specific = entry.provider_specific_entry; - if (typeof specific !== "object" || specific === null) return 1; +function fastMultiplier(entry: LiteLlmEntry): number | null { + const specific = entry["provider_specific_entry"]; + if (typeof specific !== "object" || specific === null) return null; const fast = finiteNumber((specific as Record)["fast"]); - return fast !== null && fast > 0 ? fast : 1; + return fast !== null && fast > 0 ? fast : null; } /** @@ -94,21 +142,19 @@ export function parseRateTable(document: unknown): RateTable { for (const [name, raw] of Object.entries(document as Record)) { if (typeof raw !== "object" || raw === null) continue; const entry = raw as LiteLlmEntry; - const input = finiteNumber(entry.input_cost_per_token); - const output = finiteNumber(entry.output_cost_per_token); - if (input === null || output === null) continue; + const standard = readTokenRates(entry, ""); + if (standard === null) continue; const key = normalizeRateKey(name); if (key.length === 0) continue; + const multiple = fastMultiplier(entry); table.set(key, { - inputCostPerToken: input, - outputCostPerToken: output, - // Anthropic bills cache reads at a discount and cache writes at a - // premium. When a model omits them, cached input is priced as plain - // input rather than as free. - cacheReadCostPerToken: finiteNumber(entry.cache_read_input_token_cost) ?? input, - cacheCreationCostPerToken: finiteNumber(entry.cache_creation_input_token_cost) ?? input, - fastMultiplier: fastMultiplier(entry), + ...standard, + fast: + multiple === null + ? readTokenRates(entry, "_priority", standard) + : scaleTokenRates(standard, multiple), + ultrafast: readTokenRates(entry, "_ultrafast", standard), }); } @@ -139,6 +185,11 @@ export function parseRateTable(document: unknown): RateTable { return table; } +/** The rates a request at `speed` bills at. */ +function ratesAt(rate: ModelRate, speed: UsageSpeed): TokenRates { + return (speed === "standard" ? null : rate[speed]) ?? rate; +} + function normalizeRateKey(model: string): string { return model.trim().toLowerCase(); } @@ -199,16 +250,36 @@ export function lookupRate(table: RateTable, model: string): ModelRate | null { } /** The parts of a transcript record that decide its price. */ -export type PricedRecord = Pick; +export type PricedRecord = Pick; export interface PricedUsage { readonly costUsd: number; readonly costSource: UsageCostSource; + /** `costUsd` by token category, or `null` when no rates are known to split it. */ + readonly categoryCostUsd: UsageCategoryCost | null; + /** What `costUsd` exceeds the same tokens at standard rates. `0` without rates. */ + readonly speedPremiumUsd: number; +} + +function costByCategory(totals: UsageTokenTotals, rates: TokenRates): UsageCategoryCost { + return { + input: totals.uncachedInputTokens * rates.inputCostPerToken, + cacheRead: totals.cachedInputTokens * rates.cacheReadCostPerToken, + cacheWrite: totals.cacheCreationTokens * rates.cacheCreationCostPerToken, + output: totals.outputTokens * rates.outputCostPerToken, + }; +} + +function sumCategories(cost: UsageCategoryCost): number { + return cost.input + cost.cacheRead + cost.cacheWrite + cost.output; } /** * Prices one record's tokens. * + * A provider-reported cost is kept as is, and split by category and speed in + * proportion to the model's list rates when those are known. + * * `reasoningTokens` is intentionally not charged separately: it is already * counted inside `outputTokens`. */ @@ -219,22 +290,37 @@ export function priceUsage( ): PricedUsage { const { model, totals, reportedCostUsd } = record; const override = overrides?.get(model.trim()); - if (override === undefined && reportedCostUsd !== null && Number.isFinite(reportedCostUsd)) { - return { costUsd: reportedCostUsd, costSource: "providerReported" }; - } - + const reported = + override === undefined && reportedCostUsd !== null && Number.isFinite(reportedCostUsd) + ? reportedCostUsd + : null; + const unsplit = (costUsd: number, costSource: UsageCostSource): PricedUsage => ({ + costUsd, + costSource, + categoryCostUsd: null, + speedPremiumUsd: 0, + }); const rate = override ?? lookupRate(table, model); - if (rate === null) return { costUsd: 0, costSource: "unpriced" }; - - const standardCostUsd = - totals.uncachedInputTokens * rate.inputCostPerToken + - totals.cachedInputTokens * rate.cacheReadCostPerToken + - totals.cacheCreationTokens * rate.cacheCreationCostPerToken + - totals.outputTokens * rate.outputCostPerToken; + if (rate === null) { + return reported === null ? unsplit(0, "unpriced") : unsplit(reported, "providerReported"); + } + const listCost = costByCategory(totals, ratesAt(rate, record.speed)); + const listCostUsd = sumCategories(listCost); + if (reported !== null && listCostUsd <= 0) return unsplit(reported, "providerReported"); + const premiumUsd = + record.speed === "standard" ? 0 : listCostUsd - sumCategories(costByCategory(totals, rate)); + const scale = reported === null ? 1 : reported / listCostUsd; return { - costUsd: standardCostUsd * (record.fast ? rate.fastMultiplier : 1), - costSource: "modelPriced", + costUsd: reported ?? listCostUsd, + costSource: reported === null ? "modelPriced" : "providerReported", + categoryCostUsd: { + input: listCost.input * scale, + cacheRead: listCost.cacheRead * scale, + cacheWrite: listCost.cacheWrite * scale, + output: listCost.output * scale, + }, + speedPremiumUsd: premiumUsd * scale, }; } @@ -249,9 +335,6 @@ export function cacheSavingsUsd( ): number { const rate = overrides?.get(record.model.trim()) ?? lookupRate(table, record.model); if (rate === null) return 0; - return ( - record.totals.cachedInputTokens * - (rate.inputCostPerToken - rate.cacheReadCostPerToken) * - (record.fast ? rate.fastMultiplier : 1) - ); + const rates = ratesAt(rate, record.speed); + return record.totals.cachedInputTokens * (rates.inputCostPerToken - rates.cacheReadCostPerToken); } diff --git a/apps/server/src/usage/usageScanCache.test.ts b/apps/server/src/usage/usageScanCache.test.ts index 8f32b01f2..167ba402b 100644 --- a/apps/server/src/usage/usageScanCache.test.ts +++ b/apps/server/src/usage/usageScanCache.test.ts @@ -24,7 +24,7 @@ function record(overrides: Partial = {}): UsageRecord { reasoningTokens: 0, }, reportedCostUsd: null, - fast: false, + speed: "standard", dedupeKey: "msg_1:", ...overrides, }; @@ -61,7 +61,7 @@ describe("scan cache round trip", () => { [ "/a.jsonl", 100, - [record(), record({ dedupeKey: "msg_2:", model: "claude-opus-5-5", fast: true })], + [record(), record({ dedupeKey: "msg_2:", model: "claude-opus-5-5", speed: "fast" })], ], ["/b.jsonl", 200, [record({ sessionId: "session-b", reportedCostUsd: 1.5 })]], ]); @@ -79,11 +79,14 @@ describe("scan cache round trip", () => { size: 80, mtimeMs: 400, provider: "codex", - records: [record({ provider: "codex", model: "gpt-5.2-codex", dedupeKey: null })], + records: [ + record({ provider: "codex", model: "gpt-6-astra", dedupeKey: null, speed: "ultrafast" }), + ], tailRecords: [], position: position({ codexState: { - model: "gpt-5.2-codex", + model: "gpt-6-astra", + speed: "ultrafast", sessionId: "session-c", lastUsageSignature: '{"input_tokens":1}', sawSessionMeta: true, @@ -128,8 +131,8 @@ describe("scan cache round trip", () => { expect(decodeScanCache(JSON.parse(JSON.stringify(poisoned))).has("/a.jsonl")).toBe(false); }); - it("drops an entry whose fast flag is not 0 or 1", () => { - const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record({ fast: true })]]])); + it("drops an entry whose speed is not a known index", () => { + const encoded = encodeScanCache(cacheWith([["/a.jsonl", 100, [record({ speed: "fast" })]]])); const row = encoded.files["/a.jsonl"]!.r[0]!; const poisoned = { ...encoded, @@ -140,16 +143,16 @@ describe("scan cache round trip", () => { }); it("migrates v3 rows and keeps Claude rescan pending across persistence", () => { - const original = cacheWith([["/deleted.jsonl", 100, [record({ fast: true })]]]); + const original = cacheWith([["/deleted.jsonl", 100, [record({ speed: "fast" })]]]); original.set("/codex.jsonl", { ...original.get("/deleted.jsonl")!, provider: "codex", - records: [record({ provider: "codex", fast: false })], + records: [record({ provider: "codex", speed: "standard" })], }); original.set("/deleted.db", { ...original.get("/deleted.jsonl")!, provider: "antigravity", - records: [record({ provider: "antigravity", fast: false })], + records: [record({ provider: "antigravity", speed: "standard" })], }); const encoded = encodeScanCache(original); const files = Object.fromEntries( @@ -164,16 +167,46 @@ describe("scan cache round trip", () => { ); const migrated = decodeScanCache({ ...encoded, version: 3, files }); - expect(migrated.get("/deleted.jsonl")?.records[0]?.fast).toBe(false); + expect(migrated.get("/deleted.jsonl")?.records[0]?.speed).toBe("standard"); expect(migrated.get("/deleted.jsonl")?.needsFastRescan).toBe(true); expect(migrated.get("/codex.jsonl")?.records).toEqual(original.get("/codex.jsonl")?.records); - expect(migrated.get("/codex.jsonl")?.needsFastRescan).toBeUndefined(); + // v3 Codex rows predate service tiers: kept, but re-parsed when live. + expect(migrated.get("/codex.jsonl")?.needsFastRescan).toBe(true); + expect(migrated.get("/deleted.db")?.needsFastRescan).toBeUndefined(); expect(migrated.get("/deleted.db")?.records).toEqual(original.get("/deleted.db")?.records); expect(decodeScanCache(encodeScanCache(migrated)).get("/deleted.jsonl")?.needsFastRescan).toBe( true, ); }); + it("migrates v4 rows, keeping fast speed and rescanning Codex rollouts", () => { + const original = cacheWith([["/claude.jsonl", 100, [record({ speed: "fast" })]]]); + original.set("/codex.jsonl", { + ...original.get("/claude.jsonl")!, + provider: "codex", + records: [record({ provider: "codex", speed: "standard", dedupeKey: null })], + position: { + resumeOffset: 100, + guardLength: 0, + guardHash: 0, + codexState: null, + }, + }); + const migrated = decodeScanCache( + JSON.parse(JSON.stringify({ ...encodeScanCache(original), version: 4 })), + ); + + expect(migrated.get("/claude.jsonl")?.records[0]?.speed).toBe("fast"); + expect(migrated.get("/claude.jsonl")?.needsFastRescan).toBeUndefined(); + expect(migrated.get("/codex.jsonl")?.records).toEqual(original.get("/codex.jsonl")?.records); + expect(migrated.get("/codex.jsonl")?.needsFastRescan).toBe(true); + expect(migrated.get("/codex.jsonl")?.position.resumeOffset).toBe(0); + // The pending rescan survives a round trip through the current version. + expect(decodeScanCache(encodeScanCache(migrated)).get("/codex.jsonl")?.needsFastRescan).toBe( + true, + ); + }); + it("interns repeated model and session strings", () => { const encoded = encodeScanCache( cacheWith([["/a.jsonl", 100, [record(), record({ dedupeKey: "msg_2:" }), record()]]]), diff --git a/apps/server/src/usage/usageScanCache.ts b/apps/server/src/usage/usageScanCache.ts index 6e409d4ae..80d7ff5d3 100644 --- a/apps/server/src/usage/usageScanCache.ts +++ b/apps/server/src/usage/usageScanCache.ts @@ -17,7 +17,7 @@ import type { UsageProviderKind } from "@t3tools/contracts"; import { GUARD_LENGTH, type TranscriptParsePosition } from "./usageTranscriptReader.ts"; -import type { CodexScanState, UsageRecord } from "./usageTranscripts.ts"; +import type { CodexScanState, UsageRecord, UsageSpeed } from "./usageTranscripts.ts"; // v2: Codex fork-copy suppression changed what a file parses to, so v1 // entries would keep serving double-counted records forever. @@ -25,7 +25,26 @@ import type { CodexScanState, UsageRecord } from "./usageTranscripts.ts"; // re-parses only its appended bytes instead of starting over. // v4: records carry Claude fast mode. v3 rows are migrated rather than dropped: // deleted transcripts have no other copy of their historical usage. -const USAGE_SCAN_CACHE_VERSION = 4 as const; +// v5: Codex records carry their service tier. v4 rows store speed the same +// way (0 standard, 1 fast), so they load as-is; v3 and v4 Codex entries keep +// their history but are fully re-parsed when the rollout still exists. +const USAGE_SCAN_CACHE_VERSION = 5 as const; + +/** + * Each cache version writes its own file in the state directory. An older + * server sharing that directory cannot read a newer cache and would replace + * it, dropping saved usage for deleted transcripts. Separate files keep both. + * A v5 server reads the legacy (v3/v4) file once, when its own file is missing. + */ +export const SCAN_CACHE_FILE_NAME = "usage-scan-cache-v5.json"; +export const LEGACY_SCAN_CACHE_FILE_NAME = "usage-scan-cache.json"; + +/** Serialised as the index into this list. */ +const SPEEDS: readonly UsageSpeed[] = ["standard", "fast", "ultrafast"]; + +function isSpeed(value: unknown): value is UsageSpeed { + return SPEEDS.some((speed) => speed === value); +} export interface CachedFile { readonly size: number; @@ -40,7 +59,10 @@ export interface CachedFile { */ readonly tailRecords: readonly UsageRecord[]; readonly position: TranscriptParsePosition; - /** A v3 Claude entry must be fully parsed if its transcript still exists. */ + /** + * A migrated entry whose rows predate speed tracking (v3 Claude, v3/v4 + * Codex) must be fully parsed if its transcript still exists. + */ readonly needsFastRescan?: boolean; } @@ -62,7 +84,7 @@ type SerializedRecord = readonly [ reasoningTokens: number, dedupeKey: string | null, reportedCostUsd: number | null, - fast: 0 | 1, + speed: number, ]; interface SerializedFile { @@ -116,7 +138,7 @@ export function encodeScanCache(cache: ScanCache): SerializedCache { record.totals.reasoningTokens, record.dedupeKey, record.reportedCostUsd, - record.fast ? 1 : 0, + SPEEDS.indexOf(record.speed), ]; const files: Record = {}; @@ -153,8 +175,10 @@ export function decodeScanCache(document: unknown): ScanCache { if (typeof document !== "object" || document === null) return cache; const root = document as Partial; - if (root.version !== USAGE_SCAN_CACHE_VERSION && root.version !== 3) return cache; - const legacy = root.version === 3; + const version = root.version; + if (version !== USAGE_SCAN_CACHE_VERSION && version !== 4 && version !== 3) return cache; + // v3 rows have no speed column. + const legacy = version === 3; if (!isRecordArray(root.models) || !isRecordArray(root.sessions)) return cache; if (typeof root.files !== "object" || root.files === null) return cache; @@ -187,8 +211,15 @@ export function decodeScanCache(document: unknown): ScanCache { reasoning, dedupeKey, reportedCostUsd, - fast, + speedIndex, ] = row as SerializedRecord; + // v3 never stored speed. Retained rows are conservatively standard; a + // live transcript is fully reparsed before it becomes warm. + const speed: UsageSpeed | undefined = legacy + ? "standard" + : typeof speedIndex === "number" + ? SPEEDS[speedIndex] + : undefined; const model = typeof modelIndex === "number" ? models[modelIndex] : undefined; if ( @@ -200,7 +231,7 @@ export function decodeScanCache(document: unknown): ScanCache { !Number.isFinite(cacheCreation) || !Number.isFinite(output) || !Number.isFinite(reasoning) || - (!legacy && fast !== 0 && fast !== 1) + speed === undefined ) { return null; } @@ -218,9 +249,7 @@ export function decodeScanCache(document: unknown): ScanCache { reasoningTokens: reasoning, }, reportedCostUsd: typeof reportedCostUsd === "number" ? reportedCostUsd : null, - // v3 never stored speed. Retained rows are conservatively standard; - // a live Claude transcript is fully reparsed before it becomes warm. - fast: !legacy && fast === 1, + speed, dedupeKey: typeof dedupeKey === "string" ? dedupeKey : null, }); } @@ -259,7 +288,11 @@ export function decodeScanCache(document: unknown): ScanCache { ) { continue; } - const codexState = decodeCodexState(entry.cs); + // v3/v4 Codex records predate service tiers, so they all priced as + // standard. Keep them, because the rollout may be gone, but mark the entry + // so a live rollout re-parses whole before it is served warm again. + const legacyCodex = entry.p === "codex" && version !== USAGE_SCAN_CACHE_VERSION; + const codexState = legacyCodex ? null : decodeCodexState(entry.cs); if (codexState === undefined) continue; const provider: UsageProviderKind = entry.p; @@ -273,14 +306,12 @@ export function decodeScanCache(document: unknown): ScanCache { provider, records, tailRecords, - position: { - resumeOffset: entry.o, - guardLength: entry.gl, - guardHash: entry.gh, - codexState, - }, - ...(legacy && provider === "claude" ? { needsFastRescan: true } : {}), - ...(!legacy && entry.fr === 1 ? { needsFastRescan: true } : {}), + position: legacyCodex + ? { resumeOffset: 0, guardLength: 0, guardHash: 0, codexState: null } + : { resumeOffset: entry.o, guardLength: entry.gl, guardHash: entry.gh, codexState }, + ...((legacy && provider === "claude") || legacyCodex || (!legacy && entry.fr === 1) + ? { needsFastRescan: true } + : {}), }); } @@ -298,6 +329,7 @@ function decodeCodexState(value: unknown): CodexScanState | null | undefined { const state = value as Partial; if ( typeof state.model !== "string" || + !isSpeed(state.speed) || typeof state.sessionId !== "string" || (state.lastUsageSignature !== null && typeof state.lastUsageSignature !== "string") || typeof state.sawSessionMeta !== "boolean" || @@ -309,6 +341,7 @@ function decodeCodexState(value: unknown): CodexScanState | null | undefined { } return { model: state.model, + speed: state.speed, sessionId: state.sessionId, lastUsageSignature: state.lastUsageSignature ?? null, sawSessionMeta: state.sawSessionMeta, diff --git a/apps/server/src/usage/usageTranscriptReader.test.ts b/apps/server/src/usage/usageTranscriptReader.test.ts index 5feb68b2f..ee547a6a2 100644 --- a/apps/server/src/usage/usageTranscriptReader.test.ts +++ b/apps/server/src/usage/usageTranscriptReader.test.ts @@ -49,6 +49,14 @@ function codexModelLine(model: string): string { })}\n`; } +function codexTierLine(serviceTier: string): string { + return `${JSON.stringify({ + type: "event_msg", + timestamp: "2026-08-01T10:00:01Z", + payload: { type: "thread_settings_applied", thread_settings: { service_tier: serviceTier } }, + })}\n`; +} + function codexUsageLine(outputTokens: number, secondsOffset: number): string { return `${JSON.stringify({ type: "event_msg", @@ -84,19 +92,24 @@ describe("readTranscriptRecords resume", () => { it("carries the Codex reducer state across the resume boundary", async () => { const path = NodePath.join(dir, "rollout.jsonl"); - await NodeFSP.writeFile(path, codexMetaLine() + codexModelLine("gpt-5.2-codex")); + await NodeFSP.writeFile( + path, + codexMetaLine() + codexModelLine("gpt-5.2-codex") + codexTierLine("ultrafast"), + ); const first = await readTranscriptRecords(path, "codex"); assert.isNotNull(first); assert.strictEqual(first.records.length, 0); - // The appended usage event has no turn_context or session_meta of its own; - // model and session must come from the state captured before the boundary. + // The appended usage event has no turn_context, thread settings, or + // session_meta of its own; model, tier, and session must come from the + // state captured before the boundary. await NodeFSP.appendFile(path, codexUsageLine(9, 5)); const second = await readTranscriptRecords(path, "codex", first.position); assert.isNotNull(second); assert.isTrue(second.resumed); assert.strictEqual(second.records.length, 1); assert.strictEqual(second.records[0]?.model, "gpt-5.2-codex"); + assert.strictEqual(second.records[0]?.speed, "ultrafast"); assert.strictEqual(second.records[0]?.sessionId, "codex-session-1"); }); diff --git a/apps/server/src/usage/usageTranscriptReader.ts b/apps/server/src/usage/usageTranscriptReader.ts index faa686990..4d1702375 100644 --- a/apps/server/src/usage/usageTranscriptReader.ts +++ b/apps/server/src/usage/usageTranscriptReader.ts @@ -108,6 +108,7 @@ const USAGE_FIELDS: Record<"claude" | "codex" | "grok", SelectedFields> = { id: true, session_id: true, model: true, + thread_settings: { service_tier: true }, forked_from_id: true, source: { subagent: { thread_spawn: { parent_thread_id: true } } }, info: { last_token_usage: true }, @@ -245,9 +246,10 @@ async function guardMatches( * still match, so only appended lines are read; otherwise the whole file is * re-parsed from the start and `resumed` reports `false`. * - * Codex carries the active model on `turn_context` lines that hold no usage of - * their own, so those still have to pass through the reducer to keep model - * attribution correct. + * Codex carries the active model on `turn_context` lines and the service tier + * on `thread_settings_applied` lines. Neither holds usage of its own, but both + * still have to pass through the reducer to keep attribution and pricing + * correct. */ export async function readTranscriptRecords( filePath: string, @@ -283,6 +285,7 @@ export async function readTranscriptRecords( if ( !mightCarryUsage(line, provider) && !line.includes('"turn_context"') && + !line.includes('"thread_settings_applied"') && !line.includes('"session_meta"') ) { return; diff --git a/apps/server/src/usage/usageTranscriptStreaming.test.ts b/apps/server/src/usage/usageTranscriptStreaming.test.ts index 2187fd4aa..fd23593e5 100644 --- a/apps/server/src/usage/usageTranscriptStreaming.test.ts +++ b/apps/server/src/usage/usageTranscriptStreaming.test.ts @@ -122,7 +122,7 @@ describe("large usage records", () => { reasoningTokens: 0, }, reportedCostUsd: 0.25, - fast: true, + speed: "fast", dedupeKey: "m1:r-m1", }, ]); diff --git a/apps/server/src/usage/usageTranscripts.test.ts b/apps/server/src/usage/usageTranscripts.test.ts index 5c57996f1..bd5603ce2 100644 --- a/apps/server/src/usage/usageTranscripts.test.ts +++ b/apps/server/src/usage/usageTranscripts.test.ts @@ -53,15 +53,15 @@ describe("parseClaudeLine", () => { reasoningTokens: 0, }); expect(record?.dedupeKey).toBe("msg_1:"); - expect(record?.fast).toBe(false); + expect(record?.speed).toBe("standard"); }); it("marks fast-mode requests", () => { const line = (speed: string) => parseClaudeLine(claudeLine({ messageId: "msg_1", contentType: "text", speed })); - expect(line("fast")?.fast).toBe(true); - expect(line("standard")?.fast).toBe(false); + expect(line("fast")?.speed).toBe("fast"); + expect(line("standard")?.speed).toBe("standard"); }); it("gives every content block of one message the same dedupe key", () => { @@ -148,6 +148,27 @@ describe("parseCodexLine", () => { expect(parseCodexLine(tokenCount(100, 0, 10, 0), state)).not.toBeNull(); }); + it("carries the service tier from the latest thread settings", () => { + const settings = (thread_settings: Record) => + JSON.stringify({ + type: "event_msg", + timestamp: "2026-08-01T05:17:42.000Z", + payload: { type: "thread_settings_applied", thread_settings }, + }); + const state = initialCodexScanState(); + parseCodexLine(turnContext, state); + const speedAfter = (line: string, output: number) => { + parseCodexLine(line, state); + return parseCodexLine(tokenCount(100, 0, output, 0), state)?.speed; + }; + + expect(parseCodexLine(tokenCount(100, 0, 1, 0), state)?.speed).toBe("standard"); + expect(speedAfter(settings({ service_tier: "ultrafast" }), 2)).toBe("ultrafast"); + expect(speedAfter(settings({ service_tier: "priority" }), 3)).toBe("fast"); + // Codex omits the field when no tier was requested. + expect(speedAfter(settings({ model: "gpt-6-astra" }), 4)).toBe("standard"); + }); + // A forked/subagent rollout opens with the parent's history copied in and // every line re-stamped to the fork instant, then the ancestors' session // metas. Counting those again multiplied usage ~1.85x on real data (#5758). diff --git a/apps/server/src/usage/usageTranscripts.ts b/apps/server/src/usage/usageTranscripts.ts index 199bcefba..25b108b4f 100644 --- a/apps/server/src/usage/usageTranscripts.ts +++ b/apps/server/src/usage/usageTranscripts.ts @@ -8,6 +8,13 @@ */ import type { UsageProviderKind, UsageTokenTotals } from "@t3tools/contracts"; +/** + * Billing speed of a request. Faster speeds bill at a model-specific premium. + * Claude fast mode and Codex `priority` are `fast`; Codex `ultrafast` is its + * own, more expensive tier. + */ +export type UsageSpeed = "standard" | "fast" | "ultrafast"; + export interface UsageRecord { readonly provider: UsageProviderKind; readonly timestampMs: number; @@ -15,11 +22,8 @@ export interface UsageRecord { readonly sessionId: string; readonly totals: UsageTokenTotals; readonly reportedCostUsd: number | null; - /** - * Whether the request ran in fast mode, which bills at a model-specific - * multiple of the standard rate. Only Claude Code records this. - */ - readonly fast: boolean; + /** Only Claude Code and Codex record a speed; other providers are `standard`. */ + readonly speed: UsageSpeed; /** * Key for cross-file de-duplication, or `null` when the record is inherently * unique and needs no dedup. @@ -154,7 +158,7 @@ export function parseClaudeRecord(parsed: unknown): UsageRecord | null { reasoningTokens: 0, }, reportedCostUsd: typeof cost === "number" && Number.isFinite(cost) ? cost : null, - fast: usageRecord["speed"] === "fast", + speed: usageRecord["speed"] === "fast" ? "fast" : "standard", dedupeKey, }; } @@ -166,12 +170,14 @@ export function parseClaudeRecord(parsed: unknown): UsageRecord | null { /** * Rolling state for a single Codex rollout file. * - * Codex `token_count` events carry no model, so the model is carried forward - * from the most recent `turn_context`. Sessions that switch models mid-run - * attribute correctly from the switch onward. + * Codex `token_count` events carry no model or service tier, so both are + * carried forward: the model from the most recent `turn_context`, the tier from + * the most recent `thread_settings_applied`. Sessions that switch either + * mid-run attribute correctly from the switch onward. */ export interface CodexScanState { model: string; + speed: UsageSpeed; sessionId: string; lastUsageSignature: string | null; sawSessionMeta: boolean; @@ -183,6 +189,7 @@ export interface CodexScanState { export function initialCodexScanState(): CodexScanState { return { model: "", + speed: "standard", sessionId: "", lastUsageSignature: null, sawSessionMeta: false, @@ -260,6 +267,14 @@ export function parseCodexRecord(parsed: unknown, state: CodexScanState): UsageR return null; } + if (payloadType === "thread_settings_applied") { + const settings = payloadRecord["thread_settings"]; + if (typeof settings === "object" && settings !== null) { + state.speed = codexSpeed((settings as Record)["service_tier"]); + } + return null; + } + if (payloadType !== "token_count") return null; const info = payloadRecord["info"]; @@ -318,13 +333,24 @@ export function parseCodexRecord(parsed: unknown, state: CodexScanState): UsageR totals, // Codex does not report cost in the rollout. reportedCostUsd: null, - fast: false, + speed: state.speed, // Events surviving the fork-copy suppression above are unique to this // rollout, so they need no global dedup. dedupeKey: null, }; } +/** + * Maps a Codex `service_tier` to its billing speed. Codex omits the field when + * no tier was requested, which bills as standard, as do `default` and + * `standard`. `fast` is accepted as an alias of `priority`. + */ +function codexSpeed(serviceTier: unknown): UsageSpeed { + if (serviceTier === "priority" || serviceTier === "fast") return "fast"; + if (serviceTier === "ultrafast") return "ultrafast"; + return "standard"; +} + /* -------------------------------------------------------------------------- */ /* Grok Build */ /* -------------------------------------------------------------------------- */ @@ -452,7 +478,7 @@ export function parseGrokRecord(parsed: unknown): readonly UsageRecord[] { sessionId, totals: grokTotalsToUsage(topLevel), reportedCostUsd: grokCostTicksToUsd(topLevel.costUsdTicks), - fast: false, + speed: "standard", // No prompt id means we cannot tell two same-second updates apart. dedupeKey: promptId === null ? null : `${sessionId}:${promptId}:grok`, }, @@ -499,7 +525,7 @@ export function parseGrokRecord(parsed: unknown): readonly UsageRecord[] { sessionId, totals, reportedCostUsd, - fast: false, + speed: "standard", dedupeKey: promptId === null ? null : `${sessionId}:${promptId}:${entry.model}`, }); } diff --git a/apps/web/src/components/usage/UsageModelDialog.tsx b/apps/web/src/components/usage/UsageModelDialog.tsx new file mode 100644 index 000000000..38ef1c3af --- /dev/null +++ b/apps/web/src/components/usage/UsageModelDialog.tsx @@ -0,0 +1,168 @@ +import { formatPercent, formatTokens, formatUsd } from "@t3tools/shared/usageFormat"; +import { isModelCostUnknown, type ModelTotals } from "@t3tools/shared/usageMerge"; +import { useMemo } from "react"; + +import { mergeAnsweredUsage, type EnvironmentUsageStatus } from "../../state/usage"; +import { ProviderInstanceIcon } from "../chat/ProviderInstanceIcon"; +import { Button } from "../ui/button"; +import { + Dialog, + DialogDescription, + DialogFooter, + DialogHeader, + DialogPanel, + DialogPopup, + DialogTitle, +} from "../ui/dialog"; +import { UsageProviderChart, type UsageChartMetric } from "./UsageProviderChart"; +import { UsageShareBar } from "./UsageShareBar"; +import { + cacheHitRate, + costPerMillionTokens, + costTypeSegments, + SPEED_COST_FOOTNOTE, + speedCostSegments, + tokenTypeSegments, +} from "./usageBreakdown"; +import { PROVIDER_PRESENTATION } from "./usageProviders"; + +export interface UsageChartWindow { + readonly days: readonly string[]; + readonly hours: readonly string[]; + readonly resolution: "day" | "hour"; + readonly timeZone: string; + readonly referenceTime: string | undefined; +} + +/** + * One model's usage in the current window, opened from the Breakdown list. + * The trend and mixes come from the same merge as the page, narrowed to this + * model's buckets. + */ +export function UsageModelDialog({ + model, + environments, + metric, + chartWindow, + onSetPrice, + onClose, +}: { + readonly model: ModelTotals; + readonly environments: readonly EnvironmentUsageStatus[]; + readonly metric: UsageChartMetric; + readonly chartWindow: UsageChartWindow; + readonly onSetPrice: () => void; + readonly onClose: () => void; +}) { + const usage = useMemo( + () => + mergeAnsweredUsage( + environments, + (bucket) => bucket.provider === model.provider && bucket.model === model.model, + ), + [environments, model.provider, model.model], + ); + const providers = useMemo(() => [model.provider], [model.provider]); + const presentation = PROVIDER_PRESENTATION[model.provider]; + const costUnknown = isModelCostUnknown(model); + const hitRate = cacheHitRate(model); + const perMillion = costPerMillionTokens(model); + const stats = [ + { label: "Cost", value: costUnknown ? "Unpriced" : formatUsd(model.costUsd) }, + { label: "Tokens", value: formatTokens(model.totalTokens) }, + perMillion === null ? null : { label: "Per 1M tokens", value: formatUsd(perMillion) }, + hitRate === null ? null : { label: "Cache hit", value: formatPercent(hitRate) }, + ].filter((stat) => stat !== null); + + return ( + { + if (!open) onClose(); + }} + > + + +
+ + {model.model} +
+ + {presentation.label} + {costUnknown ? "" : ` · ${formatPercent(model.costShare)} of cost`} + +
+ +
+
+ {stats.map((stat) => ( +
+ {stat.label} + {stat.value} +
+ ))} +
+ + {/* Unpriced cost is unknown, not zero, so its trend shows tokens. */} + + +
+ {costUnknown ? null : ( + + )} + + {usage.speedCost.fast + usage.speedCost.ultrafast > 0 ? ( + } + /> + ) : null} +
+
+
+ {model.unpricedTokens > 0 ? ( + + + {formatTokens(model.unpricedTokens)} tokens have no known price + + + + ) : null} +
+
+ ); +} + +/** What the faster speeds cost above standard rates, beside the speed bar. */ +export function SpeedPremium({ premiumUsd }: { readonly premiumUsd: number }) { + return ( + + Premium {formatUsd(premiumUsd)} + + ); +} diff --git a/apps/web/src/components/usage/UsagePage.test.tsx b/apps/web/src/components/usage/UsagePage.test.tsx index 321103256..01b06d393 100644 --- a/apps/web/src/components/usage/UsagePage.test.tsx +++ b/apps/web/src/components/usage/UsagePage.test.tsx @@ -96,14 +96,24 @@ const providerTotals = (codex: number, claude: number) => ["claude", { costUsd: claude, totalTokens: claude * 1_000 }], ] as const); +const tokensOf = (uncachedInputTokens: number) => ({ + uncachedInputTokens, + cachedInputTokens: 0, + cacheCreationTokens: 0, + outputTokens: 0, + reasoningTokens: 0, +}); + const modelTotals = Object.freeze([ { model: "expensive-model", provider: "claude" as const, costUsd: 10, totalTokens: 100, + tokens: tokensOf(100), records: 1, unpricedRecords: 0, + unpricedTokens: 0, costShare: 10 / 16, }, { @@ -111,8 +121,10 @@ const modelTotals = Object.freeze([ provider: "codex" as const, costUsd: 5, totalTokens: 1_000, + tokens: tokensOf(1_000), records: 1, unpricedRecords: 0, + unpricedTokens: 0, costShare: 5 / 16, }, { @@ -120,8 +132,10 @@ const modelTotals = Object.freeze([ provider: "codex" as const, costUsd: 1, totalTokens: 1_000, + tokens: tokensOf(1_000), records: 1, unpricedRecords: 0, + unpricedTokens: 0, costShare: 1 / 16, }, { @@ -129,8 +143,10 @@ const modelTotals = Object.freeze([ provider: "codex" as const, costUsd: 0, totalTokens: 500, + tokens: tokensOf(500), records: 2, unpricedRecords: 2, + unpricedTokens: 500, costShare: 0, }, ]); diff --git a/apps/web/src/components/usage/UsagePage.tsx b/apps/web/src/components/usage/UsagePage.tsx index 8081f6f3b..866ea0e4d 100644 --- a/apps/web/src/components/usage/UsagePage.tsx +++ b/apps/web/src/components/usage/UsagePage.tsx @@ -76,6 +76,15 @@ import { WorkspacePageHeader } from "../WorkspacePageHeader"; import { UsageLimitsSection } from "./UsageLimits"; import { UsagePriceOverrides } from "./UsagePriceOverrides"; import { UsageProviderChart } from "./UsageProviderChart"; +import { SpeedPremium, UsageModelDialog } from "./UsageModelDialog"; +import { UsageShareBar } from "./UsageShareBar"; +import { + costTypeSegments, + sortModelsByTokens, + SPEED_COST_FOOTNOTE, + speedCostSegments, + tokenTypeSegments, +} from "./usageBreakdown"; import { METRIC_OPTIONS, WINDOW_OPTIONS, @@ -125,6 +134,8 @@ export function UsagePage() { const [limitsNow, setLimitsNow] = useState(() => Date.now()); const refreshingRef = useRef(false); const [breakdown, setBreakdown] = useState<"model" | "time">("model"); + const [priceDialog, setPriceDialog] = useState<{ readonly model?: string } | null>(null); + const [selectedModelKey, setSelectedModelKey] = useState(null); const [selectedEnvironmentIds, setSelectedEnvironmentIds] = useState | null>(null); const { days: windowDays, window } = windowSelection; @@ -158,9 +169,7 @@ export function UsagePage() { const breakdownModels = useMemo( () => breakdown === "model" && metric === "tokens" - ? merged.models.toSorted( - (left, right) => right.totalTokens - left.totalTokens || right.costUsd - left.costUsd, - ) + ? sortModelsByTokens(merged.models) : merged.models, [breakdown, merged.models, metric], ); @@ -168,6 +177,14 @@ export function UsagePage() { // OpenCode disabled by default, so listing every known provider buries the one // real row under four "0 sessions · $0.00" rows for most installs. const activeProviders = useMemo(() => providersWithUsage(merged.providers), [merged.providers]); + const selectedModel = + selectedModelKey === null + ? undefined + : merged.models.find((model) => `${model.provider}:${model.model}` === selectedModelKey); + const breakdownPeak = breakdownModels.reduce( + (peak, model) => Math.max(peak, metric === "tokens" ? model.totalTokens : model.costUsd), + 0, + ); const timeValueColumnWidth = `${60 / (activeProviders.length + 2)}%`; const selectWindow = (days: number) => { @@ -316,6 +333,7 @@ export function UsagePage() { duplicateSources={merged.duplicateSources} contractMismatches={merged.contractMismatches} approximateEnvironments={merged.approximateEnvironments} + onOpenModelPrices={() => setPriceDialog({})} /> @@ -574,6 +592,35 @@ export function UsagePage() { + {merged.totalTokens > 0 ? ( +
+ {metric === "tokens" ? ( + + ) : ( + <> + + {merged.speedCost.fast + merged.speedCost.ultrafast > 0 ? ( + } + /> + ) : null} + + )} +
+ ) : null} +

Breakdown

@@ -600,55 +647,73 @@ export function UsagePage() {
{breakdown === "model" ? ( - - - - - - - +
- - - - - + + + + + + {breakdownModels.length === 0 ? ( - ) : ( - breakdownModels.map((model) => ( - - - - - - - )) + breakdownModels.map((model, index) => { + const key = `${model.provider}:${model.model}`; + const value = metric === "tokens" ? model.totalTokens : model.costUsd; + return ( + + + + + + + + ); + }) )}
ModelCostShareTokens
#ModelCostShareTokens
+ No activity in this window.
- - - {model.model} - - - {isModelCostUnknown(model) ? ( - Unpriced - ) : ( - formatUsd(model.costUsd) - )} - - {isModelCostUnknown(model) ? "—" : formatPercent(model.costShare)} - - {formatTokens(model.totalTokens)} -
{index + 1} + {/* The button's overlay makes the whole row open the model. + Focus shows as the row's hover fill, not a ring. */} + +
+
0 && breakdownPeak > 0 + ? `max(0.5rem, ${(value / breakdownPeak) * 100}%)` + : 0, + backgroundColor: + PROVIDER_PRESENTATION[model.provider].color, + }} + /> +
+
+ {isModelCostUnknown(model) ? ( + Unpriced + ) : ( + formatUsd(model.costUsd) + )} + + {isModelCostUnknown(model) ? "" : formatPercent(model.costShare)} + {formatTokens(model.totalTokens)}
@@ -721,6 +786,35 @@ export function UsagePage() { + {selectedModel !== undefined && !showingLimits ? ( + { + setSelectedModelKey(null); + setPriceDialog({ model: selectedModel.model }); + }} + onClose={() => setSelectedModelKey(null)} + /> + ) : null} + {priceDialog ? ( + { + if (!open) setPriceDialog(null); + }} + /> + ) : null} ); } @@ -833,6 +927,7 @@ function UsageEnvironmentFilter({ duplicateSources, contractMismatches, approximateEnvironments, + onOpenModelPrices, }: { readonly environments: readonly EnvironmentUsageStatus[]; readonly selectedEnvironments: readonly EnvironmentUsageStatus[]; @@ -843,8 +938,8 @@ function UsageEnvironmentFilter({ readonly duplicateSources: readonly string[]; readonly contractMismatches: MergedUsage["contractMismatches"]; readonly approximateEnvironments: readonly string[]; + readonly onOpenModelPrices: () => void; }) { - const [modelPricesOpen, setModelPricesOpen] = useState(false); const allSelected = selectedEnvironmentIds === null; const label = allSelected ? "All environments" @@ -869,126 +964,116 @@ function UsageEnvironmentFilter({ ); return ( - <> - - - {label} - - {showUsageStatus && pendingCount > 0 ? ( - <> - - - {pendingCount} {pendingCount === 1 ? "environment" : "environments"} still - scanning - {isPartial ? "; totals are partial" : ""} - - - ) : showUsageStatus && hasIssue ? ( - 0 - ? "Some usage sources could not report complete usage" - : "Some environments could not report usage" - } - /> - ) : ( - - )} - - - - onSelectionChange(checked ? null : new Set())} - > - All environments - - - {environments.map((environment) => { - const checked = - selectedEnvironmentIds === null || - selectedEnvironmentIds.has(environment.environmentId); - const status = - environment.error !== null - ? "Unavailable" - : environment.summary !== null && - !isCompatibleUsageContractVersion( - environment.summary.contractVersion, - USAGE_CONTRACT_VERSION, - ) - ? "Update required" - : environment.summary === null - ? "Scanning…" - : environment.isPending - ? "Refreshing…" - : collectUsageSourceWarnings([environment]).length > 0 - ? "Partial history" - : "Ready"; - return ( - { - const next = new Set(selectedEnvironments.map((entry) => entry.environmentId)); - if (nextChecked) next.add(environment.environmentId); - else next.delete(environment.environmentId); - onSelectionChange(next.size === environments.length ? null : next); - }} - > - - {environment.label} - {showUsageStatus ? ( - - {status} - - ) : null} - - - ); - })} - {environments.length === 0 ? ( -

No environments connected.

- ) : null} - {showUsageStatus && isPartial ? ( -

- Totals are partial while selected environments scan. -

- ) : null} - {showUsageStatus ? ( - + + {label} + + {showUsageStatus && pendingCount > 0 ? ( + <> + + + {pendingCount} {pendingCount === 1 ? "environment" : "environments"} still scanning + {isPartial ? "; totals are partial" : ""} + + + ) : showUsageStatus && hasIssue ? ( + 0 + ? "Some usage sources could not report complete usage" + : "Some environments could not report usage" + } /> - ) : null} - - setModelPricesOpen(true)}> - - Model prices - -
-
- {modelPricesOpen ? ( - - ) : null} - + ) : ( + + )} + + + + onSelectionChange(checked ? null : new Set())} + > + All environments + + + {environments.map((environment) => { + const checked = + selectedEnvironmentIds === null || + selectedEnvironmentIds.has(environment.environmentId); + const status = + environment.error !== null + ? "Unavailable" + : environment.summary !== null && + !isCompatibleUsageContractVersion( + environment.summary.contractVersion, + USAGE_CONTRACT_VERSION, + ) + ? "Update required" + : environment.summary === null + ? "Scanning…" + : environment.isPending + ? "Refreshing…" + : collectUsageSourceWarnings([environment]).length > 0 + ? "Partial history" + : "Ready"; + return ( + { + const next = new Set(selectedEnvironments.map((entry) => entry.environmentId)); + if (nextChecked) next.add(environment.environmentId); + else next.delete(environment.environmentId); + onSelectionChange(next.size === environments.length ? null : next); + }} + > + + {environment.label} + {showUsageStatus ? ( + + {status} + + ) : null} + + + ); + })} + {environments.length === 0 ? ( +

No environments connected.

+ ) : null} + {showUsageStatus && isPartial ? ( +

+ Totals are partial while selected environments scan. +

+ ) : null} + {showUsageStatus ? ( + + ) : null} + + + + Model prices + +
+ ); } diff --git a/apps/web/src/components/usage/UsagePriceOverrides.tsx b/apps/web/src/components/usage/UsagePriceOverrides.tsx index bb57c3acc..50825754b 100644 --- a/apps/web/src/components/usage/UsagePriceOverrides.tsx +++ b/apps/web/src/components/usage/UsagePriceOverrides.tsx @@ -34,12 +34,14 @@ import { Tooltip, TooltipTrigger, TooltipPopup } from "../ui/tooltip"; import { USAGE_PRICE_FIELDS } from "./usagePriceForm"; import { isEmptyUsagePriceDraft, + PREFILLED_PRICE_DRAFT_ID, usagePriceCell, usagePriceRows, usagePriceTableChanges, usagePriceTableErrors, type UsagePriceDraft, type UsagePriceField, + withoutSupersededPrefill, } from "./usagePriceTable"; import { writeUsagePrices, @@ -92,13 +94,16 @@ type SaveAttempt = { >; }; +/** Edits custom model prices. `initialModel` opens with a new row for that model. */ export function UsagePriceOverrides({ usage, initialSelectedEnvironmentIds, + initialModel, onOpenChange, }: { readonly usage: readonly EnvironmentUsageStatus[]; readonly initialSelectedEnvironmentIds: ReadonlySet | null; + readonly initialModel?: string | undefined; readonly onOpenChange: (open: boolean) => void; }) { const environments = useAtomValue(priceTargetsAtom); @@ -106,7 +111,11 @@ export function UsagePriceOverrides({ const selected = environments.filter( (environment) => selectedIds === null || selectedIds.has(environment.environmentId), ); - const [drafts, setDrafts] = useState([]); + const [draftState, setDrafts] = useState(() => + initialModel === undefined + ? [] + : [{ id: PREFILLED_PRICE_DRAFT_ID, model: initialModel, isNew: true, values: {} }], + ); const [pending, setPending] = useState(false); const [attempt, setAttempt] = useState(null); const focusRowRef = useRef(null); @@ -125,6 +134,8 @@ export function UsagePriceOverrides({ .flatMap((environment) => environment.summary?.buckets.map((bucket) => bucket.model) ?? []), ]), ].sort(); + // A model that already has a custom price somewhere is edited in its existing row. + const drafts = withoutSupersededPrefill(draftState, customModels); const rows = usagePriceRows(customModels, drafts, { saving: attempt !== null }); const stagedDrafts = drafts.filter((draft) => !isEmptyUsagePriceDraft(draft)); const errors = usagePriceTableErrors(selected, stagedDrafts); diff --git a/apps/web/src/components/usage/UsageShareBar.tsx b/apps/web/src/components/usage/UsageShareBar.tsx new file mode 100644 index 000000000..55ca4cce3 --- /dev/null +++ b/apps/web/src/components/usage/UsageShareBar.tsx @@ -0,0 +1,76 @@ +import { formatPercent } from "@t3tools/shared/usageFormat"; +import type { ReactNode } from "react"; + +import { Tooltip, TooltipPopup, TooltipTrigger } from "../ui/tooltip"; + +export interface ShareSegment { + readonly label: string; + readonly value: number; + readonly color: string; +} + +/** + * One part-to-whole bar with its legend, for cost or tokens split by type or + * speed. Empty segments are left out, and nothing renders without a total. + */ +export function UsageShareBar({ + label, + segments, + format, + aside, + footnote, +}: { + readonly label: string; + readonly segments: readonly ShareSegment[]; + readonly format: (value: number) => string; + readonly aside?: ReactNode; + readonly footnote?: string; +}) { + const visible = segments.filter((segment) => segment.value > 0); + const total = visible.reduce((sum, segment) => sum + segment.value, 0); + if (total <= 0) return null; + + return ( +
+
+

{label}

+ {aside} +
+
`${segment.label} ${format(segment.value)}`).join(", ")}`} + className="flex h-2 gap-0.5" + > + {visible.map((segment) => ( + + + } + /> + + {segment.label} · {format(segment.value)} · {formatPercent(segment.value / total)} + + + ))} +
+
+ {visible.map((segment) => ( + + + {segment.label} + {format(segment.value)} + + ))} +
+ {footnote ?

{footnote}

: null} +
+ ); +} diff --git a/apps/web/src/components/usage/usageBreakdown.test.ts b/apps/web/src/components/usage/usageBreakdown.test.ts new file mode 100644 index 000000000..e4463e8ad --- /dev/null +++ b/apps/web/src/components/usage/usageBreakdown.test.ts @@ -0,0 +1,114 @@ +import type { ModelTotals } from "@t3tools/shared/usageMerge"; +import { describe, expect, it } from "vite-plus/test"; + +import { + cacheHitRate, + costPerMillionTokens, + costTypeSegments, + sortModelsByTokens, + speedCostSegments, + tokenTypeSegments, +} from "./usageBreakdown"; + +const model = ( + name: string, + totalTokens: number, + costUsd: number, + overrides: Partial = {}, +): ModelTotals => ({ + model: name, + provider: "codex", + costUsd, + totalTokens, + tokens: { + uncachedInputTokens: totalTokens, + cachedInputTokens: 0, + cacheCreationTokens: 0, + outputTokens: 0, + reasoningTokens: 0, + }, + records: 1, + unpricedRecords: 0, + unpricedTokens: 0, + costShare: 0, + ...overrides, +}); + +describe("sortModelsByTokens", () => { + it("sorts by tokens, breaks ties by cost, and leaves the input alone", () => { + const models = [ + model("lower-cost", 100, 1), + model("more-tokens", 200, 2), + model("higher-cost", 100, 3), + ]; + + expect(sortModelsByTokens(models).map((item) => item.model)).toEqual([ + "more-tokens", + "higher-cost", + "lower-cost", + ]); + expect(models.map((item) => item.model)).toEqual(["lower-cost", "more-tokens", "higher-cost"]); + }); +}); + +describe("model rates", () => { + it("counts cache writes as misses and leaves unpriced tokens out of $/1M", () => { + const mixed = model("mixed", 4_000_000, 6, { + tokens: { + uncachedInputTokens: 1_000_000, + cachedInputTokens: 1_000_000, + cacheCreationTokens: 2_000_000, + outputTokens: 0, + reasoningTokens: 0, + }, + records: 4, + unpricedRecords: 1, + unpricedTokens: 1_000_000, + }); + + expect(cacheHitRate(mixed)).toBe(0.25); + expect(costPerMillionTokens(mixed)).toBe(2); + expect(costPerMillionTokens({ ...mixed, unpricedRecords: 4 })).toBeNull(); + }); +}); + +describe("share segments", () => { + const values = (segments: readonly { label: string; value: number }[]) => + segments.map(({ label, value }) => [label, value]); + + it("splits cost by type, treating sub-cent unsplit cost as rounding", () => { + const cost = { input: 1, cacheRead: 2, cacheWrite: 3, output: 4, unsplit: 0.004 }; + + expect(values(costTypeSegments(cost))).toEqual([ + ["Input", 1], + ["Cache read", 2], + ["Cache write", 3], + ["Output", 4], + ["Other", 0], + ]); + expect(costTypeSegments({ ...cost, unsplit: 5 }).at(-1)?.value).toBe(5); + }); + + it("orders speeds by price and tokens by type", () => { + expect(values(speedCostSegments({ standard: 9, fast: 6, ultrafast: 2, premium: 4 }))).toEqual([ + ["Standard", 9], + ["Fast", 6], + ["Ultrafast", 2], + ]); + expect( + values( + tokenTypeSegments({ + uncachedInputTokens: 1, + cachedInputTokens: 2, + cacheCreationTokens: 3, + outputTokens: 4, + }), + ), + ).toEqual([ + ["Input", 1], + ["Cache read", 2], + ["Cache write", 3], + ["Output", 4], + ]); + }); +}); diff --git a/apps/web/src/components/usage/usageBreakdown.ts b/apps/web/src/components/usage/usageBreakdown.ts new file mode 100644 index 000000000..cc9e53112 --- /dev/null +++ b/apps/web/src/components/usage/usageBreakdown.ts @@ -0,0 +1,84 @@ +import type { UsageTokenTotals } from "@t3tools/contracts"; +import { + isModelCostUnknown, + type CategoryCost, + type ModelTotals, + type SpeedCost, +} from "@t3tools/shared/usageMerge"; + +import type { ShareSegment } from "./UsageShareBar"; + +export function sortModelsByTokens(models: readonly ModelTotals[]) { + return models.toSorted( + (left, right) => right.totalTokens - left.totalTokens || right.costUsd - left.costUsd, + ); +} + +/** + * Share of a model's input read from cache, or `null` without input. Cache + * writes count as misses: that input was processed in full. + */ +export function cacheHitRate({ tokens }: ModelTotals): number | null { + const input = tokens.uncachedInputTokens + tokens.cachedInputTokens + tokens.cacheCreationTokens; + return input === 0 ? null : tokens.cachedInputTokens / input; +} + +/** Effective USD per million priced tokens, or `null` when none were priced. */ +export function costPerMillionTokens(model: ModelTotals): number | null { + const pricedTokens = model.totalTokens - model.unpricedTokens; + return pricedTokens <= 0 || isModelCostUnknown(model) + ? null + : (model.costUsd / pricedTokens) * 1_000_000; +} + +/** A neutral step between background and foreground, so mixes never borrow a provider's color. */ +const ink = (percent: number) => + `color-mix(in oklab, var(--foreground) ${percent}%, var(--background))`; + +/** Adjacent segments stay above the 15 ΔE separation floor in both themes. */ +const TYPE_COLORS = { + input: ink(60), + cacheRead: ink(30), + cacheWrite: ink(72), + output: ink(100), + other: ink(44), +}; + +export function costTypeSegments(cost: CategoryCost): readonly ShareSegment[] { + return [ + { label: "Input", value: cost.input, color: TYPE_COLORS.input }, + { label: "Cache read", value: cost.cacheRead, color: TYPE_COLORS.cacheRead }, + { label: "Cache write", value: cost.cacheWrite, color: TYPE_COLORS.cacheWrite }, + { label: "Output", value: cost.output, color: TYPE_COLORS.output }, + // Reported cost with no rates to split it, or from older servers. Below a + // cent it is rounding, not usage. + { label: "Other", value: cost.unsplit >= 0.005 ? cost.unsplit : 0, color: TYPE_COLORS.other }, + ]; +} + +export function tokenTypeSegments( + tokens: Omit, +): readonly ShareSegment[] { + return [ + { label: "Input", value: tokens.uncachedInputTokens, color: TYPE_COLORS.input }, + { label: "Cache read", value: tokens.cachedInputTokens, color: TYPE_COLORS.cacheRead }, + { label: "Cache write", value: tokens.cacheCreationTokens, color: TYPE_COLORS.cacheWrite }, + { label: "Output", value: tokens.outputTokens, color: TYPE_COLORS.output }, + ]; +} + +/** + * Servers from before the speed split report no speed figures, and the merge + * counts all of their cost as standard. + */ +export const SPEED_COST_FOOTNOTE = + "Servers that predate speed tracking count all cost as Standard."; + +/** Speeds are ordered by price, so they brighten from standard to ultrafast. */ +export function speedCostSegments(cost: SpeedCost): readonly ShareSegment[] { + return [ + { label: "Standard", value: cost.standard, color: ink(34) }, + { label: "Fast", value: cost.fast, color: ink(66) }, + { label: "Ultrafast", value: cost.ultrafast, color: ink(100) }, + ]; +} diff --git a/apps/web/src/components/usage/usagePriceTable.test.ts b/apps/web/src/components/usage/usagePriceTable.test.ts index a01eaa1c0..deba2ce38 100644 --- a/apps/web/src/components/usage/usagePriceTable.test.ts +++ b/apps/web/src/components/usage/usagePriceTable.test.ts @@ -6,6 +6,8 @@ import { usagePriceTableChanges, usagePriceTableErrors, type UsagePriceDraft, + PREFILLED_PRICE_DRAFT_ID, + withoutSupersededPrefill, } from "./usagePriceTable"; import type { UsagePriceTarget } from "./usagePriceTargets"; @@ -209,3 +211,23 @@ describe("price table rows", () => { ]); }); }); + +describe("prefilled Set price draft", () => { + const prefill = (values: Record = {}) => ({ + id: PREFILLED_PRICE_DRAFT_ID, + model: "custom-model", + isNew: true as const, + values, + }); + + it("yields to an existing row once prices load, unless it was edited", () => { + // Prices not loaded yet: the prefilled row shows. + expect(withoutSupersededPrefill([prefill()], [])).toHaveLength(1); + // Prices loaded with an override for that model: edit the existing row instead. + expect(withoutSupersededPrefill([prefill()], ["custom-model"])).toEqual([]); + // Typed rates are never discarded; the duplicate-row error explains instead. + expect( + withoutSupersededPrefill([prefill({ inputCostPerMillionTokens: "2" })], ["custom-model"]), + ).toHaveLength(1); + }); +}); diff --git a/apps/web/src/components/usage/usagePriceTable.ts b/apps/web/src/components/usage/usagePriceTable.ts index 517494524..34efd0e9b 100644 --- a/apps/web/src/components/usage/usagePriceTable.ts +++ b/apps/web/src/components/usage/usagePriceTable.ts @@ -24,6 +24,26 @@ export function isEmptyUsagePriceDraft(draft: UsagePriceDraft) { ); } +/** Row id of the draft a "Set price" shortcut opens the table with. */ +export const PREFILLED_PRICE_DRAFT_ID = "new:initial"; + +/** + * Hides the prefilled draft while it is untouched and its model already has a + * custom row. Prices can load after the dialog opens, so this is decided on + * every render rather than once when the draft is created. + */ +export function withoutSupersededPrefill( + drafts: readonly UsagePriceDraft[], + customModels: readonly string[], +): readonly UsagePriceDraft[] { + return drafts.filter( + (draft) => + draft.id !== PREFILLED_PRICE_DRAFT_ID || + !customModels.includes(draft.model.trim()) || + Object.values(draft.values).some((value) => (value ?? "").trim() !== ""), + ); +} + function modelPrice(target: UsagePriceTarget, model: string) { return target.prices && Object.hasOwn(target.prices, model) ? target.prices[model] : undefined; } diff --git a/apps/web/src/state/usage.test.tsx b/apps/web/src/state/usage.test.tsx index 20eb78cdf..6052a283f 100644 --- a/apps/web/src/state/usage.test.tsx +++ b/apps/web/src/state/usage.test.tsx @@ -3,7 +3,7 @@ import { act, useLayoutEffect } from "react"; import { create, type ReactTestRenderer } from "react-test-renderer"; import { afterEach, beforeEach, describe, expect, it, vi } from "vite-plus/test"; -import { useUsage, type EnvironmentUsageStatus, type UsageView } from "./usage"; +import { mergeAnsweredUsage, useUsage, type EnvironmentUsageStatus, type UsageView } from "./usage"; const testState = vi.hoisted(() => ({ environments: [] as EnvironmentUsageStatus[] })); const rateRefresh = vi.hoisted(() => ({ @@ -231,3 +231,26 @@ describe("usage environment selection", () => { expect(latest.isPartial).toBe(false); }); }); + +describe("mergeAnsweredUsage", () => { + it("narrows per-source buckets to one model, as the model dialog does", () => { + const base = environment("a", 1); + const summary = base.summary!; + const modelA = summary.buckets[0]!; + const modelB = { ...modelA, model: "b", costUsd: 5 }; + const status: EnvironmentUsageStatus = { + ...base, + summary: { + ...summary, + buckets: [modelA, modelB], + // v7+ servers attribute buckets per source; the merge reads these. + sources: summary.sources.map((source) => ({ ...source, buckets: [modelA, modelB] })), + }, + }; + + expect(mergeAnsweredUsage([status]).costUsd).toBe(6); + const onlyA = mergeAnsweredUsage([status], (bucket) => bucket.model === "a"); + expect(onlyA.costUsd).toBe(1); + expect(onlyA.models.map((model) => model.model)).toEqual(["a"]); + }); +}); diff --git a/apps/web/src/state/usage.ts b/apps/web/src/state/usage.ts index 8b4ebabf3..3abf4b369 100644 --- a/apps/web/src/state/usage.ts +++ b/apps/web/src/state/usage.ts @@ -10,6 +10,7 @@ import { useAtomValue } from "@effect/atom-react"; import { USAGE_CONTRACT_VERSION, type EnvironmentId, + type UsageBucket, type UsageSummary, type UsageSummaryInput, } from "@t3tools/contracts"; @@ -18,7 +19,12 @@ import * as Option from "effect/Option"; import { AsyncResult, Atom } from "effect/unstable/reactivity"; import { useCallback, useMemo, useRef, useState } from "react"; -import { mergeUsage, type EnvironmentUsage, type MergedUsage } from "@t3tools/shared/usageMerge"; +import { + mergeUsage, + narrowUsageSummary, + type EnvironmentUsage, + type MergedUsage, +} from "@t3tools/shared/usageMerge"; import { appAtomRegistry } from "../rpc/atomRegistry"; import { environmentPresentations } from "./presentation"; import { serverEnvironment } from "./server"; @@ -73,6 +79,30 @@ export interface UsageView { readonly refresh: (input?: UsageSummaryInput) => Promise; } +/** + * Merges every environment that has answered. `keepBucket` narrows the merge, + * for example to one model; source ownership still applies, so the result + * matches that slice of the full merge. Session counts are per directory and + * are not narrowed. + */ +export function mergeAnsweredUsage( + environments: readonly EnvironmentUsageStatus[], + keepBucket?: (bucket: UsageBucket) => boolean, +): MergedUsage { + const answered: EnvironmentUsage[] = environments.flatMap(({ environmentId, label, summary }) => + summary === null + ? [] + : [ + { + environmentId, + label, + summary: keepBucket === undefined ? summary : narrowUsageSummary(summary, keepBucket), + }, + ], + ); + return mergeUsage(answered, USAGE_CONTRACT_VERSION); +} + export function useUsage( input: UsageSummaryInput, selectedEnvironmentIds: ReadonlySet | null = null, @@ -118,20 +148,7 @@ export function useUsage( [selectedEnvironments, windowKey], ); - const merged = useMemo(() => { - const answered: EnvironmentUsage[] = selectedEnvironments.flatMap((environment) => - environment.summary === null - ? [] - : [ - { - environmentId: environment.environmentId, - label: environment.label, - summary: environment.summary, - }, - ], - ); - return mergeUsage(answered, USAGE_CONTRACT_VERSION); - }, [selectedEnvironments]); + const merged = useMemo(() => mergeAnsweredUsage(selectedEnvironments), [selectedEnvironments]); const answeredCount = selectedEnvironments.filter( (environment) => environment.summary !== null, diff --git a/docs/user/usage.md b/docs/user/usage.md index 4e8daacec..f11e8e23a 100644 --- a/docs/user/usage.md +++ b/docs/user/usage.md @@ -8,7 +8,11 @@ desktop when the terminal is not focused. Customize `usage.open` in **Usage** combines Codex, Claude Code, Grok Build, Antigravity, and local OpenCode session history from your connected environments. It shows token use, cache savings, provider shares, model breakdowns, and estimated -API-equivalent cost. These estimates are not your subscription bill. +API-equivalent cost, split by token type and by speed. These estimates are not your subscription bill. +**Premium** is what Fast and Ultrafast requests cost above standard rates. Servers that predate +speed tracking report no speed split, so all of their cost counts as Standard. Cost that cannot be +split, such as a provider-reported cost for a model without public rates, shows as **Other**. +Select a model under **Breakdown** to see its trend, cache hit rate, and cost per million tokens. Totals depend on the history available on each server. Grok turns without a saved completed-turn record are missing from the totals. @@ -48,6 +52,8 @@ On web or desktop, open the environment dropdown on **Usage**, then choose **Mod edit, or reset a model's estimated price. **Apply to** starts with your current Usage filter; choose all environments or individual destinations. Enter the exact model ID and USD rates per million input and output tokens, including for models without public pricing. +When a model on **Usage** has no known price, select it under **Breakdown** and choose +**Set price** to open this table with that model added. Cache read and cache write rates are optional and use the input rate when blank. Enter `0` for tokens that are free. Saved prices replace automatic pricing for all of that environment's history and are diff --git a/packages/contracts/src/usage.ts b/packages/contracts/src/usage.ts index 44fd05310..447a1ce5e 100644 --- a/packages/contracts/src/usage.ts +++ b/packages/contracts/src/usage.ts @@ -22,6 +22,7 @@ import { ForwardCompatibleArray, NonNegativeInt, TrimmedNonEmptyString } from ". * rather than failing the whole page. * New provider and bucket literal variants are additive and can be skipped by * older clients without changing the version. + * Optional bucket fields are additive too: older clients ignore them. */ export const USAGE_CONTRACT_VERSION = 8 as const; @@ -109,6 +110,18 @@ export const UsageTokenTotals = Schema.Struct({ }); export type UsageTokenTotals = typeof UsageTokenTotals.Type; +/** + * A bucket's cost split by token category, in USD. A provider-reported cost is + * split in proportion to the model's list rates. + */ +export const UsageCategoryCost = Schema.Struct({ + input: Schema.Number, + cacheRead: Schema.Number, + cacheWrite: Schema.Number, + output: Schema.Number, +}); +export type UsageCategoryCost = typeof UsageCategoryCost.Type; + /** * One `(day, hourStart?, provider, model)` cell. `hourStart` is the UTC start * instant of a rolling bucket and is present only for hourly requests. @@ -131,6 +144,16 @@ export const UsageBucket = Schema.Struct({ * rather than derived on the client. */ cacheSavingsUsd: Schema.Number, + /** + * `costUsd` by token category. Cost with no known rates stays out of it, and + * it is absent when nothing could be split or the server predates it. + */ + categoryCostUsd: Schema.optional(UsageCategoryCost), + /** Cost of fast and ultrafast requests. Absent when zero; the rest is standard. */ + fastCostUsd: Schema.optional(Schema.Number), + ultrafastCostUsd: Schema.optional(Schema.Number), + /** What fast and ultrafast requests cost above the standard rate. Absent when zero. */ + speedPremiumUsd: Schema.optional(Schema.Number), costSource: UsageCostSource, /** Distinct assistant responses, after de-duplication. */ records: NonNegativeInt, diff --git a/packages/shared/src/usageMerge.test.ts b/packages/shared/src/usageMerge.test.ts index c46435956..ce3d95cdd 100644 --- a/packages/shared/src/usageMerge.test.ts +++ b/packages/shared/src/usageMerge.test.ts @@ -10,7 +10,12 @@ import { import { describe, expect, it } from "vite-plus/test"; import * as Schema from "effect/Schema"; -import { isModelCostUnknown, mergeUsage, type EnvironmentUsage } from "./usageMerge.ts"; +import { + isModelCostUnknown, + mergeUsage, + narrowUsageSummary, + type EnvironmentUsage, +} from "./usageMerge.ts"; const decodeSummary = Schema.decodeUnknownSync(UsageSummary); const encodeSummary = Schema.encodeSync(UsageSummary); @@ -582,6 +587,117 @@ describe("mergeUsage", () => { ]); }); + it("narrows per-source buckets so a one-model slice excludes other models", () => { + const a = bucket({ model: "model-a", costUsd: 1 }); + const b = bucket({ model: "model-b", costUsd: 5 }); + const full = summary( + [a, b], + [{ provider: "claude", hostId: "mac", homePath: "/a/.claude", buckets: [a, b] }], + ); + const onlyA = (usage: typeof full) => + mergeUsage( + [ + environment( + "env-a", + narrowUsageSummary(usage, (entry) => entry.model === "model-a"), + ), + ], + USAGE_CONTRACT_VERSION, + ); + + // v7+ servers: the merge reads the per-source buckets. + expect(onlyA(full).costUsd).toBe(1); + expect(onlyA(full).models.map((model) => model.model)).toEqual(["model-a"]); + // Older servers without per-source buckets narrow through the provider-wide list. + const legacy = summary([a, b], [{ provider: "claude", hostId: "mac", homePath: "/a/.claude" }]); + expect(onlyA(legacy).costUsd).toBe(1); + expect(narrowUsageSummary(legacy, () => true).sources[0]).not.toHaveProperty("buckets"); + }); + + it("splits cost by category and speed, counting older servers as unsplit standard cost", () => { + const merged = mergeUsage( + [ + environment( + "env-a", + summary( + [ + bucket({ + costUsd: 10, + categoryCostUsd: { input: 1, cacheRead: 2, cacheWrite: 3, output: 4 }, + fastCostUsd: 6, + speedPremiumUsd: 3, + }), + bucket({ + provider: "codex", + model: "unknown-model", + costUsd: 0, + costSource: "unpriced", + unpricedRecords: 5, + }), + // Reported cost on one record, no rates for the other four. + bucket({ + provider: "codex", + model: "partly-reported", + costUsd: 0, + unpricedRecords: 4, + }), + ], + [ + { provider: "claude", hostId: "mac", homePath: "/a/.claude" }, + { provider: "codex", hostId: "mac", homePath: "/a/.codex" }, + ], + ), + ), + environment( + "env-b", + summary( + [bucket({ costUsd: 5 })], + [{ provider: "claude", hostId: "linux", homePath: "/b/.claude" }], + USAGE_MERGE_COMPATIBLE_SINCE, + ), + ), + ], + USAGE_CONTRACT_VERSION, + ); + + expect(merged.categoryCost).toEqual({ + input: 1, + cacheRead: 2, + cacheWrite: 3, + output: 4, + unsplit: 5, + }); + expect(merged.speedCost).toEqual({ standard: 9, fast: 6, ultrafast: 0, premium: 3 }); + expect(merged.models.map(({ model, unpricedTokens }) => [model, unpricedTokens])).toEqual([ + ["claude-fable-5", 0], + ["unknown-model", 1160], + ["partly-reported", 928], + ]); + }); + + it("orders models by cost descending", () => { + const merged = mergeUsage( + [ + environment( + "env-a", + summary( + [ + bucket({ provider: "claude", model: "lower-cost", costUsd: 4 }), + bucket({ provider: "codex", model: "higher-cost", costUsd: 9 }), + ], + [ + { provider: "claude", hostId: "mac", homePath: "/a/.claude" }, + { provider: "codex", hostId: "mac", homePath: "/a/.codex" }, + ], + ), + ), + ], + USAGE_CONTRACT_VERSION, + ); + + expect(merged.models.map((model) => model.model)).toEqual(["higher-cost", "lower-cost"]); + }); + it("keeps two machines apart when hostname and home path collide", () => { // Every Mac resolves /Users/theo/.claude, so a hostname clash used to make // one machine's usage vanish. Filesystem identity separates them. diff --git a/packages/shared/src/usageMerge.ts b/packages/shared/src/usageMerge.ts index 14e85e059..43c0158b1 100644 --- a/packages/shared/src/usageMerge.ts +++ b/packages/shared/src/usageMerge.ts @@ -14,6 +14,7 @@ import { type UsageSource, type UsageSourceFingerprint, type UsageSummary, + type UsageTokenTotals, } from "@t3tools/contracts"; export interface EnvironmentUsage { @@ -37,12 +38,18 @@ export interface ModelTotals { readonly provider: UsageProviderKind; readonly costUsd: number; readonly totalTokens: number; + readonly tokens: UsageTokenTotals; readonly records: number; /** * Records whose tokens are counted here but which contributed nothing to * `costUsd`. When it equals `records` the cost is unknown, not zero. */ readonly unpricedRecords: number; + /** + * Tokens with no known rates, which a custom price would cover. A cell that + * mixes these with reported costs counts its tokens by record share. + */ + readonly unpricedTokens: number; readonly costShare: number; } @@ -76,6 +83,27 @@ export interface CostQuality { readonly cacheSavingsUsd: number; } +/** + * `costUsd` by token category. `unsplit` is cost no rates could split, + * including all cost from servers that predate the split. + */ +export interface CategoryCost { + readonly input: number; + readonly cacheRead: number; + readonly cacheWrite: number; + readonly output: number; + readonly unsplit: number; +} + +/** `costUsd` by request speed. Servers that predate speeds count as standard. */ +export interface SpeedCost { + readonly standard: number; + readonly fast: number; + readonly ultrafast: number; + /** What fast and ultrafast requests cost above the standard rate. */ + readonly premium: number; +} + export interface UsageContractMismatch { readonly environmentId: EnvironmentId; readonly direction: "serverBehind" | "clientBehind"; @@ -97,6 +125,8 @@ export interface MergedUsage { readonly daily: readonly DailyTotals[]; readonly hourly: readonly HourlyTotals[]; readonly costQuality: CostQuality; + readonly categoryCost: CategoryCost; + readonly speedCost: SpeedCost; /** Environments whose data was dropped as a duplicate of another's. */ readonly duplicateSources: readonly string[]; readonly contributingEnvironments: readonly EnvironmentId[]; @@ -416,6 +446,24 @@ function claimSources(environments: readonly EnvironmentUsage[]): { }; } +/** + * Narrows a summary to the buckets `keep` accepts, for example one model. + * Per-source buckets (v7+) are narrowed too: the merge prefers them over the + * provider-wide list, so filtering only `buckets` would leave the slice whole. + */ +export function narrowUsageSummary( + summary: UsageSummary, + keep: (bucket: UsageBucket) => boolean, +): UsageSummary { + return { + ...summary, + buckets: summary.buckets.filter(keep), + sources: summary.sources.map((source) => + source.buckets === undefined ? source : { ...source, buckets: source.buckets.filter(keep) }, + ), + }; +} + /** Sources this environment owns after fingerprint claims, plus their buckets. */ function ownedContribution( environment: EnvironmentUsage, @@ -518,6 +566,8 @@ const EMPTY_MERGED: MergedUsage = { unpricedShare: 0, cacheSavingsUsd: 0, }, + categoryCost: { input: 0, cacheRead: 0, cacheWrite: 0, output: 0, unsplit: 0 }, + speedCost: { standard: 0, fast: 0, ultrafast: 0, premium: 0 }, duplicateSources: [], contributingEnvironments: [], contractMismatches: [], @@ -577,6 +627,8 @@ export function mergeUsage( let cacheSavingsUsd = 0; let providerReportedRecords = 0; let unpricedRecords = 0; + const categoryCost = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 }; + const speedCost = { fast: 0, ultrafast: 0, premium: 0 }; const providerAccumulator = new Map< UsageProviderKind, @@ -588,8 +640,10 @@ export function mergeUsage( provider: UsageProviderKind; costUsd: number; totalTokens: number; + tokens: UsageTokenTotals; records: number; unpricedRecords: number; + unpricedTokens: number; } >(); const dailyAccumulator = new Map< @@ -649,6 +703,15 @@ export function mergeUsage( records += bucket.records; unpricedRecords += bucket.unpricedRecords; if (bucket.costSource === "providerReported") providerReportedRecords += bucket.records; + if (bucket.categoryCostUsd !== undefined) { + categoryCost.input += bucket.categoryCostUsd.input; + categoryCost.cacheRead += bucket.categoryCostUsd.cacheRead; + categoryCost.cacheWrite += bucket.categoryCostUsd.cacheWrite; + categoryCost.output += bucket.categoryCostUsd.output; + } + speedCost.fast += bucket.fastCostUsd ?? 0; + speedCost.ultrafast += bucket.ultrafastCostUsd ?? 0; + speedCost.premium += bucket.speedPremiumUsd ?? 0; const provider = providerAccumulator.get(bucket.provider) ?? { costUsd: 0, @@ -666,13 +729,31 @@ export function mergeUsage( provider: bucket.provider, costUsd: 0, totalTokens: 0, + tokens: { + uncachedInputTokens: 0, + cachedInputTokens: 0, + cacheCreationTokens: 0, + outputTokens: 0, + reasoningTokens: 0, + }, records: 0, unpricedRecords: 0, + unpricedTokens: 0, }; model.costUsd += bucket.costUsd; model.totalTokens += tokens; + model.tokens = { + uncachedInputTokens: model.tokens.uncachedInputTokens + bucket.totals.uncachedInputTokens, + cachedInputTokens: model.tokens.cachedInputTokens + bucket.totals.cachedInputTokens, + cacheCreationTokens: model.tokens.cacheCreationTokens + bucket.totals.cacheCreationTokens, + outputTokens: model.tokens.outputTokens + bucket.totals.outputTokens, + reasoningTokens: model.tokens.reasoningTokens + bucket.totals.reasoningTokens, + }; model.records += bucket.records; model.unpricedRecords += bucket.unpricedRecords; + if (bucket.records > 0) { + model.unpricedTokens += (tokens * bucket.unpricedRecords) / bucket.records; + } modelAccumulator.set(modelKey, model); const day = dailyAccumulator.get(bucket.day) ?? { @@ -730,8 +811,10 @@ export function mergeUsage( provider: totals.provider, costUsd: totals.costUsd, totalTokens: totals.totalTokens, + tokens: totals.tokens, records: totals.records, unpricedRecords: totals.unpricedRecords, + unpricedTokens: totals.unpricedTokens, costShare: costUsd === 0 ? 0 : totals.costUsd / costUsd, })) .sort((a, b) => b.costUsd - a.costUsd || b.totalTokens - a.totalTokens); @@ -770,6 +853,22 @@ export function mergeUsage( records === 0 ? 0 : (records - providerReportedRecords - unpricedRecords) / records, cacheSavingsUsd, }, + // Clamped so float error never shows as a negative remainder. + categoryCost: { + ...categoryCost, + unsplit: Math.max( + 0, + costUsd - + categoryCost.input - + categoryCost.cacheRead - + categoryCost.cacheWrite - + categoryCost.output, + ), + }, + speedCost: { + ...speedCost, + standard: Math.max(0, costUsd - speedCost.fast - speedCost.ultrafast), + }, duplicateSources: duplicates, contributingEnvironments, contractMismatches,