feat(cost-optimization): add by-model view to cache leakage table with plain-language columns

Adds a By virtual key / By model toggle to the cache leakage table. The model
view aggregates the daily activity model breakdown and is scoped to Anthropic
(Claude) models, which support prompt caching. Renames the columns to plain
language: Uncached input, Cache hit rate, and Potential savings (replacing
Realized caching savings and Est. savings left), with a tooltip on Potential
savings that spells out how it is calculated
This commit is contained in:
Tin Chi Lo 2026-07-23 12:24:26 -07:00
parent bd73ca8c64
commit d6d52d95e5
4 changed files with 212 additions and 78 deletions

View file

@ -1,4 +1,4 @@
import { render } from "@testing-library/react";
import { fireEvent, render } from "@testing-library/react";
import { describe, expect, it, vi } from "vitest";
import type { DailyData, KeyMetricWithMetadata, SpendMetrics } from "@/components/UsagePage/types";
@ -41,6 +41,24 @@ const dayWithKeys = (date: string, apiKeys: Record<string, KeyMetricWithMetadata
},
});
const dayWithModels = (date: string, models: Record<string, Partial<SpendMetrics>>): DailyData => ({
date,
metrics: baseMetrics({}),
breakdown: {
models: Object.fromEntries(
Object.entries(models).map(([name, m]) => [
name,
{ metrics: baseMetrics(m), metadata: {}, api_key_breakdown: {} },
]),
),
model_groups: {},
mcp_servers: {},
providers: {},
api_keys: {},
entities: {},
},
});
const renderWith = (results: DailyData[]) =>
render(
<CacheLeakageCard
@ -67,13 +85,27 @@ describe("CacheLeakageCard", () => {
expect(getByText("0.0%")).toBeInTheDocument();
expect(getByText("90.0%")).toBeInTheDocument();
[
"Input tokens in the selected range that were neither read from nor written to the prompt cache",
"Share of this key's total input tokens that were served from the prompt cache",
"Dollars this key actually saved because cached input was billed at the discounted cache-read rate",
"Approximate dollars this key could still save if its uncached input had hit the cache at the portfolio's realized discount",
"Input tokens you sent in this range that weren't served from or written to the cache",
"Share of your input tokens that were served from the cache",
"About how much you'd save if this uncached input used prompt caching. Estimated as uncached input tokens times the per-token discount your cached traffic already gets (realized cache savings ÷ cache-read tokens).",
].forEach((info) => expect(getByLabelText(info)).toBeInTheDocument());
});
it("switches to the model view and lists only Anthropic models", () => {
const { getByText, queryByText } = renderWith([
dayWithModels("2026-07-12", {
"claude-sonnet-5": { prompt_tokens: 5000, cache_read_input_tokens: 0 },
"gpt-4o": { prompt_tokens: 8000, cache_read_input_tokens: 0 },
}),
]);
fireEvent.click(getByText("By model"));
expect(getByText("Cache leakage by model")).toBeInTheDocument();
expect(getByText("claude-sonnet-5")).toBeInTheDocument();
expect(queryByText("gpt-4o")).not.toBeInTheDocument();
});
it("shows an empty state when no key used tokens in the range", () => {
const { getByText, queryByRole } = renderWith([dayWithKeys("2026-07-12", {})]);

View file

@ -1,14 +1,15 @@
"use client";
import React, { useMemo } from "react";
import React, { useMemo, useState } from "react";
import { Info } from "lucide-react";
import AdvancedDatePicker from "@/components/shared/advanced_date_picker";
import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card";
import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table";
import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs";
import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip";
import { formatNumberWithCommas } from "@/utils/dataUtils";
import { computeCacheLeakage, pct, usd } from "./costOptimizationUtils";
import { CacheLeakageDimension, computeCacheLeakage, pct, usd } from "./costOptimizationUtils";
import { DailyActivityRange } from "./useDailyActivityRange";
interface CacheLeakageCardProps {
@ -19,19 +20,22 @@ const HeadWithInfo = ({ label, info }: { label: string; info: string }) => (
<span className="inline-flex items-center gap-1">
{label}
<Tooltip>
<TooltipTrigger asChild>
<span className="inline-flex" aria-label={info}>
<Info className="h-3 w-3 text-gray-400" />
</span>
<TooltipTrigger render={<span className="inline-flex" aria-label={info} />}>
<Info className="h-3 w-3 text-gray-400" />
</TooltipTrigger>
<TooltipContent>{info}</TooltipContent>
<TooltipContent className="max-w-xs">{info}</TooltipContent>
</Tooltip>
</span>
);
const CacheLeakageCard: React.FC<CacheLeakageCardProps> = ({ activity }) => {
const { dateValue, onDateChange, results, loading, isFetchingMore } = activity;
const leakage = useMemo(() => computeCacheLeakage(results), [results]);
const [dimension, setDimension] = useState<CacheLeakageDimension>("key");
const leakage = useMemo(() => computeCacheLeakage(results, dimension), [results, dimension]);
const subject = dimension === "model" ? "Models" : "Keys";
const firstColumn = dimension === "model" ? "Model" : "Key";
const emptyNoun = dimension === "model" ? "model" : "key";
return (
<TooltipProvider delay={300}>
@ -39,64 +43,67 @@ const CacheLeakageCard: React.FC<CacheLeakageCardProps> = ({ activity }) => {
<CardHeader>
<div className="flex flex-wrap items-start justify-between gap-4">
<div>
<CardTitle>Cache leakage by virtual key</CardTitle>
<CardTitle>Cache leakage by {dimension === "model" ? "model" : "virtual key"}</CardTitle>
<p className="mt-1 text-sm text-muted-foreground">
Keys sending large volumes of uncached prompt tokens with a low cache-hit ratio are likely missing
prompt caching. Estimated savings left is approximate: uncached prompt tokens priced at the
portfolio&apos;s realized cache-read discount.
{subject} sending large volumes of uncached input with a low cache hit rate are likely missing prompt
caching. Potential savings is approximate: uncached input priced at the realized cache-read discount.
{dimension === "model" ? " Limited to Anthropic (Claude) models, which support prompt caching." : ""}
</p>
</div>
<AdvancedDatePicker value={dateValue} onValueChange={onDateChange} />
</div>
<Tabs
value={dimension}
onValueChange={(value) => setDimension(value === "model" ? "model" : "key")}
className="mt-3"
>
<TabsList>
<TabsTrigger value="key">By virtual key</TabsTrigger>
<TabsTrigger value="model">By model</TabsTrigger>
</TabsList>
</Tabs>
</CardHeader>
<CardContent>
{leakage.rows.length === 0 ? (
<p className="py-8 text-center text-sm text-muted-foreground">
{loading || isFetchingMore ? "Loading..." : "No key usage in this range."}
{loading || isFetchingMore ? "Loading..." : `No ${emptyNoun} usage in this range.`}
</p>
) : (
<Table>
<TableHeader>
<TableRow>
<TableHead>Key</TableHead>
<TableHead>{firstColumn}</TableHead>
<TableHead className="text-right">
<HeadWithInfo
label="Uncached prompt tokens"
info="Input tokens in the selected range that were neither read from nor written to the prompt cache"
label="Uncached input"
info="Input tokens you sent in this range that weren't served from or written to the cache"
/>
</TableHead>
<TableHead className="text-right">
<HeadWithInfo
label="Cache hit ratio"
info="Share of this key's total input tokens that were served from the prompt cache"
label="Cache hit rate"
info="Share of your input tokens that were served from the cache"
/>
</TableHead>
<TableHead className="text-right">
<HeadWithInfo
label="Realized caching savings"
info="Dollars this key actually saved because cached input was billed at the discounted cache-read rate"
/>
</TableHead>
<TableHead className="text-right">
<HeadWithInfo
label="Est. savings left"
info="Approximate dollars this key could still save if its uncached input had hit the cache at the portfolio's realized discount"
label="Potential savings"
info="About how much you'd save if this uncached input used prompt caching. Estimated as uncached input tokens times the per-token discount your cached traffic already gets (realized cache savings ÷ cache-read tokens)."
/>
</TableHead>
</TableRow>
</TableHeader>
<TableBody>
{leakage.rows.map((row) => (
<TableRow key={row.apiKey}>
<TableRow key={row.id}>
<TableCell className="font-medium">
{row.keyAlias || `${row.apiKey.slice(0, 8)}...`}
{row.teamId && <span className="ml-1 text-xs text-muted-foreground">({row.teamId})</span>}
{row.label}
{row.sublabel && <span className="ml-1 text-xs text-muted-foreground">({row.sublabel})</span>}
</TableCell>
<TableCell className="text-right">{formatNumberWithCommas(row.uncachedPromptTokens)}</TableCell>
<TableCell className="text-right">{pct(row.cacheHitRatio)}</TableCell>
<TableCell className="text-right">{usd(row.realizedCachingSavings)}</TableCell>
<TableCell className="text-right">
{row.estSavingsLeft == null ? "—" : usd(row.estSavingsLeft)}
{row.potentialSavings == null ? "—" : usd(row.potentialSavings)}
</TableCell>
</TableRow>
))}

View file

@ -2,7 +2,7 @@ import { describe, expect, it } from "vitest";
import type { DailyData, SpendMetrics } from "@/components/UsagePage/types";
import type { ToolSpendDailyEntry, ToolSpendEntry } from "@/components/networking";
import { buildDailyToolSeries, computeCacheLeakage, topToolsBySpend } from "./costOptimizationUtils";
import { buildDailyToolSeries, computeCacheLeakage, isAnthropicModel, topToolsBySpend } from "./costOptimizationUtils";
const metrics = (overrides: Partial<SpendMetrics>): SpendMetrics => ({
spend: 0,
@ -38,6 +38,21 @@ const day = (
},
});
const modelDay = (date: string, models: Record<string, Partial<SpendMetrics>>): DailyData => ({
date,
metrics: metrics({}),
breakdown: {
models: Object.fromEntries(
Object.entries(models).map(([name, m]) => [name, { metrics: metrics(m), metadata: {}, api_key_breakdown: {} }]),
),
model_groups: {},
mcp_servers: {},
providers: {},
entities: {},
api_keys: {},
},
});
describe("computeCacheLeakage", () => {
it("aggregates a key's tokens and savings across multiple days", () => {
const results = [
@ -76,8 +91,8 @@ describe("computeCacheLeakage", () => {
];
const { rows, discountPerToken } = computeCacheLeakage(results);
expect(discountPerToken).toBeCloseTo(0.002, 6);
expect(rows.map((r) => r.keyAlias)).toEqual(["leaker"]);
expect(rows[0].estSavingsLeft).toBeCloseTo(1.0, 6);
expect(rows.map((r) => r.label)).toEqual(["leaker"]);
expect(rows[0].potentialSavings).toBeCloseTo(1.0, 6);
});
it("returns null estimate and ranks by uncached tokens when nobody used caching", () => {
@ -89,8 +104,8 @@ describe("computeCacheLeakage", () => {
];
const { rows, discountPerToken } = computeCacheLeakage(results);
expect(discountPerToken).toBeNull();
expect(rows.map((r) => r.keyAlias)).toEqual(["big", "small"]);
expect(rows.every((r) => r.estSavingsLeft === null)).toBe(true);
expect(rows.map((r) => r.label)).toEqual(["big", "small"]);
expect(rows.every((r) => r.potentialSavings === null)).toBe(true);
});
it("computes cache hit ratio against total prompt tokens and clamps inconsistent data at zero", () => {
@ -101,7 +116,7 @@ describe("computeCacheLeakage", () => {
}),
];
const { rows } = computeCacheLeakage(results);
expect(rows.map((r) => r.keyAlias)).toEqual(["mixed"]);
expect(rows.map((r) => r.label)).toEqual(["mixed"]);
expect(rows[0].cacheHitRatio).toBeCloseTo(0.75, 6);
expect(rows[0].uncachedPromptTokens).toBe(250);
});
@ -110,11 +125,63 @@ describe("computeCacheLeakage", () => {
const keys = Object.fromEntries(
Array.from({ length: 15 }, (_, i) => [`h${i}`, { alias: `k${i}`, metrics: { prompt_tokens: i + 1 } }]),
);
const { rows } = computeCacheLeakage([day("2026-07-01", keys)], 5);
const { rows } = computeCacheLeakage([day("2026-07-01", keys)], "key", 5);
expect(rows).toHaveLength(5);
});
});
describe("computeCacheLeakage by model", () => {
it("aggregates only Anthropic models and ignores other providers", () => {
const models: Record<string, Partial<SpendMetrics>> = {
"claude-sonnet-5": { prompt_tokens: 10000, cache_read_input_tokens: 0 },
"anthropic/claude-haiku-4-5": { prompt_tokens: 4000, cache_read_input_tokens: 0 },
"bedrock/anthropic.claude-3-5-sonnet": { prompt_tokens: 2000, cache_read_input_tokens: 0 },
"gpt-4o": { prompt_tokens: 9000, cache_read_input_tokens: 0 },
"deepseek-chat": { prompt_tokens: 8000, cache_read_input_tokens: 0 },
};
const { rows } = computeCacheLeakage([modelDay("2026-07-01", models)], "model");
expect(rows.map((r) => r.id)).toEqual([
"claude-sonnet-5",
"anthropic/claude-haiku-4-5",
"bedrock/anthropic.claude-3-5-sonnet",
]);
});
it("labels model rows by model name with no sublabel", () => {
const results = [modelDay("2026-07-01", { "claude-sonnet-5": { prompt_tokens: 1000 } })];
const { rows } = computeCacheLeakage(results, "model");
expect(rows[0].label).toBe("claude-sonnet-5");
expect(rows[0].sublabel).toBeNull();
});
it("prices model leakage at the Anthropic realized cache-read discount", () => {
const results = [
modelDay("2026-07-01", {
"claude-sonnet-5": { prompt_tokens: 1000, cache_read_input_tokens: 1000, prompt_caching_savings_spend: 2.0 },
"claude-haiku-4-5": { prompt_tokens: 500 },
}),
];
const { rows, discountPerToken } = computeCacheLeakage(results, "model");
expect(discountPerToken).toBeCloseTo(0.002, 6);
expect(rows.map((r) => r.id)).toEqual(["claude-haiku-4-5"]);
expect(rows[0].potentialSavings).toBeCloseTo(1.0, 6);
});
});
describe("isAnthropicModel", () => {
it("matches Claude-family models across providers and rejects others", () => {
const anthropic = [
"claude-sonnet-5",
"anthropic/claude-haiku-4-5",
"bedrock/anthropic.claude-3-5-sonnet",
"vertex_ai/claude-opus-4-8",
];
const others = ["gpt-4o", "deepseek-chat", "gemini-2.5-pro", "mistral-large"];
expect(anthropic.every(isAnthropicModel)).toBe(true);
expect(others.some(isAnthropicModel)).toBe(false);
});
});
describe("buildDailyToolSeries", () => {
const daily: ToolSpendDailyEntry[] = [
{ date: "2026-07-01", tool_name: "search", spend: 1.0, call_count: 1 },

View file

@ -1,4 +1,4 @@
import { DailyData } from "@/components/UsagePage/types";
import { DailyData, SpendMetrics } from "@/components/UsagePage/types";
import { ToolSpendDailyEntry, ToolSpendEntry } from "@/components/networking";
import { formatNumberWithCommas } from "@/utils/dataUtils";
@ -9,15 +9,15 @@ export const usd = (value: number): string => {
export const pct = (ratio: number): string => `${formatNumberWithCommas(ratio * 100, 1)}%`;
export type CacheLeakageDimension = "key" | "model";
export interface CacheLeakageRow {
apiKey: string;
keyAlias: string | null;
teamId: string | null;
id: string;
label: string;
sublabel: string | null;
uncachedPromptTokens: number;
cacheReadTokens: number;
cacheHitRatio: number;
realizedCachingSavings: number;
estSavingsLeft: number | null;
potentialSavings: number | null;
}
export interface CacheLeakageResult {
@ -25,8 +25,10 @@ export interface CacheLeakageResult {
discountPerToken: number | null;
}
interface KeyAccumulator {
keyAlias: string | null;
export const isAnthropicModel = (model: string): boolean => /claude|anthropic/i.test(model);
interface LeakageAccumulator {
alias: string | null;
teamId: string | null;
promptTokens: number;
cacheReadTokens: number;
@ -34,8 +36,8 @@ interface KeyAccumulator {
realizedCachingSavings: number;
}
const emptyAccumulator = (): KeyAccumulator => ({
keyAlias: null,
const emptyAccumulator = (): LeakageAccumulator => ({
alias: null,
teamId: null,
promptTokens: 0,
cacheReadTokens: 0,
@ -43,26 +45,54 @@ const emptyAccumulator = (): KeyAccumulator => ({
realizedCachingSavings: 0,
});
export const computeCacheLeakage = (results: readonly DailyData[], limit = 10): CacheLeakageResult => {
const byKey = new Map<string, KeyAccumulator>();
const addMetrics = (
acc: LeakageAccumulator,
m: SpendMetrics,
alias: string | null,
teamId: string | null,
): LeakageAccumulator => ({
alias: acc.alias ?? alias,
teamId: acc.teamId ?? teamId,
promptTokens: acc.promptTokens + (m.prompt_tokens ?? 0),
cacheReadTokens: acc.cacheReadTokens + (m.cache_read_input_tokens ?? 0),
cacheCreationTokens: acc.cacheCreationTokens + (m.cache_creation_input_tokens ?? 0),
realizedCachingSavings: acc.realizedCachingSavings + (m.prompt_caching_savings_spend ?? 0),
});
const aggregateByKey = (results: readonly DailyData[]): Map<string, LeakageAccumulator> => {
const byKey = new Map<string, LeakageAccumulator>();
for (const day of results) {
const apiKeys = day.breakdown?.api_keys ?? {};
for (const [apiKey, entry] of Object.entries(apiKeys)) {
for (const [apiKey, entry] of Object.entries(day.breakdown?.api_keys ?? {})) {
const acc = byKey.get(apiKey) ?? emptyAccumulator();
const m = entry.metrics;
const next: KeyAccumulator = {
keyAlias: acc.keyAlias ?? entry.metadata?.key_alias ?? null,
teamId: acc.teamId ?? entry.metadata?.team_id ?? null,
promptTokens: acc.promptTokens + (m.prompt_tokens ?? 0),
cacheReadTokens: acc.cacheReadTokens + (m.cache_read_input_tokens ?? 0),
cacheCreationTokens: acc.cacheCreationTokens + (m.cache_creation_input_tokens ?? 0),
realizedCachingSavings: acc.realizedCachingSavings + (m.prompt_caching_savings_spend ?? 0),
};
byKey.set(apiKey, next);
byKey.set(
apiKey,
addMetrics(acc, entry.metrics, entry.metadata?.key_alias ?? null, entry.metadata?.team_id ?? null),
);
}
}
return byKey;
};
const totals = [...byKey.values()].reduce(
const aggregateByModel = (results: readonly DailyData[]): Map<string, LeakageAccumulator> => {
const byModel = new Map<string, LeakageAccumulator>();
for (const day of results) {
for (const [model, entry] of Object.entries(day.breakdown?.models ?? {})) {
if (!isAnthropicModel(model)) continue;
const acc = byModel.get(model) ?? emptyAccumulator();
byModel.set(model, addMetrics(acc, entry.metrics, null, null));
}
}
return byModel;
};
export const computeCacheLeakage = (
results: readonly DailyData[],
dimension: CacheLeakageDimension = "key",
limit = 10,
): CacheLeakageResult => {
const byEntity = dimension === "model" ? aggregateByModel(results) : aggregateByKey(results);
const totals = [...byEntity.values()].reduce(
(agg, a) => ({
cacheReadTokens: agg.cacheReadTokens + a.cacheReadTokens,
realizedCachingSavings: agg.realizedCachingSavings + a.realizedCachingSavings,
@ -71,25 +101,23 @@ export const computeCacheLeakage = (results: readonly DailyData[], limit = 10):
);
const discountPerToken = totals.cacheReadTokens > 0 ? totals.realizedCachingSavings / totals.cacheReadTokens : null;
const rows: CacheLeakageRow[] = [...byKey.entries()]
.map(([apiKey, a]) => {
const rows: CacheLeakageRow[] = [...byEntity.entries()]
.map(([id, a]) => {
const uncachedPromptTokens = Math.max(0, a.promptTokens - a.cacheReadTokens - a.cacheCreationTokens);
return {
apiKey,
keyAlias: a.keyAlias,
teamId: a.teamId,
id,
label: dimension === "model" ? id : a.alias ?? `${id.slice(0, 8)}...`,
sublabel: dimension === "model" ? null : a.teamId,
uncachedPromptTokens,
cacheReadTokens: a.cacheReadTokens,
cacheHitRatio: a.promptTokens > 0 ? a.cacheReadTokens / a.promptTokens : 0,
realizedCachingSavings: a.realizedCachingSavings,
estSavingsLeft: discountPerToken != null ? uncachedPromptTokens * discountPerToken : null,
potentialSavings: discountPerToken != null ? uncachedPromptTokens * discountPerToken : null,
};
})
.filter((row) => row.uncachedPromptTokens > 0);
const sorted = rows.sort((x, y) =>
discountPerToken != null
? (y.estSavingsLeft ?? 0) - (x.estSavingsLeft ?? 0)
? (y.potentialSavings ?? 0) - (x.potentialSavings ?? 0)
: y.uncachedPromptTokens - x.uncachedPromptTokens,
);