From d6d52d95e5a359c70f733bd77e539c7f419cfc1f Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Thu, 23 Jul 2026 12:24:26 -0700 Subject: [PATCH] feat(cost-optimization): add by-model view to cache leakage table with plain-language columns Adds a By virtual key / By model toggle to the cache leakage table. The model view aggregates the daily activity model breakdown and is scoped to Anthropic (Claude) models, which support prompt caching. Renames the columns to plain language: Uncached input, Cache hit rate, and Potential savings (replacing Realized caching savings and Est. savings left), with a tooltip on Potential savings that spells out how it is calculated --- .../_components/CacheLeakageCard.test.tsx | 42 +++++++- .../_components/CacheLeakageCard.tsx | 69 +++++++------ .../_components/costOptimizationUtils.test.ts | 81 +++++++++++++-- .../_components/costOptimizationUtils.ts | 98 ++++++++++++------- 4 files changed, 212 insertions(+), 78 deletions(-) diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.test.tsx index 1d82fbb48ea..bb40c341786 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.test.tsx @@ -1,4 +1,4 @@ -import { render } from "@testing-library/react"; +import { fireEvent, render } from "@testing-library/react"; import { describe, expect, it, vi } from "vitest"; import type { DailyData, KeyMetricWithMetadata, SpendMetrics } from "@/components/UsagePage/types"; @@ -41,6 +41,24 @@ const dayWithKeys = (date: string, apiKeys: Record>): DailyData => ({ + date, + metrics: baseMetrics({}), + breakdown: { + models: Object.fromEntries( + Object.entries(models).map(([name, m]) => [ + name, + { metrics: baseMetrics(m), metadata: {}, api_key_breakdown: {} }, + ]), + ), + model_groups: {}, + mcp_servers: {}, + providers: {}, + api_keys: {}, + entities: {}, + }, +}); + const renderWith = (results: DailyData[]) => render( { expect(getByText("0.0%")).toBeInTheDocument(); expect(getByText("90.0%")).toBeInTheDocument(); [ - "Input tokens in the selected range that were neither read from nor written to the prompt cache", - "Share of this key's total input tokens that were served from the prompt cache", - "Dollars this key actually saved because cached input was billed at the discounted cache-read rate", - "Approximate dollars this key could still save if its uncached input had hit the cache at the portfolio's realized discount", + "Input tokens you sent in this range that weren't served from or written to the cache", + "Share of your input tokens that were served from the cache", + "About how much you'd save if this uncached input used prompt caching. Estimated as uncached input tokens times the per-token discount your cached traffic already gets (realized cache savings ÷ cache-read tokens).", ].forEach((info) => expect(getByLabelText(info)).toBeInTheDocument()); }); + it("switches to the model view and lists only Anthropic models", () => { + const { getByText, queryByText } = renderWith([ + dayWithModels("2026-07-12", { + "claude-sonnet-5": { prompt_tokens: 5000, cache_read_input_tokens: 0 }, + "gpt-4o": { prompt_tokens: 8000, cache_read_input_tokens: 0 }, + }), + ]); + + fireEvent.click(getByText("By model")); + + expect(getByText("Cache leakage by model")).toBeInTheDocument(); + expect(getByText("claude-sonnet-5")).toBeInTheDocument(); + expect(queryByText("gpt-4o")).not.toBeInTheDocument(); + }); + it("shows an empty state when no key used tokens in the range", () => { const { getByText, queryByRole } = renderWith([dayWithKeys("2026-07-12", {})]); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.tsx index 359ce502b07..ad1ff303dbc 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/CacheLeakageCard.tsx @@ -1,14 +1,15 @@ "use client"; -import React, { useMemo } from "react"; +import React, { useMemo, useState } from "react"; import { Info } from "lucide-react"; import AdvancedDatePicker from "@/components/shared/advanced_date_picker"; import { Card, CardContent, CardHeader, CardTitle } from "@/components/ui/card"; import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table"; +import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs"; import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip"; import { formatNumberWithCommas } from "@/utils/dataUtils"; -import { computeCacheLeakage, pct, usd } from "./costOptimizationUtils"; +import { CacheLeakageDimension, computeCacheLeakage, pct, usd } from "./costOptimizationUtils"; import { DailyActivityRange } from "./useDailyActivityRange"; interface CacheLeakageCardProps { @@ -19,19 +20,22 @@ const HeadWithInfo = ({ label, info }: { label: string; info: string }) => ( {label} - - - - + }> + - {info} + {info} ); const CacheLeakageCard: React.FC = ({ activity }) => { const { dateValue, onDateChange, results, loading, isFetchingMore } = activity; - const leakage = useMemo(() => computeCacheLeakage(results), [results]); + const [dimension, setDimension] = useState("key"); + const leakage = useMemo(() => computeCacheLeakage(results, dimension), [results, dimension]); + + const subject = dimension === "model" ? "Models" : "Keys"; + const firstColumn = dimension === "model" ? "Model" : "Key"; + const emptyNoun = dimension === "model" ? "model" : "key"; return ( @@ -39,64 +43,67 @@ const CacheLeakageCard: React.FC = ({ activity }) => {
- Cache leakage by virtual key + Cache leakage by {dimension === "model" ? "model" : "virtual key"}

- Keys sending large volumes of uncached prompt tokens with a low cache-hit ratio are likely missing - prompt caching. Estimated savings left is approximate: uncached prompt tokens priced at the - portfolio's realized cache-read discount. + {subject} sending large volumes of uncached input with a low cache hit rate are likely missing prompt + caching. Potential savings is approximate: uncached input priced at the realized cache-read discount. + {dimension === "model" ? " Limited to Anthropic (Claude) models, which support prompt caching." : ""}

+ setDimension(value === "model" ? "model" : "key")} + className="mt-3" + > + + By virtual key + By model + +
{leakage.rows.length === 0 ? (

- {loading || isFetchingMore ? "Loading..." : "No key usage in this range."} + {loading || isFetchingMore ? "Loading..." : `No ${emptyNoun} usage in this range.`}

) : ( - Key + {firstColumn} - - - {leakage.rows.map((row) => ( - + - {row.keyAlias || `${row.apiKey.slice(0, 8)}...`} - {row.teamId && ({row.teamId})} + {row.label} + {row.sublabel && ({row.sublabel})} {formatNumberWithCommas(row.uncachedPromptTokens)} {pct(row.cacheHitRatio)} - {usd(row.realizedCachingSavings)} - {row.estSavingsLeft == null ? "—" : usd(row.estSavingsLeft)} + {row.potentialSavings == null ? "—" : usd(row.potentialSavings)} ))} diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.test.ts b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.test.ts index 562552ffb50..2f1558465d5 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.test.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.test.ts @@ -2,7 +2,7 @@ import { describe, expect, it } from "vitest"; import type { DailyData, SpendMetrics } from "@/components/UsagePage/types"; import type { ToolSpendDailyEntry, ToolSpendEntry } from "@/components/networking"; -import { buildDailyToolSeries, computeCacheLeakage, topToolsBySpend } from "./costOptimizationUtils"; +import { buildDailyToolSeries, computeCacheLeakage, isAnthropicModel, topToolsBySpend } from "./costOptimizationUtils"; const metrics = (overrides: Partial): SpendMetrics => ({ spend: 0, @@ -38,6 +38,21 @@ const day = ( }, }); +const modelDay = (date: string, models: Record>): DailyData => ({ + date, + metrics: metrics({}), + breakdown: { + models: Object.fromEntries( + Object.entries(models).map(([name, m]) => [name, { metrics: metrics(m), metadata: {}, api_key_breakdown: {} }]), + ), + model_groups: {}, + mcp_servers: {}, + providers: {}, + entities: {}, + api_keys: {}, + }, +}); + describe("computeCacheLeakage", () => { it("aggregates a key's tokens and savings across multiple days", () => { const results = [ @@ -76,8 +91,8 @@ describe("computeCacheLeakage", () => { ]; const { rows, discountPerToken } = computeCacheLeakage(results); expect(discountPerToken).toBeCloseTo(0.002, 6); - expect(rows.map((r) => r.keyAlias)).toEqual(["leaker"]); - expect(rows[0].estSavingsLeft).toBeCloseTo(1.0, 6); + expect(rows.map((r) => r.label)).toEqual(["leaker"]); + expect(rows[0].potentialSavings).toBeCloseTo(1.0, 6); }); it("returns null estimate and ranks by uncached tokens when nobody used caching", () => { @@ -89,8 +104,8 @@ describe("computeCacheLeakage", () => { ]; const { rows, discountPerToken } = computeCacheLeakage(results); expect(discountPerToken).toBeNull(); - expect(rows.map((r) => r.keyAlias)).toEqual(["big", "small"]); - expect(rows.every((r) => r.estSavingsLeft === null)).toBe(true); + expect(rows.map((r) => r.label)).toEqual(["big", "small"]); + expect(rows.every((r) => r.potentialSavings === null)).toBe(true); }); it("computes cache hit ratio against total prompt tokens and clamps inconsistent data at zero", () => { @@ -101,7 +116,7 @@ describe("computeCacheLeakage", () => { }), ]; const { rows } = computeCacheLeakage(results); - expect(rows.map((r) => r.keyAlias)).toEqual(["mixed"]); + expect(rows.map((r) => r.label)).toEqual(["mixed"]); expect(rows[0].cacheHitRatio).toBeCloseTo(0.75, 6); expect(rows[0].uncachedPromptTokens).toBe(250); }); @@ -110,11 +125,63 @@ describe("computeCacheLeakage", () => { const keys = Object.fromEntries( Array.from({ length: 15 }, (_, i) => [`h${i}`, { alias: `k${i}`, metrics: { prompt_tokens: i + 1 } }]), ); - const { rows } = computeCacheLeakage([day("2026-07-01", keys)], 5); + const { rows } = computeCacheLeakage([day("2026-07-01", keys)], "key", 5); expect(rows).toHaveLength(5); }); }); +describe("computeCacheLeakage by model", () => { + it("aggregates only Anthropic models and ignores other providers", () => { + const models: Record> = { + "claude-sonnet-5": { prompt_tokens: 10000, cache_read_input_tokens: 0 }, + "anthropic/claude-haiku-4-5": { prompt_tokens: 4000, cache_read_input_tokens: 0 }, + "bedrock/anthropic.claude-3-5-sonnet": { prompt_tokens: 2000, cache_read_input_tokens: 0 }, + "gpt-4o": { prompt_tokens: 9000, cache_read_input_tokens: 0 }, + "deepseek-chat": { prompt_tokens: 8000, cache_read_input_tokens: 0 }, + }; + const { rows } = computeCacheLeakage([modelDay("2026-07-01", models)], "model"); + expect(rows.map((r) => r.id)).toEqual([ + "claude-sonnet-5", + "anthropic/claude-haiku-4-5", + "bedrock/anthropic.claude-3-5-sonnet", + ]); + }); + + it("labels model rows by model name with no sublabel", () => { + const results = [modelDay("2026-07-01", { "claude-sonnet-5": { prompt_tokens: 1000 } })]; + const { rows } = computeCacheLeakage(results, "model"); + expect(rows[0].label).toBe("claude-sonnet-5"); + expect(rows[0].sublabel).toBeNull(); + }); + + it("prices model leakage at the Anthropic realized cache-read discount", () => { + const results = [ + modelDay("2026-07-01", { + "claude-sonnet-5": { prompt_tokens: 1000, cache_read_input_tokens: 1000, prompt_caching_savings_spend: 2.0 }, + "claude-haiku-4-5": { prompt_tokens: 500 }, + }), + ]; + const { rows, discountPerToken } = computeCacheLeakage(results, "model"); + expect(discountPerToken).toBeCloseTo(0.002, 6); + expect(rows.map((r) => r.id)).toEqual(["claude-haiku-4-5"]); + expect(rows[0].potentialSavings).toBeCloseTo(1.0, 6); + }); +}); + +describe("isAnthropicModel", () => { + it("matches Claude-family models across providers and rejects others", () => { + const anthropic = [ + "claude-sonnet-5", + "anthropic/claude-haiku-4-5", + "bedrock/anthropic.claude-3-5-sonnet", + "vertex_ai/claude-opus-4-8", + ]; + const others = ["gpt-4o", "deepseek-chat", "gemini-2.5-pro", "mistral-large"]; + expect(anthropic.every(isAnthropicModel)).toBe(true); + expect(others.some(isAnthropicModel)).toBe(false); + }); +}); + describe("buildDailyToolSeries", () => { const daily: ToolSpendDailyEntry[] = [ { date: "2026-07-01", tool_name: "search", spend: 1.0, call_count: 1 }, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.ts b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.ts index 2868bdac880..30f851bfeca 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/costOptimizationUtils.ts @@ -1,4 +1,4 @@ -import { DailyData } from "@/components/UsagePage/types"; +import { DailyData, SpendMetrics } from "@/components/UsagePage/types"; import { ToolSpendDailyEntry, ToolSpendEntry } from "@/components/networking"; import { formatNumberWithCommas } from "@/utils/dataUtils"; @@ -9,15 +9,15 @@ export const usd = (value: number): string => { export const pct = (ratio: number): string => `${formatNumberWithCommas(ratio * 100, 1)}%`; +export type CacheLeakageDimension = "key" | "model"; + export interface CacheLeakageRow { - apiKey: string; - keyAlias: string | null; - teamId: string | null; + id: string; + label: string; + sublabel: string | null; uncachedPromptTokens: number; - cacheReadTokens: number; cacheHitRatio: number; - realizedCachingSavings: number; - estSavingsLeft: number | null; + potentialSavings: number | null; } export interface CacheLeakageResult { @@ -25,8 +25,10 @@ export interface CacheLeakageResult { discountPerToken: number | null; } -interface KeyAccumulator { - keyAlias: string | null; +export const isAnthropicModel = (model: string): boolean => /claude|anthropic/i.test(model); + +interface LeakageAccumulator { + alias: string | null; teamId: string | null; promptTokens: number; cacheReadTokens: number; @@ -34,8 +36,8 @@ interface KeyAccumulator { realizedCachingSavings: number; } -const emptyAccumulator = (): KeyAccumulator => ({ - keyAlias: null, +const emptyAccumulator = (): LeakageAccumulator => ({ + alias: null, teamId: null, promptTokens: 0, cacheReadTokens: 0, @@ -43,26 +45,54 @@ const emptyAccumulator = (): KeyAccumulator => ({ realizedCachingSavings: 0, }); -export const computeCacheLeakage = (results: readonly DailyData[], limit = 10): CacheLeakageResult => { - const byKey = new Map(); +const addMetrics = ( + acc: LeakageAccumulator, + m: SpendMetrics, + alias: string | null, + teamId: string | null, +): LeakageAccumulator => ({ + alias: acc.alias ?? alias, + teamId: acc.teamId ?? teamId, + promptTokens: acc.promptTokens + (m.prompt_tokens ?? 0), + cacheReadTokens: acc.cacheReadTokens + (m.cache_read_input_tokens ?? 0), + cacheCreationTokens: acc.cacheCreationTokens + (m.cache_creation_input_tokens ?? 0), + realizedCachingSavings: acc.realizedCachingSavings + (m.prompt_caching_savings_spend ?? 0), +}); + +const aggregateByKey = (results: readonly DailyData[]): Map => { + const byKey = new Map(); for (const day of results) { - const apiKeys = day.breakdown?.api_keys ?? {}; - for (const [apiKey, entry] of Object.entries(apiKeys)) { + for (const [apiKey, entry] of Object.entries(day.breakdown?.api_keys ?? {})) { const acc = byKey.get(apiKey) ?? emptyAccumulator(); - const m = entry.metrics; - const next: KeyAccumulator = { - keyAlias: acc.keyAlias ?? entry.metadata?.key_alias ?? null, - teamId: acc.teamId ?? entry.metadata?.team_id ?? null, - promptTokens: acc.promptTokens + (m.prompt_tokens ?? 0), - cacheReadTokens: acc.cacheReadTokens + (m.cache_read_input_tokens ?? 0), - cacheCreationTokens: acc.cacheCreationTokens + (m.cache_creation_input_tokens ?? 0), - realizedCachingSavings: acc.realizedCachingSavings + (m.prompt_caching_savings_spend ?? 0), - }; - byKey.set(apiKey, next); + byKey.set( + apiKey, + addMetrics(acc, entry.metrics, entry.metadata?.key_alias ?? null, entry.metadata?.team_id ?? null), + ); } } + return byKey; +}; - const totals = [...byKey.values()].reduce( +const aggregateByModel = (results: readonly DailyData[]): Map => { + const byModel = new Map(); + for (const day of results) { + for (const [model, entry] of Object.entries(day.breakdown?.models ?? {})) { + if (!isAnthropicModel(model)) continue; + const acc = byModel.get(model) ?? emptyAccumulator(); + byModel.set(model, addMetrics(acc, entry.metrics, null, null)); + } + } + return byModel; +}; + +export const computeCacheLeakage = ( + results: readonly DailyData[], + dimension: CacheLeakageDimension = "key", + limit = 10, +): CacheLeakageResult => { + const byEntity = dimension === "model" ? aggregateByModel(results) : aggregateByKey(results); + + const totals = [...byEntity.values()].reduce( (agg, a) => ({ cacheReadTokens: agg.cacheReadTokens + a.cacheReadTokens, realizedCachingSavings: agg.realizedCachingSavings + a.realizedCachingSavings, @@ -71,25 +101,23 @@ export const computeCacheLeakage = (results: readonly DailyData[], limit = 10): ); const discountPerToken = totals.cacheReadTokens > 0 ? totals.realizedCachingSavings / totals.cacheReadTokens : null; - const rows: CacheLeakageRow[] = [...byKey.entries()] - .map(([apiKey, a]) => { + const rows: CacheLeakageRow[] = [...byEntity.entries()] + .map(([id, a]) => { const uncachedPromptTokens = Math.max(0, a.promptTokens - a.cacheReadTokens - a.cacheCreationTokens); return { - apiKey, - keyAlias: a.keyAlias, - teamId: a.teamId, + id, + label: dimension === "model" ? id : a.alias ?? `${id.slice(0, 8)}...`, + sublabel: dimension === "model" ? null : a.teamId, uncachedPromptTokens, - cacheReadTokens: a.cacheReadTokens, cacheHitRatio: a.promptTokens > 0 ? a.cacheReadTokens / a.promptTokens : 0, - realizedCachingSavings: a.realizedCachingSavings, - estSavingsLeft: discountPerToken != null ? uncachedPromptTokens * discountPerToken : null, + potentialSavings: discountPerToken != null ? uncachedPromptTokens * discountPerToken : null, }; }) .filter((row) => row.uncachedPromptTokens > 0); const sorted = rows.sort((x, y) => discountPerToken != null - ? (y.estSavingsLeft ?? 0) - (x.estSavingsLeft ?? 0) + ? (y.potentialSavings ?? 0) - (x.potentialSavings ?? 0) : y.uncachedPromptTokens - x.uncachedPromptTokens, );