feat(ui): surface provider prompt-cache tokens in Cache Analytics

The Caching page's cache-hit and cached-token views only reflect LiteLLM's
own response cache (LiteLLM_SpendLogs.cache_hit). Provider prompt caching
(e.g. Anthropic cache_read_input_tokens) never sets that flag, so those
tokens were invisible and the overall numbers did not add up for
anthropic_messages traffic.

Add a /global/activity/cache_hits/prompt_caching endpoint that reads
cache_read_input_tokens / cache_creation_input_tokens from the daily spend
rollup (the only place they are persisted) and a Provider Prompt Caching
section on the dashboard with read/creation token stats and a per-model
chart, kept separate from the response-cache metrics.

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
Mubashir Osmani 2026-07-17 22:59:41 +00:00
parent c5b4456401
commit 1c89dec3fe
6 changed files with 419 additions and 5 deletions

View file

@ -4,6 +4,7 @@ from typing import List, Optional
import fastapi
from fastapi import APIRouter, Depends, HTTPException, status
from pydantic import BaseModel, TypeAdapter
from litellm.proxy._types import *
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
@ -11,6 +12,18 @@ from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
router = APIRouter()
class PromptCacheActivityRow(BaseModel):
api_key: str
model: str | None = None
prompt_tokens: int
cache_read_input_tokens: int
cache_creation_input_tokens: int
api_requests: int
_PROMPT_CACHE_ROWS_ADAPTER = TypeAdapter(list[PromptCacheActivityRow])
@router.get(
"/global/activity/cache_hits",
tags=["Budget & Spend Tracking"],
@ -103,3 +116,66 @@ async def get_global_activity(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"error": str(e)},
)
@router.get(
"/global/activity/cache_hits/prompt_caching",
tags=["Budget & Spend Tracking"],
dependencies=[Depends(user_api_key_auth)],
responses={
200: {"model": list[PromptCacheActivityRow]},
},
include_in_schema=False,
)
async def get_prompt_cache_activity(
start_date: str | None = fastapi.Query(
default=None,
description="Time from which to start viewing prompt-cache usage",
),
end_date: str | None = fastapi.Query(
default=None,
description="Time till which to view prompt-cache usage",
),
) -> list[PromptCacheActivityRow]:
if start_date is None or end_date is None:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={"error": "Please provide start_date and end_date"},
)
from litellm.proxy.proxy_server import prisma_client
if prisma_client is None:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={
"error": "Database not connected. Connect a database to your proxy - https://docs.litellm.ai/docs/simple_proxy#managing-auth---virtual-keys"
},
)
sql_query = """
SELECT
CASE
WHEN vt."key_alias" IS NOT NULL THEN vt."key_alias"
ELSE 'Unnamed Key'
END AS api_key,
ds."model",
SUM(ds."prompt_tokens")::bigint AS prompt_tokens,
SUM(ds."cache_read_input_tokens")::bigint AS cache_read_input_tokens,
SUM(ds."cache_creation_input_tokens")::bigint AS cache_creation_input_tokens,
SUM(ds."api_requests")::bigint AS api_requests
FROM "LiteLLM_DailyUserSpend" ds
LEFT JOIN "LiteLLM_VerificationToken" vt ON ds."api_key" = vt."token"
WHERE
ds."date" >= $1
AND ds."date" <= $2
GROUP BY
vt."key_alias",
ds."model"
"""
db_response = await prisma_client.db.query_raw(sql_query, start_date, end_date)
if db_response is None:
return []
return _PROMPT_CACHE_ROWS_ADAPTER.validate_python(db_response)

View file

@ -0,0 +1,110 @@
from __future__ import annotations
import sys
from pathlib import Path
from unittest.mock import AsyncMock, MagicMock
import pytest
from fastapi import FastAPI
from fastapi.testclient import TestClient
sys.path.insert(0, str(Path(__file__).resolve().parents[4]))
from litellm.proxy import proxy_server
from litellm.proxy.analytics_endpoints.analytics_endpoints import router
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
@pytest.fixture
def client() -> TestClient:
app = FastAPI()
app.include_router(router)
app.dependency_overrides[user_api_key_auth] = lambda: MagicMock()
return TestClient(app, raise_server_exceptions=False)
def _set_prisma(monkeypatch: pytest.MonkeyPatch, query_raw: AsyncMock) -> None:
pc = MagicMock()
pc.db.query_raw = query_raw
monkeypatch.setattr(proxy_server, "prisma_client", pc)
def test_prompt_cache_activity_returns_provider_cache_tokens(client, monkeypatch):
query_raw = AsyncMock(
return_value=[
{
"api_key": "prod-key",
"model": "anthropic/claude-haiku-4-5",
"prompt_tokens": 8865,
"cache_read_input_tokens": 4402,
"cache_creation_input_tokens": 4402,
"api_requests": 5,
}
]
)
_set_prisma(monkeypatch, query_raw)
response = client.get(
"/global/activity/cache_hits/prompt_caching",
params={"start_date": "2026-07-10", "end_date": "2026-07-18"},
)
assert response.status_code == 200
assert response.json() == [
{
"api_key": "prod-key",
"model": "anthropic/claude-haiku-4-5",
"prompt_tokens": 8865,
"cache_read_input_tokens": 4402,
"cache_creation_input_tokens": 4402,
"api_requests": 5,
}
]
def test_prompt_cache_activity_reads_daily_spend_cache_columns(client, monkeypatch):
query_raw = AsyncMock(return_value=[])
_set_prisma(monkeypatch, query_raw)
client.get(
"/global/activity/cache_hits/prompt_caching",
params={"start_date": "2026-07-10", "end_date": "2026-07-18"},
)
sql = query_raw.await_args.args[0]
assert "LiteLLM_DailyUserSpend" in sql
assert "cache_read_input_tokens" in sql
assert "cache_creation_input_tokens" in sql
assert query_raw.await_args.args[1:] == ("2026-07-10", "2026-07-18")
def test_prompt_cache_activity_empty_response(client, monkeypatch):
_set_prisma(monkeypatch, AsyncMock(return_value=None))
response = client.get(
"/global/activity/cache_hits/prompt_caching",
params={"start_date": "2026-07-10", "end_date": "2026-07-18"},
)
assert response.status_code == 200
assert response.json() == []
def test_prompt_cache_activity_requires_dates(client, monkeypatch):
_set_prisma(monkeypatch, AsyncMock(return_value=[]))
response = client.get("/global/activity/cache_hits/prompt_caching")
assert response.status_code == 400
def test_prompt_cache_activity_no_prisma(client, monkeypatch):
monkeypatch.setattr(proxy_server, "prisma_client", None)
response = client.get(
"/global/activity/cache_hits/prompt_caching",
params={"start_date": "2026-07-10", "end_date": "2026-07-18"},
)
assert response.status_code == 400
assert "Database not connected" in response.text

View file

@ -4,16 +4,29 @@ import { screen, waitFor, within } from "@testing-library/react";
import { renderWithProviders } from "../../../../../tests/test-utils";
import CacheDashboard from "./cache_dashboard";
const { adminGlobalCacheActivity, cachingHealthCheckCall } = vi.hoisted(() => ({
const { adminGlobalCacheActivity, adminGlobalPromptCacheActivity, cachingHealthCheckCall } = vi.hoisted(() => ({
adminGlobalCacheActivity: vi.fn(),
adminGlobalPromptCacheActivity: vi.fn(),
cachingHealthCheckCall: vi.fn(),
}));
vi.mock("@/components/networking", () => ({
adminGlobalCacheActivity,
adminGlobalPromptCacheActivity,
cachingHealthCheckCall,
}));
const promptCacheActivity = [
{
api_key: "sk-1",
model: "claude-haiku-4-5",
prompt_tokens: 80000,
cache_read_input_tokens: 71000,
cache_creation_input_tokens: 4000,
api_requests: 12,
},
];
const cacheActivity = [
{
api_key: "sk-1",
@ -46,8 +59,12 @@ const findChartCards = async () => {
expect(document.querySelectorAll("path.recharts-rectangle").length).toBeGreaterThan(0);
});
const cards = Array.from(document.querySelectorAll('[data-slot="card"]'));
expect(cards).toHaveLength(2);
return { requestsCard: cards[0] as HTMLElement, tokensCard: cards[1] as HTMLElement };
expect(cards).toHaveLength(3);
return {
requestsCard: cards[0] as HTMLElement,
tokensCard: cards[1] as HTMLElement,
promptCacheCard: cards[2] as HTMLElement,
};
};
const barFills = (card: HTMLElement) =>
@ -67,6 +84,23 @@ describe("CacheDashboard cache analytics charts", () => {
beforeEach(() => {
vi.clearAllMocks();
adminGlobalCacheActivity.mockResolvedValue(cacheActivity);
adminGlobalPromptCacheActivity.mockResolvedValue(promptCacheActivity);
});
it("surfaces provider prompt-cache read/creation tokens the response-cache view omits", async () => {
renderDashboard();
const { promptCacheCard } = await findChartCards();
expect(screen.getByText("Provider Prompt Caching")).toBeInTheDocument();
expect(screen.getByText("71K")).toBeInTheDocument();
expect(within(promptCacheCard).getByText("Prompt Cache Input Tokens by Model")).toBeInTheDocument();
expect(legendFillByCategory(promptCacheCard)).toEqual({
"Uncached Input Tokens": "var(--color-sky-500, #0ea5e9)",
"Cache Read Input Tokens": "var(--color-teal-500, #14b8a6)",
"Cache Creation Input Tokens": "var(--color-indigo-500, #6366f1)",
});
expect(within(promptCacheCard).getAllByText("claude-haiku-4-5").length).toBeGreaterThan(0);
});
it("renders both chart card titles", async () => {

View file

@ -13,14 +13,19 @@ import {
TabPanels,
Text,
} from "@tremor/react";
import React, { useEffect, useState } from "react";
import React, { useEffect, useMemo, useState } from "react";
import NotificationsManager from "@/components/molecules/notifications_manager";
import UsageDatePicker from "@/components/shared/usage_date_picker";
import { BarChart } from "@/components/shared/charts";
import { Card as ChartCard, CardContent, CardHeader, CardTitle } from "@/components/ui/card";
import { RefreshIcon } from "@heroicons/react/outline";
import { adminGlobalCacheActivity, cachingHealthCheckCall } from "@/components/networking";
import {
adminGlobalCacheActivity,
adminGlobalPromptCacheActivity,
cachingHealthCheckCall,
PromptCacheActivityItem,
} from "@/components/networking";
// Import the new component
import { CacheHealthTab } from "./cache_health";
@ -70,6 +75,13 @@ type uiData = {
"Generated Completion Tokens": number;
};
type promptCacheUiData = {
name: string;
"Uncached Input Tokens": number;
"Cache Read Input Tokens": number;
"Cache Creation Input Tokens": number;
};
interface CacheHealthResponse {
status?: string;
cache_type?: string;
@ -102,6 +114,7 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
const [selectedApiKeys, setSelectedApiKeys] = useState<string[]>([]);
const [selectedModels, setSelectedModels] = useState<string[]>([]);
const [data, setData] = useState<cacheDataItem[]>([]);
const [promptCacheData, setPromptCacheData] = useState<PromptCacheActivityItem[]>([]);
const [cachedResponses, setCachedResponses] = useState("0");
const [cachedTokens, setCachedTokens] = useState("0");
const [cacheHitRatio, setCacheHitRatio] = useState("0");
@ -125,6 +138,12 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
formatDateWithoutTZ(dateValue.to),
);
setData(response);
const promptCacheResponse = await adminGlobalPromptCacheActivity(
accessToken,
formatDateWithoutTZ(dateValue.from),
formatDateWithoutTZ(dateValue.to),
);
setPromptCacheData(promptCacheResponse);
};
fetchData();
@ -148,6 +167,14 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
);
setData(new_cache_data);
let new_prompt_cache_data = await adminGlobalPromptCacheActivity(
accessToken,
formatDateWithoutTZ(startTime),
formatDateWithoutTZ(endTime),
);
setPromptCacheData(new_prompt_cache_data);
};
useEffect(() => {
@ -224,6 +251,36 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
setFilteredData(processedData);
}, [selectedApiKeys, selectedModels, dateValue, data]);
const { promptCacheChart, promptCacheReadTokens, promptCacheCreationTokens } = useMemo(() => {
const rows = promptCacheData
.filter((item) => selectedApiKeys.length === 0 || selectedApiKeys.includes(item.api_key))
.filter((item) => selectedModels.length === 0 || (item.model !== null && selectedModels.includes(item.model)));
const byModel = new Map<string, promptCacheUiData>();
for (const item of rows) {
const cacheRead = item.cache_read_input_tokens || 0;
const cacheCreation = item.cache_creation_input_tokens || 0;
const uncached = Math.max((item.prompt_tokens || 0) - cacheRead - cacheCreation, 0);
const name = item.model || "Unknown";
const existing = byModel.get(name);
byModel.set(name, {
name,
"Uncached Input Tokens": (existing?.["Uncached Input Tokens"] ?? 0) + uncached,
"Cache Read Input Tokens": (existing?.["Cache Read Input Tokens"] ?? 0) + cacheRead,
"Cache Creation Input Tokens": (existing?.["Cache Creation Input Tokens"] ?? 0) + cacheCreation,
});
}
const sum = (pick: (item: PromptCacheActivityItem) => number) =>
rows.reduce((total, item) => total + (pick(item) || 0), 0);
return {
promptCacheChart: Array.from(byModel.values()),
promptCacheReadTokens: valueFormatterNumbers(sum((item) => item.cache_read_input_tokens)),
promptCacheCreationTokens: valueFormatterNumbers(sum((item) => item.cache_creation_input_tokens)),
};
}, [selectedApiKeys, selectedModels, promptCacheData]);
const handleRefreshClick = () => {
// Update the 'lastRefreshed' state to the current date and time
const currentDate = new Date();
@ -385,6 +442,56 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
/>
</CardContent>
</ChartCard>
<div className="mt-8">
<p className="text-tremor-title font-semibold text-tremor-content-strong dark:text-dark-tremor-content-strong">
Provider Prompt Caching
</p>
<p className="text-tremor-default text-tremor-content dark:text-dark-tremor-content mt-1">
Input tokens cached by the provider (e.g. Anthropic prompt caching). Tracked separately from
LiteLLM&apos;s response cache above.
</p>
</div>
<div className="grid grid-cols-1 gap-6 sm:grid-cols-2 mt-4">
<Card>
<p className="text-tremor-default font-medium text-tremor-content dark:text-dark-tremor-content">
Cache Read Input Tokens
</p>
<div className="mt-2 flex items-baseline space-x-2.5">
<p className="text-tremor-metric font-semibold text-tremor-content-strong dark:text-dark-tremor-content-strong">
{promptCacheReadTokens}
</p>
</div>
</Card>
<Card>
<p className="text-tremor-default font-medium text-tremor-content dark:text-dark-tremor-content">
Cache Creation Input Tokens
</p>
<div className="mt-2 flex items-baseline space-x-2.5">
<p className="text-tremor-metric font-semibold text-tremor-content-strong dark:text-dark-tremor-content-strong">
{promptCacheCreationTokens}
</p>
</div>
</Card>
</div>
<ChartCard className="mt-4">
<CardHeader>
<CardTitle className="text-base font-semibold">Prompt Cache Input Tokens by Model</CardTitle>
</CardHeader>
<CardContent>
<BarChart
data={promptCacheChart}
stack={true}
index="name"
valueFormatter={valueFormatterNumbers}
categories={["Uncached Input Tokens", "Cache Read Input Tokens", "Cache Creation Input Tokens"]}
colors={["sky", "teal", "indigo"]}
yAxisWidth={48}
/>
</CardContent>
</ChartCard>
</Card>
</TabPanel>
<TabPanel>

View file

@ -2163,6 +2163,27 @@ export const adminGlobalCacheActivity = async (
}
};
export interface PromptCacheActivityItem {
api_key: string;
model: string | null;
prompt_tokens: number;
cache_read_input_tokens: number;
cache_creation_input_tokens: number;
api_requests: number;
}
export const adminGlobalPromptCacheActivity = async (
accessToken: string,
startTime: string | undefined,
endTime: string | undefined,
): Promise<PromptCacheActivityItem[]> => {
const query = startTime && endTime ? { start_date: startTime, end_date: endTime } : undefined;
return await apiClient.get<PromptCacheActivityItem[]>(`/global/activity/cache_hits/prompt_caching`, {
accessToken,
query,
});
};
export const adminGlobalActivityPerModel = async (
accessToken: string,
startTime: string | undefined,

View file

@ -4385,6 +4385,23 @@ export interface paths {
patch?: never;
trace?: never;
};
"/global/activity/cache_hits/prompt_caching": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/** Get Prompt Cache Activity */
get: operations["get_prompt_cache_activity_global_activity_cache_hits_prompt_caching_get"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/global/activity/exceptions": {
parameters: {
query?: never;
@ -29590,6 +29607,21 @@ export interface components {
prompt_id: string;
prompt_info?: components["schemas"]["PromptInfo"] | null;
};
/** PromptCacheActivityRow */
PromptCacheActivityRow: {
/** Api Key */
api_key: string;
/** Api Requests */
api_requests: number;
/** Cache Creation Input Tokens */
cache_creation_input_tokens: number;
/** Cache Read Input Tokens */
cache_read_input_tokens: number;
/** Model */
model?: string | null;
/** Prompt Tokens */
prompt_tokens: number;
};
/** PromptInfo */
PromptInfo: {
/**
@ -40202,6 +40234,40 @@ export interface operations {
};
};
};
get_prompt_cache_activity_global_activity_cache_hits_prompt_caching_get: {
parameters: {
query?: {
/** @description Time from which to start viewing prompt-cache usage */
start_date?: string | null;
/** @description Time till which to view prompt-cache usage */
end_date?: string | null;
};
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["PromptCacheActivityRow"][];
};
};
/** @description Validation Error */
422: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["HTTPValidationError"];
};
};
};
};
get_global_activity_exceptions_global_activity_exceptions_get: {
parameters: {
query: {