From 1c89dec3feb74c4d834ed8c214fdc51eaa1e7ed9 Mon Sep 17 00:00:00 2001 From: Mubashir Osmani Date: Fri, 17 Jul 2026 22:59:41 +0000 Subject: [PATCH] feat(ui): surface provider prompt-cache tokens in Cache Analytics The Caching page's cache-hit and cached-token views only reflect LiteLLM's own response cache (LiteLLM_SpendLogs.cache_hit). Provider prompt caching (e.g. Anthropic cache_read_input_tokens) never sets that flag, so those tokens were invisible and the overall numbers did not add up for anthropic_messages traffic. Add a /global/activity/cache_hits/prompt_caching endpoint that reads cache_read_input_tokens / cache_creation_input_tokens from the daily spend rollup (the only place they are persisted) and a Provider Prompt Caching section on the dashboard with read/creation token stats and a per-model chart, kept separate from the response-cache metrics. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../analytics_endpoints.py | 76 ++++++++++++ .../test_analytics_endpoints.py | 110 +++++++++++++++++ .../_components/cache_dashboard.test.tsx | 40 ++++++- .../caching/_components/cache_dashboard.tsx | 111 +++++++++++++++++- .../src/components/networking.tsx | 21 ++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 66 +++++++++++ 6 files changed, 419 insertions(+), 5 deletions(-) create mode 100644 tests/test_litellm/proxy/analytics_endpoints/test_analytics_endpoints.py diff --git a/litellm/proxy/analytics_endpoints/analytics_endpoints.py b/litellm/proxy/analytics_endpoints/analytics_endpoints.py index 4c1ff31e5a1..8e6f9922470 100644 --- a/litellm/proxy/analytics_endpoints/analytics_endpoints.py +++ b/litellm/proxy/analytics_endpoints/analytics_endpoints.py @@ -4,6 +4,7 @@ from typing import List, Optional import fastapi from fastapi import APIRouter, Depends, HTTPException, status +from pydantic import BaseModel, TypeAdapter from litellm.proxy._types import * from litellm.proxy.auth.user_api_key_auth import user_api_key_auth @@ -11,6 +12,18 @@ from litellm.proxy.auth.user_api_key_auth import user_api_key_auth router = APIRouter() +class PromptCacheActivityRow(BaseModel): + api_key: str + model: str | None = None + prompt_tokens: int + cache_read_input_tokens: int + cache_creation_input_tokens: int + api_requests: int + + +_PROMPT_CACHE_ROWS_ADAPTER = TypeAdapter(list[PromptCacheActivityRow]) + + @router.get( "/global/activity/cache_hits", tags=["Budget & Spend Tracking"], @@ -103,3 +116,66 @@ async def get_global_activity( status_code=status.HTTP_400_BAD_REQUEST, detail={"error": str(e)}, ) + + +@router.get( + "/global/activity/cache_hits/prompt_caching", + tags=["Budget & Spend Tracking"], + dependencies=[Depends(user_api_key_auth)], + responses={ + 200: {"model": list[PromptCacheActivityRow]}, + }, + include_in_schema=False, +) +async def get_prompt_cache_activity( + start_date: str | None = fastapi.Query( + default=None, + description="Time from which to start viewing prompt-cache usage", + ), + end_date: str | None = fastapi.Query( + default=None, + description="Time till which to view prompt-cache usage", + ), +) -> list[PromptCacheActivityRow]: + if start_date is None or end_date is None: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail={"error": "Please provide start_date and end_date"}, + ) + + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail={ + "error": "Database not connected. Connect a database to your proxy - https://docs.litellm.ai/docs/simple_proxy#managing-auth---virtual-keys" + }, + ) + + sql_query = """ + SELECT + CASE + WHEN vt."key_alias" IS NOT NULL THEN vt."key_alias" + ELSE 'Unnamed Key' + END AS api_key, + ds."model", + SUM(ds."prompt_tokens")::bigint AS prompt_tokens, + SUM(ds."cache_read_input_tokens")::bigint AS cache_read_input_tokens, + SUM(ds."cache_creation_input_tokens")::bigint AS cache_creation_input_tokens, + SUM(ds."api_requests")::bigint AS api_requests + FROM "LiteLLM_DailyUserSpend" ds + LEFT JOIN "LiteLLM_VerificationToken" vt ON ds."api_key" = vt."token" + WHERE + ds."date" >= $1 + AND ds."date" <= $2 + GROUP BY + vt."key_alias", + ds."model" + """ + db_response = await prisma_client.db.query_raw(sql_query, start_date, end_date) + + if db_response is None: + return [] + + return _PROMPT_CACHE_ROWS_ADAPTER.validate_python(db_response) diff --git a/tests/test_litellm/proxy/analytics_endpoints/test_analytics_endpoints.py b/tests/test_litellm/proxy/analytics_endpoints/test_analytics_endpoints.py new file mode 100644 index 00000000000..b830563378b --- /dev/null +++ b/tests/test_litellm/proxy/analytics_endpoints/test_analytics_endpoints.py @@ -0,0 +1,110 @@ +from __future__ import annotations + +import sys +from pathlib import Path +from unittest.mock import AsyncMock, MagicMock + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient + +sys.path.insert(0, str(Path(__file__).resolve().parents[4])) + +from litellm.proxy import proxy_server +from litellm.proxy.analytics_endpoints.analytics_endpoints import router +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + + +@pytest.fixture +def client() -> TestClient: + app = FastAPI() + app.include_router(router) + app.dependency_overrides[user_api_key_auth] = lambda: MagicMock() + return TestClient(app, raise_server_exceptions=False) + + +def _set_prisma(monkeypatch: pytest.MonkeyPatch, query_raw: AsyncMock) -> None: + pc = MagicMock() + pc.db.query_raw = query_raw + monkeypatch.setattr(proxy_server, "prisma_client", pc) + + +def test_prompt_cache_activity_returns_provider_cache_tokens(client, monkeypatch): + query_raw = AsyncMock( + return_value=[ + { + "api_key": "prod-key", + "model": "anthropic/claude-haiku-4-5", + "prompt_tokens": 8865, + "cache_read_input_tokens": 4402, + "cache_creation_input_tokens": 4402, + "api_requests": 5, + } + ] + ) + _set_prisma(monkeypatch, query_raw) + + response = client.get( + "/global/activity/cache_hits/prompt_caching", + params={"start_date": "2026-07-10", "end_date": "2026-07-18"}, + ) + + assert response.status_code == 200 + assert response.json() == [ + { + "api_key": "prod-key", + "model": "anthropic/claude-haiku-4-5", + "prompt_tokens": 8865, + "cache_read_input_tokens": 4402, + "cache_creation_input_tokens": 4402, + "api_requests": 5, + } + ] + + +def test_prompt_cache_activity_reads_daily_spend_cache_columns(client, monkeypatch): + query_raw = AsyncMock(return_value=[]) + _set_prisma(monkeypatch, query_raw) + + client.get( + "/global/activity/cache_hits/prompt_caching", + params={"start_date": "2026-07-10", "end_date": "2026-07-18"}, + ) + + sql = query_raw.await_args.args[0] + assert "LiteLLM_DailyUserSpend" in sql + assert "cache_read_input_tokens" in sql + assert "cache_creation_input_tokens" in sql + assert query_raw.await_args.args[1:] == ("2026-07-10", "2026-07-18") + + +def test_prompt_cache_activity_empty_response(client, monkeypatch): + _set_prisma(monkeypatch, AsyncMock(return_value=None)) + + response = client.get( + "/global/activity/cache_hits/prompt_caching", + params={"start_date": "2026-07-10", "end_date": "2026-07-18"}, + ) + + assert response.status_code == 200 + assert response.json() == [] + + +def test_prompt_cache_activity_requires_dates(client, monkeypatch): + _set_prisma(monkeypatch, AsyncMock(return_value=[])) + + response = client.get("/global/activity/cache_hits/prompt_caching") + + assert response.status_code == 400 + + +def test_prompt_cache_activity_no_prisma(client, monkeypatch): + monkeypatch.setattr(proxy_server, "prisma_client", None) + + response = client.get( + "/global/activity/cache_hits/prompt_caching", + params={"start_date": "2026-07-10", "end_date": "2026-07-18"}, + ) + + assert response.status_code == 400 + assert "Database not connected" in response.text diff --git a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx index 17d14cd7fac..7a3448352bb 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx @@ -4,16 +4,29 @@ import { screen, waitFor, within } from "@testing-library/react"; import { renderWithProviders } from "../../../../../tests/test-utils"; import CacheDashboard from "./cache_dashboard"; -const { adminGlobalCacheActivity, cachingHealthCheckCall } = vi.hoisted(() => ({ +const { adminGlobalCacheActivity, adminGlobalPromptCacheActivity, cachingHealthCheckCall } = vi.hoisted(() => ({ adminGlobalCacheActivity: vi.fn(), + adminGlobalPromptCacheActivity: vi.fn(), cachingHealthCheckCall: vi.fn(), })); vi.mock("@/components/networking", () => ({ adminGlobalCacheActivity, + adminGlobalPromptCacheActivity, cachingHealthCheckCall, })); +const promptCacheActivity = [ + { + api_key: "sk-1", + model: "claude-haiku-4-5", + prompt_tokens: 80000, + cache_read_input_tokens: 71000, + cache_creation_input_tokens: 4000, + api_requests: 12, + }, +]; + const cacheActivity = [ { api_key: "sk-1", @@ -46,8 +59,12 @@ const findChartCards = async () => { expect(document.querySelectorAll("path.recharts-rectangle").length).toBeGreaterThan(0); }); const cards = Array.from(document.querySelectorAll('[data-slot="card"]')); - expect(cards).toHaveLength(2); - return { requestsCard: cards[0] as HTMLElement, tokensCard: cards[1] as HTMLElement }; + expect(cards).toHaveLength(3); + return { + requestsCard: cards[0] as HTMLElement, + tokensCard: cards[1] as HTMLElement, + promptCacheCard: cards[2] as HTMLElement, + }; }; const barFills = (card: HTMLElement) => @@ -67,6 +84,23 @@ describe("CacheDashboard cache analytics charts", () => { beforeEach(() => { vi.clearAllMocks(); adminGlobalCacheActivity.mockResolvedValue(cacheActivity); + adminGlobalPromptCacheActivity.mockResolvedValue(promptCacheActivity); + }); + + it("surfaces provider prompt-cache read/creation tokens the response-cache view omits", async () => { + renderDashboard(); + const { promptCacheCard } = await findChartCards(); + + expect(screen.getByText("Provider Prompt Caching")).toBeInTheDocument(); + expect(screen.getByText("71K")).toBeInTheDocument(); + + expect(within(promptCacheCard).getByText("Prompt Cache Input Tokens by Model")).toBeInTheDocument(); + expect(legendFillByCategory(promptCacheCard)).toEqual({ + "Uncached Input Tokens": "var(--color-sky-500, #0ea5e9)", + "Cache Read Input Tokens": "var(--color-teal-500, #14b8a6)", + "Cache Creation Input Tokens": "var(--color-indigo-500, #6366f1)", + }); + expect(within(promptCacheCard).getAllByText("claude-haiku-4-5").length).toBeGreaterThan(0); }); it("renders both chart card titles", async () => { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx index b8e8dc8adb1..25c95b82f26 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx @@ -13,14 +13,19 @@ import { TabPanels, Text, } from "@tremor/react"; -import React, { useEffect, useState } from "react"; +import React, { useEffect, useMemo, useState } from "react"; import NotificationsManager from "@/components/molecules/notifications_manager"; import UsageDatePicker from "@/components/shared/usage_date_picker"; import { BarChart } from "@/components/shared/charts"; import { Card as ChartCard, CardContent, CardHeader, CardTitle } from "@/components/ui/card"; import { RefreshIcon } from "@heroicons/react/outline"; -import { adminGlobalCacheActivity, cachingHealthCheckCall } from "@/components/networking"; +import { + adminGlobalCacheActivity, + adminGlobalPromptCacheActivity, + cachingHealthCheckCall, + PromptCacheActivityItem, +} from "@/components/networking"; // Import the new component import { CacheHealthTab } from "./cache_health"; @@ -70,6 +75,13 @@ type uiData = { "Generated Completion Tokens": number; }; +type promptCacheUiData = { + name: string; + "Uncached Input Tokens": number; + "Cache Read Input Tokens": number; + "Cache Creation Input Tokens": number; +}; + interface CacheHealthResponse { status?: string; cache_type?: string; @@ -102,6 +114,7 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole const [selectedApiKeys, setSelectedApiKeys] = useState([]); const [selectedModels, setSelectedModels] = useState([]); const [data, setData] = useState([]); + const [promptCacheData, setPromptCacheData] = useState([]); const [cachedResponses, setCachedResponses] = useState("0"); const [cachedTokens, setCachedTokens] = useState("0"); const [cacheHitRatio, setCacheHitRatio] = useState("0"); @@ -125,6 +138,12 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole formatDateWithoutTZ(dateValue.to), ); setData(response); + const promptCacheResponse = await adminGlobalPromptCacheActivity( + accessToken, + formatDateWithoutTZ(dateValue.from), + formatDateWithoutTZ(dateValue.to), + ); + setPromptCacheData(promptCacheResponse); }; fetchData(); @@ -148,6 +167,14 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole ); setData(new_cache_data); + + let new_prompt_cache_data = await adminGlobalPromptCacheActivity( + accessToken, + formatDateWithoutTZ(startTime), + formatDateWithoutTZ(endTime), + ); + + setPromptCacheData(new_prompt_cache_data); }; useEffect(() => { @@ -224,6 +251,36 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole setFilteredData(processedData); }, [selectedApiKeys, selectedModels, dateValue, data]); + const { promptCacheChart, promptCacheReadTokens, promptCacheCreationTokens } = useMemo(() => { + const rows = promptCacheData + .filter((item) => selectedApiKeys.length === 0 || selectedApiKeys.includes(item.api_key)) + .filter((item) => selectedModels.length === 0 || (item.model !== null && selectedModels.includes(item.model))); + + const byModel = new Map(); + for (const item of rows) { + const cacheRead = item.cache_read_input_tokens || 0; + const cacheCreation = item.cache_creation_input_tokens || 0; + const uncached = Math.max((item.prompt_tokens || 0) - cacheRead - cacheCreation, 0); + const name = item.model || "Unknown"; + const existing = byModel.get(name); + byModel.set(name, { + name, + "Uncached Input Tokens": (existing?.["Uncached Input Tokens"] ?? 0) + uncached, + "Cache Read Input Tokens": (existing?.["Cache Read Input Tokens"] ?? 0) + cacheRead, + "Cache Creation Input Tokens": (existing?.["Cache Creation Input Tokens"] ?? 0) + cacheCreation, + }); + } + + const sum = (pick: (item: PromptCacheActivityItem) => number) => + rows.reduce((total, item) => total + (pick(item) || 0), 0); + + return { + promptCacheChart: Array.from(byModel.values()), + promptCacheReadTokens: valueFormatterNumbers(sum((item) => item.cache_read_input_tokens)), + promptCacheCreationTokens: valueFormatterNumbers(sum((item) => item.cache_creation_input_tokens)), + }; + }, [selectedApiKeys, selectedModels, promptCacheData]); + const handleRefreshClick = () => { // Update the 'lastRefreshed' state to the current date and time const currentDate = new Date(); @@ -385,6 +442,56 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole /> + +
+

+ Provider Prompt Caching +

+

+ Input tokens cached by the provider (e.g. Anthropic prompt caching). Tracked separately from + LiteLLM's response cache above. +

+
+ +
+ +

+ Cache Read Input Tokens +

+
+

+ {promptCacheReadTokens} +

+
+
+ +

+ Cache Creation Input Tokens +

+
+

+ {promptCacheCreationTokens} +

+
+
+
+ + + + Prompt Cache Input Tokens by Model + + + + + diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index d44a491b840..cff7bac699d 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -2163,6 +2163,27 @@ export const adminGlobalCacheActivity = async ( } }; +export interface PromptCacheActivityItem { + api_key: string; + model: string | null; + prompt_tokens: number; + cache_read_input_tokens: number; + cache_creation_input_tokens: number; + api_requests: number; +} + +export const adminGlobalPromptCacheActivity = async ( + accessToken: string, + startTime: string | undefined, + endTime: string | undefined, +): Promise => { + const query = startTime && endTime ? { start_date: startTime, end_date: endTime } : undefined; + return await apiClient.get(`/global/activity/cache_hits/prompt_caching`, { + accessToken, + query, + }); +}; + export const adminGlobalActivityPerModel = async ( accessToken: string, startTime: string | undefined, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 0d8f55164f9..0a04c360365 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -4385,6 +4385,23 @@ export interface paths { patch?: never; trace?: never; }; + "/global/activity/cache_hits/prompt_caching": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** Get Prompt Cache Activity */ + get: operations["get_prompt_cache_activity_global_activity_cache_hits_prompt_caching_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/global/activity/exceptions": { parameters: { query?: never; @@ -29590,6 +29607,21 @@ export interface components { prompt_id: string; prompt_info?: components["schemas"]["PromptInfo"] | null; }; + /** PromptCacheActivityRow */ + PromptCacheActivityRow: { + /** Api Key */ + api_key: string; + /** Api Requests */ + api_requests: number; + /** Cache Creation Input Tokens */ + cache_creation_input_tokens: number; + /** Cache Read Input Tokens */ + cache_read_input_tokens: number; + /** Model */ + model?: string | null; + /** Prompt Tokens */ + prompt_tokens: number; + }; /** PromptInfo */ PromptInfo: { /** @@ -40202,6 +40234,40 @@ export interface operations { }; }; }; + get_prompt_cache_activity_global_activity_cache_hits_prompt_caching_get: { + parameters: { + query?: { + /** @description Time from which to start viewing prompt-cache usage */ + start_date?: string | null; + /** @description Time till which to view prompt-cache usage */ + end_date?: string | null; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["PromptCacheActivityRow"][]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; get_global_activity_exceptions_global_activity_exceptions_get: { parameters: { query: {