diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py
index 06395a3c3cc..41a4a977d46 100644
--- a/litellm/proxy/spend_tracking/spend_management_endpoints.py
+++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py
@@ -152,6 +152,7 @@ class _SessionSpendRow(TypedDict):
session_total_spend: float
mcp_tool_call_count: int
mcp_tool_call_spend: float
+ session_cache_hit_count: ReadOnly[int]
class _SpendSumAggregate(TypedDict, total=False):
@@ -2239,6 +2240,10 @@ async def ui_view_spend_logs(
status_filter: str | None = fastapi.Query(
default=None, description="Filter logs by status (e.g., success, failure)"
),
+ cache_hit: str | None = fastapi.Query(
+ default=None,
+ description="Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss)",
+ ),
model: str | None = fastapi.Query(default=None, description="Filter logs by model"),
model_id: str | None = fastapi.Query(
default=None,
@@ -2311,6 +2316,13 @@ async def ui_view_spend_logs(
param="sort_order",
code=status.HTTP_400_BAD_REQUEST,
)
+ if cache_hit is not None and cache_hit.lower() not in {"true", "false"}:
+ raise ProxyException(
+ message=f"Invalid cache_hit: {cache_hit}. Must be one of: true, false",
+ type="bad_request",
+ param="cache_hit",
+ code=status.HTTP_400_BAD_REQUEST,
+ )
try:
is_admin_view: Final = _is_admin_view_safe(user_api_key_dict=user_api_key_dict)
@@ -2542,6 +2554,12 @@ async def ui_view_spend_logs(
sql_params.append(f"%{like_escaped_session_id}%")
p += 1
+ if cache_hit is not None:
+ if cache_hit.lower() == "true":
+ sql_conditions.append("LOWER(cache_hit) = 'true'")
+ else:
+ sql_conditions.append("(cache_hit IS NULL OR LOWER(cache_hit) <> 'true')")
+
# Status filter
if status_filter is not None:
if status_filter == "success":
@@ -4093,7 +4111,8 @@ async def _build_ui_spend_logs_response(
)::int AS mcp_tool_call_count,
COALESCE(SUM(spend) FILTER (
WHERE call_type IN ('call_mcp_tool', 'list_mcp_tools')
- ), 0)::double precision AS mcp_tool_call_spend
+ ), 0)::double precision AS mcp_tool_call_spend,
+ COUNT(*) FILTER (WHERE LOWER(cache_hit) = 'true')::int AS session_cache_hit_count
FROM "LiteLLM_SpendLogs"
WHERE session_id = ANY($1::text[])
AND api_key = ANY($2::text[])
@@ -4107,6 +4126,7 @@ async def _build_ui_spend_logs_response(
"session_total_spend": float(row.get("session_total_spend") or 0.0),
"mcp_tool_call_count": int(row.get("mcp_tool_call_count") or 0),
"mcp_tool_call_spend": float(row.get("mcp_tool_call_spend") or 0.0),
+ "session_cache_hit_count": int(row.get("session_cache_hit_count") or 0),
}
for row in rows
if row.get("session_id")
@@ -4129,6 +4149,7 @@ async def _build_ui_spend_logs_response(
if session_stats["mcp_tool_call_count"]:
row_dict["mcp_tool_call_count"] = session_stats["mcp_tool_call_count"]
row_dict["mcp_tool_call_spend"] = session_stats["mcp_tool_call_spend"]
+ row_dict["session_cache_hit_count"] = session_stats["session_cache_hit_count"]
enriched.append(row_dict)
response_data: list = enriched
else:
diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py
index 8c15ead8983..3689021793d 100644
--- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py
+++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py
@@ -104,6 +104,10 @@ def _reconstruct_ui_where_from_sql(sql_query, params):
where["OR"] = where.get("OR", []) + [{"multi_team": True}]
elif "status = 'success'" in cond:
where["OR"] = where.get("OR", []) + [{"status": "success"}]
+ elif cond == "LOWER(cache_hit) = 'true'":
+ where["cache_hit"] = {"is_hit": True}
+ elif "cache_hit IS NULL" in cond:
+ where["cache_hit"] = {"is_hit": False}
elif sess:
where["session_id"] = {"contains": str(params[int(sess.group(1)) - 1]).strip("%")}
elif status:
@@ -2302,6 +2306,100 @@ async def test_ui_view_spend_logs_with_status(client, monkeypatch):
app.dependency_overrides.pop(ps.user_api_key_auth, None)
+@pytest.mark.asyncio
+async def test_ui_view_spend_logs_with_cache_hit(client, monkeypatch):
+ mock_spend_logs = [
+ {
+ "id": "log1",
+ "request_id": "req1",
+ "api_key": "sk-test-key",
+ "user": "test_user_1",
+ "team_id": "team1",
+ "spend": 0.05,
+ "startTime": datetime.datetime.now(timezone.utc).isoformat(),
+ "model": "gpt-3.5-turbo",
+ "cache_hit": "True",
+ },
+ {
+ "id": "log2",
+ "request_id": "req2",
+ "api_key": "sk-test-key",
+ "user": "test_user_2",
+ "team_id": "team1",
+ "spend": 0.10,
+ "startTime": datetime.datetime.now(timezone.utc).isoformat(),
+ "model": "gpt-4",
+ "cache_hit": "False",
+ },
+ ]
+
+ def filter_by_cache_hit(where):
+ cache_cond = where.get("cache_hit")
+ if cache_cond is None:
+ return mock_spend_logs
+ if cache_cond["is_hit"]:
+ return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() == "true"]
+ return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() != "true"]
+
+ monkeypatch.setattr(
+ "litellm.proxy.proxy_server.prisma_client",
+ make_ui_spend_logs_mock_prisma(mock_spend_logs, filter_by_cache_hit),
+ )
+
+ start_date, end_date = _default_date_range()
+
+ app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth(
+ user_role=LitellmUserRoles.PROXY_ADMIN
+ )
+ try:
+ response = client.get(
+ "/spend/logs/ui",
+ params={
+ "cache_hit": "true",
+ "start_date": start_date,
+ "end_date": end_date,
+ },
+ headers={"Authorization": "Bearer sk-test"},
+ )
+
+ assert response.status_code == 200
+ data = response.json()
+ assert data["total"] == 1
+ assert len(data["data"]) == 1
+ assert data["data"][0]["request_id"] == "req1"
+
+ response = client.get(
+ "/spend/logs/ui",
+ params={
+ "cache_hit": "false",
+ "start_date": start_date,
+ "end_date": end_date,
+ },
+ headers={"Authorization": "Bearer sk-test"},
+ )
+
+ assert response.status_code == 200
+ data = response.json()
+ assert data["total"] == 1
+ assert len(data["data"]) == 1
+ assert data["data"][0]["request_id"] == "req2"
+
+ response = client.get(
+ "/spend/logs/ui",
+ params={
+ "cache_hit": "maybe",
+ "start_date": start_date,
+ "end_date": end_date,
+ },
+ headers={"Authorization": "Bearer sk-test"},
+ )
+
+ assert response.status_code == 400
+ assert "cache_hit" in response.text
+ finally:
+ app.dependency_overrides.pop(ps.user_api_key_auth, None)
+
+
@pytest.mark.asyncio
async def test_ui_view_spend_logs_with_model(client, monkeypatch):
mock_spend_logs = [
@@ -3749,6 +3847,62 @@ async def test_build_ui_spend_logs_response_sums_multi_round_session_spend():
assert call_args[2] == [api_key]
+@pytest.mark.asyncio
+async def test_build_ui_spend_logs_response_session_cache_hit_count():
+ """
+ Each row of a session must carry session_cache_hit_count aggregated across
+ the whole session so the UI can show how many requests in the session were
+ served from the response cache.
+ """
+ from litellm.proxy.spend_tracking.spend_management_endpoints import (
+ _build_ui_spend_logs_response,
+ )
+
+ session_id = "sess-cache-hits"
+ api_key = "hashed-key-xyz"
+ dict_rows = [
+ {"request_id": "req-1", "session_id": session_id, "call_type": "completion", "api_key": api_key},
+ {"request_id": "req-2", "session_id": session_id, "call_type": "completion", "api_key": api_key},
+ {"request_id": "req-3", "session_id": None, "call_type": "completion", "api_key": api_key},
+ ]
+
+ mock_prisma = MagicMock()
+ mock_prisma.db.litellm_spendlogs.group_by = AsyncMock(
+ return_value=[{"session_id": session_id, "_count": {"session_id": 2}}]
+ )
+ mock_prisma.db.query_raw = AsyncMock(
+ return_value=[
+ {
+ "session_id": session_id,
+ "session_total_spend": 0.05,
+ "mcp_tool_call_count": 0,
+ "mcp_tool_call_spend": 0.0,
+ "session_cache_hit_count": 2,
+ }
+ ]
+ )
+
+ result = await _build_ui_spend_logs_response(
+ prisma_client=mock_prisma,
+ data=dict_rows,
+ total_records=3,
+ page=1,
+ page_size=50,
+ total_pages=1,
+ enrich_session_counts=True,
+ )
+
+ rows = result["data"]
+ assert rows[0]["session_cache_hit_count"] == 2
+ assert rows[1]["session_cache_hit_count"] == 2
+ assert "session_cache_hit_count" not in rows[2]
+
+ # The aggregate SQL must actually compute the cache-hit count.
+ _, call_args, _ = mock_prisma.db.query_raw.mock_calls[0]
+ assert "session_cache_hit_count" in call_args[0]
+ assert "LOWER(cache_hit) = 'true'" in call_args[0]
+
+
# ---------------------------------------------------------------------------
# Tests for /spend/logs team-member permission
# ---------------------------------------------------------------------------
diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx
index 5b6d70b4771..f3c6a030619 100644
--- a/ui/litellm-dashboard/src/components/networking.tsx
+++ b/ui/litellm-dashboard/src/components/networking.tsx
@@ -2002,6 +2002,8 @@ interface UiSpendLogsParams {
user_id?: string;
end_user?: string;
status_filter?: string;
+ /** Filter by response cache result: "true" (cache hit) or "false" (cache miss) */
+ cache_hit?: string;
/** Filter by model name (e.g. "gpt-4") */
model?: string;
/** Filter by model ID (litellm model deployment id) */
diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx
index aab8d2b8cb9..e48fd555dda 100644
--- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx
@@ -359,6 +359,19 @@ describe("LogDetailContent", () => {
expect(screen.queryByText("Response Cache")).not.toBeInTheDocument();
});
+ it("should display the Cache Key next to the Response Cache result", () => {
+ render();
+
+ expect(screen.getByText("Cache Key")).toBeInTheDocument();
+ expect(screen.getByText("abc123cachekey")).toBeInTheDocument();
+ });
+
+ it("should hide the Cache Key row when caching is off", () => {
+ render();
+
+ expect(screen.queryByText("Cache Key")).not.toBeInTheDocument();
+ });
+
it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => {
render(
): number | und
const RESPONSE_CACHE_TOOLTIP =
"Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed.";
const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching";
+const CACHE_KEY_TOOLTIP =
+ "The key LiteLLM computed for this request in the response cache. Requests with the same cache key share a cached response; a different key means the request content did not match any cached entry.";
const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching";
function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) {
@@ -436,6 +438,13 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata:
)}
+ {showResponseCache && logEntry.cache_key && logEntry.cache_key !== "Cache OFF" && (
+ }
+ >
+
+
+ )}
{promptCacheReadTokens > 0 && (
AGENT_CALL_TYPES.includes(row.call_type)).length;
const mcpCount = sessionLogs.filter((row) => MCP_CALL_TYPES.includes(row.call_type)).length;
+ const cacheHitCount = sessionLogs.filter((row) => String(row.cache_hit ?? "").toLowerCase() === "true").length;
const logsForList = isSessionMode ? sessionLogs : currentLog ? [currentLog] : [];
const leftPanelId = isSessionMode ? sessionId || "" : currentLog?.request_id || "";
const leftPanelDisplayId = leftPanelId.length > 14 ? `${leftPanelId.slice(0, 11)}...` : leftPanelId;
@@ -383,7 +384,8 @@ export function LogDetailsDrawer({
{isSessionMode && (
<>
·
- {sessionDurationSeconds}s
+ {sessionDurationSeconds}s·
+ {cacheHitCount}/{logsForList.length} cached
>
)}
diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx
index af6a6d1f178..d6adbcdd55a 100644
--- a/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx
@@ -31,6 +31,12 @@ const STATUS_FILTER_ITEMS = [
{ value: "success", label: "Success" },
{ value: "failure", label: "Failure" },
] as const;
+
+const CACHE_HIT_FILTER_ITEMS = [
+ { value: ALL_VALUE, label: "All Requests" },
+ { value: "true", label: "Cache Hit" },
+ { value: "false", label: "Cache Miss" },
+] as const;
const PAGE_SIZE = 50;
const asString = (value: unknown): string => (typeof value === "string" ? value : "");
@@ -364,6 +370,27 @@ export function RequestLogsFilters({ get, set, teams, logsWindow }: RequestLogsF
/>
+
+
+
+
0 && `${sessionLlmCount} LLM`,
sessionAgentCount > 0 && `${sessionAgentCount} Agent`,
sessionMcpCount > 0 && `${sessionMcpCount} MCP`,
+ log.session_cache_hit_count != null && `${log.session_cache_hit_count} cache hit`,
].filter(Boolean);
return ;
},
diff --git a/ui/litellm-dashboard/src/components/view_logs/columns.tsx b/ui/litellm-dashboard/src/components/view_logs/columns.tsx
index d4f784bf165..eef957922d7 100644
--- a/ui/litellm-dashboard/src/components/view_logs/columns.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/columns.tsx
@@ -42,6 +42,7 @@ export type LogEntry = {
request_duration_ms?: number;
session_total_count?: number;
session_total_spend?: number;
+ session_cache_hit_count?: number;
mcp_tool_call_count?: number;
mcp_tool_call_spend?: number;
session_llm_count?: number;
diff --git a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx
index 26c5bda1593..f474d4218f6 100644
--- a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx
@@ -82,6 +82,7 @@ describe("useLogFilterLogic", () => {
{ id: LOG_FILTER_IDS.SESSION_ID, value: "sess-1", param: "session_id" },
{ id: LOG_FILTER_IDS.END_USER, value: "end-user-1", param: "end_user" },
{ id: LOG_FILTER_IDS.STATUS, value: "failure", param: "status_filter" },
+ { id: LOG_FILTER_IDS.CACHE_HIT, value: "true", param: "cache_hit" },
{ id: LOG_FILTER_IDS.MODEL_ID, value: "model-uuid-1", param: "model_id" },
{ id: LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL, value: "gpt-4o", param: "model" },
{ id: LOG_FILTER_IDS.KEY_ALIAS, value: "alias-1", param: "key_alias" },
diff --git a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx
index 474f51e93b3..f24b8364a08 100644
--- a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx
+++ b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx
@@ -26,6 +26,7 @@ export const LOG_FILTER_IDS = {
ERROR_MESSAGE: "error_message",
KEY_HASH: "key_hash",
SESSION_ID: "session_id",
+ CACHE_HIT: "cache_hit",
MODEL_ID: "model_id",
PUBLIC_MODEL_OR_SEARCH_TOOL: "model",
REQUEST_ID: "request_id",
@@ -42,6 +43,7 @@ export const LOG_FILTER_LABELS: Record = {
[LOG_FILTER_IDS.ERROR_MESSAGE]: "Error Message",
[LOG_FILTER_IDS.KEY_HASH]: "Key Hash",
[LOG_FILTER_IDS.SESSION_ID]: "Session ID",
+ [LOG_FILTER_IDS.CACHE_HIT]: "Cache Hit",
[LOG_FILTER_IDS.MODEL_ID]: "Model",
[LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL]: "Public model / search tool",
};
@@ -167,6 +169,7 @@ export function useLogFilterLogic({
user_id: userIdFilter,
end_user: getFilterValue(columnFilters, LOG_FILTER_IDS.END_USER),
status_filter: getFilterValue(columnFilters, LOG_FILTER_IDS.STATUS),
+ cache_hit: getFilterValue(columnFilters, LOG_FILTER_IDS.CACHE_HIT),
model_id: getFilterValue(columnFilters, LOG_FILTER_IDS.MODEL_ID),
model: getFilterValue(columnFilters, LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL),
key_alias: getFilterValue(columnFilters, LOG_FILTER_IDS.KEY_ALIAS),
diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts
index 4ccf7b59fbc..d9f5af3f22a 100644
--- a/ui/litellm-dashboard/src/lib/http/schema.d.ts
+++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts
@@ -53273,6 +53273,8 @@ export interface operations {
page_size?: number;
/** @description Filter logs by status (e.g., success, failure) */
status_filter?: string | null;
+ /** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */
+ cache_hit?: string | null;
/** @description Filter logs by model */
model?: string | null;
/** @description Filter logs by model ID (litellm model deployment id) */
@@ -53381,6 +53383,8 @@ export interface operations {
page_size?: number;
/** @description Filter logs by status (e.g., success, failure) */
status_filter?: string | null;
+ /** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */
+ cache_hit?: string | null;
/** @description Filter logs by model */
model?: string | null;
/** @description Filter logs by model ID (litellm model deployment id) */