diff --git a/litellm/proxy/spend_tracking/spend_management_endpoints.py b/litellm/proxy/spend_tracking/spend_management_endpoints.py index 06395a3c3cc..41a4a977d46 100644 --- a/litellm/proxy/spend_tracking/spend_management_endpoints.py +++ b/litellm/proxy/spend_tracking/spend_management_endpoints.py @@ -152,6 +152,7 @@ class _SessionSpendRow(TypedDict): session_total_spend: float mcp_tool_call_count: int mcp_tool_call_spend: float + session_cache_hit_count: ReadOnly[int] class _SpendSumAggregate(TypedDict, total=False): @@ -2239,6 +2240,10 @@ async def ui_view_spend_logs( status_filter: str | None = fastapi.Query( default=None, description="Filter logs by status (e.g., success, failure)" ), + cache_hit: str | None = fastapi.Query( + default=None, + description="Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss)", + ), model: str | None = fastapi.Query(default=None, description="Filter logs by model"), model_id: str | None = fastapi.Query( default=None, @@ -2311,6 +2316,13 @@ async def ui_view_spend_logs( param="sort_order", code=status.HTTP_400_BAD_REQUEST, ) + if cache_hit is not None and cache_hit.lower() not in {"true", "false"}: + raise ProxyException( + message=f"Invalid cache_hit: {cache_hit}. Must be one of: true, false", + type="bad_request", + param="cache_hit", + code=status.HTTP_400_BAD_REQUEST, + ) try: is_admin_view: Final = _is_admin_view_safe(user_api_key_dict=user_api_key_dict) @@ -2542,6 +2554,12 @@ async def ui_view_spend_logs( sql_params.append(f"%{like_escaped_session_id}%") p += 1 + if cache_hit is not None: + if cache_hit.lower() == "true": + sql_conditions.append("LOWER(cache_hit) = 'true'") + else: + sql_conditions.append("(cache_hit IS NULL OR LOWER(cache_hit) <> 'true')") + # Status filter if status_filter is not None: if status_filter == "success": @@ -4093,7 +4111,8 @@ async def _build_ui_spend_logs_response( )::int AS mcp_tool_call_count, COALESCE(SUM(spend) FILTER ( WHERE call_type IN ('call_mcp_tool', 'list_mcp_tools') - ), 0)::double precision AS mcp_tool_call_spend + ), 0)::double precision AS mcp_tool_call_spend, + COUNT(*) FILTER (WHERE LOWER(cache_hit) = 'true')::int AS session_cache_hit_count FROM "LiteLLM_SpendLogs" WHERE session_id = ANY($1::text[]) AND api_key = ANY($2::text[]) @@ -4107,6 +4126,7 @@ async def _build_ui_spend_logs_response( "session_total_spend": float(row.get("session_total_spend") or 0.0), "mcp_tool_call_count": int(row.get("mcp_tool_call_count") or 0), "mcp_tool_call_spend": float(row.get("mcp_tool_call_spend") or 0.0), + "session_cache_hit_count": int(row.get("session_cache_hit_count") or 0), } for row in rows if row.get("session_id") @@ -4129,6 +4149,7 @@ async def _build_ui_spend_logs_response( if session_stats["mcp_tool_call_count"]: row_dict["mcp_tool_call_count"] = session_stats["mcp_tool_call_count"] row_dict["mcp_tool_call_spend"] = session_stats["mcp_tool_call_spend"] + row_dict["session_cache_hit_count"] = session_stats["session_cache_hit_count"] enriched.append(row_dict) response_data: list = enriched else: diff --git a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py index 8c15ead8983..3689021793d 100644 --- a/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py +++ b/tests/test_litellm/proxy/spend_tracking/test_spend_management_endpoints.py @@ -104,6 +104,10 @@ def _reconstruct_ui_where_from_sql(sql_query, params): where["OR"] = where.get("OR", []) + [{"multi_team": True}] elif "status = 'success'" in cond: where["OR"] = where.get("OR", []) + [{"status": "success"}] + elif cond == "LOWER(cache_hit) = 'true'": + where["cache_hit"] = {"is_hit": True} + elif "cache_hit IS NULL" in cond: + where["cache_hit"] = {"is_hit": False} elif sess: where["session_id"] = {"contains": str(params[int(sess.group(1)) - 1]).strip("%")} elif status: @@ -2302,6 +2306,100 @@ async def test_ui_view_spend_logs_with_status(client, monkeypatch): app.dependency_overrides.pop(ps.user_api_key_auth, None) +@pytest.mark.asyncio +async def test_ui_view_spend_logs_with_cache_hit(client, monkeypatch): + mock_spend_logs = [ + { + "id": "log1", + "request_id": "req1", + "api_key": "sk-test-key", + "user": "test_user_1", + "team_id": "team1", + "spend": 0.05, + "startTime": datetime.datetime.now(timezone.utc).isoformat(), + "model": "gpt-3.5-turbo", + "cache_hit": "True", + }, + { + "id": "log2", + "request_id": "req2", + "api_key": "sk-test-key", + "user": "test_user_2", + "team_id": "team1", + "spend": 0.10, + "startTime": datetime.datetime.now(timezone.utc).isoformat(), + "model": "gpt-4", + "cache_hit": "False", + }, + ] + + def filter_by_cache_hit(where): + cache_cond = where.get("cache_hit") + if cache_cond is None: + return mock_spend_logs + if cache_cond["is_hit"]: + return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() == "true"] + return [log for log in mock_spend_logs if str(log.get("cache_hit", "")).lower() != "true"] + + monkeypatch.setattr( + "litellm.proxy.proxy_server.prisma_client", + make_ui_spend_logs_mock_prisma(mock_spend_logs, filter_by_cache_hit), + ) + + start_date, end_date = _default_date_range() + + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN + ) + try: + response = client.get( + "/spend/logs/ui", + params={ + "cache_hit": "true", + "start_date": start_date, + "end_date": end_date, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + + assert response.status_code == 200 + data = response.json() + assert data["total"] == 1 + assert len(data["data"]) == 1 + assert data["data"][0]["request_id"] == "req1" + + response = client.get( + "/spend/logs/ui", + params={ + "cache_hit": "false", + "start_date": start_date, + "end_date": end_date, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + + assert response.status_code == 200 + data = response.json() + assert data["total"] == 1 + assert len(data["data"]) == 1 + assert data["data"][0]["request_id"] == "req2" + + response = client.get( + "/spend/logs/ui", + params={ + "cache_hit": "maybe", + "start_date": start_date, + "end_date": end_date, + }, + headers={"Authorization": "Bearer sk-test"}, + ) + + assert response.status_code == 400 + assert "cache_hit" in response.text + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + @pytest.mark.asyncio async def test_ui_view_spend_logs_with_model(client, monkeypatch): mock_spend_logs = [ @@ -3749,6 +3847,62 @@ async def test_build_ui_spend_logs_response_sums_multi_round_session_spend(): assert call_args[2] == [api_key] +@pytest.mark.asyncio +async def test_build_ui_spend_logs_response_session_cache_hit_count(): + """ + Each row of a session must carry session_cache_hit_count aggregated across + the whole session so the UI can show how many requests in the session were + served from the response cache. + """ + from litellm.proxy.spend_tracking.spend_management_endpoints import ( + _build_ui_spend_logs_response, + ) + + session_id = "sess-cache-hits" + api_key = "hashed-key-xyz" + dict_rows = [ + {"request_id": "req-1", "session_id": session_id, "call_type": "completion", "api_key": api_key}, + {"request_id": "req-2", "session_id": session_id, "call_type": "completion", "api_key": api_key}, + {"request_id": "req-3", "session_id": None, "call_type": "completion", "api_key": api_key}, + ] + + mock_prisma = MagicMock() + mock_prisma.db.litellm_spendlogs.group_by = AsyncMock( + return_value=[{"session_id": session_id, "_count": {"session_id": 2}}] + ) + mock_prisma.db.query_raw = AsyncMock( + return_value=[ + { + "session_id": session_id, + "session_total_spend": 0.05, + "mcp_tool_call_count": 0, + "mcp_tool_call_spend": 0.0, + "session_cache_hit_count": 2, + } + ] + ) + + result = await _build_ui_spend_logs_response( + prisma_client=mock_prisma, + data=dict_rows, + total_records=3, + page=1, + page_size=50, + total_pages=1, + enrich_session_counts=True, + ) + + rows = result["data"] + assert rows[0]["session_cache_hit_count"] == 2 + assert rows[1]["session_cache_hit_count"] == 2 + assert "session_cache_hit_count" not in rows[2] + + # The aggregate SQL must actually compute the cache-hit count. + _, call_args, _ = mock_prisma.db.query_raw.mock_calls[0] + assert "session_cache_hit_count" in call_args[0] + assert "LOWER(cache_hit) = 'true'" in call_args[0] + + # --------------------------------------------------------------------------- # Tests for /spend/logs team-member permission # --------------------------------------------------------------------------- diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index 5b6d70b4771..f3c6a030619 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -2002,6 +2002,8 @@ interface UiSpendLogsParams { user_id?: string; end_user?: string; status_filter?: string; + /** Filter by response cache result: "true" (cache hit) or "false" (cache miss) */ + cache_hit?: string; /** Filter by model name (e.g. "gpt-4") */ model?: string; /** Filter by model ID (litellm model deployment id) */ diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx index aab8d2b8cb9..e48fd555dda 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx @@ -359,6 +359,19 @@ describe("LogDetailContent", () => { expect(screen.queryByText("Response Cache")).not.toBeInTheDocument(); }); + it("should display the Cache Key next to the Response Cache result", () => { + render(); + + expect(screen.getByText("Cache Key")).toBeInTheDocument(); + expect(screen.getByText("abc123cachekey")).toBeInTheDocument(); + }); + + it("should hide the Cache Key row when caching is off", () => { + render(); + + expect(screen.queryByText("Cache Key")).not.toBeInTheDocument(); + }); + it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => { render( ): number | und const RESPONSE_CACHE_TOOLTIP = "Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed."; const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching"; +const CACHE_KEY_TOOLTIP = + "The key LiteLLM computed for this request in the response cache. Requests with the same cache key share a cached response; a different key means the request content did not match any cached entry."; const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching"; function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) { @@ -436,6 +438,13 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: )} + {showResponseCache && logEntry.cache_key && logEntry.cache_key !== "Cache OFF" && ( + } + > + + + )} {promptCacheReadTokens > 0 && ( AGENT_CALL_TYPES.includes(row.call_type)).length; const mcpCount = sessionLogs.filter((row) => MCP_CALL_TYPES.includes(row.call_type)).length; + const cacheHitCount = sessionLogs.filter((row) => String(row.cache_hit ?? "").toLowerCase() === "true").length; const logsForList = isSessionMode ? sessionLogs : currentLog ? [currentLog] : []; const leftPanelId = isSessionMode ? sessionId || "" : currentLog?.request_id || ""; const leftPanelDisplayId = leftPanelId.length > 14 ? `${leftPanelId.slice(0, 11)}...` : leftPanelId; @@ -383,7 +384,8 @@ export function LogDetailsDrawer({ {isSessionMode && ( <> · - {sessionDurationSeconds}s + {sessionDurationSeconds}s· + {cacheHitCount}/{logsForList.length} cached )} diff --git a/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx b/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx index af6a6d1f178..d6adbcdd55a 100644 --- a/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/RequestLogsFilters.tsx @@ -31,6 +31,12 @@ const STATUS_FILTER_ITEMS = [ { value: "success", label: "Success" }, { value: "failure", label: "Failure" }, ] as const; + +const CACHE_HIT_FILTER_ITEMS = [ + { value: ALL_VALUE, label: "All Requests" }, + { value: "true", label: "Cache Hit" }, + { value: "false", label: "Cache Miss" }, +] as const; const PAGE_SIZE = 50; const asString = (value: unknown): string => (typeof value === "string" ? value : ""); @@ -364,6 +370,27 @@ export function RequestLogsFilters({ get, set, teams, logsWindow }: RequestLogsF /> + + + + 0 && `${sessionLlmCount} LLM`, sessionAgentCount > 0 && `${sessionAgentCount} Agent`, sessionMcpCount > 0 && `${sessionMcpCount} MCP`, + log.session_cache_hit_count != null && `${log.session_cache_hit_count} cache hit`, ].filter(Boolean); return ; }, diff --git a/ui/litellm-dashboard/src/components/view_logs/columns.tsx b/ui/litellm-dashboard/src/components/view_logs/columns.tsx index d4f784bf165..eef957922d7 100644 --- a/ui/litellm-dashboard/src/components/view_logs/columns.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/columns.tsx @@ -42,6 +42,7 @@ export type LogEntry = { request_duration_ms?: number; session_total_count?: number; session_total_spend?: number; + session_cache_hit_count?: number; mcp_tool_call_count?: number; mcp_tool_call_spend?: number; session_llm_count?: number; diff --git a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx index 26c5bda1593..f474d4218f6 100644 --- a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.test.tsx @@ -82,6 +82,7 @@ describe("useLogFilterLogic", () => { { id: LOG_FILTER_IDS.SESSION_ID, value: "sess-1", param: "session_id" }, { id: LOG_FILTER_IDS.END_USER, value: "end-user-1", param: "end_user" }, { id: LOG_FILTER_IDS.STATUS, value: "failure", param: "status_filter" }, + { id: LOG_FILTER_IDS.CACHE_HIT, value: "true", param: "cache_hit" }, { id: LOG_FILTER_IDS.MODEL_ID, value: "model-uuid-1", param: "model_id" }, { id: LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL, value: "gpt-4o", param: "model" }, { id: LOG_FILTER_IDS.KEY_ALIAS, value: "alias-1", param: "key_alias" }, diff --git a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx index 474f51e93b3..f24b8364a08 100644 --- a/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/log_filter_logic.tsx @@ -26,6 +26,7 @@ export const LOG_FILTER_IDS = { ERROR_MESSAGE: "error_message", KEY_HASH: "key_hash", SESSION_ID: "session_id", + CACHE_HIT: "cache_hit", MODEL_ID: "model_id", PUBLIC_MODEL_OR_SEARCH_TOOL: "model", REQUEST_ID: "request_id", @@ -42,6 +43,7 @@ export const LOG_FILTER_LABELS: Record = { [LOG_FILTER_IDS.ERROR_MESSAGE]: "Error Message", [LOG_FILTER_IDS.KEY_HASH]: "Key Hash", [LOG_FILTER_IDS.SESSION_ID]: "Session ID", + [LOG_FILTER_IDS.CACHE_HIT]: "Cache Hit", [LOG_FILTER_IDS.MODEL_ID]: "Model", [LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL]: "Public model / search tool", }; @@ -167,6 +169,7 @@ export function useLogFilterLogic({ user_id: userIdFilter, end_user: getFilterValue(columnFilters, LOG_FILTER_IDS.END_USER), status_filter: getFilterValue(columnFilters, LOG_FILTER_IDS.STATUS), + cache_hit: getFilterValue(columnFilters, LOG_FILTER_IDS.CACHE_HIT), model_id: getFilterValue(columnFilters, LOG_FILTER_IDS.MODEL_ID), model: getFilterValue(columnFilters, LOG_FILTER_IDS.PUBLIC_MODEL_OR_SEARCH_TOOL), key_alias: getFilterValue(columnFilters, LOG_FILTER_IDS.KEY_ALIAS), diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 4ccf7b59fbc..d9f5af3f22a 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -53273,6 +53273,8 @@ export interface operations { page_size?: number; /** @description Filter logs by status (e.g., success, failure) */ status_filter?: string | null; + /** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */ + cache_hit?: string | null; /** @description Filter logs by model */ model?: string | null; /** @description Filter logs by model ID (litellm model deployment id) */ @@ -53381,6 +53383,8 @@ export interface operations { page_size?: number; /** @description Filter logs by status (e.g., success, failure) */ status_filter?: string | null; + /** @description Filter logs by response cache result: 'true' (served from cache) or 'false' (cache miss) */ + cache_hit?: string | null; /** @description Filter logs by model */ model?: string | null; /** @description Filter logs by model ID (litellm model deployment id) */