diff --git a/litellm/constants.py b/litellm/constants.py index 78aba30f9c0..765bbfe1e54 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -147,6 +147,7 @@ LITELLM_UI_ALLOW_HEADERS: Final = [ "x-litellm-adaptive-router-model", "x-litellm-applied-guardrails", "x-litellm-guardrail-scan-id", + "x-litellm-cache-key", ] # Gemini model-specific minimal thinking budget constants diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py index 3383527e932..b9a31acca96 100644 --- a/tests/test_litellm/proxy/test_proxy_server.py +++ b/tests/test_litellm/proxy/test_proxy_server.py @@ -78,6 +78,16 @@ def client_no_auth(): return TestClient(app) +def test_cors_exposes_cache_key_header_to_browser_js(): + from fastapi.middleware.cors import CORSMiddleware + + from litellm.constants import LITELLM_UI_ALLOW_HEADERS + + cors_middleware = next(m for m in app.user_middleware if m.cls is CORSMiddleware) + assert cors_middleware.kwargs["expose_headers"] is LITELLM_UI_ALLOW_HEADERS + assert "x-litellm-cache-key" in cors_middleware.kwargs["expose_headers"] + + def test_login_v2_returns_redirect_url_and_sets_cookie(monkeypatch): mock_login_result = {"user_id": "test-user"} mock_prisma_client = MagicMock() diff --git a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx index 5afc94eb043..f31e1839739 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx @@ -33,4 +33,22 @@ describe("ResponseMetrics prompt cache chips", () => { expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument(); expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument(); }); + + it("shows the response cache indicator instead of the provider cache chips on a response-cache hit", () => { + render( + , + ); + + expect(screen.getByText("Response Cache: Hit")).toBeInTheDocument(); + expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument(); + expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument(); + }); + + it("does not show the response cache indicator when the flag is absent", () => { + render(); + + expect(screen.queryByText(/Response Cache/)).not.toBeInTheDocument(); + }); }); diff --git a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx index 3e7f2884b23..ec62d0618d7 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx @@ -7,12 +7,16 @@ import { DatabaseBackup, DollarSign, Hash, + History, Lightbulb, Wrench, } from "lucide-react"; import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip"; import { PROMPT_CACHE_CREATION_TOOLTIP, PROMPT_CACHE_READ_TOOLTIP } from "@/utils/promptCacheUsage"; +const RESPONSE_CACHE_TOOLTIP = + "This response was replayed from LiteLLM's response cache. The request never reached the provider, so it did not read from or write to the provider's own prompt cache."; + export interface TokenUsage { completionTokens?: number; promptTokens?: number; @@ -21,6 +25,7 @@ export interface TokenUsage { cacheReadTokens?: number; cacheCreationTokens?: number; cost?: number; + servedFromResponseCache?: boolean; } interface ResponseMetricsProps { @@ -51,7 +56,22 @@ function MetricItem({ label, tooltip, icon, value }: MetricItemProps) { ); } +function ResponseCacheIndicator() { + return ( +