From 1cc70f84c0a981f347c84b3264396d717e5bd64d Mon Sep 17 00:00:00 2001 From: ryan-crabbe-berri Date: Tue, 21 Jul 2026 14:26:26 -0700 Subject: [PATCH] fix(ui): distinguish response cache from provider prompt caching (#34138) * fix(ui): distinguish response cache from provider prompt caching The log detail drawer labeled LiteLLM's response cache result as "Cache Hit" and rendered a red "false" tag next to provider prompt cache token counts, which read as prompt caching being broken. The row is now labeled "Response Cache" with an explanatory tooltip, shows a neutral "Miss" tag instead of a red one, and the prompt cache token rows are prefixed with "Prompt Cache" and get their own tooltips. Cost breakdown line items get the same prefix. The Caching dashboard only reports response cache analytics but never said so; it is renamed to "Response Cache" in the sidebar, gains a scope description pointing to the Usage page and Logs for prompt caching, and the ambiguous "Cached Tokens" stat card is renamed to "Cached Completion Tokens". * test(ui): cover renamed Response Cache sidebar item in e2e The sidebar navigation spec now clicks the renamed "Response Cache" item and asserts it routes to /ui/caching, and the menu label fixture maps the new label while keeping "Caching" as a legacy alias. Verified by running sidebar.spec.ts through run_e2e.sh (full harness: built UI served by the proxy, seeded postgres); both tests pass. * feat(ui): link cache tooltips and dashboard description to docs The Response Cache tooltip links to the proxy caching docs and the two prompt cache token tooltips link to the prompt caching docs, so users can jump straight to the explanation of whichever mechanism they are looking at. The Response Cache dashboard description links both docs pages the same way. --- .../e2e_tests/fixtures/menuMappings.ts | 3 +- .../tests/navigation/sidebar.spec.ts | 11 +- ui/litellm-dashboard/eslint-suppressions.json | 2 +- .../_components/cache_dashboard.test.tsx | 16 +++ .../caching/_components/cache_dashboard.tsx | 24 +++- .../src/components/leftnav.tsx | 8 +- .../view_logs/CostBreakdownViewer.test.tsx | 21 ++++ .../view_logs/CostBreakdownViewer.tsx | 4 +- .../LogDetailContent.test.tsx | 87 +++++++++++++-- .../LogDetailsDrawer/LogDetailContent.tsx | 104 +++++++++++++----- 10 files changed, 240 insertions(+), 40 deletions(-) diff --git a/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts b/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts index 4a4bb64c8ed..d6e7ea86982 100644 --- a/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts +++ b/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts @@ -26,7 +26,8 @@ export const menuLabelToPage: Record = { "Cost Tracking": Page.CostTracking, "UI Theme": Page.UiTheme, // Experimental submenu items - Caching: Page.Caching, + "Response Cache": Page.Caching, + Caching: Page.Caching, // Legacy label support Prompts: Page.Prompts, Budgets: Page.Budgets, "API Playground": Page.TransformRequest, diff --git a/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts b/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts index 7e42d07ae7c..b220dc09ae2 100644 --- a/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts +++ b/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts @@ -8,7 +8,16 @@ import { MIGRATED_E2E_PAGES } from "../../fixtures/migratedPages"; import type { Page as PlaywrightPage } from "@playwright/test"; const sidebarButtons = { - [Role.ProxyAdmin]: ["Virtual Keys", "Playground", "Models", "Usage", "Teams", "Internal Users", "AI Hub"], + [Role.ProxyAdmin]: [ + "Virtual Keys", + "Playground", + "Models", + "Usage", + "Teams", + "Internal Users", + "AI Hub", + "Response Cache", + ], }; /** Migrated pages live at a path route; legacy pages keep the ?page= query param. */ diff --git a/ui/litellm-dashboard/eslint-suppressions.json b/ui/litellm-dashboard/eslint-suppressions.json index d596897c4c9..01208e994a7 100644 --- a/ui/litellm-dashboard/eslint-suppressions.json +++ b/ui/litellm-dashboard/eslint-suppressions.json @@ -2207,7 +2207,7 @@ }, "src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx": { "no-nested-ternary": { - "count": 4 + "count": 3 } }, "src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx": { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx index 17d14cd7fac..13472a3d1df 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx @@ -76,6 +76,22 @@ describe("CacheDashboard cache analytics charts", () => { expect(screen.getByText("Cached Completion Tokens vs Generated Completion Tokens")).toBeInTheDocument(); }); + it("scopes the analytics tab to the response cache, not provider prompt caching", async () => { + renderDashboard(); + + expect(await screen.findByText(/is not shown here/)).toBeInTheDocument(); + expect(screen.getByRole("link", { name: "response cache" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/proxy/caching", + ); + expect(screen.getByRole("link", { name: "prompt caching" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/completion/prompt_caching", + ); + expect(screen.queryByText("Cached Tokens")).not.toBeInTheDocument(); + expect(screen.getAllByText("Cached Completion Tokens").length).toBeGreaterThan(0); + }); + it("renders the requests chart with each category legend-bound to its fill and stacked in order", async () => { renderDashboard(); const { requestsCard } = await findChartCards(); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx index b8e8dc8adb1..51c0b85cedb 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx @@ -282,6 +282,28 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole + + Analytics for LiteLLM's{" "} + + response cache + {" "} + (e.g. Redis / in-memory): requests answered from cache without calling the LLM provider. Provider-side{" "} + + prompt caching + {" "} + (cached input tokens from Anthropic, OpenAI, etc.) is not shown here; see "Prompt Caching + Metrics" on the Usage page or individual requests in the Logs page. + = ({ accessToken, token, userRole

- Cached Tokens + Cached Completion Tokens

diff --git a/ui/litellm-dashboard/src/components/leftnav.tsx b/ui/litellm-dashboard/src/components/leftnav.tsx index b0d534e9b7c..355db8bdfc7 100644 --- a/ui/litellm-dashboard/src/components/leftnav.tsx +++ b/ui/litellm-dashboard/src/components/leftnav.tsx @@ -244,7 +244,13 @@ const menuGroups: MenuGroup[] = [ icon: , external_url: "https://models.litellm.ai/cookbook", }, - { key: "caching", page: "caching", label: "Caching", icon: , roles: all_admin_roles }, + { + key: "caching", + page: "caching", + label: "Response Cache", + icon: , + roles: all_admin_roles, + }, { key: "experimental", page: "experimental", diff --git a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx index 6ad3b696dea..d484185f873 100644 --- a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx @@ -174,6 +174,27 @@ describe("CostBreakdownViewer", () => { expect(screen.getByText("Azure Model Router Flat Cost:")).toBeInTheDocument(); }); + it("labels provider prompt cache line items as Prompt Cache Read/Write Cost", async () => { + renderWithProviders( + , + ); + + await expandCostBreakdown(); + + expect(screen.getByText("Prompt Cache Read Cost:")).toBeInTheDocument(); + expect(screen.getByText("Prompt Cache Write Cost:")).toBeInTheDocument(); + }); + it("shows '(Cached)' in the header when cacheHit is true", () => { const breakdown: CostBreakdown = { input_cost: 0.001, diff --git a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx index 0a48e12b0e7..96ac965cf8b 100644 --- a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx @@ -136,7 +136,7 @@ export const CostBreakdownViewer: React.FC = ({

{(costBreakdown?.cache_read_cost ?? 0) > 0 && (
- Cache Read Cost: + Prompt Cache Read Cost: {formatCost(isCached ? 0 : costBreakdown?.cache_read_cost)} {(cacheReadTokens ?? 0) > 0 && ( @@ -149,7 +149,7 @@ export const CostBreakdownViewer: React.FC = ({ )} {(costBreakdown?.cache_creation_cost ?? 0) > 0 && (
- Cache Write Cost: + Prompt Cache Write Cost: {formatCost(isCached ? 0 : costBreakdown?.cache_creation_cost)} {(cacheCreationTokens ?? 0) > 0 && ( diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx index 85a38e26977..868d9fcfabe 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx @@ -255,26 +255,97 @@ describe("LogDetailContent", () => { expect(screen.getByText("2 masked")).toBeInTheDocument(); }); - it("should display cache hit information when cache_hit is true", () => { + it("should display a green Response Cache 'Hit' tag when the response cache served the request", () => { + render(); + + expect(screen.getByText("Response Cache")).toBeInTheDocument(); + expect(screen.getByText("Hit").closest(".ant-tag")).toHaveClass("ant-tag-green"); + }); + + it("should show prompt cache tokens without an alarming red tag when only provider prompt caching occurred", () => { render( , ); - expect(screen.getByText("Cache Hit")).toBeInTheDocument(); - expect(screen.getByText("true")).toBeInTheDocument(); - expect(screen.getByText("Cache Read Tokens")).toBeInTheDocument(); - expect(screen.getByText("100")).toBeInTheDocument(); + expect(screen.getByText("Prompt Cache Read Tokens")).toBeInTheDocument(); + expect(screen.getByText("34,462")).toBeInTheDocument(); + expect(screen.getByText("Prompt Cache Creation Tokens")).toBeInTheDocument(); + expect(screen.getByText("83")).toBeInTheDocument(); + expect(screen.getByText("Miss").closest(".ant-tag")).not.toHaveClass("ant-tag-red"); + expect(screen.queryByText("Cache Hit")).not.toBeInTheDocument(); + }); + + it("should display Prompt Cache Creation Tokens even when there are no cache read tokens", () => { + render( + , + ); + + expect(screen.getByText("Prompt Cache Creation Tokens")).toBeInTheDocument(); + expect(screen.getByText("83")).toBeInTheDocument(); + }); + + it("should link the Response Cache tooltip to the response caching docs", async () => { + const user = userEvent.setup(); + render(); + + const label = screen.getByText("Response Cache").closest(".ant-space") as HTMLElement; + await user.hover(within(label).getByRole("img", { name: "info-circle" })); + + expect(await screen.findByRole("link", { name: "Docs" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/proxy/caching", + ); + }); + + it("should link the prompt cache tooltips to the prompt caching docs", async () => { + const user = userEvent.setup(); + render( + , + ); + + const label = screen.getByText("Prompt Cache Read Tokens").closest(".ant-space") as HTMLElement; + await user.hover(within(label).getByRole("img", { name: "info-circle" })); + + expect(await screen.findByRole("link", { name: "Docs" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/completion/prompt_caching", + ); + }); + + it("should hide the Response Cache row when cache_hit is not a true/false value", () => { + render(); + + expect(screen.queryByText("Response Cache")).not.toBeInTheDocument(); }); it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => { diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx index d4521afddf6..2752cccce56 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx @@ -1,5 +1,6 @@ import { useState } from "react"; -import { Typography, Descriptions, Card, Tag, Tabs, Alert, Collapse, Radio, Space, Spin } from "antd"; +import { Typography, Descriptions, Card, Tag, Tabs, Alert, Collapse, Radio, Space, Spin, Tooltip } from "antd"; +import { InfoCircleOutlined } from "@ant-design/icons"; import moment from "moment"; import { LogEntry } from "../columns"; import { formatNumberWithCommas } from "@/utils/dataUtils"; @@ -278,6 +279,40 @@ function getUncachedInputTextTokens(metadata: Record): number | und return Number.isFinite(n) ? n : undefined; } +const RESPONSE_CACHE_TOOLTIP = + "Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed."; +const PROMPT_CACHE_READ_TOOLTIP = + "Input tokens read from the LLM provider's prompt cache (e.g. Anthropic / OpenAI), billed at a discounted rate. Reported by the provider."; +const PROMPT_CACHE_CREATION_TOOLTIP = + "Input tokens written to the LLM provider's prompt cache for reuse by later requests."; +const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching"; +const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching"; + +function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) { + return ( + + {label} + + {tooltip}{" "} + + Docs + + + } + > + + + + ); +} + function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: Record }) { const completionStartTime = logEntry.completionStartTime; const ttftMs = @@ -285,14 +320,11 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: ? new Date(completionStartTime).getTime() - new Date(logEntry.startTime).getTime() : null; - const hasCacheActivity = - logEntry.cache_hit || - (metadata?.additional_usage_values?.cache_read_input_tokens && - metadata.additional_usage_values.cache_read_input_tokens > 0); - - const cacheHitValue = String(logEntry.cache_hit ?? "None"); - const cacheHitColor = - cacheHitValue.toLowerCase() === "true" ? "green" : cacheHitValue.toLowerCase() === "false" ? "red" : "default"; + const responseCacheValue = String(logEntry.cache_hit ?? "").toLowerCase(); + const isResponseCacheHit = responseCacheValue === "true"; + const showResponseCache = isResponseCacheHit || responseCacheValue === "false"; + const promptCacheReadTokens = Number(metadata?.additional_usage_values?.cache_read_input_tokens) || 0; + const promptCacheCreationTokens = Number(metadata?.additional_usage_values?.cache_creation_input_tokens) || 0; const uncachedInputTokens = getUncachedInputTextTokens(metadata); const showAnthropicMessagesInputOutput = @@ -326,22 +358,44 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: {(ttftMs / 1000).toFixed(3)} s )} - {hasCacheActivity && ( - <> - - {cacheHitValue} - - {metadata?.additional_usage_values?.cache_read_input_tokens > 0 && ( - - {formatNumberWithCommas(metadata.additional_usage_values.cache_read_input_tokens)} - - )} - {metadata?.additional_usage_values?.cache_creation_input_tokens > 0 && ( - - {formatNumberWithCommas(metadata.additional_usage_values.cache_creation_input_tokens)} - - )} - + {showResponseCache && ( + + } + > + {isResponseCacheHit ? "Hit" : "Miss"} + + )} + {promptCacheReadTokens > 0 && ( + + } + > + {formatNumberWithCommas(promptCacheReadTokens)} + + )} + {promptCacheCreationTokens > 0 && ( + + } + > + {formatNumberWithCommas(promptCacheCreationTokens)} + )} {metadata?.litellm_overhead_time_ms !== undefined && metadata.litellm_overhead_time_ms !== null && (