fix(ui): distinguish response cache from provider prompt caching (#34138)

* fix(ui): distinguish response cache from provider prompt caching

The log detail drawer labeled LiteLLM's response cache result as
"Cache Hit" and rendered a red "false" tag next to provider prompt
cache token counts, which read as prompt caching being broken. The
row is now labeled "Response Cache" with an explanatory tooltip,
shows a neutral "Miss" tag instead of a red one, and the prompt
cache token rows are prefixed with "Prompt Cache" and get their own
tooltips. Cost breakdown line items get the same prefix.

The Caching dashboard only reports response cache analytics but never
said so; it is renamed to "Response Cache" in the sidebar, gains a
scope description pointing to the Usage page and Logs for prompt
caching, and the ambiguous "Cached Tokens" stat card is renamed to
"Cached Completion Tokens".

* test(ui): cover renamed Response Cache sidebar item in e2e

The sidebar navigation spec now clicks the renamed "Response Cache"
item and asserts it routes to /ui/caching, and the menu label fixture
maps the new label while keeping "Caching" as a legacy alias.
Verified by running sidebar.spec.ts through run_e2e.sh (full harness:
built UI served by the proxy, seeded postgres); both tests pass.

* feat(ui): link cache tooltips and dashboard description to docs

The Response Cache tooltip links to the proxy caching docs and the
two prompt cache token tooltips link to the prompt caching docs, so
users can jump straight to the explanation of whichever mechanism
they are looking at. The Response Cache dashboard description links
both docs pages the same way.
This commit is contained in:
ryan-crabbe-berri 2026-07-21 14:26:26 -07:00 • committed by GitHub
parent b47fe730a4
commit 1cc70f84c0
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
10 changed files with 240 additions and 40 deletions

View file

@ -26,7 +26,8 @@ export const menuLabelToPage: Record<string, Page> = {
"Cost Tracking": Page.CostTracking,
"UI Theme": Page.UiTheme,
// Experimental submenu items
Caching: Page.Caching,
"Response Cache": Page.Caching,
Caching: Page.Caching, // Legacy label support
Prompts: Page.Prompts,
Budgets: Page.Budgets,
"API Playground": Page.TransformRequest,

View file

@ -8,7 +8,16 @@ import { MIGRATED_E2E_PAGES } from "../../fixtures/migratedPages";
import type { Page as PlaywrightPage } from "@playwright/test";
const sidebarButtons = {
[Role.ProxyAdmin]: ["Virtual Keys", "Playground", "Models", "Usage", "Teams", "Internal Users", "AI Hub"],
[Role.ProxyAdmin]: [
"Virtual Keys",
"Playground",
"Models",
"Usage",
"Teams",
"Internal Users",
"AI Hub",
"Response Cache",
],
};
/** Migrated pages live at a path route; legacy pages keep the ?page= query param. */

View file

@ -2207,7 +2207,7 @@
},
"src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx": {
"no-nested-ternary": {
"count": 4
"count": 3
}
},
"src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx": {

View file

@ -76,6 +76,22 @@ describe("CacheDashboard cache analytics charts", () => {
expect(screen.getByText("Cached Completion Tokens vs Generated Completion Tokens")).toBeInTheDocument();
});
it("scopes the analytics tab to the response cache, not provider prompt caching", async () => {
renderDashboard();
expect(await screen.findByText(/is not shown here/)).toBeInTheDocument();
expect(screen.getByRole("link", { name: "response cache" })).toHaveAttribute(
"href",
"https://docs.litellm.ai/docs/proxy/caching",
);
expect(screen.getByRole("link", { name: "prompt caching" })).toHaveAttribute(
"href",
"https://docs.litellm.ai/docs/completion/prompt_caching",
);
expect(screen.queryByText("Cached Tokens")).not.toBeInTheDocument();
expect(screen.getAllByText("Cached Completion Tokens").length).toBeGreaterThan(0);
});
it("renders the requests chart with each category legend-bound to its fill and stacked in order", async () => {
renderDashboard();
const { requestsCard } = await findChartCards();

View file

@ -282,6 +282,28 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
<TabPanels>
<TabPanel>
<Card>
<Text className="text-tremor-content dark:text-dark-tremor-content">
Analytics for LiteLLM&apos;s{" "}
<a
href="https://docs.litellm.ai/docs/proxy/caching"
target="_blank"
rel="noreferrer"
className="underline"
>
response cache
</a>{" "}
(e.g. Redis / in-memory): requests answered from cache without calling the LLM provider. Provider-side{" "}
<a
href="https://docs.litellm.ai/docs/completion/prompt_caching"
target="_blank"
rel="noreferrer"
className="underline"
>
prompt caching
</a>{" "}
(cached input tokens from Anthropic, OpenAI, etc.) is not shown here; see &quot;Prompt Caching
Metrics&quot; on the Usage page or individual requests in the Logs page.
</Text>
<Grid numItems={3} className="gap-4 mt-4">
<Col>
<MultiSelect
@ -340,7 +362,7 @@ const CacheDashboard: React.FC<CachePageProps> = ({ accessToken, token, userRole
<Card>
<p className="text-tremor-default font-medium text-tremor-content dark:text-dark-tremor-content">
Cached Tokens
Cached Completion Tokens
</p>
<div className="mt-2 flex items-baseline space-x-2.5">
<p className="text-tremor-metric font-semibold text-tremor-content-strong dark:text-dark-tremor-content-strong">

View file

@ -244,7 +244,13 @@ const menuGroups: MenuGroup[] = [
icon: <BookOpen {...ICON} />,
external_url: "https://models.litellm.ai/cookbook",
},
{ key: "caching", page: "caching", label: "Caching", icon: <Database {...ICON} />, roles: all_admin_roles },
{
key: "caching",
page: "caching",
label: "Response Cache",
icon: <Database {...ICON} />,
roles: all_admin_roles,
},
{
key: "experimental",
page: "experimental",

View file

@ -174,6 +174,27 @@ describe("CostBreakdownViewer", () => {
expect(screen.getByText("Azure Model Router Flat Cost:")).toBeInTheDocument();
});
it("labels provider prompt cache line items as Prompt Cache Read/Write Cost", async () => {
renderWithProviders(
<CostBreakdownViewer
costBreakdown={{
input_cost: 0.01,
output_cost: 0.02,
cache_read_cost: 0.001,
cache_creation_cost: 0.002,
}}
totalSpend={0.03}
cacheReadTokens={100}
cacheCreationTokens={50}
/>,
);
await expandCostBreakdown();
expect(screen.getByText("Prompt Cache Read Cost:")).toBeInTheDocument();
expect(screen.getByText("Prompt Cache Write Cost:")).toBeInTheDocument();
});
it("shows '(Cached)' in the header when cacheHit is true", () => {
const breakdown: CostBreakdown = {
input_cost: 0.001,

View file

@ -136,7 +136,7 @@ export const CostBreakdownViewer: React.FC<CostBreakdownViewerProps> = ({
</div>
{(costBreakdown?.cache_read_cost ?? 0) > 0 && (
<div className="flex text-sm">
<span className="text-gray-600 font-medium w-1/3">Cache Read Cost:</span>
<span className="text-gray-600 font-medium w-1/3">Prompt Cache Read Cost:</span>
<span className="text-gray-900">
{formatCost(isCached ? 0 : costBreakdown?.cache_read_cost)}
{(cacheReadTokens ?? 0) > 0 && (
@ -149,7 +149,7 @@ export const CostBreakdownViewer: React.FC<CostBreakdownViewerProps> = ({
)}
{(costBreakdown?.cache_creation_cost ?? 0) > 0 && (
<div className="flex text-sm">
<span className="text-gray-600 font-medium w-1/3">Cache Write Cost:</span>
<span className="text-gray-600 font-medium w-1/3">Prompt Cache Write Cost:</span>
<span className="text-gray-900">
{formatCost(isCached ? 0 : costBreakdown?.cache_creation_cost)}
{(cacheCreationTokens ?? 0) > 0 && (

View file

@ -255,26 +255,97 @@ describe("LogDetailContent", () => {
expect(screen.getByText("2 masked")).toBeInTheDocument();
});
it("should display cache hit information when cache_hit is true", () => {
it("should display a green Response Cache 'Hit' tag when the response cache served the request", () => {
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "True" })} />);
expect(screen.getByText("Response Cache")).toBeInTheDocument();
expect(screen.getByText("Hit").closest(".ant-tag")).toHaveClass("ant-tag-green");
});
it("should show prompt cache tokens without an alarming red tag when only provider prompt caching occurred", () => {
render(
<LogDetailContent
logEntry={createLogEntry({
cache_hit: "true",
cache_hit: "False",
metadata: {
status: "success",
additional_usage_values: {
cache_read_input_tokens: 100,
cache_creation_input_tokens: 0,
cache_read_input_tokens: 34462,
cache_creation_input_tokens: 83,
},
},
})}
/>,
);
expect(screen.getByText("Cache Hit")).toBeInTheDocument();
expect(screen.getByText("true")).toBeInTheDocument();
expect(screen.getByText("Cache Read Tokens")).toBeInTheDocument();
expect(screen.getByText("100")).toBeInTheDocument();
expect(screen.getByText("Prompt Cache Read Tokens")).toBeInTheDocument();
expect(screen.getByText("34,462")).toBeInTheDocument();
expect(screen.getByText("Prompt Cache Creation Tokens")).toBeInTheDocument();
expect(screen.getByText("83")).toBeInTheDocument();
expect(screen.getByText("Miss").closest(".ant-tag")).not.toHaveClass("ant-tag-red");
expect(screen.queryByText("Cache Hit")).not.toBeInTheDocument();
});
it("should display Prompt Cache Creation Tokens even when there are no cache read tokens", () => {
render(
<LogDetailContent
logEntry={createLogEntry({
cache_hit: "None",
metadata: {
status: "success",
additional_usage_values: {
cache_read_input_tokens: 0,
cache_creation_input_tokens: 83,
},
},
})}
/>,
);
expect(screen.getByText("Prompt Cache Creation Tokens")).toBeInTheDocument();
expect(screen.getByText("83")).toBeInTheDocument();
});
it("should link the Response Cache tooltip to the response caching docs", async () => {
const user = userEvent.setup();
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "True" })} />);
const label = screen.getByText("Response Cache").closest(".ant-space") as HTMLElement;
await user.hover(within(label).getByRole("img", { name: "info-circle" }));
expect(await screen.findByRole("link", { name: "Docs" })).toHaveAttribute(
"href",
"https://docs.litellm.ai/docs/proxy/caching",
);
});
it("should link the prompt cache tooltips to the prompt caching docs", async () => {
const user = userEvent.setup();
render(
<LogDetailContent
logEntry={createLogEntry({
cache_hit: "None",
metadata: {
status: "success",
additional_usage_values: { cache_read_input_tokens: 100 },
},
})}
/>,
);
const label = screen.getByText("Prompt Cache Read Tokens").closest(".ant-space") as HTMLElement;
await user.hover(within(label).getByRole("img", { name: "info-circle" }));
expect(await screen.findByRole("link", { name: "Docs" })).toHaveAttribute(
"href",
"https://docs.litellm.ai/docs/completion/prompt_caching",
);
});
it("should hide the Response Cache row when cache_hit is not a true/false value", () => {
render(<LogDetailContent logEntry={createLogEntry({ cache_hit: "None" })} />);
expect(screen.queryByText("Response Cache")).not.toBeInTheDocument();
});
it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => {

View file

@ -1,5 +1,6 @@
import { useState } from "react";
import { Typography, Descriptions, Card, Tag, Tabs, Alert, Collapse, Radio, Space, Spin } from "antd";
import { Typography, Descriptions, Card, Tag, Tabs, Alert, Collapse, Radio, Space, Spin, Tooltip } from "antd";
import { InfoCircleOutlined } from "@ant-design/icons";
import moment from "moment";
import { LogEntry } from "../columns";
import { formatNumberWithCommas } from "@/utils/dataUtils";
@ -278,6 +279,40 @@ function getUncachedInputTextTokens(metadata: Record<string, any>): number | und
return Number.isFinite(n) ? n : undefined;
}
const RESPONSE_CACHE_TOOLTIP =
"Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed.";
const PROMPT_CACHE_READ_TOOLTIP =
"Input tokens read from the LLM provider's prompt cache (e.g. Anthropic / OpenAI), billed at a discounted rate. Reported by the provider.";
const PROMPT_CACHE_CREATION_TOOLTIP =
"Input tokens written to the LLM provider's prompt cache for reuse by later requests.";
const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching";
const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching";
function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) {
return (
<Space size={4}>
{label}
<Tooltip
title={
<>
{tooltip}{" "}
<a
href={docsUrl}
target="_blank"
rel="noreferrer"
style={{ color: "#91caff", textDecoration: "underline" }}
>
Docs
</a>
</>
}
>
<InfoCircleOutlined style={{ color: "#8c8c8c" }} />
</Tooltip>
</Space>
);
}
function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: Record<string, any> }) {
const completionStartTime = logEntry.completionStartTime;
const ttftMs =
@ -285,14 +320,11 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata:
? new Date(completionStartTime).getTime() - new Date(logEntry.startTime).getTime()
: null;
const hasCacheActivity =
logEntry.cache_hit ||
(metadata?.additional_usage_values?.cache_read_input_tokens &&
metadata.additional_usage_values.cache_read_input_tokens > 0);
const cacheHitValue = String(logEntry.cache_hit ?? "None");
const cacheHitColor =
cacheHitValue.toLowerCase() === "true" ? "green" : cacheHitValue.toLowerCase() === "false" ? "red" : "default";
const responseCacheValue = String(logEntry.cache_hit ?? "").toLowerCase();
const isResponseCacheHit = responseCacheValue === "true";
const showResponseCache = isResponseCacheHit || responseCacheValue === "false";
const promptCacheReadTokens = Number(metadata?.additional_usage_values?.cache_read_input_tokens) || 0;
const promptCacheCreationTokens = Number(metadata?.additional_usage_values?.cache_creation_input_tokens) || 0;
const uncachedInputTokens = getUncachedInputTextTokens(metadata);
const showAnthropicMessagesInputOutput =
@ -326,22 +358,44 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata:
<Descriptions.Item label="Time to First Token">{(ttftMs / 1000).toFixed(3)} s</Descriptions.Item>
)}
{hasCacheActivity && (
<>
<Descriptions.Item label="Cache Hit">
<Tag color={cacheHitColor}>{cacheHitValue}</Tag>
</Descriptions.Item>
{metadata?.additional_usage_values?.cache_read_input_tokens > 0 && (
<Descriptions.Item label="Cache Read Tokens">
{formatNumberWithCommas(metadata.additional_usage_values.cache_read_input_tokens)}
</Descriptions.Item>
)}
{metadata?.additional_usage_values?.cache_creation_input_tokens > 0 && (
<Descriptions.Item label="Cache Creation Tokens">
{formatNumberWithCommas(metadata.additional_usage_values.cache_creation_input_tokens)}
</Descriptions.Item>
)}
</>
{showResponseCache && (
<Descriptions.Item
label={
<MetricLabel
label="Response Cache"
tooltip={RESPONSE_CACHE_TOOLTIP}
docsUrl={RESPONSE_CACHE_DOCS_URL}
/>
}
>
<Tag color={isResponseCacheHit ? "green" : "default"}>{isResponseCacheHit ? "Hit" : "Miss"}</Tag>
</Descriptions.Item>
)}
{promptCacheReadTokens > 0 && (
<Descriptions.Item
label={
<MetricLabel
label="Prompt Cache Read Tokens"
tooltip={PROMPT_CACHE_READ_TOOLTIP}
docsUrl={PROMPT_CACHE_DOCS_URL}
/>
}
>
{formatNumberWithCommas(promptCacheReadTokens)}
</Descriptions.Item>
)}
{promptCacheCreationTokens > 0 && (
<Descriptions.Item
label={
<MetricLabel
label="Prompt Cache Creation Tokens"
tooltip={PROMPT_CACHE_CREATION_TOOLTIP}
docsUrl={PROMPT_CACHE_DOCS_URL}
/>
}
>
{formatNumberWithCommas(promptCacheCreationTokens)}
</Descriptions.Item>
)}
{metadata?.litellm_overhead_time_ms !== undefined && metadata.litellm_overhead_time_ms !== null && (