diff --git a/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts b/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts index 4a4bb64c8ed..d6e7ea86982 100644 --- a/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts +++ b/ui/litellm-dashboard/e2e_tests/fixtures/menuMappings.ts @@ -26,7 +26,8 @@ export const menuLabelToPage: Record = { "Cost Tracking": Page.CostTracking, "UI Theme": Page.UiTheme, // Experimental submenu items - Caching: Page.Caching, + "Response Cache": Page.Caching, + Caching: Page.Caching, // Legacy label support Prompts: Page.Prompts, Budgets: Page.Budgets, "API Playground": Page.TransformRequest, diff --git a/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts b/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts index 7e42d07ae7c..b220dc09ae2 100644 --- a/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts +++ b/ui/litellm-dashboard/e2e_tests/tests/navigation/sidebar.spec.ts @@ -8,7 +8,16 @@ import { MIGRATED_E2E_PAGES } from "../../fixtures/migratedPages"; import type { Page as PlaywrightPage } from "@playwright/test"; const sidebarButtons = { - [Role.ProxyAdmin]: ["Virtual Keys", "Playground", "Models", "Usage", "Teams", "Internal Users", "AI Hub"], + [Role.ProxyAdmin]: [ + "Virtual Keys", + "Playground", + "Models", + "Usage", + "Teams", + "Internal Users", + "AI Hub", + "Response Cache", + ], }; /** Migrated pages live at a path route; legacy pages keep the ?page= query param. */ diff --git a/ui/litellm-dashboard/eslint-suppressions.json b/ui/litellm-dashboard/eslint-suppressions.json index d596897c4c9..01208e994a7 100644 --- a/ui/litellm-dashboard/eslint-suppressions.json +++ b/ui/litellm-dashboard/eslint-suppressions.json @@ -2207,7 +2207,7 @@ }, "src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx": { "no-nested-ternary": { - "count": 4 + "count": 3 } }, "src/components/view_logs/LogDetailsDrawer/LogDetailsDrawer.tsx": { diff --git a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx index 17d14cd7fac..13472a3d1df 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.test.tsx @@ -76,6 +76,22 @@ describe("CacheDashboard cache analytics charts", () => { expect(screen.getByText("Cached Completion Tokens vs Generated Completion Tokens")).toBeInTheDocument(); }); + it("scopes the analytics tab to the response cache, not provider prompt caching", async () => { + renderDashboard(); + + expect(await screen.findByText(/is not shown here/)).toBeInTheDocument(); + expect(screen.getByRole("link", { name: "response cache" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/proxy/caching", + ); + expect(screen.getByRole("link", { name: "prompt caching" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/completion/prompt_caching", + ); + expect(screen.queryByText("Cached Tokens")).not.toBeInTheDocument(); + expect(screen.getAllByText("Cached Completion Tokens").length).toBeGreaterThan(0); + }); + it("renders the requests chart with each category legend-bound to its fill and stacked in order", async () => { renderDashboard(); const { requestsCard } = await findChartCards(); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx index b8e8dc8adb1..51c0b85cedb 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/caching/_components/cache_dashboard.tsx @@ -282,6 +282,28 @@ const CacheDashboard: React.FC = ({ accessToken, token, userRole + + Analytics for LiteLLM's{" "} + + response cache + {" "} + (e.g. Redis / in-memory): requests answered from cache without calling the LLM provider. Provider-side{" "} + + prompt caching + {" "} + (cached input tokens from Anthropic, OpenAI, etc.) is not shown here; see "Prompt Caching + Metrics" on the Usage page or individual requests in the Logs page. + = ({ accessToken, token, userRole

- Cached Tokens + Cached Completion Tokens

diff --git a/ui/litellm-dashboard/src/components/leftnav.tsx b/ui/litellm-dashboard/src/components/leftnav.tsx index b0d534e9b7c..355db8bdfc7 100644 --- a/ui/litellm-dashboard/src/components/leftnav.tsx +++ b/ui/litellm-dashboard/src/components/leftnav.tsx @@ -244,7 +244,13 @@ const menuGroups: MenuGroup[] = [ icon: , external_url: "https://models.litellm.ai/cookbook", }, - { key: "caching", page: "caching", label: "Caching", icon: , roles: all_admin_roles }, + { + key: "caching", + page: "caching", + label: "Response Cache", + icon: , + roles: all_admin_roles, + }, { key: "experimental", page: "experimental", diff --git a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx index 6ad3b696dea..d484185f873 100644 --- a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.test.tsx @@ -174,6 +174,27 @@ describe("CostBreakdownViewer", () => { expect(screen.getByText("Azure Model Router Flat Cost:")).toBeInTheDocument(); }); + it("labels provider prompt cache line items as Prompt Cache Read/Write Cost", async () => { + renderWithProviders( + , + ); + + await expandCostBreakdown(); + + expect(screen.getByText("Prompt Cache Read Cost:")).toBeInTheDocument(); + expect(screen.getByText("Prompt Cache Write Cost:")).toBeInTheDocument(); + }); + it("shows '(Cached)' in the header when cacheHit is true", () => { const breakdown: CostBreakdown = { input_cost: 0.001, diff --git a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx index 0a48e12b0e7..96ac965cf8b 100644 --- a/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/CostBreakdownViewer.tsx @@ -136,7 +136,7 @@ export const CostBreakdownViewer: React.FC = ({

{(costBreakdown?.cache_read_cost ?? 0) > 0 && (
- Cache Read Cost: + Prompt Cache Read Cost: {formatCost(isCached ? 0 : costBreakdown?.cache_read_cost)} {(cacheReadTokens ?? 0) > 0 && ( @@ -149,7 +149,7 @@ export const CostBreakdownViewer: React.FC = ({ )} {(costBreakdown?.cache_creation_cost ?? 0) > 0 && (
- Cache Write Cost: + Prompt Cache Write Cost: {formatCost(isCached ? 0 : costBreakdown?.cache_creation_cost)} {(cacheCreationTokens ?? 0) > 0 && ( diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx index 85a38e26977..868d9fcfabe 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.test.tsx @@ -255,26 +255,97 @@ describe("LogDetailContent", () => { expect(screen.getByText("2 masked")).toBeInTheDocument(); }); - it("should display cache hit information when cache_hit is true", () => { + it("should display a green Response Cache 'Hit' tag when the response cache served the request", () => { + render(); + + expect(screen.getByText("Response Cache")).toBeInTheDocument(); + expect(screen.getByText("Hit").closest(".ant-tag")).toHaveClass("ant-tag-green"); + }); + + it("should show prompt cache tokens without an alarming red tag when only provider prompt caching occurred", () => { render( , ); - expect(screen.getByText("Cache Hit")).toBeInTheDocument(); - expect(screen.getByText("true")).toBeInTheDocument(); - expect(screen.getByText("Cache Read Tokens")).toBeInTheDocument(); - expect(screen.getByText("100")).toBeInTheDocument(); + expect(screen.getByText("Prompt Cache Read Tokens")).toBeInTheDocument(); + expect(screen.getByText("34,462")).toBeInTheDocument(); + expect(screen.getByText("Prompt Cache Creation Tokens")).toBeInTheDocument(); + expect(screen.getByText("83")).toBeInTheDocument(); + expect(screen.getByText("Miss").closest(".ant-tag")).not.toHaveClass("ant-tag-red"); + expect(screen.queryByText("Cache Hit")).not.toBeInTheDocument(); + }); + + it("should display Prompt Cache Creation Tokens even when there are no cache read tokens", () => { + render( + , + ); + + expect(screen.getByText("Prompt Cache Creation Tokens")).toBeInTheDocument(); + expect(screen.getByText("83")).toBeInTheDocument(); + }); + + it("should link the Response Cache tooltip to the response caching docs", async () => { + const user = userEvent.setup(); + render(); + + const label = screen.getByText("Response Cache").closest(".ant-space") as HTMLElement; + await user.hover(within(label).getByRole("img", { name: "info-circle" })); + + expect(await screen.findByRole("link", { name: "Docs" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/proxy/caching", + ); + }); + + it("should link the prompt cache tooltips to the prompt caching docs", async () => { + const user = userEvent.setup(); + render( + , + ); + + const label = screen.getByText("Prompt Cache Read Tokens").closest(".ant-space") as HTMLElement; + await user.hover(within(label).getByRole("img", { name: "info-circle" })); + + expect(await screen.findByRole("link", { name: "Docs" })).toHaveAttribute( + "href", + "https://docs.litellm.ai/docs/completion/prompt_caching", + ); + }); + + it("should hide the Response Cache row when cache_hit is not a true/false value", () => { + render(); + + expect(screen.queryByText("Response Cache")).not.toBeInTheDocument(); }); it("should display LiteLLM Overhead when litellm_overhead_time_ms is in metadata", () => { diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx index d4521afddf6..2752cccce56 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/LogDetailContent.tsx @@ -1,5 +1,6 @@ import { useState } from "react"; -import { Typography, Descriptions, Card, Tag, Tabs, Alert, Collapse, Radio, Space, Spin } from "antd"; +import { Typography, Descriptions, Card, Tag, Tabs, Alert, Collapse, Radio, Space, Spin, Tooltip } from "antd"; +import { InfoCircleOutlined } from "@ant-design/icons"; import moment from "moment"; import { LogEntry } from "../columns"; import { formatNumberWithCommas } from "@/utils/dataUtils"; @@ -278,6 +279,40 @@ function getUncachedInputTextTokens(metadata: Record): number | und return Number.isFinite(n) ? n : undefined; } +const RESPONSE_CACHE_TOOLTIP = + "Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed."; +const PROMPT_CACHE_READ_TOOLTIP = + "Input tokens read from the LLM provider's prompt cache (e.g. Anthropic / OpenAI), billed at a discounted rate. Reported by the provider."; +const PROMPT_CACHE_CREATION_TOOLTIP = + "Input tokens written to the LLM provider's prompt cache for reuse by later requests."; +const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching"; +const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching"; + +function MetricLabel({ label, tooltip, docsUrl }: { label: string; tooltip: string; docsUrl: string }) { + return ( + + {label} + + {tooltip}{" "} + + Docs + + + } + > + + + + ); +} + function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: Record }) { const completionStartTime = logEntry.completionStartTime; const ttftMs = @@ -285,14 +320,11 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: ? new Date(completionStartTime).getTime() - new Date(logEntry.startTime).getTime() : null; - const hasCacheActivity = - logEntry.cache_hit || - (metadata?.additional_usage_values?.cache_read_input_tokens && - metadata.additional_usage_values.cache_read_input_tokens > 0); - - const cacheHitValue = String(logEntry.cache_hit ?? "None"); - const cacheHitColor = - cacheHitValue.toLowerCase() === "true" ? "green" : cacheHitValue.toLowerCase() === "false" ? "red" : "default"; + const responseCacheValue = String(logEntry.cache_hit ?? "").toLowerCase(); + const isResponseCacheHit = responseCacheValue === "true"; + const showResponseCache = isResponseCacheHit || responseCacheValue === "false"; + const promptCacheReadTokens = Number(metadata?.additional_usage_values?.cache_read_input_tokens) || 0; + const promptCacheCreationTokens = Number(metadata?.additional_usage_values?.cache_creation_input_tokens) || 0; const uncachedInputTokens = getUncachedInputTextTokens(metadata); const showAnthropicMessagesInputOutput = @@ -326,22 +358,44 @@ function MetricsSection({ logEntry, metadata }: { logEntry: LogEntry; metadata: {(ttftMs / 1000).toFixed(3)} s )} - {hasCacheActivity && ( - <> - - {cacheHitValue} - - {metadata?.additional_usage_values?.cache_read_input_tokens > 0 && ( - - {formatNumberWithCommas(metadata.additional_usage_values.cache_read_input_tokens)} - - )} - {metadata?.additional_usage_values?.cache_creation_input_tokens > 0 && ( - - {formatNumberWithCommas(metadata.additional_usage_values.cache_creation_input_tokens)} - - )} - + {showResponseCache && ( + + } + > + {isResponseCacheHit ? "Hit" : "Miss"} + + )} + {promptCacheReadTokens > 0 && ( + + } + > + {formatNumberWithCommas(promptCacheReadTokens)} + + )} + {promptCacheCreationTokens > 0 && ( + + } + > + {formatNumberWithCommas(promptCacheCreationTokens)} + )} {metadata?.litellm_overhead_time_ms !== undefined && metadata.litellm_overhead_time_ms !== null && (