From b72dab8049f3a8c6da5033c866acca0ab96bda95 Mon Sep 17 00:00:00 2001 From: Yassin Kortam Date: Thu, 13 Aug 2026 16:58:14 -0700 Subject: [PATCH] feat(ui): show provider prompt cache tokens in chat response metrics (#36827) The chat metrics bar reported In/Out/Reasoning/Total/cost only, so a playground user had no signal that provider prompt caching worked. The cached-token counts were already visible in the Logs drawer, which meant the answer to "does caching work here" lived on a different page. Adds cacheReadTokens and cacheCreationTokens to TokenUsage and renders them as two chips, reusing the prompt-cache tooltip wording already introduced for the Logs drawer so both surfaces say the same thing. A single helper, extractPromptCacheTokens, normalizes the three usage shapes the playground consumes: Anthropic Messages (cache_read_input_tokens / cache_creation_input_tokens), chat completions (prompt_tokens_details) and the Responses API (input_tokens_details). All three producers call it instead of parsing per surface. Counts that are absent, zero or non-finite are dropped, so providers without prompt caching render exactly what they render today. --- .../llm_calls/anthropic_messages.test.tsx | 59 +++++++++++++++++ .../llm_calls/anthropic_messages.tsx | 2 + .../chat_ui/ResponseMetrics.test.tsx | 36 ++++++++++ .../components/chat_ui/ResponseMetrics.tsx | 44 ++++++++++++- .../src/components/chat_ui/types.ts | 10 +-- .../llm_calls/chat_completion.test.tsx | 65 +++++++++++++++++++ .../components/llm_calls/chat_completion.tsx | 2 + .../llm_calls/responses_api.test.tsx | 56 ++++++++++++++++ .../components/llm_calls/responses_api.tsx | 2 + .../LogDetailsDrawer/LogDetailContent.tsx | 5 +- .../src/utils/promptCacheUsage.test.ts | 39 +++++++++++ .../src/utils/promptCacheUsage.ts | 37 +++++++++++ 12 files changed, 345 insertions(+), 12 deletions(-) create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.test.tsx create mode 100644 ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx create mode 100644 ui/litellm-dashboard/src/utils/promptCacheUsage.test.ts create mode 100644 ui/litellm-dashboard/src/utils/promptCacheUsage.ts diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.test.tsx new file mode 100644 index 00000000000..995104d4b7f --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.test.tsx @@ -0,0 +1,59 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { makeAnthropicMessagesRequest } from "./anthropic_messages"; +import type { TokenUsage } from "@/components/chat_ui/ResponseMetrics"; + +vi.mock("@/components/networking", () => ({ + getProxyBaseUrl: vi.fn(() => "https://example.com"), +})); + +const mockMessagesStream = vi.fn(); + +vi.mock("@anthropic-ai/sdk", () => ({ + default: vi.fn(() => ({ messages: { stream: mockMessagesStream } })), +})); + +describe("anthropic_messages prompt cache usage", () => { + const captureUsage = async (usage: Record): Promise => { + async function* mockStream() { + yield { + type: "message_delta", + usage: { input_tokens: 5000, output_tokens: 2, ...usage }, + }; + } + mockMessagesStream.mockReturnValue(mockStream()); + + const onUsageData = vi.fn(); + await makeAnthropicMessagesRequest( + [{ role: "user", content: "Hello" }], + vi.fn(), + "claude-haiku-4-5", + "test-token", + undefined, + undefined, + undefined, + undefined, + onUsageData, + ); + + expect(onUsageData).toHaveBeenCalledTimes(1); + return onUsageData.mock.calls[0][0] as TokenUsage; + }; + + afterEach(() => { + vi.clearAllMocks(); + }); + + it("surfaces read and creation tokens from Anthropic-shape usage", async () => { + await expect( + captureUsage({ cache_read_input_tokens: 4695, cache_creation_input_tokens: 1234 }), + ).resolves.toMatchObject({ cacheReadTokens: 4695, cacheCreationTokens: 1234, promptTokens: 5000 }); + }); + + it("omits cache fields entirely when Anthropic reports no prompt caching", async () => { + const usageData = await captureUsage({}); + + expect(usageData).not.toHaveProperty("cacheReadTokens"); + expect(usageData).not.toHaveProperty("cacheCreationTokens"); + expect(usageData.promptTokens).toBe(5000); + }); +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.tsx b/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.tsx index 4319315396a..4facf8a2046 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/playground/llm_calls/anthropic_messages.tsx @@ -5,6 +5,7 @@ import { buildMcpToolBlocks } from "@/components/llm_calls/mcp_tool_blocks"; import { MCPServer, MCPToolset } from "@/components/mcp_tools/types"; import { getProxyBaseUrl } from "@/components/networking"; import NotificationManager from "@/components/molecules/notifications_manager"; +import { extractPromptCacheTokens } from "@/utils/promptCacheUsage"; export async function makeAnthropicMessagesRequest( messages: MessageType[], @@ -109,6 +110,7 @@ export async function makeAnthropicMessagesRequest( completionTokens: usage.output_tokens, promptTokens: usage.input_tokens, totalTokens: usage.input_tokens + usage.output_tokens, + ...extractPromptCacheTokens(usage), }; onUsageData(usageData); } diff --git a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx new file mode 100644 index 00000000000..5afc94eb043 --- /dev/null +++ b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx @@ -0,0 +1,36 @@ +import { render, screen } from "@testing-library/react"; +import { describe, it, expect } from "vitest"; +import ResponseMetrics, { type TokenUsage } from "./ResponseMetrics"; + +const baseUsage: TokenUsage = { promptTokens: 5000, completionTokens: 12, totalTokens: 5012 }; + +describe("ResponseMetrics prompt cache chips", () => { + it("renders both cache chips when the provider reports reads and writes", () => { + render(); + + expect(screen.getByText("Cache Read: 4695")).toBeInTheDocument(); + expect(screen.getByText("Cache Write: 1234")).toBeInTheDocument(); + }); + + it("renders only the read chip when the provider reports reads alone", () => { + render(); + + expect(screen.getByText("Cache Read: 4695")).toBeInTheDocument(); + expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument(); + }); + + it("renders no cache chips for a provider that reports no cache fields", () => { + render(); + + expect(screen.getByText("In: 5000")).toBeInTheDocument(); + expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument(); + expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument(); + }); + + it("renders no cache chips when the provider reports zero cache tokens", () => { + render(); + + expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument(); + expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx index 58069d43ade..9795201125a 100644 --- a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx +++ b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx @@ -1,12 +1,25 @@ import React from "react"; -import { ArrowDownToLine, ArrowUpFromLine, Clock, DollarSign, Hash, Lightbulb, Wrench } from "lucide-react"; +import { + ArrowDownToLine, + ArrowUpFromLine, + Clock, + Database, + DatabaseBackup, + DollarSign, + Hash, + Lightbulb, + Wrench, +} from "lucide-react"; import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip"; +import { PROMPT_CACHE_CREATION_TOOLTIP, PROMPT_CACHE_READ_TOOLTIP } from "@/utils/promptCacheUsage"; export interface TokenUsage { completionTokens?: number; promptTokens?: number; totalTokens?: number; reasoningTokens?: number; + cacheReadTokens?: number; + cacheCreationTokens?: number; cost?: number; } @@ -38,6 +51,33 @@ function MetricItem({ label, tooltip, icon, value }: MetricItemProps) { ); } +function PromptCacheChips({ usage }: { usage?: TokenUsage }) { + const readTokens = usage?.cacheReadTokens ?? 0; + const creationTokens = usage?.cacheCreationTokens ?? 0; + + return ( + <> + {readTokens > 0 && ( +