diff --git a/litellm/constants.py b/litellm/constants.py
index 78aba30f9c0..765bbfe1e54 100644
--- a/litellm/constants.py
+++ b/litellm/constants.py
@@ -147,6 +147,7 @@ LITELLM_UI_ALLOW_HEADERS: Final = [
"x-litellm-adaptive-router-model",
"x-litellm-applied-guardrails",
"x-litellm-guardrail-scan-id",
+ "x-litellm-cache-key",
]
# Gemini model-specific minimal thinking budget constants
diff --git a/tests/test_litellm/proxy/test_proxy_server.py b/tests/test_litellm/proxy/test_proxy_server.py
index 3383527e932..b9a31acca96 100644
--- a/tests/test_litellm/proxy/test_proxy_server.py
+++ b/tests/test_litellm/proxy/test_proxy_server.py
@@ -78,6 +78,16 @@ def client_no_auth():
return TestClient(app)
+def test_cors_exposes_cache_key_header_to_browser_js():
+ from fastapi.middleware.cors import CORSMiddleware
+
+ from litellm.constants import LITELLM_UI_ALLOW_HEADERS
+
+ cors_middleware = next(m for m in app.user_middleware if m.cls is CORSMiddleware)
+ assert cors_middleware.kwargs["expose_headers"] is LITELLM_UI_ALLOW_HEADERS
+ assert "x-litellm-cache-key" in cors_middleware.kwargs["expose_headers"]
+
+
def test_login_v2_returns_redirect_url_and_sets_cookie(monkeypatch):
mock_login_result = {"user_id": "test-user"}
mock_prisma_client = MagicMock()
diff --git a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx
index 5afc94eb043..f31e1839739 100644
--- a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx
+++ b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.test.tsx
@@ -33,4 +33,22 @@ describe("ResponseMetrics prompt cache chips", () => {
expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument();
expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
});
+
+ it("shows the response cache indicator instead of the provider cache chips on a response-cache hit", () => {
+ render(
+ ,
+ );
+
+ expect(screen.getByText("Response Cache: Hit")).toBeInTheDocument();
+ expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument();
+ expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
+ });
+
+ it("does not show the response cache indicator when the flag is absent", () => {
+ render();
+
+ expect(screen.queryByText(/Response Cache/)).not.toBeInTheDocument();
+ });
});
diff --git a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx
index 3e7f2884b23..ec62d0618d7 100644
--- a/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx
+++ b/ui/litellm-dashboard/src/components/chat_ui/ResponseMetrics.tsx
@@ -7,12 +7,16 @@ import {
DatabaseBackup,
DollarSign,
Hash,
+ History,
Lightbulb,
Wrench,
} from "lucide-react";
import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip";
import { PROMPT_CACHE_CREATION_TOOLTIP, PROMPT_CACHE_READ_TOOLTIP } from "@/utils/promptCacheUsage";
+const RESPONSE_CACHE_TOOLTIP =
+ "This response was replayed from LiteLLM's response cache. The request never reached the provider, so it did not read from or write to the provider's own prompt cache.";
+
export interface TokenUsage {
completionTokens?: number;
promptTokens?: number;
@@ -21,6 +25,7 @@ export interface TokenUsage {
cacheReadTokens?: number;
cacheCreationTokens?: number;
cost?: number;
+ servedFromResponseCache?: boolean;
}
interface ResponseMetricsProps {
@@ -51,7 +56,22 @@ function MetricItem({ label, tooltip, icon, value }: MetricItemProps) {
);
}
+function ResponseCacheIndicator() {
+ return (
+ }
+ value="Hit"
+ />
+ );
+}
+
function PromptCacheChips({ usage }: { usage?: TokenUsage }) {
+ if (usage?.servedFromResponseCache) {
+ return ;
+ }
+
const readTokens = usage?.cacheReadTokens ?? 0;
const creationTokens = usage?.cacheCreationTokens ?? 0;
diff --git a/ui/litellm-dashboard/src/components/llm_calls/chat_completion.test.tsx b/ui/litellm-dashboard/src/components/llm_calls/chat_completion.test.tsx
index c91de3f6b20..bc4b4e3a351 100644
--- a/ui/litellm-dashboard/src/components/llm_calls/chat_completion.test.tsx
+++ b/ui/litellm-dashboard/src/components/llm_calls/chat_completion.test.tsx
@@ -24,6 +24,10 @@ vi.mock("openai", () => ({
},
}));
+const nonStreamingResponse = (data: unknown, headers: Record = {}) => ({
+ withResponse: async () => ({ data, response: { headers: new Headers(headers) } }),
+});
+
describe("chat_completion", () => {
const mockUpdateUI = vi.fn();
const mockChatHistory = [{ role: "user", content: "Hello" }];
@@ -226,25 +230,27 @@ describe("chat_completion", () => {
});
it("should send a non-streaming request and render the whole message at once when streaming is disabled", async () => {
- mockCreate.mockResolvedValueOnce({
- id: "chatcmpl-1",
- object: "chat.completion",
- created: 1,
- model: "gpt-4",
- choices: [
- {
- index: 0,
- finish_reason: "stop",
- message: { role: "assistant", content: "Hello there" },
+ mockCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ id: "chatcmpl-1",
+ object: "chat.completion",
+ created: 1,
+ model: "gpt-4",
+ choices: [
+ {
+ index: 0,
+ finish_reason: "stop",
+ message: { role: "assistant", content: "Hello there" },
+ },
+ ],
+ usage: {
+ completion_tokens: 2,
+ prompt_tokens: 5,
+ total_tokens: 7,
+ cost: 0.25,
},
- ],
- usage: {
- completion_tokens: 2,
- prompt_tokens: 5,
- total_tokens: 7,
- cost: 0.25,
- },
- });
+ }),
+ );
const onTimingData = vi.fn();
const onUsageData = vi.fn();
@@ -298,24 +304,26 @@ describe("chat_completion", () => {
});
it("should surface reasoning content and MCP metadata from a non-streaming response", async () => {
- mockCreate.mockResolvedValueOnce({
- model: "gpt-4",
- choices: [
- {
- index: 0,
- finish_reason: "stop",
- message: {
- role: "assistant",
- content: "done",
- reasoning_content: "thinking",
- provider_specific_fields: {
- mcp_tool_calls: [{ id: "call_1", function: { name: "search_docs", arguments: "{}" } }],
- mcp_call_results: [{ tool_call_id: "call_1", result: "found it" }],
+ mockCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ model: "gpt-4",
+ choices: [
+ {
+ index: 0,
+ finish_reason: "stop",
+ message: {
+ role: "assistant",
+ content: "done",
+ reasoning_content: "thinking",
+ provider_specific_fields: {
+ mcp_tool_calls: [{ id: "call_1", function: { name: "search_docs", arguments: "{}" } }],
+ mcp_call_results: [{ tool_call_id: "call_1", result: "found it" }],
+ },
},
},
- },
- ],
- });
+ ],
+ }),
+ );
const onReasoningContent = vi.fn();
const onMCPEvent = vi.fn();
@@ -459,3 +467,137 @@ describe("chat_completion prompt cache usage", () => {
expect(usageData).not.toHaveProperty("cacheCreationTokens");
});
});
+
+describe("chat_completion response cache", () => {
+ const mockUpdateUI = vi.fn();
+ const mockChatHistory = [{ role: "user", content: "Hello" }];
+
+ afterEach(() => {
+ vi.clearAllMocks();
+ });
+
+ it("flags a non-streaming response-cache hit even though it replays provider prompt-cache usage", async () => {
+ mockCreate.mockReturnValueOnce(
+ nonStreamingResponse(
+ {
+ id: "chatcmpl-replayed",
+ model: "gpt-4",
+ choices: [{ index: 0, finish_reason: "stop", message: { role: "assistant", content: "Hello there" } }],
+ usage: {
+ completion_tokens: 2,
+ prompt_tokens: 5000,
+ total_tokens: 5002,
+ prompt_tokens_details: { cached_tokens: 4695 },
+ },
+ },
+ { "x-litellm-cache-key": "cache-key-abc" },
+ ),
+ );
+
+ const onUsageData = vi.fn();
+
+ await makeOpenAIChatCompletionRequest(
+ mockChatHistory,
+ mockUpdateUI,
+ "gpt-4",
+ "test-token",
+ undefined, // tags
+ undefined, // signal
+ undefined, // onReasoningContent
+ undefined, // onTimingData
+ onUsageData,
+ undefined, // traceId
+ undefined, // vector_store_ids
+ undefined, // guardrails
+ undefined, // policies
+ undefined, // selectedMCPServers
+ undefined, // onImageGenerated
+ undefined, // onSearchResults
+ undefined, // temperature
+ undefined, // max_tokens
+ undefined, // onTotalLatency
+ undefined, // customBaseUrl
+ undefined, // mcpServers
+ undefined, // mcpServerToolRestrictions
+ undefined, // onMCPEvent
+ undefined, // mockTestFallbacks
+ undefined, // mcpToolsets
+ false, // streamingEnabled
+ );
+
+ expect(onUsageData).toHaveBeenCalledWith(
+ expect.objectContaining({ cacheReadTokens: 4695, servedFromResponseCache: true }),
+ );
+ });
+
+ it("does not flag a non-streaming response that missed the response cache", async () => {
+ mockCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ id: "chatcmpl-fresh",
+ model: "gpt-4",
+ choices: [{ index: 0, finish_reason: "stop", message: { role: "assistant", content: "Hello there" } }],
+ usage: { completion_tokens: 2, prompt_tokens: 5, total_tokens: 7 },
+ }),
+ );
+
+ const onUsageData = vi.fn();
+
+ await makeOpenAIChatCompletionRequest(
+ mockChatHistory,
+ mockUpdateUI,
+ "gpt-4",
+ "test-token",
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ onUsageData,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ false, // streamingEnabled
+ );
+
+ expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }));
+ });
+
+ it("never flags a streaming response, even when the proxy reports a cache key", async () => {
+ async function* mockStream() {
+ yield {
+ choices: [{ delta: {}, index: 0 }],
+ model: "gpt-4",
+ usage: { completion_tokens: 2, prompt_tokens: 5, total_tokens: 7 },
+ };
+ }
+ mockCreate.mockResolvedValueOnce(mockStream());
+
+ const onUsageData = vi.fn();
+
+ await makeOpenAIChatCompletionRequest(
+ mockChatHistory,
+ mockUpdateUI,
+ "gpt-4",
+ "test-token",
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ onUsageData,
+ );
+
+ expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }));
+ });
+});
diff --git a/ui/litellm-dashboard/src/components/llm_calls/chat_completion.tsx b/ui/litellm-dashboard/src/components/llm_calls/chat_completion.tsx
index ea66006349b..cd0852bc06e 100644
--- a/ui/litellm-dashboard/src/components/llm_calls/chat_completion.tsx
+++ b/ui/litellm-dashboard/src/components/llm_calls/chat_completion.tsx
@@ -73,6 +73,7 @@ export async function makeOpenAIChatCompletionRequest(
const startTime = Date.now();
let firstTokenReceived = false;
let timeToFirstToken: number | undefined = undefined;
+ let servedFromResponseCache = false;
// Track MCP metadata cumulatively across chunks
let mcpMetadata: {
@@ -143,7 +144,13 @@ export async function makeOpenAIChatCompletionRequest(
{ ...requestBody, stream: true, stream_options: { include_usage: true } },
{ signal },
)
- : [completionAsSingleChunk(await client.chat.completions.create({ ...requestBody, stream: false }, { signal }))];
+ : await (async () => {
+ const nonStreamingResponse = await client.chat.completions
+ .create({ ...requestBody, stream: false }, { signal })
+ .withResponse();
+ servedFromResponseCache = nonStreamingResponse.response.headers.get("x-litellm-cache-key") !== null;
+ return [completionAsSingleChunk(nonStreamingResponse.data)];
+ })();
for await (const chunk of response) {
// Process content and measure time to first token
@@ -228,6 +235,7 @@ export async function makeOpenAIChatCompletionRequest(
promptTokens: chunkWithUsage.usage.prompt_tokens,
totalTokens: chunkWithUsage.usage.total_tokens,
...extractPromptCacheTokens(chunkWithUsage.usage),
+ ...(servedFromResponseCache ? { servedFromResponseCache: true } : {}),
};
// Check for reasoning tokens
diff --git a/ui/litellm-dashboard/src/components/llm_calls/responses_api.test.tsx b/ui/litellm-dashboard/src/components/llm_calls/responses_api.test.tsx
index 033813397fc..0b94c093acd 100644
--- a/ui/litellm-dashboard/src/components/llm_calls/responses_api.test.tsx
+++ b/ui/litellm-dashboard/src/components/llm_calls/responses_api.test.tsx
@@ -20,6 +20,10 @@ vi.mock("openai", () => ({
},
}));
+const nonStreamingResponse = (data: unknown, headers: Record = {}) => ({
+ withResponse: async () => ({ data, response: { headers: new Headers(headers) } }),
+});
+
describe("responses_api", () => {
const mockUpdateTextUI = vi.fn();
const messages: MessageType[] = [{ role: "user", content: "Hello" }];
@@ -71,19 +75,21 @@ describe("responses_api", () => {
});
it("should send a non-streaming request and render the whole output at once when streaming is disabled", async () => {
- mockResponsesCreate.mockResolvedValueOnce({
- id: "resp_456",
- output: [
- {
- type: "message",
- content: [
- { type: "output_text", text: "Full " },
- { type: "output_text", text: "answer" },
- ],
- },
- ],
- usage: { output_tokens: 3, input_tokens: 4, total_tokens: 7 },
- });
+ mockResponsesCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ id: "resp_456",
+ output: [
+ {
+ type: "message",
+ content: [
+ { type: "output_text", text: "Full " },
+ { type: "output_text", text: "answer" },
+ ],
+ },
+ ],
+ usage: { output_tokens: 3, input_tokens: 4, total_tokens: 7 },
+ }),
+ );
const onTimingData = vi.fn();
const onUsageData = vi.fn();
@@ -162,10 +168,12 @@ describe("responses_api", () => {
expect(onTotalLatency).toHaveBeenCalledTimes(1);
expect(onTotalLatency).toHaveBeenLastCalledWith(expect.any(Number));
- mockResponsesCreate.mockResolvedValueOnce({
- id: "resp_latency",
- output: [{ type: "message", content: [{ type: "output_text", text: "Answer" }] }],
- });
+ mockResponsesCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ id: "resp_latency",
+ output: [{ type: "message", content: [{ type: "output_text", text: "Answer" }] }],
+ }),
+ );
await callWithStreaming(false);
expect(onTotalLatency).toHaveBeenCalledTimes(2);
@@ -224,14 +232,16 @@ describe("responses_api", () => {
});
it("should replay MCP output items as events for a non-streaming response", async () => {
- mockResponsesCreate.mockResolvedValueOnce({
- id: "resp_789",
- output: [
- { type: "mcp_call", id: "mcp_1", name: "search_docs", arguments: "{}", output: "found it" },
- { type: "message", content: [{ type: "output_text", text: "Answer" }] },
- ],
- usage: { output_tokens: 1, input_tokens: 1, total_tokens: 2 },
- });
+ mockResponsesCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ id: "resp_789",
+ output: [
+ { type: "mcp_call", id: "mcp_1", name: "search_docs", arguments: "{}", output: "found it" },
+ { type: "message", content: [{ type: "output_text", text: "Answer" }] },
+ ],
+ usage: { output_tokens: 1, input_tokens: 1, total_tokens: 2 },
+ }),
+ );
const onMCPEvent = vi.fn();
const onUsageData = vi.fn();
@@ -413,3 +423,131 @@ describe("responses_api prompt cache usage", () => {
});
});
});
+
+describe("responses_api response cache", () => {
+ const mockUpdateTextUI = vi.fn();
+ const messages: MessageType[] = [{ role: "user", content: "Hello" }];
+
+ afterEach(() => {
+ vi.clearAllMocks();
+ });
+
+ it("flags a non-streaming response-cache hit even though it replays provider prompt-cache usage", async () => {
+ mockResponsesCreate.mockReturnValueOnce(
+ nonStreamingResponse(
+ {
+ id: "resp_replayed",
+ output: [{ type: "message", content: [{ type: "output_text", text: "Full answer" }] }],
+ usage: {
+ output_tokens: 2,
+ input_tokens: 5000,
+ total_tokens: 5002,
+ input_tokens_details: { cached_tokens: 4695 },
+ },
+ },
+ { "x-litellm-cache-key": "cache-key-abc" },
+ ),
+ );
+
+ const onUsageData = vi.fn();
+
+ await makeOpenAIResponsesRequest(
+ messages,
+ mockUpdateTextUI,
+ "gpt-4",
+ "test-token",
+ undefined, // tags
+ undefined, // signal
+ undefined, // onReasoningContent
+ undefined, // onTimingData
+ onUsageData,
+ undefined, // traceId
+ undefined, // vector_store_ids
+ undefined, // guardrails
+ undefined, // policies
+ undefined, // selectedMCPServers
+ undefined, // previousResponseId
+ undefined, // onResponseId
+ undefined, // onMCPEvent
+ undefined, // codeInterpreterEnabled
+ undefined, // onCodeInterpreterResult
+ undefined, // customBaseUrl
+ undefined, // mcpServers
+ undefined, // mcpServerToolRestrictions
+ undefined, // mcpToolsets
+ false, // streamingEnabled
+ );
+
+ expect(onUsageData).toHaveBeenCalledWith(
+ expect.objectContaining({ cacheReadTokens: 4695, servedFromResponseCache: true }),
+ "",
+ );
+ });
+
+ it("does not flag a non-streaming response that missed the response cache", async () => {
+ mockResponsesCreate.mockReturnValueOnce(
+ nonStreamingResponse({
+ id: "resp_fresh",
+ output: [{ type: "message", content: [{ type: "output_text", text: "Full answer" }] }],
+ usage: { output_tokens: 2, input_tokens: 5, total_tokens: 7 },
+ }),
+ );
+
+ const onUsageData = vi.fn();
+
+ await makeOpenAIResponsesRequest(
+ messages,
+ mockUpdateTextUI,
+ "gpt-4",
+ "test-token",
+ undefined, // tags
+ undefined, // signal
+ undefined, // onReasoningContent
+ undefined, // onTimingData
+ onUsageData,
+ undefined, // traceId
+ undefined, // vector_store_ids
+ undefined, // guardrails
+ undefined, // policies
+ undefined, // selectedMCPServers
+ undefined, // previousResponseId
+ undefined, // onResponseId
+ undefined, // onMCPEvent
+ undefined, // codeInterpreterEnabled
+ undefined, // onCodeInterpreterResult
+ undefined, // customBaseUrl
+ undefined, // mcpServers
+ undefined, // mcpServerToolRestrictions
+ undefined, // mcpToolsets
+ false, // streamingEnabled
+ );
+
+ expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }), "");
+ });
+
+ it("never flags a streaming response, even when the proxy reports a cache key", async () => {
+ async function* mockStream() {
+ yield {
+ type: "response.completed",
+ response: { id: "resp_stream", usage: { output_tokens: 2, input_tokens: 5, total_tokens: 7 } },
+ };
+ }
+ mockResponsesCreate.mockResolvedValueOnce(mockStream());
+
+ const onUsageData = vi.fn();
+
+ await makeOpenAIResponsesRequest(
+ messages,
+ mockUpdateTextUI,
+ "gpt-4",
+ "test-token",
+ undefined,
+ undefined,
+ undefined,
+ undefined,
+ onUsageData,
+ );
+
+ expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }), "");
+ });
+});
diff --git a/ui/litellm-dashboard/src/components/llm_calls/responses_api.tsx b/ui/litellm-dashboard/src/components/llm_calls/responses_api.tsx
index 8d71a4e29a8..7ab76488504 100644
--- a/ui/litellm-dashboard/src/components/llm_calls/responses_api.tsx
+++ b/ui/litellm-dashboard/src/components/llm_calls/responses_api.tsx
@@ -116,6 +116,7 @@ export async function makeOpenAIResponsesRequest(
try {
const startTime = Date.now();
let firstTokenReceived = false;
+ let servedFromResponseCache = false;
// Format messages for the API
const formattedInput = messages.map((message) => {
@@ -202,7 +203,15 @@ export async function makeOpenAIResponsesRequest(
// Create request to OpenAI responses API
// Use 'any' type to avoid TypeScript issues with the experimental API
- const response = await (client as any).responses.create({ ...requestBody, stream: streamingEnabled }, { signal });
+ const response = streamingEnabled
+ ? await (client as any).responses.create({ ...requestBody, stream: true }, { signal })
+ : await (async () => {
+ const nonStreamingResponse = await (client as any).responses
+ .create({ ...requestBody, stream: false }, { signal })
+ .withResponse();
+ servedFromResponseCache = nonStreamingResponse.response.headers.get("x-litellm-cache-key") !== null;
+ return nonStreamingResponse.data;
+ })();
const events = streamingEnabled ? response : responseAsEvents(response);
let mcpToolUsed = "";
@@ -292,6 +301,7 @@ export async function makeOpenAIResponsesRequest(
promptTokens: usage.input_tokens,
totalTokens: usage.total_tokens,
...extractPromptCacheTokens(usage),
+ ...(servedFromResponseCache ? { servedFromResponseCache: true } : {}),
};
// Add reasoning tokens if available