mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(dashboard): don't show a stale provider prompt-cache chip on a response-cache hit (#37951)
* fix(dashboard): don't show a stale provider prompt-cache chip on a response-cache hit The playground's non-streaming chat completion and responses paths replayed a cache hit's original usage payload verbatim, so ResponseMetrics kept rendering the provider's prompt-cache-write/read chips using token counts from the original request. Detect the hit via the x-litellm-cache-key response header and render a Response Cache indicator instead. * fix(dashboard): expose x-litellm-cache-key through CORS for the playground cache-hit indicator
This commit is contained in:
parent
27ca05a707
commit
104fe73113
8 changed files with 408 additions and 61 deletions
|
|
@ -147,6 +147,7 @@ LITELLM_UI_ALLOW_HEADERS: Final = [
|
|||
"x-litellm-adaptive-router-model",
|
||||
"x-litellm-applied-guardrails",
|
||||
"x-litellm-guardrail-scan-id",
|
||||
"x-litellm-cache-key",
|
||||
]
|
||||
|
||||
# Gemini model-specific minimal thinking budget constants
|
||||
|
|
|
|||
|
|
@ -78,6 +78,16 @@ def client_no_auth():
|
|||
return TestClient(app)
|
||||
|
||||
|
||||
def test_cors_exposes_cache_key_header_to_browser_js():
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
|
||||
from litellm.constants import LITELLM_UI_ALLOW_HEADERS
|
||||
|
||||
cors_middleware = next(m for m in app.user_middleware if m.cls is CORSMiddleware)
|
||||
assert cors_middleware.kwargs["expose_headers"] is LITELLM_UI_ALLOW_HEADERS
|
||||
assert "x-litellm-cache-key" in cors_middleware.kwargs["expose_headers"]
|
||||
|
||||
|
||||
def test_login_v2_returns_redirect_url_and_sets_cookie(monkeypatch):
|
||||
mock_login_result = {"user_id": "test-user"}
|
||||
mock_prisma_client = MagicMock()
|
||||
|
|
|
|||
|
|
@ -33,4 +33,22 @@ describe("ResponseMetrics prompt cache chips", () => {
|
|||
expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows the response cache indicator instead of the provider cache chips on a response-cache hit", () => {
|
||||
render(
|
||||
<ResponseMetrics
|
||||
usage={{ ...baseUsage, cacheReadTokens: 4695, cacheCreationTokens: 1234, servedFromResponseCache: true }}
|
||||
/>,
|
||||
);
|
||||
|
||||
expect(screen.getByText("Response Cache: Hit")).toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("does not show the response cache indicator when the flag is absent", () => {
|
||||
render(<ResponseMetrics usage={baseUsage} />);
|
||||
|
||||
expect(screen.queryByText(/Response Cache/)).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -7,12 +7,16 @@ import {
|
|||
DatabaseBackup,
|
||||
DollarSign,
|
||||
Hash,
|
||||
History,
|
||||
Lightbulb,
|
||||
Wrench,
|
||||
} from "lucide-react";
|
||||
import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip";
|
||||
import { PROMPT_CACHE_CREATION_TOOLTIP, PROMPT_CACHE_READ_TOOLTIP } from "@/utils/promptCacheUsage";
|
||||
|
||||
const RESPONSE_CACHE_TOOLTIP =
|
||||
"This response was replayed from LiteLLM's response cache. The request never reached the provider, so it did not read from or write to the provider's own prompt cache.";
|
||||
|
||||
export interface TokenUsage {
|
||||
completionTokens?: number;
|
||||
promptTokens?: number;
|
||||
|
|
@ -21,6 +25,7 @@ export interface TokenUsage {
|
|||
cacheReadTokens?: number;
|
||||
cacheCreationTokens?: number;
|
||||
cost?: number;
|
||||
servedFromResponseCache?: boolean;
|
||||
}
|
||||
|
||||
interface ResponseMetricsProps {
|
||||
|
|
@ -51,7 +56,22 @@ function MetricItem({ label, tooltip, icon, value }: MetricItemProps) {
|
|||
);
|
||||
}
|
||||
|
||||
function ResponseCacheIndicator() {
|
||||
return (
|
||||
<MetricItem
|
||||
label="Response Cache"
|
||||
tooltip={RESPONSE_CACHE_TOOLTIP}
|
||||
icon={<History className="size-3" aria-hidden="true" />}
|
||||
value="Hit"
|
||||
/>
|
||||
);
|
||||
}
|
||||
|
||||
function PromptCacheChips({ usage }: { usage?: TokenUsage }) {
|
||||
if (usage?.servedFromResponseCache) {
|
||||
return <ResponseCacheIndicator />;
|
||||
}
|
||||
|
||||
const readTokens = usage?.cacheReadTokens ?? 0;
|
||||
const creationTokens = usage?.cacheCreationTokens ?? 0;
|
||||
|
||||
|
|
|
|||
|
|
@ -24,6 +24,10 @@ vi.mock("openai", () => ({
|
|||
},
|
||||
}));
|
||||
|
||||
const nonStreamingResponse = (data: unknown, headers: Record<string, string> = {}) => ({
|
||||
withResponse: async () => ({ data, response: { headers: new Headers(headers) } }),
|
||||
});
|
||||
|
||||
describe("chat_completion", () => {
|
||||
const mockUpdateUI = vi.fn();
|
||||
const mockChatHistory = [{ role: "user", content: "Hello" }];
|
||||
|
|
@ -226,25 +230,27 @@ describe("chat_completion", () => {
|
|||
});
|
||||
|
||||
it("should send a non-streaming request and render the whole message at once when streaming is disabled", async () => {
|
||||
mockCreate.mockResolvedValueOnce({
|
||||
id: "chatcmpl-1",
|
||||
object: "chat.completion",
|
||||
created: 1,
|
||||
model: "gpt-4",
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
finish_reason: "stop",
|
||||
message: { role: "assistant", content: "Hello there" },
|
||||
mockCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
id: "chatcmpl-1",
|
||||
object: "chat.completion",
|
||||
created: 1,
|
||||
model: "gpt-4",
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
finish_reason: "stop",
|
||||
message: { role: "assistant", content: "Hello there" },
|
||||
},
|
||||
],
|
||||
usage: {
|
||||
completion_tokens: 2,
|
||||
prompt_tokens: 5,
|
||||
total_tokens: 7,
|
||||
cost: 0.25,
|
||||
},
|
||||
],
|
||||
usage: {
|
||||
completion_tokens: 2,
|
||||
prompt_tokens: 5,
|
||||
total_tokens: 7,
|
||||
cost: 0.25,
|
||||
},
|
||||
});
|
||||
}),
|
||||
);
|
||||
|
||||
const onTimingData = vi.fn();
|
||||
const onUsageData = vi.fn();
|
||||
|
|
@ -298,24 +304,26 @@ describe("chat_completion", () => {
|
|||
});
|
||||
|
||||
it("should surface reasoning content and MCP metadata from a non-streaming response", async () => {
|
||||
mockCreate.mockResolvedValueOnce({
|
||||
model: "gpt-4",
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
finish_reason: "stop",
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "done",
|
||||
reasoning_content: "thinking",
|
||||
provider_specific_fields: {
|
||||
mcp_tool_calls: [{ id: "call_1", function: { name: "search_docs", arguments: "{}" } }],
|
||||
mcp_call_results: [{ tool_call_id: "call_1", result: "found it" }],
|
||||
mockCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
model: "gpt-4",
|
||||
choices: [
|
||||
{
|
||||
index: 0,
|
||||
finish_reason: "stop",
|
||||
message: {
|
||||
role: "assistant",
|
||||
content: "done",
|
||||
reasoning_content: "thinking",
|
||||
provider_specific_fields: {
|
||||
mcp_tool_calls: [{ id: "call_1", function: { name: "search_docs", arguments: "{}" } }],
|
||||
mcp_call_results: [{ tool_call_id: "call_1", result: "found it" }],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
],
|
||||
}),
|
||||
);
|
||||
|
||||
const onReasoningContent = vi.fn();
|
||||
const onMCPEvent = vi.fn();
|
||||
|
|
@ -459,3 +467,137 @@ describe("chat_completion prompt cache usage", () => {
|
|||
expect(usageData).not.toHaveProperty("cacheCreationTokens");
|
||||
});
|
||||
});
|
||||
|
||||
describe("chat_completion response cache", () => {
|
||||
const mockUpdateUI = vi.fn();
|
||||
const mockChatHistory = [{ role: "user", content: "Hello" }];
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("flags a non-streaming response-cache hit even though it replays provider prompt-cache usage", async () => {
|
||||
mockCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse(
|
||||
{
|
||||
id: "chatcmpl-replayed",
|
||||
model: "gpt-4",
|
||||
choices: [{ index: 0, finish_reason: "stop", message: { role: "assistant", content: "Hello there" } }],
|
||||
usage: {
|
||||
completion_tokens: 2,
|
||||
prompt_tokens: 5000,
|
||||
total_tokens: 5002,
|
||||
prompt_tokens_details: { cached_tokens: 4695 },
|
||||
},
|
||||
},
|
||||
{ "x-litellm-cache-key": "cache-key-abc" },
|
||||
),
|
||||
);
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
|
||||
await makeOpenAIChatCompletionRequest(
|
||||
mockChatHistory,
|
||||
mockUpdateUI,
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined, // tags
|
||||
undefined, // signal
|
||||
undefined, // onReasoningContent
|
||||
undefined, // onTimingData
|
||||
onUsageData,
|
||||
undefined, // traceId
|
||||
undefined, // vector_store_ids
|
||||
undefined, // guardrails
|
||||
undefined, // policies
|
||||
undefined, // selectedMCPServers
|
||||
undefined, // onImageGenerated
|
||||
undefined, // onSearchResults
|
||||
undefined, // temperature
|
||||
undefined, // max_tokens
|
||||
undefined, // onTotalLatency
|
||||
undefined, // customBaseUrl
|
||||
undefined, // mcpServers
|
||||
undefined, // mcpServerToolRestrictions
|
||||
undefined, // onMCPEvent
|
||||
undefined, // mockTestFallbacks
|
||||
undefined, // mcpToolsets
|
||||
false, // streamingEnabled
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledWith(
|
||||
expect.objectContaining({ cacheReadTokens: 4695, servedFromResponseCache: true }),
|
||||
);
|
||||
});
|
||||
|
||||
it("does not flag a non-streaming response that missed the response cache", async () => {
|
||||
mockCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
id: "chatcmpl-fresh",
|
||||
model: "gpt-4",
|
||||
choices: [{ index: 0, finish_reason: "stop", message: { role: "assistant", content: "Hello there" } }],
|
||||
usage: { completion_tokens: 2, prompt_tokens: 5, total_tokens: 7 },
|
||||
}),
|
||||
);
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
|
||||
await makeOpenAIChatCompletionRequest(
|
||||
mockChatHistory,
|
||||
mockUpdateUI,
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
onUsageData,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
false, // streamingEnabled
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }));
|
||||
});
|
||||
|
||||
it("never flags a streaming response, even when the proxy reports a cache key", async () => {
|
||||
async function* mockStream() {
|
||||
yield {
|
||||
choices: [{ delta: {}, index: 0 }],
|
||||
model: "gpt-4",
|
||||
usage: { completion_tokens: 2, prompt_tokens: 5, total_tokens: 7 },
|
||||
};
|
||||
}
|
||||
mockCreate.mockResolvedValueOnce(mockStream());
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
|
||||
await makeOpenAIChatCompletionRequest(
|
||||
mockChatHistory,
|
||||
mockUpdateUI,
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
onUsageData,
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }));
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -73,6 +73,7 @@ export async function makeOpenAIChatCompletionRequest(
|
|||
const startTime = Date.now();
|
||||
let firstTokenReceived = false;
|
||||
let timeToFirstToken: number | undefined = undefined;
|
||||
let servedFromResponseCache = false;
|
||||
|
||||
// Track MCP metadata cumulatively across chunks
|
||||
let mcpMetadata: {
|
||||
|
|
@ -143,7 +144,13 @@ export async function makeOpenAIChatCompletionRequest(
|
|||
{ ...requestBody, stream: true, stream_options: { include_usage: true } },
|
||||
{ signal },
|
||||
)
|
||||
: [completionAsSingleChunk(await client.chat.completions.create({ ...requestBody, stream: false }, { signal }))];
|
||||
: await (async () => {
|
||||
const nonStreamingResponse = await client.chat.completions
|
||||
.create({ ...requestBody, stream: false }, { signal })
|
||||
.withResponse();
|
||||
servedFromResponseCache = nonStreamingResponse.response.headers.get("x-litellm-cache-key") !== null;
|
||||
return [completionAsSingleChunk(nonStreamingResponse.data)];
|
||||
})();
|
||||
|
||||
for await (const chunk of response) {
|
||||
// Process content and measure time to first token
|
||||
|
|
@ -228,6 +235,7 @@ export async function makeOpenAIChatCompletionRequest(
|
|||
promptTokens: chunkWithUsage.usage.prompt_tokens,
|
||||
totalTokens: chunkWithUsage.usage.total_tokens,
|
||||
...extractPromptCacheTokens(chunkWithUsage.usage),
|
||||
...(servedFromResponseCache ? { servedFromResponseCache: true } : {}),
|
||||
};
|
||||
|
||||
// Check for reasoning tokens
|
||||
|
|
|
|||
|
|
@ -20,6 +20,10 @@ vi.mock("openai", () => ({
|
|||
},
|
||||
}));
|
||||
|
||||
const nonStreamingResponse = (data: unknown, headers: Record<string, string> = {}) => ({
|
||||
withResponse: async () => ({ data, response: { headers: new Headers(headers) } }),
|
||||
});
|
||||
|
||||
describe("responses_api", () => {
|
||||
const mockUpdateTextUI = vi.fn();
|
||||
const messages: MessageType[] = [{ role: "user", content: "Hello" }];
|
||||
|
|
@ -71,19 +75,21 @@ describe("responses_api", () => {
|
|||
});
|
||||
|
||||
it("should send a non-streaming request and render the whole output at once when streaming is disabled", async () => {
|
||||
mockResponsesCreate.mockResolvedValueOnce({
|
||||
id: "resp_456",
|
||||
output: [
|
||||
{
|
||||
type: "message",
|
||||
content: [
|
||||
{ type: "output_text", text: "Full " },
|
||||
{ type: "output_text", text: "answer" },
|
||||
],
|
||||
},
|
||||
],
|
||||
usage: { output_tokens: 3, input_tokens: 4, total_tokens: 7 },
|
||||
});
|
||||
mockResponsesCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
id: "resp_456",
|
||||
output: [
|
||||
{
|
||||
type: "message",
|
||||
content: [
|
||||
{ type: "output_text", text: "Full " },
|
||||
{ type: "output_text", text: "answer" },
|
||||
],
|
||||
},
|
||||
],
|
||||
usage: { output_tokens: 3, input_tokens: 4, total_tokens: 7 },
|
||||
}),
|
||||
);
|
||||
|
||||
const onTimingData = vi.fn();
|
||||
const onUsageData = vi.fn();
|
||||
|
|
@ -162,10 +168,12 @@ describe("responses_api", () => {
|
|||
expect(onTotalLatency).toHaveBeenCalledTimes(1);
|
||||
expect(onTotalLatency).toHaveBeenLastCalledWith(expect.any(Number));
|
||||
|
||||
mockResponsesCreate.mockResolvedValueOnce({
|
||||
id: "resp_latency",
|
||||
output: [{ type: "message", content: [{ type: "output_text", text: "Answer" }] }],
|
||||
});
|
||||
mockResponsesCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
id: "resp_latency",
|
||||
output: [{ type: "message", content: [{ type: "output_text", text: "Answer" }] }],
|
||||
}),
|
||||
);
|
||||
|
||||
await callWithStreaming(false);
|
||||
expect(onTotalLatency).toHaveBeenCalledTimes(2);
|
||||
|
|
@ -224,14 +232,16 @@ describe("responses_api", () => {
|
|||
});
|
||||
|
||||
it("should replay MCP output items as events for a non-streaming response", async () => {
|
||||
mockResponsesCreate.mockResolvedValueOnce({
|
||||
id: "resp_789",
|
||||
output: [
|
||||
{ type: "mcp_call", id: "mcp_1", name: "search_docs", arguments: "{}", output: "found it" },
|
||||
{ type: "message", content: [{ type: "output_text", text: "Answer" }] },
|
||||
],
|
||||
usage: { output_tokens: 1, input_tokens: 1, total_tokens: 2 },
|
||||
});
|
||||
mockResponsesCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
id: "resp_789",
|
||||
output: [
|
||||
{ type: "mcp_call", id: "mcp_1", name: "search_docs", arguments: "{}", output: "found it" },
|
||||
{ type: "message", content: [{ type: "output_text", text: "Answer" }] },
|
||||
],
|
||||
usage: { output_tokens: 1, input_tokens: 1, total_tokens: 2 },
|
||||
}),
|
||||
);
|
||||
|
||||
const onMCPEvent = vi.fn();
|
||||
const onUsageData = vi.fn();
|
||||
|
|
@ -413,3 +423,131 @@ describe("responses_api prompt cache usage", () => {
|
|||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("responses_api response cache", () => {
|
||||
const mockUpdateTextUI = vi.fn();
|
||||
const messages: MessageType[] = [{ role: "user", content: "Hello" }];
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("flags a non-streaming response-cache hit even though it replays provider prompt-cache usage", async () => {
|
||||
mockResponsesCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse(
|
||||
{
|
||||
id: "resp_replayed",
|
||||
output: [{ type: "message", content: [{ type: "output_text", text: "Full answer" }] }],
|
||||
usage: {
|
||||
output_tokens: 2,
|
||||
input_tokens: 5000,
|
||||
total_tokens: 5002,
|
||||
input_tokens_details: { cached_tokens: 4695 },
|
||||
},
|
||||
},
|
||||
{ "x-litellm-cache-key": "cache-key-abc" },
|
||||
),
|
||||
);
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
|
||||
await makeOpenAIResponsesRequest(
|
||||
messages,
|
||||
mockUpdateTextUI,
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined, // tags
|
||||
undefined, // signal
|
||||
undefined, // onReasoningContent
|
||||
undefined, // onTimingData
|
||||
onUsageData,
|
||||
undefined, // traceId
|
||||
undefined, // vector_store_ids
|
||||
undefined, // guardrails
|
||||
undefined, // policies
|
||||
undefined, // selectedMCPServers
|
||||
undefined, // previousResponseId
|
||||
undefined, // onResponseId
|
||||
undefined, // onMCPEvent
|
||||
undefined, // codeInterpreterEnabled
|
||||
undefined, // onCodeInterpreterResult
|
||||
undefined, // customBaseUrl
|
||||
undefined, // mcpServers
|
||||
undefined, // mcpServerToolRestrictions
|
||||
undefined, // mcpToolsets
|
||||
false, // streamingEnabled
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledWith(
|
||||
expect.objectContaining({ cacheReadTokens: 4695, servedFromResponseCache: true }),
|
||||
"",
|
||||
);
|
||||
});
|
||||
|
||||
it("does not flag a non-streaming response that missed the response cache", async () => {
|
||||
mockResponsesCreate.mockReturnValueOnce(
|
||||
nonStreamingResponse({
|
||||
id: "resp_fresh",
|
||||
output: [{ type: "message", content: [{ type: "output_text", text: "Full answer" }] }],
|
||||
usage: { output_tokens: 2, input_tokens: 5, total_tokens: 7 },
|
||||
}),
|
||||
);
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
|
||||
await makeOpenAIResponsesRequest(
|
||||
messages,
|
||||
mockUpdateTextUI,
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined, // tags
|
||||
undefined, // signal
|
||||
undefined, // onReasoningContent
|
||||
undefined, // onTimingData
|
||||
onUsageData,
|
||||
undefined, // traceId
|
||||
undefined, // vector_store_ids
|
||||
undefined, // guardrails
|
||||
undefined, // policies
|
||||
undefined, // selectedMCPServers
|
||||
undefined, // previousResponseId
|
||||
undefined, // onResponseId
|
||||
undefined, // onMCPEvent
|
||||
undefined, // codeInterpreterEnabled
|
||||
undefined, // onCodeInterpreterResult
|
||||
undefined, // customBaseUrl
|
||||
undefined, // mcpServers
|
||||
undefined, // mcpServerToolRestrictions
|
||||
undefined, // mcpToolsets
|
||||
false, // streamingEnabled
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }), "");
|
||||
});
|
||||
|
||||
it("never flags a streaming response, even when the proxy reports a cache key", async () => {
|
||||
async function* mockStream() {
|
||||
yield {
|
||||
type: "response.completed",
|
||||
response: { id: "resp_stream", usage: { output_tokens: 2, input_tokens: 5, total_tokens: 7 } },
|
||||
};
|
||||
}
|
||||
mockResponsesCreate.mockResolvedValueOnce(mockStream());
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
|
||||
await makeOpenAIResponsesRequest(
|
||||
messages,
|
||||
mockUpdateTextUI,
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
onUsageData,
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledWith(expect.not.objectContaining({ servedFromResponseCache: true }), "");
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -116,6 +116,7 @@ export async function makeOpenAIResponsesRequest(
|
|||
try {
|
||||
const startTime = Date.now();
|
||||
let firstTokenReceived = false;
|
||||
let servedFromResponseCache = false;
|
||||
|
||||
// Format messages for the API
|
||||
const formattedInput = messages.map((message) => {
|
||||
|
|
@ -202,7 +203,15 @@ export async function makeOpenAIResponsesRequest(
|
|||
|
||||
// Create request to OpenAI responses API
|
||||
// Use 'any' type to avoid TypeScript issues with the experimental API
|
||||
const response = await (client as any).responses.create({ ...requestBody, stream: streamingEnabled }, { signal });
|
||||
const response = streamingEnabled
|
||||
? await (client as any).responses.create({ ...requestBody, stream: true }, { signal })
|
||||
: await (async () => {
|
||||
const nonStreamingResponse = await (client as any).responses
|
||||
.create({ ...requestBody, stream: false }, { signal })
|
||||
.withResponse();
|
||||
servedFromResponseCache = nonStreamingResponse.response.headers.get("x-litellm-cache-key") !== null;
|
||||
return nonStreamingResponse.data;
|
||||
})();
|
||||
const events = streamingEnabled ? response : responseAsEvents(response);
|
||||
|
||||
let mcpToolUsed = "";
|
||||
|
|
@ -292,6 +301,7 @@ export async function makeOpenAIResponsesRequest(
|
|||
promptTokens: usage.input_tokens,
|
||||
totalTokens: usage.total_tokens,
|
||||
...extractPromptCacheTokens(usage),
|
||||
...(servedFromResponseCache ? { servedFromResponseCache: true } : {}),
|
||||
};
|
||||
|
||||
// Add reasoning tokens if available
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue