mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-28 01:32:17 +00:00
The chat metrics bar reported In/Out/Reasoning/Total/cost only, so a playground user had no signal that provider prompt caching worked. The cached-token counts were already visible in the Logs drawer, which meant the answer to "does caching work here" lived on a different page. Adds cacheReadTokens and cacheCreationTokens to TokenUsage and renders them as two chips, reusing the prompt-cache tooltip wording already introduced for the Logs drawer so both surfaces say the same thing. A single helper, extractPromptCacheTokens, normalizes the three usage shapes the playground consumes: Anthropic Messages (cache_read_input_tokens / cache_creation_input_tokens), chat completions (prompt_tokens_details) and the Responses API (input_tokens_details). All three producers call it instead of parsing per surface. Counts that are absent, zero or non-finite are dropped, so providers without prompt caching render exactly what they render today.
39 lines
1.5 KiB
TypeScript
39 lines
1.5 KiB
TypeScript
import { describe, expect, it } from "vitest";
|
|
import { extractPromptCacheTokens } from "./promptCacheUsage";
|
|
|
|
describe("extractPromptCacheTokens", () => {
|
|
it("reads the Anthropic Messages shape", () => {
|
|
expect(
|
|
extractPromptCacheTokens({ cache_read_input_tokens: 5678, cache_creation_input_tokens: 1234 }),
|
|
).toStrictEqual({ cacheReadTokens: 5678, cacheCreationTokens: 1234 });
|
|
});
|
|
|
|
it("reads the chat completions shape", () => {
|
|
expect(
|
|
extractPromptCacheTokens({ prompt_tokens_details: { cached_tokens: 4695, cache_write_tokens: 0 } }),
|
|
).toStrictEqual({ cacheReadTokens: 4695 });
|
|
});
|
|
|
|
it("reads the Responses API shape", () => {
|
|
expect(
|
|
extractPromptCacheTokens({ input_tokens_details: { cached_tokens: 0, cache_write_tokens: 4695 } }),
|
|
).toStrictEqual({ cacheCreationTokens: 4695 });
|
|
});
|
|
|
|
it("returns nothing for usage without cache fields", () => {
|
|
expect(extractPromptCacheTokens({})).toStrictEqual({});
|
|
expect(extractPromptCacheTokens(undefined)).toStrictEqual({});
|
|
expect(extractPromptCacheTokens(null)).toStrictEqual({});
|
|
});
|
|
|
|
it("drops zero and non-finite counts so non-caching providers render nothing", () => {
|
|
expect(
|
|
extractPromptCacheTokens({
|
|
cache_read_input_tokens: 0,
|
|
cache_creation_input_tokens: null,
|
|
prompt_tokens_details: { cached_tokens: 0, cache_write_tokens: 0 },
|
|
}),
|
|
).toStrictEqual({});
|
|
expect(extractPromptCacheTokens({ cache_read_input_tokens: Number.NaN })).toStrictEqual({});
|
|
});
|
|
});
|