mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
feat(ui): show provider prompt cache tokens in chat response metrics (#36827)
The chat metrics bar reported In/Out/Reasoning/Total/cost only, so a playground user had no signal that provider prompt caching worked. The cached-token counts were already visible in the Logs drawer, which meant the answer to "does caching work here" lived on a different page. Adds cacheReadTokens and cacheCreationTokens to TokenUsage and renders them as two chips, reusing the prompt-cache tooltip wording already introduced for the Logs drawer so both surfaces say the same thing. A single helper, extractPromptCacheTokens, normalizes the three usage shapes the playground consumes: Anthropic Messages (cache_read_input_tokens / cache_creation_input_tokens), chat completions (prompt_tokens_details) and the Responses API (input_tokens_details). All three producers call it instead of parsing per surface. Counts that are absent, zero or non-finite are dropped, so providers without prompt caching render exactly what they render today.
This commit is contained in:
parent
4bc27f1664
commit
b72dab8049
12 changed files with 345 additions and 12 deletions
|
|
@ -0,0 +1,59 @@
|
|||
import { afterEach, describe, expect, it, vi } from "vitest";
|
||||
import { makeAnthropicMessagesRequest } from "./anthropic_messages";
|
||||
import type { TokenUsage } from "@/components/chat_ui/ResponseMetrics";
|
||||
|
||||
vi.mock("@/components/networking", () => ({
|
||||
getProxyBaseUrl: vi.fn(() => "https://example.com"),
|
||||
}));
|
||||
|
||||
const mockMessagesStream = vi.fn();
|
||||
|
||||
vi.mock("@anthropic-ai/sdk", () => ({
|
||||
default: vi.fn(() => ({ messages: { stream: mockMessagesStream } })),
|
||||
}));
|
||||
|
||||
describe("anthropic_messages prompt cache usage", () => {
|
||||
const captureUsage = async (usage: Record<string, unknown>): Promise<TokenUsage> => {
|
||||
async function* mockStream() {
|
||||
yield {
|
||||
type: "message_delta",
|
||||
usage: { input_tokens: 5000, output_tokens: 2, ...usage },
|
||||
};
|
||||
}
|
||||
mockMessagesStream.mockReturnValue(mockStream());
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
await makeAnthropicMessagesRequest(
|
||||
[{ role: "user", content: "Hello" }],
|
||||
vi.fn(),
|
||||
"claude-haiku-4-5",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
onUsageData,
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledTimes(1);
|
||||
return onUsageData.mock.calls[0][0] as TokenUsage;
|
||||
};
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("surfaces read and creation tokens from Anthropic-shape usage", async () => {
|
||||
await expect(
|
||||
captureUsage({ cache_read_input_tokens: 4695, cache_creation_input_tokens: 1234 }),
|
||||
).resolves.toMatchObject({ cacheReadTokens: 4695, cacheCreationTokens: 1234, promptTokens: 5000 });
|
||||
});
|
||||
|
||||
it("omits cache fields entirely when Anthropic reports no prompt caching", async () => {
|
||||
const usageData = await captureUsage({});
|
||||
|
||||
expect(usageData).not.toHaveProperty("cacheReadTokens");
|
||||
expect(usageData).not.toHaveProperty("cacheCreationTokens");
|
||||
expect(usageData.promptTokens).toBe(5000);
|
||||
});
|
||||
});
|
||||
|
|
@ -5,6 +5,7 @@ import { buildMcpToolBlocks } from "@/components/llm_calls/mcp_tool_blocks";
|
|||
import { MCPServer, MCPToolset } from "@/components/mcp_tools/types";
|
||||
import { getProxyBaseUrl } from "@/components/networking";
|
||||
import NotificationManager from "@/components/molecules/notifications_manager";
|
||||
import { extractPromptCacheTokens } from "@/utils/promptCacheUsage";
|
||||
|
||||
export async function makeAnthropicMessagesRequest(
|
||||
messages: MessageType[],
|
||||
|
|
@ -109,6 +110,7 @@ export async function makeAnthropicMessagesRequest(
|
|||
completionTokens: usage.output_tokens,
|
||||
promptTokens: usage.input_tokens,
|
||||
totalTokens: usage.input_tokens + usage.output_tokens,
|
||||
...extractPromptCacheTokens(usage),
|
||||
};
|
||||
onUsageData(usageData);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,36 @@
|
|||
import { render, screen } from "@testing-library/react";
|
||||
import { describe, it, expect } from "vitest";
|
||||
import ResponseMetrics, { type TokenUsage } from "./ResponseMetrics";
|
||||
|
||||
const baseUsage: TokenUsage = { promptTokens: 5000, completionTokens: 12, totalTokens: 5012 };
|
||||
|
||||
describe("ResponseMetrics prompt cache chips", () => {
|
||||
it("renders both cache chips when the provider reports reads and writes", () => {
|
||||
render(<ResponseMetrics usage={{ ...baseUsage, cacheReadTokens: 4695, cacheCreationTokens: 1234 }} />);
|
||||
|
||||
expect(screen.getByText("Cache Read: 4695")).toBeInTheDocument();
|
||||
expect(screen.getByText("Cache Write: 1234")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders only the read chip when the provider reports reads alone", () => {
|
||||
render(<ResponseMetrics usage={{ ...baseUsage, cacheReadTokens: 4695 }} />);
|
||||
|
||||
expect(screen.getByText("Cache Read: 4695")).toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders no cache chips for a provider that reports no cache fields", () => {
|
||||
render(<ResponseMetrics usage={baseUsage} />);
|
||||
|
||||
expect(screen.getByText("In: 5000")).toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders no cache chips when the provider reports zero cache tokens", () => {
|
||||
render(<ResponseMetrics usage={{ ...baseUsage, cacheReadTokens: 0, cacheCreationTokens: 0 }} />);
|
||||
|
||||
expect(screen.queryByText(/Cache Read/)).not.toBeInTheDocument();
|
||||
expect(screen.queryByText(/Cache Write/)).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
@ -1,12 +1,25 @@
|
|||
import React from "react";
|
||||
import { ArrowDownToLine, ArrowUpFromLine, Clock, DollarSign, Hash, Lightbulb, Wrench } from "lucide-react";
|
||||
import {
|
||||
ArrowDownToLine,
|
||||
ArrowUpFromLine,
|
||||
Clock,
|
||||
Database,
|
||||
DatabaseBackup,
|
||||
DollarSign,
|
||||
Hash,
|
||||
Lightbulb,
|
||||
Wrench,
|
||||
} from "lucide-react";
|
||||
import { Tooltip, TooltipContent, TooltipTrigger } from "@/components/ui/tooltip";
|
||||
import { PROMPT_CACHE_CREATION_TOOLTIP, PROMPT_CACHE_READ_TOOLTIP } from "@/utils/promptCacheUsage";
|
||||
|
||||
export interface TokenUsage {
|
||||
completionTokens?: number;
|
||||
promptTokens?: number;
|
||||
totalTokens?: number;
|
||||
reasoningTokens?: number;
|
||||
cacheReadTokens?: number;
|
||||
cacheCreationTokens?: number;
|
||||
cost?: number;
|
||||
}
|
||||
|
||||
|
|
@ -38,6 +51,33 @@ function MetricItem({ label, tooltip, icon, value }: MetricItemProps) {
|
|||
);
|
||||
}
|
||||
|
||||
function PromptCacheChips({ usage }: { usage?: TokenUsage }) {
|
||||
const readTokens = usage?.cacheReadTokens ?? 0;
|
||||
const creationTokens = usage?.cacheCreationTokens ?? 0;
|
||||
|
||||
return (
|
||||
<>
|
||||
{readTokens > 0 && (
|
||||
<MetricItem
|
||||
label="Cache Read"
|
||||
tooltip={PROMPT_CACHE_READ_TOOLTIP}
|
||||
icon={<Database className="size-3" aria-hidden="true" />}
|
||||
value={String(readTokens)}
|
||||
/>
|
||||
)}
|
||||
|
||||
{creationTokens > 0 && (
|
||||
<MetricItem
|
||||
label="Cache Write"
|
||||
tooltip={PROMPT_CACHE_CREATION_TOOLTIP}
|
||||
icon={<DatabaseBackup className="size-3" aria-hidden="true" />}
|
||||
value={String(creationTokens)}
|
||||
/>
|
||||
)}
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
const ResponseMetrics: React.FC<ResponseMetricsProps> = ({ timeToFirstToken, totalLatency, usage, toolName }) => {
|
||||
if (!timeToFirstToken && !totalLatency && !usage) return null;
|
||||
|
||||
|
|
@ -70,6 +110,8 @@ const ResponseMetrics: React.FC<ResponseMetricsProps> = ({ timeToFirstToken, tot
|
|||
/>
|
||||
)}
|
||||
|
||||
<PromptCacheChips usage={usage} />
|
||||
|
||||
{usage?.completionTokens !== undefined && (
|
||||
<MetricItem
|
||||
label="Out"
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
import type { TokenUsage } from "./ResponseMetrics";
|
||||
|
||||
export interface VectorStoreSearchResult {
|
||||
score: number;
|
||||
content: Array<{ text: string; type: string }>;
|
||||
|
|
@ -33,13 +35,7 @@ export interface MessageType {
|
|||
reasoningContent?: string;
|
||||
timeToFirstToken?: number;
|
||||
totalLatency?: number;
|
||||
usage?: {
|
||||
completionTokens?: number;
|
||||
promptTokens?: number;
|
||||
totalTokens?: number;
|
||||
reasoningTokens?: number;
|
||||
cost?: number;
|
||||
};
|
||||
usage?: TokenUsage;
|
||||
toolName?: string;
|
||||
imagePreviewUrl?: string;
|
||||
image?: {
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { makeOpenAIChatCompletionRequest } from "./chat_completion";
|
||||
import type { TokenUsage } from "../chat_ui/ResponseMetrics";
|
||||
|
||||
vi.mock("@/components/networking", () => ({
|
||||
getProxyBaseUrl: vi.fn(() => "https://example.com"),
|
||||
|
|
@ -394,3 +395,67 @@ describe("chat_completion", () => {
|
|||
expect(callArgs).not.toHaveProperty("mock_testing_fallbacks");
|
||||
});
|
||||
});
|
||||
|
||||
describe("chat_completion prompt cache usage", () => {
|
||||
const captureUsage = async (usage: Record<string, unknown>): Promise<TokenUsage> => {
|
||||
async function* mockStream() {
|
||||
yield {
|
||||
choices: [{ delta: {}, index: 0 }],
|
||||
model: "gpt-4",
|
||||
usage: { completion_tokens: 2, prompt_tokens: 5000, total_tokens: 5002, ...usage },
|
||||
};
|
||||
}
|
||||
mockCreate.mockResolvedValue(mockStream());
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
await makeOpenAIChatCompletionRequest(
|
||||
[{ role: "user", content: "Hello" }],
|
||||
vi.fn(),
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
onUsageData,
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledTimes(1);
|
||||
return onUsageData.mock.calls[0][0] as TokenUsage;
|
||||
};
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("surfaces read and creation tokens from Anthropic-shape usage", async () => {
|
||||
await expect(
|
||||
captureUsage({ cache_read_input_tokens: 4695, cache_creation_input_tokens: 1234 }),
|
||||
).resolves.toMatchObject({ cacheReadTokens: 4695, cacheCreationTokens: 1234 });
|
||||
});
|
||||
|
||||
it("surfaces read tokens from OpenAI-shape prompt_tokens_details", async () => {
|
||||
await expect(
|
||||
captureUsage({ prompt_tokens_details: { cached_tokens: 4695, cache_write_tokens: 0 } }),
|
||||
).resolves.toMatchObject({ cacheReadTokens: 4695, promptTokens: 5000 });
|
||||
});
|
||||
|
||||
it("omits cache fields entirely for a provider that reports none", async () => {
|
||||
const usageData = await captureUsage({});
|
||||
|
||||
expect(usageData).not.toHaveProperty("cacheReadTokens");
|
||||
expect(usageData).not.toHaveProperty("cacheCreationTokens");
|
||||
expect(usageData.promptTokens).toBe(5000);
|
||||
});
|
||||
|
||||
it("omits cache fields when the provider reports zeroes", async () => {
|
||||
const usageData = await captureUsage({
|
||||
cache_read_input_tokens: 0,
|
||||
cache_creation_input_tokens: 0,
|
||||
prompt_tokens_details: { cached_tokens: 0 },
|
||||
});
|
||||
|
||||
expect(usageData).not.toHaveProperty("cacheReadTokens");
|
||||
expect(usageData).not.toHaveProperty("cacheCreationTokens");
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import { TokenUsage } from "../chat_ui/ResponseMetrics";
|
|||
import { VectorStoreSearchResponse } from "../chat_ui/types";
|
||||
import { getProxyBaseUrl } from "@/components/networking";
|
||||
import { MCPServer, MCPToolset, type MCPEvent } from "@/components/mcp_tools/types";
|
||||
import { extractPromptCacheTokens } from "@/utils/promptCacheUsage";
|
||||
|
||||
const completionAsSingleChunk = (completion: ChatCompletion): ChatCompletionChunk =>
|
||||
({
|
||||
|
|
@ -226,6 +227,7 @@ export async function makeOpenAIChatCompletionRequest(
|
|||
completionTokens: chunkWithUsage.usage.completion_tokens,
|
||||
promptTokens: chunkWithUsage.usage.prompt_tokens,
|
||||
totalTokens: chunkWithUsage.usage.total_tokens,
|
||||
...extractPromptCacheTokens(chunkWithUsage.usage),
|
||||
};
|
||||
|
||||
// Check for reasoning tokens
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
|
||||
import { makeOpenAIResponsesRequest } from "./responses_api";
|
||||
import { MessageType } from "../chat_ui/types";
|
||||
import type { TokenUsage } from "../chat_ui/ResponseMetrics";
|
||||
|
||||
vi.mock("@/components/networking", () => ({
|
||||
getProxyBaseUrl: vi.fn(() => "https://example.com"),
|
||||
|
|
@ -294,3 +295,58 @@ describe("responses_api", () => {
|
|||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("responses_api prompt cache usage", () => {
|
||||
const captureUsage = async (usage: Record<string, unknown>): Promise<TokenUsage> => {
|
||||
async function* mockStream() {
|
||||
yield {
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id: "resp_cache",
|
||||
usage: { output_tokens: 2, input_tokens: 5000, total_tokens: 5002, ...usage },
|
||||
},
|
||||
};
|
||||
}
|
||||
mockResponsesCreate.mockResolvedValue(mockStream());
|
||||
|
||||
const onUsageData = vi.fn();
|
||||
await makeOpenAIResponsesRequest(
|
||||
[{ role: "user", content: "Hello" }],
|
||||
vi.fn(),
|
||||
"gpt-4",
|
||||
"test-token",
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
onUsageData,
|
||||
);
|
||||
|
||||
expect(onUsageData).toHaveBeenCalledTimes(1);
|
||||
return onUsageData.mock.calls[0][0] as TokenUsage;
|
||||
};
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllMocks();
|
||||
});
|
||||
|
||||
it("surfaces read tokens from Responses-shape input_tokens_details", async () => {
|
||||
await expect(
|
||||
captureUsage({ input_tokens_details: { cached_tokens: 4695, cache_write_tokens: 0 } }),
|
||||
).resolves.toMatchObject({ cacheReadTokens: 4695, promptTokens: 5000 });
|
||||
});
|
||||
|
||||
it("surfaces creation tokens from Responses-shape cache writes", async () => {
|
||||
await expect(
|
||||
captureUsage({ input_tokens_details: { cached_tokens: 0, cache_write_tokens: 4695 } }),
|
||||
).resolves.toMatchObject({ cacheCreationTokens: 4695 });
|
||||
});
|
||||
|
||||
it("omits cache fields entirely for a provider that reports none", async () => {
|
||||
const usageData = await captureUsage({});
|
||||
|
||||
expect(usageData).not.toHaveProperty("cacheReadTokens");
|
||||
expect(usageData).not.toHaveProperty("cacheCreationTokens");
|
||||
expect(usageData.promptTokens).toBe(5000);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import { MessageType } from "../chat_ui/types";
|
|||
import { TokenUsage } from "../chat_ui/ResponseMetrics";
|
||||
import { getProxyBaseUrl } from "@/components/networking";
|
||||
import NotificationManager from "@/components/molecules/notifications_manager";
|
||||
import { extractPromptCacheTokens } from "@/utils/promptCacheUsage";
|
||||
import type { MCPEvent } from "@/components/mcp_tools/types";
|
||||
import { MCPServer, MCPToolset } from "@/components/mcp_tools/types";
|
||||
import {
|
||||
|
|
@ -290,6 +291,7 @@ export async function makeOpenAIResponsesRequest(
|
|||
completionTokens: usage.output_tokens,
|
||||
promptTokens: usage.input_tokens,
|
||||
totalTokens: usage.total_tokens,
|
||||
...extractPromptCacheTokens(usage),
|
||||
};
|
||||
|
||||
// Add reasoning tokens if available
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import { InfoCircleOutlined } from "@ant-design/icons";
|
|||
import moment from "moment";
|
||||
import { LogEntry } from "../columns";
|
||||
import { formatNumberWithCommas } from "@/utils/dataUtils";
|
||||
import { PROMPT_CACHE_CREATION_TOOLTIP, PROMPT_CACHE_READ_TOOLTIP } from "@/utils/promptCacheUsage";
|
||||
import GuardrailViewer from "../GuardrailViewer/GuardrailViewer";
|
||||
import EvalViewer from "../EvalViewer/EvalViewer";
|
||||
import { CostBreakdownViewer } from "../CostBreakdownViewer";
|
||||
|
|
@ -285,10 +286,6 @@ function getUncachedInputTextTokens(metadata: Record<string, any>): number | und
|
|||
|
||||
const RESPONSE_CACHE_TOOLTIP =
|
||||
"Whether this request was served from LiteLLM's response cache (e.g. Redis / in-memory), skipping the LLM provider call entirely. This is separate from provider prompt caching; a Miss here does not mean prompt caching failed.";
|
||||
const PROMPT_CACHE_READ_TOOLTIP =
|
||||
"Input tokens read from the LLM provider's prompt cache (e.g. Anthropic / OpenAI), billed at a discounted rate. Reported by the provider.";
|
||||
const PROMPT_CACHE_CREATION_TOOLTIP =
|
||||
"Input tokens written to the LLM provider's prompt cache for reuse by later requests.";
|
||||
const RESPONSE_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/proxy/caching";
|
||||
const PROMPT_CACHE_DOCS_URL = "https://docs.litellm.ai/docs/completion/prompt_caching";
|
||||
|
||||
|
|
|
|||
39
ui/litellm-dashboard/src/utils/promptCacheUsage.test.ts
Normal file
39
ui/litellm-dashboard/src/utils/promptCacheUsage.test.ts
Normal file
|
|
@ -0,0 +1,39 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
import { extractPromptCacheTokens } from "./promptCacheUsage";
|
||||
|
||||
describe("extractPromptCacheTokens", () => {
|
||||
it("reads the Anthropic Messages shape", () => {
|
||||
expect(
|
||||
extractPromptCacheTokens({ cache_read_input_tokens: 5678, cache_creation_input_tokens: 1234 }),
|
||||
).toStrictEqual({ cacheReadTokens: 5678, cacheCreationTokens: 1234 });
|
||||
});
|
||||
|
||||
it("reads the chat completions shape", () => {
|
||||
expect(
|
||||
extractPromptCacheTokens({ prompt_tokens_details: { cached_tokens: 4695, cache_write_tokens: 0 } }),
|
||||
).toStrictEqual({ cacheReadTokens: 4695 });
|
||||
});
|
||||
|
||||
it("reads the Responses API shape", () => {
|
||||
expect(
|
||||
extractPromptCacheTokens({ input_tokens_details: { cached_tokens: 0, cache_write_tokens: 4695 } }),
|
||||
).toStrictEqual({ cacheCreationTokens: 4695 });
|
||||
});
|
||||
|
||||
it("returns nothing for usage without cache fields", () => {
|
||||
expect(extractPromptCacheTokens({})).toStrictEqual({});
|
||||
expect(extractPromptCacheTokens(undefined)).toStrictEqual({});
|
||||
expect(extractPromptCacheTokens(null)).toStrictEqual({});
|
||||
});
|
||||
|
||||
it("drops zero and non-finite counts so non-caching providers render nothing", () => {
|
||||
expect(
|
||||
extractPromptCacheTokens({
|
||||
cache_read_input_tokens: 0,
|
||||
cache_creation_input_tokens: null,
|
||||
prompt_tokens_details: { cached_tokens: 0, cache_write_tokens: 0 },
|
||||
}),
|
||||
).toStrictEqual({});
|
||||
expect(extractPromptCacheTokens({ cache_read_input_tokens: Number.NaN })).toStrictEqual({});
|
||||
});
|
||||
});
|
||||
37
ui/litellm-dashboard/src/utils/promptCacheUsage.ts
Normal file
37
ui/litellm-dashboard/src/utils/promptCacheUsage.ts
Normal file
|
|
@ -0,0 +1,37 @@
|
|||
export const PROMPT_CACHE_READ_TOOLTIP =
|
||||
"Input tokens read from the LLM provider's prompt cache (e.g. Anthropic / OpenAI), billed at a discounted rate. Reported by the provider.";
|
||||
export const PROMPT_CACHE_CREATION_TOOLTIP =
|
||||
"Input tokens written to the LLM provider's prompt cache for reuse by later requests.";
|
||||
|
||||
interface CachedTokenDetails {
|
||||
cached_tokens?: number | null;
|
||||
cache_write_tokens?: number | null;
|
||||
}
|
||||
|
||||
export interface ProviderCacheUsage {
|
||||
cache_read_input_tokens?: number | null;
|
||||
cache_creation_input_tokens?: number | null;
|
||||
prompt_tokens_details?: CachedTokenDetails | null;
|
||||
input_tokens_details?: CachedTokenDetails | null;
|
||||
}
|
||||
|
||||
export interface PromptCacheTokens {
|
||||
cacheReadTokens?: number;
|
||||
cacheCreationTokens?: number;
|
||||
}
|
||||
|
||||
const positiveTokenCount = (value: number | null | undefined): number | undefined =>
|
||||
typeof value === "number" && Number.isFinite(value) && value > 0 ? value : undefined;
|
||||
|
||||
export const extractPromptCacheTokens = (usage: ProviderCacheUsage | null | undefined): PromptCacheTokens => {
|
||||
const details = usage?.prompt_tokens_details ?? usage?.input_tokens_details;
|
||||
const cacheReadTokens =
|
||||
positiveTokenCount(usage?.cache_read_input_tokens) ?? positiveTokenCount(details?.cached_tokens);
|
||||
const cacheCreationTokens =
|
||||
positiveTokenCount(usage?.cache_creation_input_tokens) ?? positiveTokenCount(details?.cache_write_tokens);
|
||||
|
||||
return {
|
||||
...(cacheReadTokens !== undefined && { cacheReadTokens }),
|
||||
...(cacheCreationTokens !== undefined && { cacheCreationTokens }),
|
||||
};
|
||||
};
|
||||
Loading…
Add table
Reference in a new issue