From 1a9aa13bd26e5c7fd80e56a956f5ef8812c176a6 Mon Sep 17 00:00:00 2001 From: Christiaan Arnoldus Date: Thu, 26 Jun 2025 16:18:49 +0200 Subject: [PATCH] Use upstream_inference_cost for OpenRouter BYOK cost calculation and show cached token count (#5145) Improve OpenRouter cache calculation and show cached tokens --- src/api/providers/openrouter.ts | 14 +++---- .../__tests__/ContextWindowProgress.spec.tsx | 1 - webview-ui/src/components/chat/ChatView.tsx | 1 - webview-ui/src/components/chat/TaskHeader.tsx | 39 +++++++++---------- .../chat/__tests__/TaskHeader.spec.tsx | 1 - 5 files changed, 23 insertions(+), 33 deletions(-) diff --git a/src/api/providers/openrouter.ts b/src/api/providers/openrouter.ts index 51d97963e7..6565daa238 100644 --- a/src/api/providers/openrouter.ts +++ b/src/api/providers/openrouter.ts @@ -48,13 +48,11 @@ interface CompletionUsage { } total_tokens?: number cost?: number - is_byok?: boolean + cost_details?: { + upstream_inference_cost?: number + } } -// with bring your own key, OpenRouter charges 5% of what it normally would: https://openrouter.ai/docs/use-cases/byok -// so we multiply the cost reported by OpenRouter to get an estimate of what the request actually cost -const BYOK_COST_MULTIPLIER = 20 - export class OpenRouterHandler extends BaseProvider implements SingleCompletionHandler { protected options: ApiHandlerOptions private client: OpenAI @@ -168,11 +166,9 @@ export class OpenRouterHandler extends BaseProvider implements SingleCompletionH type: "usage", inputTokens: lastUsage.prompt_tokens || 0, outputTokens: lastUsage.completion_tokens || 0, - // Waiting on OpenRouter to figure out what this represents in the Gemini case - // and how to best support it. - // cacheReadTokens: lastUsage.prompt_tokens_details?.cached_tokens, + cacheReadTokens: lastUsage.prompt_tokens_details?.cached_tokens, reasoningTokens: lastUsage.completion_tokens_details?.reasoning_tokens, - totalCost: (lastUsage.is_byok ? BYOK_COST_MULTIPLIER : 1) * (lastUsage.cost || 0), + totalCost: (lastUsage.cost_details?.upstream_inference_cost || 0) + (lastUsage.cost || 0), } } } diff --git a/webview-ui/src/__tests__/ContextWindowProgress.spec.tsx b/webview-ui/src/__tests__/ContextWindowProgress.spec.tsx index 5a5ff463ef..6b989a4e74 100644 --- a/webview-ui/src/__tests__/ContextWindowProgress.spec.tsx +++ b/webview-ui/src/__tests__/ContextWindowProgress.spec.tsx @@ -51,7 +51,6 @@ describe("ContextWindowProgress", () => { task: { ts: Date.now(), type: "say" as const, say: "text" as const, text: "Test task" }, tokensIn: 100, tokensOut: 50, - doesModelSupportPromptCache: true, totalCost: 0.001, contextTokens: 1000, onClose: vi.fn(), diff --git a/webview-ui/src/components/chat/ChatView.tsx b/webview-ui/src/components/chat/ChatView.tsx index 77e1d88df7..5456c5b698 100644 --- a/webview-ui/src/components/chat/ChatView.tsx +++ b/webview-ui/src/components/chat/ChatView.tsx @@ -1371,7 +1371,6 @@ const ChatViewComponent: React.ForwardRefRenderFunction} - {doesModelSupportPromptCache && - ((typeof cacheReads === "number" && cacheReads > 0) || - (typeof cacheWrites === "number" && cacheWrites > 0)) && ( -
- {t("chat:task.cache")} - {typeof cacheWrites === "number" && cacheWrites > 0 && ( - - - {formatLargeNumber(cacheWrites)} - - )} - {typeof cacheReads === "number" && cacheReads > 0 && ( - - - {formatLargeNumber(cacheReads)} - - )} -
- )} + {((typeof cacheReads === "number" && cacheReads > 0) || + (typeof cacheWrites === "number" && cacheWrites > 0)) && ( +
+ {t("chat:task.cache")} + {typeof cacheWrites === "number" && cacheWrites > 0 && ( + + + {formatLargeNumber(cacheWrites)} + + )} + {typeof cacheReads === "number" && cacheReads > 0 && ( + + + {formatLargeNumber(cacheReads)} + + )} +
+ )} {!!totalCost && (
diff --git a/webview-ui/src/components/chat/__tests__/TaskHeader.spec.tsx b/webview-ui/src/components/chat/__tests__/TaskHeader.spec.tsx index 784a263531..9acd0ad5ce 100644 --- a/webview-ui/src/components/chat/__tests__/TaskHeader.spec.tsx +++ b/webview-ui/src/components/chat/__tests__/TaskHeader.spec.tsx @@ -49,7 +49,6 @@ describe("TaskHeader", () => { task: { type: "say", ts: Date.now(), text: "Test task", images: [] }, tokensIn: 100, tokensOut: 50, - doesModelSupportPromptCache: true, totalCost: 0.05, contextTokens: 200, buttonsDisabled: false,