From ee51908373db20ff5a69a89fb458e06b7ed6e331 Mon Sep 17 00:00:00 2001 From: jaberjaber23 Date: Sun, 16 Aug 2026 03:22:29 +0300 Subject: [PATCH] fix: add cache read pricing for runinfra Qwen3.8-27B Verified live on 2026-08-16: two identical 4,217 token prompts through the public endpoint, second response reported 4,160 cached prompt tokens; the model page lists $0.01 per million cached input tokens. --- litellm/model_prices_and_context_window_backup.json | 2 ++ model_prices_and_context_window.json | 2 ++ 2 files changed, 4 insertions(+) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index e88e3f858ac..04121620479 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -48020,6 +48020,7 @@ "supports_tool_choice": true }, "runinfra/Qwen/Qwen3.8-27B": { + "cache_read_input_token_cost": 1e-08, "input_cost_per_token": 1e-07, "litellm_provider": "runinfra", "max_input_tokens": 262144, @@ -48033,6 +48034,7 @@ ], "supports_function_calling": true, "supports_native_streaming": true, + "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index e88e3f858ac..04121620479 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -48020,6 +48020,7 @@ "supports_tool_choice": true }, "runinfra/Qwen/Qwen3.8-27B": { + "cache_read_input_token_cost": 1e-08, "input_cost_per_token": 1e-07, "litellm_provider": "runinfra", "max_input_tokens": 262144, @@ -48033,6 +48034,7 @@ ], "supports_function_calling": true, "supports_native_streaming": true, + "supports_prompt_caching": true, "supports_reasoning": true, "supports_response_schema": true, "supports_tool_choice": true