mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix: add cache read pricing for runinfra Qwen3.8-27B
Verified live on 2026-08-16: two identical 4,217 token prompts through the public endpoint, second response reported 4,160 cached prompt tokens; the model page lists $0.01 per million cached input tokens.
This commit is contained in:
parent
18690ea835
commit
ee51908373
2 changed files with 4 additions and 0 deletions
|
|
@ -48020,6 +48020,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/Qwen/Qwen3.8-27B": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "runinfra",
|
||||
"max_input_tokens": 262144,
|
||||
|
|
@ -48033,6 +48034,7 @@
|
|||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
|
|
|
|||
|
|
@ -48020,6 +48020,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"runinfra/Qwen/Qwen3.8-27B": {
|
||||
"cache_read_input_token_cost": 1e-08,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"litellm_provider": "runinfra",
|
||||
"max_input_tokens": 262144,
|
||||
|
|
@ -48033,6 +48034,7 @@
|
|||
],
|
||||
"supports_function_calling": true,
|
||||
"supports_native_streaming": true,
|
||||
"supports_prompt_caching": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_response_schema": true,
|
||||
"supports_tool_choice": true
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue