mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(pricing): add cache_read_input_token_cost to 8 deepseek models
8 deepseek and openrouter/deepseek models declared only the legacy `input_cost_per_token_cache_hit` field, but the cost calculator (`litellm.litellm_core_utils.llm_cost_calc.utils._get_token_base_cost`) only reads `cache_read_input_token_cost`. As a result, prompt-cache-hit tokens were billed at $0 instead of the cache-read rate for: - deepseek/deepseek-coder - deepseek/deepseek-r1 - deepseek/deepseek-v3.2 - openrouter/deepseek/deepseek-chat-v3.1 - openrouter/deepseek/deepseek-v3.2 - openrouter/deepseek/deepseek-v3.2-exp - openrouter/deepseek/deepseek-r1 - openrouter/deepseek/deepseek-r1-0528 This patch adds `cache_read_input_token_cost` with the same value as the existing legacy field to both pricing JSONs, matching the convention already used by `deepseek/deepseek-chat`, `deepseek/deepseek-reasoner`, `deepseek/deepseek-v3`, and others. Adds a regression test that parametrizes over the 8 models and an end-to-end check that a cache hit on `deepseek/deepseek-r1` is billed at the cache-read rate via `generic_cost_per_token`.
This commit is contained in:
parent
2cd62cfb83
commit
6195f5344d
3 changed files with 126 additions and 0 deletions
|
|
@ -15333,6 +15333,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-coder": {
|
||||
"cache_read_input_token_cost": 1.4e-08,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -15347,6 +15348,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5.5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -15402,6 +15404,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v3.2": {
|
||||
"cache_read_input_token_cost": 2.8e-08,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"input_cost_per_token_cache_hit": 2.8e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -30615,6 +30618,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-chat-v3.1": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_cache_hit": 2e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30630,6 +30634,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v3.2": {
|
||||
"cache_read_input_token_cost": 2.8e-08,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"input_cost_per_token_cache_hit": 2.8e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30645,6 +30650,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v3.2-exp": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_cache_hit": 2e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30660,6 +30666,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5.5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30675,6 +30682,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-r1-0528": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
|
|||
|
|
@ -15333,6 +15333,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-coder": {
|
||||
"cache_read_input_token_cost": 1.4e-08,
|
||||
"input_cost_per_token": 1.4e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -15347,6 +15348,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5.5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -15402,6 +15404,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"deepseek/deepseek-v3.2": {
|
||||
"cache_read_input_token_cost": 2.8e-08,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"input_cost_per_token_cache_hit": 2.8e-08,
|
||||
"litellm_provider": "deepseek",
|
||||
|
|
@ -30690,6 +30693,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-chat-v3.1": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_cache_hit": 2e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30705,6 +30709,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v3.2": {
|
||||
"cache_read_input_token_cost": 2.8e-08,
|
||||
"input_cost_per_token": 2.8e-07,
|
||||
"input_cost_per_token_cache_hit": 2.8e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30720,6 +30725,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-v3.2-exp": {
|
||||
"cache_read_input_token_cost": 2e-08,
|
||||
"input_cost_per_token": 2e-07,
|
||||
"input_cost_per_token_cache_hit": 2e-08,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30735,6 +30741,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-r1": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5.5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
@ -30750,6 +30757,7 @@
|
|||
"supports_tool_choice": true
|
||||
},
|
||||
"openrouter/deepseek/deepseek-r1-0528": {
|
||||
"cache_read_input_token_cost": 1.4e-07,
|
||||
"input_cost_per_token": 5e-07,
|
||||
"input_cost_per_token_cache_hit": 1.4e-07,
|
||||
"litellm_provider": "openrouter",
|
||||
|
|
|
|||
110
tests/test_litellm/test_deepseek_cache_read_pricing.py
Normal file
110
tests/test_litellm/test_deepseek_cache_read_pricing.py
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
"""
|
||||
Regression test for DeepSeek and OpenRouter/DeepSeek models that historically
|
||||
only declared ``input_cost_per_token_cache_hit`` in the pricing JSON.
|
||||
|
||||
The cost calculator (``litellm.litellm_core_utils.llm_cost_calc.utils``) only
|
||||
reads ``cache_read_input_token_cost`` for prompt-cache-hit pricing. When a
|
||||
model defines ``input_cost_per_token_cache_hit`` without the canonical key,
|
||||
cache-hit tokens are billed at $0 because the cache-read cost resolves to
|
||||
``None`` -> ``0.0``.
|
||||
|
||||
This test ensures the two keys are kept in sync for the affected models and
|
||||
that a representative model actually bills cache-hit tokens at the cache rate.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.abspath("../.."))
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
# Models that previously only carried ``input_cost_per_token_cache_hit``.
|
||||
# Each must also carry ``cache_read_input_token_cost`` with an identical value.
|
||||
DEEPSEEK_CACHE_HIT_MODELS = [
|
||||
"deepseek/deepseek-coder",
|
||||
"deepseek/deepseek-r1",
|
||||
"deepseek/deepseek-v3.2",
|
||||
"openrouter/deepseek/deepseek-chat-v3.1",
|
||||
"openrouter/deepseek/deepseek-v3.2",
|
||||
"openrouter/deepseek/deepseek-v3.2-exp",
|
||||
"openrouter/deepseek/deepseek-r1",
|
||||
"openrouter/deepseek/deepseek-r1-0528",
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _use_local_model_cost_map(monkeypatch):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
litellm.get_model_info.cache_clear()
|
||||
yield
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", DEEPSEEK_CACHE_HIT_MODELS)
|
||||
def test_cache_read_input_token_cost_present(model):
|
||||
"""``cache_read_input_token_cost`` must be present and equal to the legacy
|
||||
``input_cost_per_token_cache_hit`` field."""
|
||||
info = litellm.model_cost[model]
|
||||
legacy = info.get("input_cost_per_token_cache_hit")
|
||||
canonical = info.get("cache_read_input_token_cost")
|
||||
assert legacy is not None, f"{model} is missing input_cost_per_token_cache_hit"
|
||||
assert canonical is not None, (
|
||||
f"{model} is missing cache_read_input_token_cost; cache-hit tokens would "
|
||||
f"be billed at $0 because the cost calculator only reads the canonical key."
|
||||
)
|
||||
assert canonical == legacy, (
|
||||
f"{model} cache_read_input_token_cost ({canonical}) must match input_cost_per_token_cache_hit ({legacy})"
|
||||
)
|
||||
|
||||
|
||||
def test_deepseek_r1_cache_hit_billed_at_cache_rate():
|
||||
"""End-to-end check that a cache hit on deepseek/deepseek-r1 is billed at
|
||||
the cache-read rate instead of the regular input rate."""
|
||||
model = "deepseek/deepseek-r1"
|
||||
info = litellm.model_cost[model]
|
||||
|
||||
prompt_tokens = 1000
|
||||
cached_tokens = 800
|
||||
text_tokens = prompt_tokens - cached_tokens
|
||||
completion_tokens = 100
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=prompt_tokens,
|
||||
completion_tokens=completion_tokens,
|
||||
total_tokens=prompt_tokens + completion_tokens,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
cached_tokens=cached_tokens,
|
||||
text_tokens=text_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
input_cost, output_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider="deepseek",
|
||||
)
|
||||
|
||||
expected_input_cost = (
|
||||
info["input_cost_per_token"] * text_tokens + info["cache_read_input_token_cost"] * cached_tokens
|
||||
)
|
||||
expected_output_cost = info["output_cost_per_token"] * completion_tokens
|
||||
|
||||
assert abs(input_cost - expected_input_cost) < 1e-12, (
|
||||
f"input cost mismatch: got {input_cost}, expected {expected_input_cost}"
|
||||
)
|
||||
assert abs(output_cost - expected_output_cost) < 1e-12, (
|
||||
f"output cost mismatch: got {output_cost}, expected {expected_output_cost}"
|
||||
)
|
||||
|
||||
# Sanity check: regression scenario (cache_read_input_token_cost missing)
|
||||
# would have billed cached tokens at the full input rate.
|
||||
naive_full_input_cost = info["input_cost_per_token"] * prompt_tokens
|
||||
assert input_cost < naive_full_input_cost, (
|
||||
"cache-hit tokens were not discounted; cache_read_input_token_cost is likely missing from this model entry."
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue