fix(anthropic): scale cache tokens by geo and speed modifiers

Anthropic stacks its pricing modifiers: prompt-caching multipliers apply on top
of the fast-mode and data-residency rates, not on top of the standard ones. A
fast-mode cache read on Claude Opus 5 therefore bills at 0.1 x $10/MTok, and a
us-residency cache write at 1.25 x $5.50/MTok.

The Anthropic cost path did the opposite: it subtracted the cache portion,
scaled what was left, then added the cache cost back unscaled, so every
cache-heavy fast-mode or regionalized request was billed low. On a fully cached
1M-token fast-mode prompt that is about $0.50 of unreported spend per request,
which also lets key and team budgets drift below actual upstream cost.

Applies the multiplier to the whole prompt and completion cost, matching how
generic_cost_per_token already handles the OpenAI regional uplift, and drops
_compute_cache_only_cost along with the hold-out it existed to support.

Ref: https://platform.claude.com/docs/en/build-with-claude/fast-mode#pricing
This commit is contained in:
mateo-berri 2026-07-24 11:17:55 -07:00
parent e7df7953bd
commit 74ae187c6b
3 changed files with 166 additions and 62 deletions

View file

@ -8,10 +8,7 @@ from typing import TYPE_CHECKING, Optional, Tuple
from pydantic import BaseModel, ValidationError
from litellm.litellm_core_utils.llm_cost_calc.utils import (
_get_token_base_cost,
_get_web_search_requests,
_parse_prompt_tokens_details,
calculate_cache_writing_cost,
generic_cost_per_token,
)
@ -19,41 +16,40 @@ if TYPE_CHECKING:
from litellm.types.utils import ModelInfo, Usage
import litellm
_UNPRICED_INFERENCE_GEOS = frozenset({"global", "not_available"})
def _compute_cache_only_cost(model_info: "ModelInfo", usage: "Usage", service_tier: str | None = None) -> float:
def _pricing_modifier_multiplier(model: str, usage: "Usage") -> float:
"""
Return only the cache-related portion of the prompt cost (cache read + cache write).
Resolve the combined geo and speed multiplier for a request.
These costs must NOT be scaled by geo/speed multipliers because the old
explicit ``fast/`` model entries carried unchanged cache rates while
multiplying only the regular input/output token costs.
Anthropic stacks these modifiers on top of the standard rates, and prompt
caching multipliers apply on top of the modified rates rather than the
unmodified ones, so a fast-mode cache read on Claude Opus 5 bills at
0.1 x $10/MTok, not 0.1 x $5/MTok. The caller therefore scales the whole
prompt cost, cache tokens included; see
https://platform.claude.com/docs/en/build-with-claude/fast-mode#pricing
Returns 1.0 when the model is unknown or carries no modifier entry.
"""
if usage.prompt_tokens_details is None:
return 0.0
try:
model_info = litellm.get_model_info(model=model, custom_llm_provider="anthropic")
except Exception:
return 1.0
prompt_tokens_details = _parse_prompt_tokens_details(usage)
(
_,
_,
cache_creation_cost,
cache_creation_cost_above_1hr,
cache_read_cost,
) = _get_token_base_cost(model_info=model_info, usage=usage, service_tier=service_tier)
modifiers = model_info.get("provider_specific_entry") or {}
if not isinstance(modifiers, dict):
return 1.0
cache_cost = float(prompt_tokens_details["cache_hit_tokens"]) * cache_read_cost
inference_geo = getattr(usage, "inference_geo", None)
geo_multiplier = (
modifiers.get(inference_geo.lower(), 1.0)
if isinstance(inference_geo, str) and inference_geo.lower() not in _UNPRICED_INFERENCE_GEOS
else 1.0
)
speed_multiplier = modifiers.get("fast", 1.0) if getattr(usage, "speed", None) == "fast" else 1.0
if (
prompt_tokens_details["cache_creation_tokens"]
or prompt_tokens_details["cache_creation_token_details"] is not None
):
cache_cost += calculate_cache_writing_cost(
cache_creation_tokens=prompt_tokens_details["cache_creation_tokens"],
cache_creation_token_details=prompt_tokens_details["cache_creation_token_details"],
cache_creation_cost_above_1hr=cache_creation_cost_above_1hr,
cache_creation_cost=cache_creation_cost,
)
return cache_cost
return float(geo_multiplier) * float(speed_multiplier)
def cost_per_token(model: str, usage: "Usage", service_tier: str | None = None) -> Tuple[float, float]:
@ -76,29 +72,11 @@ def cost_per_token(model: str, usage: "Usage", service_tier: str | None = None)
service_tier=service_tier,
)
# Apply provider_specific_entry multipliers for geo/speed routing
try:
model_info = litellm.get_model_info(model=model, custom_llm_provider="anthropic")
provider_specific_entry: dict = model_info.get("provider_specific_entry") or {}
multiplier = _pricing_modifier_multiplier(model=model, usage=usage)
if multiplier == 1.0:
return prompt_cost, completion_cost
multiplier = 1.0
if (
hasattr(usage, "inference_geo")
and usage.inference_geo
and usage.inference_geo.lower() not in ["global", "not_available"]
):
multiplier *= provider_specific_entry.get(usage.inference_geo.lower(), 1.0)
if hasattr(usage, "speed") and usage.speed == "fast":
multiplier *= provider_specific_entry.get("fast", 1.0)
if multiplier != 1.0:
cache_cost = _compute_cache_only_cost(model_info=model_info, usage=usage, service_tier=service_tier)
prompt_cost = (prompt_cost - cache_cost) * multiplier + cache_cost
completion_cost *= multiplier
except Exception:
pass
return prompt_cost, completion_cost
return prompt_cost * multiplier, completion_cost * multiplier
class _AnthropicServerToolUseProbe(BaseModel):

View file

@ -0,0 +1,128 @@
"""
Anthropic stacks its geo and speed pricing modifiers with the prompt-caching
multipliers, so cache reads and cache writes bill off the modified input rate,
not the standard one: a fast-mode cache read on Claude Opus 5 costs
0.1 x $10/MTok, and a ``us`` data-residency cache write costs 1.25 x $5.50/MTok.
Regression for the earlier behavior, which subtracted the cache portion before
applying the multiplier and added it back unscaled, understating spend on every
cache-heavy fast-mode or regionalized request.
https://platform.claude.com/docs/en/build-with-claude/fast-mode#pricing
"""
import pytest
import litellm
from litellm.types.utils import ModelResponse, PromptTokensDetailsWrapper, Usage
INPUT_COST = 5e-06
OUTPUT_COST = 2.5e-05
CACHE_READ_COST = 5e-07
CACHE_WRITE_COST = 6.25e-06
UNCACHED_PROMPT_TOKENS = 1000
CACHE_READ_TOKENS = 2000
CACHE_WRITE_TOKENS = 500
COMPLETION_TOKENS = 100
STANDARD_COST = (
UNCACHED_PROMPT_TOKENS * INPUT_COST
+ CACHE_READ_TOKENS * CACHE_READ_COST
+ CACHE_WRITE_TOKENS * CACHE_WRITE_COST
+ COMPLETION_TOKENS * OUTPUT_COST
)
@pytest.fixture(autouse=True)
def local_model_cost_map(monkeypatch):
original_model_cost = litellm.model_cost
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
litellm.get_model_info.cache_clear()
try:
yield
finally:
litellm.model_cost = original_model_cost
litellm.get_model_info.cache_clear()
PROMPT_TOKENS = UNCACHED_PROMPT_TOKENS + CACHE_READ_TOKENS + CACHE_WRITE_TOKENS
def _cost(speed: str | None = None, inference_geo: str | None = None) -> float:
"""Mirrors the Usage that ``AnthropicConfig.calculate_usage`` builds: a
``prompt_tokens`` total that already includes the cache tokens, with the
uncached remainder in ``prompt_tokens_details.text_tokens``."""
usage = Usage(
prompt_tokens=PROMPT_TOKENS,
completion_tokens=COMPLETION_TOKENS,
total_tokens=PROMPT_TOKENS + COMPLETION_TOKENS,
cache_read_input_tokens=CACHE_READ_TOKENS,
cache_creation_input_tokens=CACHE_WRITE_TOKENS,
prompt_tokens_details=PromptTokensDetailsWrapper(
cached_tokens=CACHE_READ_TOKENS,
cache_creation_tokens=CACHE_WRITE_TOKENS,
text_tokens=UNCACHED_PROMPT_TOKENS,
),
speed=speed,
inference_geo=inference_geo,
)
response = ModelResponse(
id="test-id",
created=1234567890,
model="claude-opus-5",
object="chat.completion",
choices=[],
usage=usage,
)
return litellm.completion_cost(
completion_response=response,
model="claude-opus-5",
custom_llm_provider="anthropic",
)
def test_standard_request_is_unmodified():
assert _cost() == pytest.approx(STANDARD_COST)
def test_global_geo_is_not_treated_as_a_priced_region():
assert _cost(inference_geo="global") == pytest.approx(STANDARD_COST)
def test_fast_mode_scales_cache_read_and_cache_write():
assert _cost(speed="fast") == pytest.approx(STANDARD_COST * 2.0)
def test_us_data_residency_scales_cache_read_and_cache_write():
assert _cost(inference_geo="us") == pytest.approx(STANDARD_COST * 1.1)
def test_fast_mode_and_data_residency_stack():
assert _cost(speed="fast", inference_geo="us") == pytest.approx(STANDARD_COST * 2.2)
def test_cache_tokens_carry_the_same_multiplier_as_uncached_tokens():
"""The bug was cache-specific, so pin the cache slice on its own: the delta
between a fast and a standard request must include the cache portion."""
cache_portion = CACHE_READ_TOKENS * CACHE_READ_COST + CACHE_WRITE_TOKENS * CACHE_WRITE_COST
assert _cost(speed="fast") - _cost() == pytest.approx(STANDARD_COST)
assert _cost(speed="fast") - _cost() > cache_portion
def test_unknown_model_falls_back_to_no_multiplier():
from litellm.llms.anthropic.cost_calculation import _pricing_modifier_multiplier
usage = Usage(prompt_tokens=10, completion_tokens=10, total_tokens=20, speed="fast")
assert _pricing_modifier_multiplier(model="not-a-real-claude-model", usage=usage) == 1.0
def test_model_without_modifier_entry_is_unscaled():
from litellm.llms.anthropic.cost_calculation import _pricing_modifier_multiplier
usage = Usage(prompt_tokens=10, completion_tokens=10, total_tokens=20, speed="fast")
assert _pricing_modifier_multiplier(model="claude-haiku-4-5", usage=usage) == 1.0

View file

@ -2672,11 +2672,10 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier():
Regression for the cache/tier interaction in the Anthropic geo/speed path.
When a request is served at "priority" and also carries a geo/speed
multiplier (here ``speed="fast"``), the cache portion is held out of the
multiplier so it is not scaled. That held-out cache cost must use the
served tier's cache rate; pricing it at the standard rate while the cache
embedded in ``prompt_cost`` is priced at the priority rate leaves a
``(cache_priority - cache_standard)(multiplier - 1)`` billing error.
multiplier (here ``speed="fast"``), the cache tokens are priced at the
served tier's cache rate and then scaled by the multiplier alongside every
other token, because Anthropic stacks prompt-caching multipliers on top of
the modified rate rather than the standard one.
"""
from litellm.llms.anthropic.cost_calculation import (
cost_per_token as anthropic_cost_per_token,
@ -2715,10 +2714,9 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier():
model=model, usage=usage, service_tier="priority"
)
# non-cache input priced at the priority rate and scaled by the fast
# multiplier; the 200 cache-hit tokens priced at the priority cache rate
# and held out of the multiplier
expected_prompt = (1000 - 200) * 6e-6 * 2 + 200 * 0.6e-6
# non-cache input at the priority rate and the 200 cache-hit tokens at the
# priority cache rate, both scaled by the fast multiplier
expected_prompt = ((1000 - 200) * 6e-6 + 200 * 0.6e-6) * 2
expected_completion = 500 * 30e-6 * 2
assert prompt_cost == pytest.approx(expected_prompt)
assert completion_cost == pytest.approx(expected_completion)