fix(anthropic): scale cache costs by fast mode and trust served speed

This commit is contained in:
mateo-berri 2026-08-26 11:24:12 -07:00
parent 95285c3433
commit a0d1fef89d
9 changed files with 123 additions and 86 deletions

View file

@ -712,11 +712,14 @@ class ModelResponseIterator:
def _handle_usage(self, anthropic_usage_chunk: dict | UsageDelta) -> Usage:
reasoning_content: Final = "".join(self.reasoning_content_chunks) if self.reasoning_content_chunks else None
return AnthropicConfig().calculate_usage(
usage: Final = AnthropicConfig().calculate_usage(
usage_object=cast(dict, anthropic_usage_chunk),
reasoning_content=reasoning_content,
speed=self.speed,
)
if usage.speed is not None:
self.speed = usage.speed
return usage
def _content_block_delta_helper(
self, chunk: dict

View file

@ -2279,6 +2279,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
str | None,
_usage.get("service_tier"),
)
raw_speed: Final = _usage.get("speed")
resolved_speed: Final = raw_speed if isinstance(raw_speed, str) else speed
iterations: Final[list[Any] | None] = _usage.get("iterations")
if iterations:
@ -2353,7 +2355,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
else None
),
inference_geo=inference_geo,
speed=speed,
speed=resolved_speed,
service_tier=service_tier,
)
return usage

View file

@ -8,12 +8,9 @@ from typing import TYPE_CHECKING, Final, Optional
from pydantic import BaseModel, ValidationError
from litellm.litellm_core_utils.llm_cost_calc.utils import (
_get_token_base_cost,
_get_web_search_requests,
calculate_cache_writing_cost,
generic_cost_per_token,
get_provider_specific_geo_multiplier,
parse_prompt_tokens_details,
)
if TYPE_CHECKING:
@ -21,43 +18,6 @@ if TYPE_CHECKING:
import litellm
def _compute_cache_only_cost(model_info: "ModelInfo", usage: "Usage", service_tier: str | None = None) -> float:
"""
Return only the cache-related portion of the prompt cost (cache read + cache write).
These costs must NOT be scaled by the ``fast`` speed multiplier because the old
explicit ``fast/`` model entries carried unchanged cache rates while
multiplying only the regular input/output token costs. Regional pricing, by
contrast, uplifts every token type, so the geo multiplier does scale them.
"""
if usage.prompt_tokens_details is None:
return 0.0
prompt_tokens_details: Final = parse_prompt_tokens_details(usage)
(
_,
_,
cache_creation_cost,
cache_creation_cost_above_1hr,
cache_read_cost,
) = _get_token_base_cost(model_info=model_info, usage=usage, service_tier=service_tier)
cache_cost = float(prompt_tokens_details["cache_hit_tokens"]) * cache_read_cost
if (
prompt_tokens_details["cache_creation_tokens"]
or prompt_tokens_details["cache_creation_token_details"] is not None
):
cache_cost += calculate_cache_writing_cost(
cache_creation_tokens=prompt_tokens_details["cache_creation_tokens"],
cache_creation_token_details=prompt_tokens_details["cache_creation_token_details"],
cache_creation_cost_above_1hr=cache_creation_cost_above_1hr,
cache_creation_cost=cache_creation_cost,
)
return cache_cost
def cost_per_token(model: str, usage: "Usage", service_tier: str | None = None) -> tuple[float, float]:
"""
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
@ -89,8 +49,7 @@ def cost_per_token(model: str, usage: "Usage", service_tier: str | None = None)
)
if speed_multiplier != 1.0:
cache_cost: Final = _compute_cache_only_cost(model_info=model_info, usage=usage, service_tier=service_tier)
prompt_cost = (prompt_cost - cache_cost) * speed_multiplier + cache_cost
prompt_cost *= speed_multiplier
completion_cost *= speed_multiplier
if geo_multiplier != 1.0:

View file

@ -12724,8 +12724,7 @@
"supports_tool_choice": true,
"supports_vision": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_output_config": true,
"supports_max_reasoning_effort": true,
@ -12762,8 +12761,7 @@
"supports_tool_choice": true,
"supports_vision": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_max_reasoning_effort": true,
"supports_output_config": true,
@ -12802,8 +12800,7 @@
"supports_xhigh_reasoning_effort": true,
"supports_max_reasoning_effort": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_output_config": true,
"supports_speed": true,
@ -12841,8 +12838,7 @@
"supports_xhigh_reasoning_effort": true,
"supports_max_reasoning_effort": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_output_config": true,
"supports_speed": true,

View file

@ -117,8 +117,10 @@ class AnthropicPassthroughLoggingHandler:
@staticmethod
def _cost_relevant_speed(request_body: Mapping[str, object] | None) -> str | None:
"""
Anthropic's ``speed=fast`` multiplies non-cache token cost, and only the request
carries it, so it has to reach the usage-building paths for spend to be right.
Anthropic's ``speed=fast`` multiplies token cost. The response usage carries the
served ``speed`` when the request asked for one, and ``calculate_usage`` prefers
that served value; this request-side value is the fallback when the response
omits it, so it still has to reach the usage-building paths.
"""
speed: Final = (request_body or {}).get("speed")
return speed if isinstance(speed, str) else None
@ -702,6 +704,7 @@ class AnthropicPassthroughLoggingHandler:
web_search_requests: int | None = None
tool_search_requests: int | None = None
inference_geo: str | None = None
speed_from_stream: str | None = None
stop_reason: str | None = None
found_usage = False
resolved_model = model
@ -725,6 +728,8 @@ class AnthropicPassthroughLoggingHandler:
cache_creation_1h = _cc.get("ephemeral_1h_input_tokens")
if usage.get("inference_geo") is not None:
inference_geo = usage.get("inference_geo")
if isinstance(usage.get("speed"), str):
speed_from_stream = usage.get("speed")
if usage.get("output_tokens") is not None:
output_tokens = usage.get("output_tokens")
found_usage = True
@ -745,6 +750,8 @@ class AnthropicPassthroughLoggingHandler:
cache_read = usage.get("cache_read_input_tokens")
if usage.get("inference_geo") is not None:
inference_geo = usage.get("inference_geo")
if isinstance(usage.get("speed"), str):
speed_from_stream = usage.get("speed")
found_usage = True
if not found_usage:
return None
@ -776,6 +783,8 @@ class AnthropicPassthroughLoggingHandler:
usage_object["server_tool_use"] = _server_tool_use
if inference_geo is not None:
usage_object["inference_geo"] = inference_geo
if speed_from_stream is not None:
usage_object["speed"] = speed_from_stream
usage_obj: Final = AnthropicConfig().calculate_usage(
usage_object=usage_object, reasoning_content=None, speed=speed
)

View file

@ -12724,8 +12724,7 @@
"supports_tool_choice": true,
"supports_vision": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_output_config": true,
"supports_max_reasoning_effort": true,
@ -12762,8 +12761,7 @@
"supports_tool_choice": true,
"supports_vision": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_max_reasoning_effort": true,
"supports_output_config": true,
@ -12802,8 +12800,7 @@
"supports_xhigh_reasoning_effort": true,
"supports_max_reasoning_effort": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_output_config": true,
"supports_speed": true,
@ -12841,8 +12838,7 @@
"supports_xhigh_reasoning_effort": true,
"supports_max_reasoning_effort": true,
"provider_specific_entry": {
"us": 1.1,
"fast": 6.0
"us": 1.1
},
"supports_output_config": true,
"supports_speed": true,

View file

@ -100,6 +100,48 @@ def test_calculate_usage():
assert usage._cache_read_input_tokens == 0
def test_calculate_usage_prefers_served_speed_from_response_usage():
"""
Anthropic reports the speed a request was actually served at in the response
usage (a fast request on a model without fast mode comes back
``"speed": "standard"``), so the served value must beat the requested one or
spend gets multiplied for fast service that never happened.
"""
config = AnthropicConfig()
served_standard = config.calculate_usage(
usage_object={"input_tokens": 12, "output_tokens": 1, "speed": "standard"},
reasoning_content=None,
speed="fast",
)
assert served_standard.speed == "standard"
no_response_speed = config.calculate_usage(
usage_object={"input_tokens": 12, "output_tokens": 1},
reasoning_content=None,
speed="fast",
)
assert no_response_speed.speed == "fast"
def test_streaming_iterator_persists_served_speed_across_usage_chunks():
"""
Only ``message_start`` usage carries the served speed; the final
``message_delta`` usage does not. The iterator must remember the served
value so the last usage chunk, which wins in the stream chunk builder, does
not fall back to the requested speed.
"""
from litellm.llms.anthropic.chat.handler import ModelResponseIterator
iterator = ModelResponseIterator(None, sync_stream=True, speed="fast")
start_usage = iterator._handle_usage({"input_tokens": 12, "output_tokens": 1, "speed": "standard"})
delta_usage = iterator._handle_usage({"output_tokens": 5})
assert start_usage.speed == "standard"
assert delta_usage.speed == "standard"
def test_calculate_usage_aggregates_cache_creation_split_across_iterations():
"""
In the iterations path each iteration can carry the 5m/1h cache_creation

View file

@ -2317,10 +2317,10 @@ class TestAnthropicResponseCostRecordedOnModelCallDetails:
class TestAnthropicPassthroughFastMode:
"""Anthropic charges a provider-specific multiplier for ``speed=fast``, and the
multiplier is applied off ``usage.speed``. The pass-through handler only sees the
speed in the request body, so it has to thread it into every usage-building path or
fast-mode pass-through spend is under-reported."""
"""Anthropic charges a provider-specific multiplier for ``speed=fast``, applied off
``usage.speed`` and covering every token type, cache included. The response usage
carries the served speed when the request asked for one; the request body's value is
the fallback, so the handler still threads it into every usage-building path."""
MODEL = "claude-opus-4-8"
STREAM_CHUNKS = [
@ -2358,11 +2358,7 @@ class TestAnthropicPassthroughFastMode:
return litellm.completion_cost(completion_response=response, model=f"anthropic/{self.MODEL}")
def _expected_fast_cost(self, standard_cost: float) -> float:
import litellm
model_info = litellm.get_model_info(model=self.MODEL, custom_llm_provider="anthropic")
cache_read_cost = 200 * (model_info.get("cache_read_input_token_cost") or 0.0)
return (standard_cost - cache_read_cost) * 2.0 + cache_read_cost
return standard_cost * 2.0
def test_non_streaming_applies_fast_multiplier(self):
import httpx
@ -2427,3 +2423,21 @@ class TestAnthropicPassthroughFastMode:
assert fast.usage.speed == "fast"
assert self._cost(fast) == pytest.approx(self._expected_fast_cost(self._cost(standard)))
def test_usage_only_fallback_prefers_served_speed_from_stream(self):
served_standard_chunks = [
chunk.replace('"usage": {"input_tokens": 1000', '"usage": {"speed": "standard", "input_tokens": 1000')
for chunk in self.STREAM_CHUNKS
]
served_standard = AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks(
all_chunks=served_standard_chunks,
model=self.MODEL,
speed="fast",
)
standard = AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks(
all_chunks=self.STREAM_CHUNKS,
model=self.MODEL,
)
assert served_standard.usage.speed == "standard"
assert self._cost(served_standard) == pytest.approx(self._cost(standard))

View file

@ -2692,12 +2692,10 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier(_l
"""
Regression for the cache/tier interaction in the Anthropic geo/speed path.
When a request is served at "priority" and also carries a geo/speed
multiplier (here ``speed="fast"``), the cache portion is held out of the
multiplier so it is not scaled. That held-out cache cost must use the
served tier's cache rate; pricing it at the standard rate while the cache
embedded in ``prompt_cost`` is priced at the priority rate leaves a
``(cache_priority - cache_standard)(multiplier - 1)`` billing error.
When a request is served at "priority" and also carries the ``fast`` speed
multiplier, the cache portion must be priced at the served tier's cache
rate and, per Anthropic's fast-mode pricing, scaled by the multiplier like
every other token type.
"""
from litellm.llms.anthropic.cost_calculation import (
cost_per_token as anthropic_cost_per_token,
@ -2734,10 +2732,7 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier(_l
model=model, usage=usage, service_tier="priority"
)
# non-cache input priced at the priority rate and scaled by the fast
# multiplier; the 200 cache-hit tokens priced at the priority cache rate
# and held out of the multiplier
expected_prompt = (1000 - 200) * 6e-6 * 2 + 200 * 0.6e-6
expected_prompt = ((1000 - 200) * 6e-6 + 200 * 0.6e-6) * 2
expected_completion = 500 * 30e-6 * 2
assert prompt_cost == pytest.approx(expected_prompt)
assert completion_cost == pytest.approx(expected_completion)
@ -2805,10 +2800,9 @@ def test_anthropic_geo_multiplier_applies_to_cache_tokens(_local_model_cost_map,
def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monkeypatch):
"""
The ``fast`` speed multiplier stays cache-exclusive (the old explicit
``fast/`` entries kept base cache rates) while the geo multiplier scales the
whole cost, so a fast + regional row prices as
``((non_cache * fast) + cache) * geo``.
Anthropic's fast-mode pricing doubles every token type, cache reads and
writes included, and the regional uplift stacks on top, so a fast +
regional row prices as ``(non_cache + cache) * fast * geo``.
"""
from litellm.llms.anthropic.cost_calculation import (
cost_per_token as anthropic_cost_per_token,
@ -2836,10 +2830,32 @@ def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monke
cache_cost = 2_000 * 0.5e-6 + 6_000 * 6.25e-6
non_cache_cost = 2_000 * 5e-6
assert prompt_cost == pytest.approx((non_cache_cost * 2.0 + cache_cost) * 1.1)
assert prompt_cost == pytest.approx((non_cache_cost + cache_cost) * 2.0 * 1.1)
assert completion_cost == pytest.approx(500 * 25e-6 * 2.0 * 1.1)
@pytest.mark.parametrize(
"model,expected_fast",
[
("claude-opus-5", 2.0),
("claude-opus-4-8", 2.0),
("claude-opus-4-6", None),
("claude-opus-4-6-20260205", None),
("claude-opus-4-7", None),
("claude-opus-4-7-20260416", None),
],
)
def test_anthropic_fast_multiplier_only_on_models_with_fast_mode(_local_model_cost_map, model, expected_fast):
"""
Anthropic serves fast mode on Opus 5 and Opus 4.8 only, at 2x. Opus 4.6 and
4.7 accept the ``speed`` request param but are always served standard, so a
``fast`` multiplier on their map entries overbills every request that asked
for fast and was served standard.
"""
entry = litellm.model_cost[model]
assert entry["provider_specific_entry"].get("fast") == expected_fast
def test_gemini_cache_tokens_details_no_negative_values():
"""
Test for Issue #18750: Negative text_tokens with Gemini caching