mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(anthropic): scale cache costs by fast mode and trust served speed
This commit is contained in:
parent
95285c3433
commit
a0d1fef89d
9 changed files with 123 additions and 86 deletions
|
|
@ -712,11 +712,14 @@ class ModelResponseIterator:
|
|||
|
||||
def _handle_usage(self, anthropic_usage_chunk: dict | UsageDelta) -> Usage:
|
||||
reasoning_content: Final = "".join(self.reasoning_content_chunks) if self.reasoning_content_chunks else None
|
||||
return AnthropicConfig().calculate_usage(
|
||||
usage: Final = AnthropicConfig().calculate_usage(
|
||||
usage_object=cast(dict, anthropic_usage_chunk),
|
||||
reasoning_content=reasoning_content,
|
||||
speed=self.speed,
|
||||
)
|
||||
if usage.speed is not None:
|
||||
self.speed = usage.speed
|
||||
return usage
|
||||
|
||||
def _content_block_delta_helper(
|
||||
self, chunk: dict
|
||||
|
|
|
|||
|
|
@ -2279,6 +2279,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
str | None,
|
||||
_usage.get("service_tier"),
|
||||
)
|
||||
raw_speed: Final = _usage.get("speed")
|
||||
resolved_speed: Final = raw_speed if isinstance(raw_speed, str) else speed
|
||||
|
||||
iterations: Final[list[Any] | None] = _usage.get("iterations")
|
||||
if iterations:
|
||||
|
|
@ -2353,7 +2355,7 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
else None
|
||||
),
|
||||
inference_geo=inference_geo,
|
||||
speed=speed,
|
||||
speed=resolved_speed,
|
||||
service_tier=service_tier,
|
||||
)
|
||||
return usage
|
||||
|
|
|
|||
|
|
@ -8,12 +8,9 @@ from typing import TYPE_CHECKING, Final, Optional
|
|||
from pydantic import BaseModel, ValidationError
|
||||
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import (
|
||||
_get_token_base_cost,
|
||||
_get_web_search_requests,
|
||||
calculate_cache_writing_cost,
|
||||
generic_cost_per_token,
|
||||
get_provider_specific_geo_multiplier,
|
||||
parse_prompt_tokens_details,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -21,43 +18,6 @@ if TYPE_CHECKING:
|
|||
import litellm
|
||||
|
||||
|
||||
def _compute_cache_only_cost(model_info: "ModelInfo", usage: "Usage", service_tier: str | None = None) -> float:
|
||||
"""
|
||||
Return only the cache-related portion of the prompt cost (cache read + cache write).
|
||||
|
||||
These costs must NOT be scaled by the ``fast`` speed multiplier because the old
|
||||
explicit ``fast/`` model entries carried unchanged cache rates while
|
||||
multiplying only the regular input/output token costs. Regional pricing, by
|
||||
contrast, uplifts every token type, so the geo multiplier does scale them.
|
||||
"""
|
||||
if usage.prompt_tokens_details is None:
|
||||
return 0.0
|
||||
|
||||
prompt_tokens_details: Final = parse_prompt_tokens_details(usage)
|
||||
(
|
||||
_,
|
||||
_,
|
||||
cache_creation_cost,
|
||||
cache_creation_cost_above_1hr,
|
||||
cache_read_cost,
|
||||
) = _get_token_base_cost(model_info=model_info, usage=usage, service_tier=service_tier)
|
||||
|
||||
cache_cost = float(prompt_tokens_details["cache_hit_tokens"]) * cache_read_cost
|
||||
|
||||
if (
|
||||
prompt_tokens_details["cache_creation_tokens"]
|
||||
or prompt_tokens_details["cache_creation_token_details"] is not None
|
||||
):
|
||||
cache_cost += calculate_cache_writing_cost(
|
||||
cache_creation_tokens=prompt_tokens_details["cache_creation_tokens"],
|
||||
cache_creation_token_details=prompt_tokens_details["cache_creation_token_details"],
|
||||
cache_creation_cost_above_1hr=cache_creation_cost_above_1hr,
|
||||
cache_creation_cost=cache_creation_cost,
|
||||
)
|
||||
|
||||
return cache_cost
|
||||
|
||||
|
||||
def cost_per_token(model: str, usage: "Usage", service_tier: str | None = None) -> tuple[float, float]:
|
||||
"""
|
||||
Calculates the cost per token for a given model, prompt tokens, and completion tokens.
|
||||
|
|
@ -89,8 +49,7 @@ def cost_per_token(model: str, usage: "Usage", service_tier: str | None = None)
|
|||
)
|
||||
|
||||
if speed_multiplier != 1.0:
|
||||
cache_cost: Final = _compute_cache_only_cost(model_info=model_info, usage=usage, service_tier=service_tier)
|
||||
prompt_cost = (prompt_cost - cache_cost) * speed_multiplier + cache_cost
|
||||
prompt_cost *= speed_multiplier
|
||||
completion_cost *= speed_multiplier
|
||||
|
||||
if geo_multiplier != 1.0:
|
||||
|
|
|
|||
|
|
@ -12724,8 +12724,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
|
|
@ -12762,8 +12761,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_output_config": true,
|
||||
|
|
@ -12802,8 +12800,7 @@
|
|||
"supports_xhigh_reasoning_effort": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_speed": true,
|
||||
|
|
@ -12841,8 +12838,7 @@
|
|||
"supports_xhigh_reasoning_effort": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_speed": true,
|
||||
|
|
|
|||
|
|
@ -117,8 +117,10 @@ class AnthropicPassthroughLoggingHandler:
|
|||
@staticmethod
|
||||
def _cost_relevant_speed(request_body: Mapping[str, object] | None) -> str | None:
|
||||
"""
|
||||
Anthropic's ``speed=fast`` multiplies non-cache token cost, and only the request
|
||||
carries it, so it has to reach the usage-building paths for spend to be right.
|
||||
Anthropic's ``speed=fast`` multiplies token cost. The response usage carries the
|
||||
served ``speed`` when the request asked for one, and ``calculate_usage`` prefers
|
||||
that served value; this request-side value is the fallback when the response
|
||||
omits it, so it still has to reach the usage-building paths.
|
||||
"""
|
||||
speed: Final = (request_body or {}).get("speed")
|
||||
return speed if isinstance(speed, str) else None
|
||||
|
|
@ -702,6 +704,7 @@ class AnthropicPassthroughLoggingHandler:
|
|||
web_search_requests: int | None = None
|
||||
tool_search_requests: int | None = None
|
||||
inference_geo: str | None = None
|
||||
speed_from_stream: str | None = None
|
||||
stop_reason: str | None = None
|
||||
found_usage = False
|
||||
resolved_model = model
|
||||
|
|
@ -725,6 +728,8 @@ class AnthropicPassthroughLoggingHandler:
|
|||
cache_creation_1h = _cc.get("ephemeral_1h_input_tokens")
|
||||
if usage.get("inference_geo") is not None:
|
||||
inference_geo = usage.get("inference_geo")
|
||||
if isinstance(usage.get("speed"), str):
|
||||
speed_from_stream = usage.get("speed")
|
||||
if usage.get("output_tokens") is not None:
|
||||
output_tokens = usage.get("output_tokens")
|
||||
found_usage = True
|
||||
|
|
@ -745,6 +750,8 @@ class AnthropicPassthroughLoggingHandler:
|
|||
cache_read = usage.get("cache_read_input_tokens")
|
||||
if usage.get("inference_geo") is not None:
|
||||
inference_geo = usage.get("inference_geo")
|
||||
if isinstance(usage.get("speed"), str):
|
||||
speed_from_stream = usage.get("speed")
|
||||
found_usage = True
|
||||
if not found_usage:
|
||||
return None
|
||||
|
|
@ -776,6 +783,8 @@ class AnthropicPassthroughLoggingHandler:
|
|||
usage_object["server_tool_use"] = _server_tool_use
|
||||
if inference_geo is not None:
|
||||
usage_object["inference_geo"] = inference_geo
|
||||
if speed_from_stream is not None:
|
||||
usage_object["speed"] = speed_from_stream
|
||||
usage_obj: Final = AnthropicConfig().calculate_usage(
|
||||
usage_object=usage_object, reasoning_content=None, speed=speed
|
||||
)
|
||||
|
|
|
|||
|
|
@ -12724,8 +12724,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
|
|
@ -12762,8 +12761,7 @@
|
|||
"supports_tool_choice": true,
|
||||
"supports_vision": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_output_config": true,
|
||||
|
|
@ -12802,8 +12800,7 @@
|
|||
"supports_xhigh_reasoning_effort": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_speed": true,
|
||||
|
|
@ -12841,8 +12838,7 @@
|
|||
"supports_xhigh_reasoning_effort": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"provider_specific_entry": {
|
||||
"us": 1.1,
|
||||
"fast": 6.0
|
||||
"us": 1.1
|
||||
},
|
||||
"supports_output_config": true,
|
||||
"supports_speed": true,
|
||||
|
|
|
|||
|
|
@ -100,6 +100,48 @@ def test_calculate_usage():
|
|||
assert usage._cache_read_input_tokens == 0
|
||||
|
||||
|
||||
def test_calculate_usage_prefers_served_speed_from_response_usage():
|
||||
"""
|
||||
Anthropic reports the speed a request was actually served at in the response
|
||||
usage (a fast request on a model without fast mode comes back
|
||||
``"speed": "standard"``), so the served value must beat the requested one or
|
||||
spend gets multiplied for fast service that never happened.
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
served_standard = config.calculate_usage(
|
||||
usage_object={"input_tokens": 12, "output_tokens": 1, "speed": "standard"},
|
||||
reasoning_content=None,
|
||||
speed="fast",
|
||||
)
|
||||
assert served_standard.speed == "standard"
|
||||
|
||||
no_response_speed = config.calculate_usage(
|
||||
usage_object={"input_tokens": 12, "output_tokens": 1},
|
||||
reasoning_content=None,
|
||||
speed="fast",
|
||||
)
|
||||
assert no_response_speed.speed == "fast"
|
||||
|
||||
|
||||
def test_streaming_iterator_persists_served_speed_across_usage_chunks():
|
||||
"""
|
||||
Only ``message_start`` usage carries the served speed; the final
|
||||
``message_delta`` usage does not. The iterator must remember the served
|
||||
value so the last usage chunk, which wins in the stream chunk builder, does
|
||||
not fall back to the requested speed.
|
||||
"""
|
||||
from litellm.llms.anthropic.chat.handler import ModelResponseIterator
|
||||
|
||||
iterator = ModelResponseIterator(None, sync_stream=True, speed="fast")
|
||||
|
||||
start_usage = iterator._handle_usage({"input_tokens": 12, "output_tokens": 1, "speed": "standard"})
|
||||
delta_usage = iterator._handle_usage({"output_tokens": 5})
|
||||
|
||||
assert start_usage.speed == "standard"
|
||||
assert delta_usage.speed == "standard"
|
||||
|
||||
|
||||
def test_calculate_usage_aggregates_cache_creation_split_across_iterations():
|
||||
"""
|
||||
In the iterations path each iteration can carry the 5m/1h cache_creation
|
||||
|
|
|
|||
|
|
@ -2317,10 +2317,10 @@ class TestAnthropicResponseCostRecordedOnModelCallDetails:
|
|||
|
||||
|
||||
class TestAnthropicPassthroughFastMode:
|
||||
"""Anthropic charges a provider-specific multiplier for ``speed=fast``, and the
|
||||
multiplier is applied off ``usage.speed``. The pass-through handler only sees the
|
||||
speed in the request body, so it has to thread it into every usage-building path or
|
||||
fast-mode pass-through spend is under-reported."""
|
||||
"""Anthropic charges a provider-specific multiplier for ``speed=fast``, applied off
|
||||
``usage.speed`` and covering every token type, cache included. The response usage
|
||||
carries the served speed when the request asked for one; the request body's value is
|
||||
the fallback, so the handler still threads it into every usage-building path."""
|
||||
|
||||
MODEL = "claude-opus-4-8"
|
||||
STREAM_CHUNKS = [
|
||||
|
|
@ -2358,11 +2358,7 @@ class TestAnthropicPassthroughFastMode:
|
|||
return litellm.completion_cost(completion_response=response, model=f"anthropic/{self.MODEL}")
|
||||
|
||||
def _expected_fast_cost(self, standard_cost: float) -> float:
|
||||
import litellm
|
||||
|
||||
model_info = litellm.get_model_info(model=self.MODEL, custom_llm_provider="anthropic")
|
||||
cache_read_cost = 200 * (model_info.get("cache_read_input_token_cost") or 0.0)
|
||||
return (standard_cost - cache_read_cost) * 2.0 + cache_read_cost
|
||||
return standard_cost * 2.0
|
||||
|
||||
def test_non_streaming_applies_fast_multiplier(self):
|
||||
import httpx
|
||||
|
|
@ -2427,3 +2423,21 @@ class TestAnthropicPassthroughFastMode:
|
|||
|
||||
assert fast.usage.speed == "fast"
|
||||
assert self._cost(fast) == pytest.approx(self._expected_fast_cost(self._cost(standard)))
|
||||
|
||||
def test_usage_only_fallback_prefers_served_speed_from_stream(self):
|
||||
served_standard_chunks = [
|
||||
chunk.replace('"usage": {"input_tokens": 1000', '"usage": {"speed": "standard", "input_tokens": 1000')
|
||||
for chunk in self.STREAM_CHUNKS
|
||||
]
|
||||
served_standard = AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks(
|
||||
all_chunks=served_standard_chunks,
|
||||
model=self.MODEL,
|
||||
speed="fast",
|
||||
)
|
||||
standard = AnthropicPassthroughLoggingHandler._build_usage_only_response_from_chunks(
|
||||
all_chunks=self.STREAM_CHUNKS,
|
||||
model=self.MODEL,
|
||||
)
|
||||
|
||||
assert served_standard.usage.speed == "standard"
|
||||
assert self._cost(served_standard) == pytest.approx(self._cost(standard))
|
||||
|
|
|
|||
|
|
@ -2692,12 +2692,10 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier(_l
|
|||
"""
|
||||
Regression for the cache/tier interaction in the Anthropic geo/speed path.
|
||||
|
||||
When a request is served at "priority" and also carries a geo/speed
|
||||
multiplier (here ``speed="fast"``), the cache portion is held out of the
|
||||
multiplier so it is not scaled. That held-out cache cost must use the
|
||||
served tier's cache rate; pricing it at the standard rate while the cache
|
||||
embedded in ``prompt_cost`` is priced at the priority rate leaves a
|
||||
``(cache_priority - cache_standard)(multiplier - 1)`` billing error.
|
||||
When a request is served at "priority" and also carries the ``fast`` speed
|
||||
multiplier, the cache portion must be priced at the served tier's cache
|
||||
rate and, per Anthropic's fast-mode pricing, scaled by the multiplier like
|
||||
every other token type.
|
||||
"""
|
||||
from litellm.llms.anthropic.cost_calculation import (
|
||||
cost_per_token as anthropic_cost_per_token,
|
||||
|
|
@ -2734,10 +2732,7 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier(_l
|
|||
model=model, usage=usage, service_tier="priority"
|
||||
)
|
||||
|
||||
# non-cache input priced at the priority rate and scaled by the fast
|
||||
# multiplier; the 200 cache-hit tokens priced at the priority cache rate
|
||||
# and held out of the multiplier
|
||||
expected_prompt = (1000 - 200) * 6e-6 * 2 + 200 * 0.6e-6
|
||||
expected_prompt = ((1000 - 200) * 6e-6 + 200 * 0.6e-6) * 2
|
||||
expected_completion = 500 * 30e-6 * 2
|
||||
assert prompt_cost == pytest.approx(expected_prompt)
|
||||
assert completion_cost == pytest.approx(expected_completion)
|
||||
|
|
@ -2805,10 +2800,9 @@ def test_anthropic_geo_multiplier_applies_to_cache_tokens(_local_model_cost_map,
|
|||
|
||||
def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monkeypatch):
|
||||
"""
|
||||
The ``fast`` speed multiplier stays cache-exclusive (the old explicit
|
||||
``fast/`` entries kept base cache rates) while the geo multiplier scales the
|
||||
whole cost, so a fast + regional row prices as
|
||||
``((non_cache * fast) + cache) * geo``.
|
||||
Anthropic's fast-mode pricing doubles every token type, cache reads and
|
||||
writes included, and the regional uplift stacks on top, so a fast +
|
||||
regional row prices as ``(non_cache + cache) * fast * geo``.
|
||||
"""
|
||||
from litellm.llms.anthropic.cost_calculation import (
|
||||
cost_per_token as anthropic_cost_per_token,
|
||||
|
|
@ -2836,10 +2830,32 @@ def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monke
|
|||
|
||||
cache_cost = 2_000 * 0.5e-6 + 6_000 * 6.25e-6
|
||||
non_cache_cost = 2_000 * 5e-6
|
||||
assert prompt_cost == pytest.approx((non_cache_cost * 2.0 + cache_cost) * 1.1)
|
||||
assert prompt_cost == pytest.approx((non_cache_cost + cache_cost) * 2.0 * 1.1)
|
||||
assert completion_cost == pytest.approx(500 * 25e-6 * 2.0 * 1.1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_fast",
|
||||
[
|
||||
("claude-opus-5", 2.0),
|
||||
("claude-opus-4-8", 2.0),
|
||||
("claude-opus-4-6", None),
|
||||
("claude-opus-4-6-20260205", None),
|
||||
("claude-opus-4-7", None),
|
||||
("claude-opus-4-7-20260416", None),
|
||||
],
|
||||
)
|
||||
def test_anthropic_fast_multiplier_only_on_models_with_fast_mode(_local_model_cost_map, model, expected_fast):
|
||||
"""
|
||||
Anthropic serves fast mode on Opus 5 and Opus 4.8 only, at 2x. Opus 4.6 and
|
||||
4.7 accept the ``speed`` request param but are always served standard, so a
|
||||
``fast`` multiplier on their map entries overbills every request that asked
|
||||
for fast and was served standard.
|
||||
"""
|
||||
entry = litellm.model_cost[model]
|
||||
assert entry["provider_specific_entry"].get("fast") == expected_fast
|
||||
|
||||
|
||||
def test_gemini_cache_tokens_details_no_negative_values():
|
||||
"""
|
||||
Test for Issue #18750: Negative text_tokens with Gemini caching
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue