mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(cost): price anthropic messages cache read/write tokens instead of full input rate
Anthropic-shaped usage was mapped through the Responses API usage converter, which ignores top-level cache_read_input_tokens/cache_creation_input_tokens, so cache hits on /v1/messages were billed entirely at the uncached input rate
This commit is contained in:
parent
ed366aafbe
commit
37744ca944
3 changed files with 46 additions and 1 deletions
|
|
@ -886,6 +886,8 @@ def _get_usage_object(
|
|||
return None
|
||||
if isinstance(usage_obj, Usage):
|
||||
return usage_obj
|
||||
elif isinstance(usage_obj, dict) and litellm.AnthropicConfig.is_anthropic_usage_object(usage_obj):
|
||||
return litellm.AnthropicConfig().calculate_usage(usage_object=usage_obj, reasoning_content=None)
|
||||
elif (
|
||||
usage_obj is not None
|
||||
and (isinstance(usage_obj, dict) or isinstance(usage_obj, ResponseAPIUsage))
|
||||
|
|
@ -1251,7 +1253,13 @@ def completion_cost(
|
|||
else:
|
||||
_usage = usage_obj
|
||||
|
||||
if ResponseAPILoggingUtils._is_response_api_usage(_usage):
|
||||
if litellm.AnthropicConfig.is_anthropic_usage_object(_usage):
|
||||
_usage = (
|
||||
litellm.AnthropicConfig()
|
||||
.calculate_usage(usage_object=_usage, reasoning_content=None)
|
||||
.model_dump()
|
||||
)
|
||||
elif ResponseAPILoggingUtils._is_response_api_usage(_usage):
|
||||
_usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
|
||||
_usage
|
||||
).model_dump()
|
||||
|
|
|
|||
|
|
@ -2122,6 +2122,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
compaction_blocks,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def is_anthropic_usage_object(usage_object: dict) -> bool:
|
||||
"""Anthropic reports prompt cache tokens as top-level ``cache_read_input_tokens`` /
|
||||
``cache_creation_input_tokens``; no other API surface uses those keys, and the
|
||||
Responses API mapping would silently drop them.
|
||||
"""
|
||||
if "prompt_tokens" in usage_object or "input_tokens" not in usage_object:
|
||||
return False
|
||||
return any(key in usage_object for key in ("cache_read_input_tokens", "cache_creation_input_tokens"))
|
||||
|
||||
def calculate_usage(
|
||||
self,
|
||||
usage_object: dict,
|
||||
|
|
|
|||
|
|
@ -3509,3 +3509,30 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
|
|||
assert combined_pair.prompt_tokens_details is not None
|
||||
assert combined_pair.prompt_tokens_details.cache_write_tokens == 100
|
||||
assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100
|
||||
|
||||
|
||||
def test_completion_cost_prices_anthropic_shaped_cache_read_tokens():
|
||||
"""Regression: an Anthropic /v1/messages response reports cache reads as top-level
|
||||
cache_read_input_tokens with input_tokens excluding them. Reading that usage as
|
||||
Responses API usage dropped the cache tokens and billed the whole prompt at the
|
||||
uncached input rate, overstating spend on cache hits."""
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
response = {
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "gpt-5.6-sol",
|
||||
"stop_reason": "end_turn",
|
||||
"content": [{"type": "text", "text": "1"}],
|
||||
"usage": {"input_tokens": 3, "output_tokens": 5, "cache_read_input_tokens": 4014},
|
||||
}
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model="gpt-5.6-sol",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(3 * 5e-6 + 4014 * 5e-7 + 5 * 3e-5, rel=1e-9)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue