fix(cost): price anthropic messages cache read/write tokens instead of full input rate

Anthropic-shaped usage was mapped through the Responses API usage converter, which ignores top-level cache_read_input_tokens/cache_creation_input_tokens, so cache hits on /v1/messages were billed entirely at the uncached input rate
This commit is contained in:
Devin AI 2026-07-28 16:37:49 +00:00
parent ed366aafbe
commit 37744ca944
3 changed files with 46 additions and 1 deletions

View file

@ -886,6 +886,8 @@ def _get_usage_object(
return None
if isinstance(usage_obj, Usage):
return usage_obj
elif isinstance(usage_obj, dict) and litellm.AnthropicConfig.is_anthropic_usage_object(usage_obj):
return litellm.AnthropicConfig().calculate_usage(usage_object=usage_obj, reasoning_content=None)
elif (
usage_obj is not None
and (isinstance(usage_obj, dict) or isinstance(usage_obj, ResponseAPIUsage))
@ -1251,7 +1253,13 @@ def completion_cost(
else:
_usage = usage_obj
if ResponseAPILoggingUtils._is_response_api_usage(_usage):
if litellm.AnthropicConfig.is_anthropic_usage_object(_usage):
_usage = (
litellm.AnthropicConfig()
.calculate_usage(usage_object=_usage, reasoning_content=None)
.model_dump()
)
elif ResponseAPILoggingUtils._is_response_api_usage(_usage):
_usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
_usage
).model_dump()

View file

@ -2122,6 +2122,16 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
compaction_blocks,
)
@staticmethod
def is_anthropic_usage_object(usage_object: dict) -> bool:
"""Anthropic reports prompt cache tokens as top-level ``cache_read_input_tokens`` /
``cache_creation_input_tokens``; no other API surface uses those keys, and the
Responses API mapping would silently drop them.
"""
if "prompt_tokens" in usage_object or "input_tokens" not in usage_object:
return False
return any(key in usage_object for key in ("cache_read_input_tokens", "cache_creation_input_tokens"))
def calculate_usage(
self,
usage_object: dict,

View file

@ -3509,3 +3509,30 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
assert combined_pair.prompt_tokens_details is not None
assert combined_pair.prompt_tokens_details.cache_write_tokens == 100
assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100
def test_completion_cost_prices_anthropic_shaped_cache_read_tokens():
"""Regression: an Anthropic /v1/messages response reports cache reads as top-level
cache_read_input_tokens with input_tokens excluding them. Reading that usage as
Responses API usage dropped the cache tokens and billed the whole prompt at the
uncached input rate, overstating spend on cache hits."""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
response = {
"id": "msg_1",
"type": "message",
"role": "assistant",
"model": "gpt-5.6-sol",
"stop_reason": "end_turn",
"content": [{"type": "text", "text": "1"}],
"usage": {"input_tokens": 3, "output_tokens": 5, "cache_read_input_tokens": 4014},
}
cost = litellm.completion_cost(
completion_response=response,
model="gpt-5.6-sol",
custom_llm_provider="openai",
)
assert cost == pytest.approx(3 * 5e-6 + 4014 * 5e-7 + 5 * 3e-5, rel=1e-9)