mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(cost_tracking): map cache_write_tokens on Responses API usage path
The Responses API (/v1/responses) usage transform rebuilt prompt token details and dropped OpenAI's input_tokens_details.cache_write_tokens, so gpt-5.6 cache-creation tokens were never logged or billed via that route. Map it in the transform, and make PromptTokensDetailsWrapper keep cache_write_tokens and cache_creation_tokens in sync on assignment. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
e6ec153243
commit
2f502a1bfc
4 changed files with 49 additions and 5 deletions
|
|
@ -1049,6 +1049,7 @@ class ResponseAPILoggingUtils:
|
|||
audio_tokens=getattr(response_api_usage.input_tokens_details, "audio_tokens", None),
|
||||
text_tokens=getattr(response_api_usage.input_tokens_details, "text_tokens", None),
|
||||
image_tokens=getattr(response_api_usage.input_tokens_details, "image_tokens", None),
|
||||
cache_write_tokens=getattr(response_api_usage.input_tokens_details, "cache_write_tokens", None),
|
||||
)
|
||||
completion_tokens_details: Optional[CompletionTokensDetailsWrapper] = None
|
||||
output_tokens_details = getattr(response_api_usage, "output_tokens_details", None)
|
||||
|
|
|
|||
|
|
@ -1538,18 +1538,23 @@ class PromptTokensDetailsWrapper(
|
|||
"""Number of cache write (creation) tokens sent to the model. OpenAI naming (prompt_tokens_details.cache_write_tokens); this is the canonical field."""
|
||||
|
||||
cache_creation_tokens: Optional[int] = None
|
||||
"""Number of cache creation tokens sent to the model. Anthropic/Bedrock naming; kept in sync with cache_write_tokens for backwards compatibility."""
|
||||
"""Number of cache creation tokens sent to the model. Anthropic/Bedrock naming; kept in sync with cache_write_tokens (assigning either mirrors to the other)."""
|
||||
|
||||
cache_creation_token_details: Optional[CacheCreationTokenDetails] = None
|
||||
"""Details of cache creation tokens sent to the model. Used for tracking 5m/1h cache creation tokens for Anthropic prompt caching."""
|
||||
|
||||
def __setattr__(self, name: str, value: object) -> None:
|
||||
super().__setattr__(name, value)
|
||||
if name == "cache_write_tokens":
|
||||
super().__setattr__("cache_creation_tokens", value)
|
||||
elif name == "cache_creation_tokens":
|
||||
super().__setattr__("cache_write_tokens", value)
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
cache_write_tokens = (
|
||||
self.cache_write_tokens = (
|
||||
self.cache_write_tokens if self.cache_write_tokens is not None else self.cache_creation_tokens
|
||||
)
|
||||
self.cache_write_tokens = cache_write_tokens
|
||||
self.cache_creation_tokens = cache_write_tokens
|
||||
if self.character_count is None:
|
||||
del self.character_count
|
||||
if self.image_count is None:
|
||||
|
|
@ -1676,7 +1681,6 @@ class Usage(SafeAttributeModel, CompletionUsage):
|
|||
)
|
||||
else:
|
||||
_prompt_tokens_details.cache_write_tokens = params["cache_creation_input_tokens"]
|
||||
_prompt_tokens_details.cache_creation_tokens = params["cache_creation_input_tokens"]
|
||||
|
||||
super().__init__(
|
||||
prompt_tokens=prompt_tokens or 0,
|
||||
|
|
|
|||
|
|
@ -369,6 +369,32 @@ class TestResponseAPILoggingUtils:
|
|||
assert result.completion_tokens_details.image_tokens == 272
|
||||
assert result.completion_tokens_details.text_tokens == 100
|
||||
|
||||
def test_transform_response_api_usage_maps_cache_write_tokens(self):
|
||||
"""Responses API (/v1/responses) cache-write tokens must survive the usage transform.
|
||||
|
||||
gpt-5.6 returns usage.input_tokens_details.cache_write_tokens (an extra field
|
||||
not typed on InputTokensDetails). Before the fix the transform rebuilt the token
|
||||
details and dropped it, leaving the cache-creation metric empty (LIT-4633).
|
||||
"""
|
||||
usage = {
|
||||
"input_tokens": 10062,
|
||||
"output_tokens": 16,
|
||||
"total_tokens": 10078,
|
||||
"input_tokens_details": {
|
||||
"cached_tokens": 0,
|
||||
"cache_write_tokens": 10059,
|
||||
},
|
||||
}
|
||||
|
||||
result = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(
|
||||
usage
|
||||
)
|
||||
|
||||
assert result.prompt_tokens_details is not None
|
||||
assert result.prompt_tokens_details.cache_write_tokens == 10059
|
||||
assert result.prompt_tokens_details.cache_creation_tokens == 10059
|
||||
assert result.prompt_tokens_details.cached_tokens == 0
|
||||
|
||||
def test_transform_response_api_usage_mixed_details(self):
|
||||
"""Test transformation handles mixed token details (cached + image + audio)."""
|
||||
# Setup - hypothetical usage with mixed token types
|
||||
|
|
|
|||
|
|
@ -74,6 +74,19 @@ def test_prompt_tokens_details_no_cache_write_tokens_when_absent():
|
|||
assert not hasattr(details, "cache_creation_tokens")
|
||||
|
||||
|
||||
def test_prompt_tokens_details_cache_write_creation_stay_in_sync_on_assignment():
|
||||
"""Assigning either name after construction must mirror to the other, so a
|
||||
caller that sets only one field can't leave the pair silently out of sync."""
|
||||
details = PromptTokensDetailsWrapper(cache_write_tokens=100)
|
||||
assert details.cache_write_tokens == details.cache_creation_tokens == 100
|
||||
|
||||
details.cache_write_tokens = 250
|
||||
assert details.cache_write_tokens == details.cache_creation_tokens == 250
|
||||
|
||||
details.cache_creation_tokens = 375
|
||||
assert details.cache_write_tokens == details.cache_creation_tokens == 375
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def local_model_cost_map(monkeypatch):
|
||||
original_model_cost = litellm.model_cost
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue