mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix: read cached tokens from input_tokens_details and exclude them from input_tokens
Mirrors the sibling Anthropic adapter: cache reads fall back to the OpenAI Responses usage shape (input_tokens_details.cached_tokens), and Anthropic input_tokens excludes cache read/creation tokens.
This commit is contained in:
parent
78111d8947
commit
aca69354f2
2 changed files with 28 additions and 10 deletions
|
|
@ -246,6 +246,11 @@ class AnthropicResponsesStreamWrapper:
|
|||
output_tokens = _get_field(usage, "output_tokens", 0) or 0
|
||||
cache_creation_tokens = int(_get_field(usage, "cache_creation_input_tokens", 0) or 0)
|
||||
cache_read_tokens = int(_get_field(usage, "cache_read_input_tokens", 0) or 0)
|
||||
if cache_read_tokens == 0:
|
||||
input_tokens_details = _get_field(usage, "input_tokens_details")
|
||||
if input_tokens_details is not None:
|
||||
cache_read_tokens = int(_get_field(input_tokens_details, "cached_tokens", 0) or 0)
|
||||
input_tokens = max(input_tokens - cache_read_tokens - cache_creation_tokens, 0)
|
||||
|
||||
# Check if tool_use was in the output to override stop_reason
|
||||
if response_obj is not None:
|
||||
|
|
|
|||
|
|
@ -79,15 +79,6 @@ class TestProcessEventTextDeltaWithoutOutputItemAdded:
|
|||
|
||||
|
||||
class TestDictShapedCompletedEvents:
|
||||
"""Usage, status, and output must be read from dict-shaped
|
||||
`response.completed` payloads, not only attribute-shaped ones.
|
||||
|
||||
Previously this branch used getattr-only access, so dict-shaped events
|
||||
always produced usage 0/0 (disabling spend tracking and TPM enforcement,
|
||||
https://github.com/BerriAI/litellm/issues/32086) and mapped
|
||||
`response.incomplete` to end_turn instead of max_tokens.
|
||||
"""
|
||||
|
||||
def test_dict_usage_is_extracted(self):
|
||||
chunks = _process_all(
|
||||
[
|
||||
|
|
@ -106,11 +97,33 @@ class TestDictShapedCompletedEvents:
|
|||
)
|
||||
assert chunks[0]["type"] == "message_delta"
|
||||
assert chunks[0]["usage"] == {
|
||||
"input_tokens": 11,
|
||||
"input_tokens": 4,
|
||||
"output_tokens": 42,
|
||||
"cache_read_input_tokens": 7,
|
||||
}
|
||||
|
||||
def test_openai_responses_cached_tokens_details_extracted(self):
|
||||
chunks = _process_all(
|
||||
[
|
||||
{
|
||||
"type": "response.completed",
|
||||
"response": {
|
||||
"status": "completed",
|
||||
"usage": {
|
||||
"input_tokens": 100,
|
||||
"output_tokens": 20,
|
||||
"input_tokens_details": {"cached_tokens": 30},
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
assert chunks[0]["usage"] == {
|
||||
"input_tokens": 70,
|
||||
"output_tokens": 20,
|
||||
"cache_read_input_tokens": 30,
|
||||
}
|
||||
|
||||
def test_dict_incomplete_maps_to_max_tokens(self):
|
||||
chunks = _process_all([{"type": "response.incomplete", "response": {"status": "incomplete"}}])
|
||||
assert chunks[0]["delta"]["stop_reason"] == "max_tokens"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue