mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(cost): bill undetailed cache-write tokens at the 5m rate
Streamed Anthropic server-tool requests keep the message_start 5m/1h breakdown while cache_creation_input_tokens grows across iterations. calculate_cache_writing_cost was pricing only the stale breakdown and dropping the remainder. Bill max(total - 5m - 1h, 0) at the 5m rate, matching the non-streaming aggregate path. Fixes #42663
This commit is contained in:
parent
80221911cd
commit
e76b25816f
1 changed files with 11 additions and 0 deletions
|
|
@ -891,6 +891,14 @@ def calculate_cache_writing_cost(
|
|||
) -> float:
|
||||
"""
|
||||
Adjust cost of cache creation tokens based on the cache creation token details.
|
||||
|
||||
When a 5m/1h breakdown is present but covers fewer tokens than
|
||||
``cache_creation_tokens`` (common on streamed Anthropic server-tool
|
||||
requests, where the stream merger keeps the ``message_start`` breakdown
|
||||
while the top-level total grows across iterations), bill the undetailed
|
||||
remainder at the default 5m rate — same rule
|
||||
``AnthropicConfig._aggregate_cache_creation_token_details`` uses for
|
||||
non-streaming ``usage.iterations``.
|
||||
"""
|
||||
total_cost: float = 0.0
|
||||
if cache_creation_token_details is not None:
|
||||
|
|
@ -902,6 +910,9 @@ def calculate_cache_writing_cost(
|
|||
total_cost += (
|
||||
cache_creation_tokens_1h * cache_creation_cost_above_1hr if cache_creation_tokens_1h is not None else 0.0
|
||||
)
|
||||
detailed: Final = (cache_creation_tokens_5m or 0) + (cache_creation_tokens_1h or 0)
|
||||
undetailed: Final = max(cache_creation_tokens - detailed, 0)
|
||||
total_cost += undetailed * cache_creation_cost
|
||||
else:
|
||||
total_cost += cache_creation_tokens * cache_creation_cost
|
||||
return total_cost
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue