mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Merge pull request #37377 from BerriAI/devin_ai_lit5757_dashscope_nested_cache_creation
fix(types): map nested prompt_tokens_details.cache_creation_input_tokens to cache_write_tokens
This commit is contained in:
commit
822cd4c4ea
5 changed files with 129 additions and 2 deletions
|
|
@ -1631,8 +1631,20 @@ class PromptTokensDetailsWrapper(
|
|||
|
||||
def __init__(self, *args, **kwargs) -> None:
|
||||
super().__init__(*args, **kwargs)
|
||||
extra_fields: Final = self.model_extra
|
||||
nested_cache_creation_input_tokens: Final = (
|
||||
extra_fields.get("cache_creation_input_tokens") if extra_fields is not None else None
|
||||
)
|
||||
self.cache_write_tokens = (
|
||||
self.cache_write_tokens if self.cache_write_tokens is not None else self.cache_creation_tokens
|
||||
self.cache_write_tokens
|
||||
if self.cache_write_tokens is not None
|
||||
else (
|
||||
self.cache_creation_tokens
|
||||
if self.cache_creation_tokens is not None
|
||||
else (
|
||||
nested_cache_creation_input_tokens if isinstance(nested_cache_creation_input_tokens, int) else None
|
||||
)
|
||||
)
|
||||
)
|
||||
if self.character_count is None:
|
||||
del self.character_count
|
||||
|
|
|
|||
|
|
@ -271,6 +271,47 @@ class TestDashscopeCostCalculator:
|
|||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
|
||||
|
||||
def test_dashscope_nested_cache_creation_input_tokens_bill_at_cache_write_rate(self):
|
||||
"""
|
||||
Regression (LIT-5757): DashScope nests cache_creation_input_tokens inside
|
||||
prompt_tokens_details; those tokens must bill at the tier's cache-creation
|
||||
rate instead of being folded into text tokens at the input rate.
|
||||
"""
|
||||
self._register_tiered_model(
|
||||
"dashscope/qwen-nested-cache-write-test",
|
||||
[
|
||||
{
|
||||
"range": [0, 128000],
|
||||
"input_cost_per_token": 4e-07,
|
||||
"cache_read_input_token_cost": 1.6e-07,
|
||||
"cache_creation_input_token_cost": 5e-07,
|
||||
"output_cost_per_token": 1.6e-06,
|
||||
}
|
||||
],
|
||||
)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=2059,
|
||||
completion_tokens=201,
|
||||
total_tokens=2260,
|
||||
prompt_tokens_details={
|
||||
"cached_tokens": 0,
|
||||
"text_tokens": 2059,
|
||||
"cache_type": "ephemeral",
|
||||
"cache_creation_input_tokens": 2048,
|
||||
"cache_creation": {"ephemeral_5m_input_tokens": 2048},
|
||||
},
|
||||
completion_tokens_details={"reasoning_tokens": 170},
|
||||
)
|
||||
|
||||
prompt_cost, _ = dashscope_cost_per_token(
|
||||
model="qwen-nested-cache-write-test", usage=usage
|
||||
)
|
||||
|
||||
assert math.isclose(
|
||||
prompt_cost, (2048 * 5e-07) + (11 * 4e-07), rel_tol=1e-10
|
||||
)
|
||||
|
||||
def test_dashscope_tiered_cache_creation_falls_back_to_tier_input_rate(self):
|
||||
"""
|
||||
Tiers without a cache_creation_input_token_cost bill cache-creation tokens at
|
||||
|
|
|
|||
|
|
@ -124,6 +124,30 @@ def test_get_logging_payload_maps_openai_cache_write_tokens_to_cache_creation_in
|
|||
assert additional_usage_values["prompt_tokens_details"]["cache_write_tokens"] == 800
|
||||
|
||||
|
||||
def test_get_logging_payload_maps_nested_cache_creation_input_tokens():
|
||||
"""
|
||||
Regression (LIT-5757): DashScope nests cache_creation_input_tokens inside
|
||||
prompt_tokens_details; SpendLogs must record it as cache_creation_input_tokens.
|
||||
"""
|
||||
additional_usage_values: Final = _get_additional_usage_values_for_usage(
|
||||
litellm.Usage(
|
||||
prompt_tokens=2059,
|
||||
completion_tokens=31,
|
||||
total_tokens=2090,
|
||||
prompt_tokens_details={
|
||||
"cached_tokens": 0,
|
||||
"text_tokens": 2059,
|
||||
"cache_type": "ephemeral",
|
||||
"cache_creation_input_tokens": 2048,
|
||||
"cache_creation": {"ephemeral_5m_input_tokens": 2048},
|
||||
},
|
||||
)
|
||||
)
|
||||
|
||||
assert additional_usage_values["cache_creation_input_tokens"] == 2048
|
||||
assert additional_usage_values["prompt_tokens_details"]["cache_write_tokens"] == 2048
|
||||
|
||||
|
||||
def test_get_logging_payload_preserves_anthropic_cache_creation_input_tokens():
|
||||
additional_usage_values = _get_additional_usage_values_for_usage(
|
||||
litellm.Usage(
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
import os
|
||||
import sys
|
||||
from typing import Optional
|
||||
from typing import Final, Optional
|
||||
from unittest.mock import Mock
|
||||
|
||||
import pytest
|
||||
|
|
@ -192,6 +192,32 @@ def test_transform_usage_with_cached_tokens_only():
|
|||
print("✓ Transformation works with cached_tokens only")
|
||||
|
||||
|
||||
def test_transform_usage_maps_nested_cache_creation_input_tokens():
|
||||
"""
|
||||
Regression (LIT-5757): DashScope nests cache_creation_input_tokens inside
|
||||
prompt_tokens_details; the bridge must surface it as cache_write_tokens.
|
||||
"""
|
||||
usage: Final = Usage(
|
||||
prompt_tokens=2059,
|
||||
completion_tokens=31,
|
||||
total_tokens=2090,
|
||||
prompt_tokens_details={
|
||||
"cached_tokens": 0,
|
||||
"text_tokens": 2059,
|
||||
"cache_type": "ephemeral",
|
||||
"cache_creation_input_tokens": 2048,
|
||||
"cache_creation": {"ephemeral_5m_input_tokens": 2048},
|
||||
},
|
||||
)
|
||||
|
||||
responses_usage: Final = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
usage
|
||||
)
|
||||
|
||||
assert responses_usage.input_tokens_details is not None
|
||||
assert responses_usage.input_tokens_details.cache_write_tokens == 2048
|
||||
|
||||
|
||||
def test_transform_usage_with_reasoning_tokens_only():
|
||||
"""
|
||||
Test transformation when only reasoning_tokens is provided (no cached_tokens).
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
import os
|
||||
import sys
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
|
||||
|
|
@ -75,6 +76,29 @@ def test_usage_dump():
|
|||
assert new_usage.prompt_tokens_details.web_search_requests == 1
|
||||
|
||||
|
||||
def test_prompt_tokens_details_maps_nested_cache_creation_input_tokens():
|
||||
"""Regression (LIT-5757): DashScope nests the Anthropic-spelled
|
||||
cache_creation_input_tokens inside prompt_tokens_details. It must populate
|
||||
the canonical cache_write_tokens/cache_creation_tokens pair, without
|
||||
overriding an explicitly provided canonical value."""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper
|
||||
|
||||
nested: Final = PromptTokensDetailsWrapper(
|
||||
cached_tokens=0, text_tokens=2059, cache_creation_input_tokens=2048
|
||||
)
|
||||
assert nested.cache_write_tokens == 2048
|
||||
assert nested.cache_creation_tokens == 2048
|
||||
|
||||
explicit: Final = PromptTokensDetailsWrapper(
|
||||
cache_write_tokens=100, cache_creation_input_tokens=2048
|
||||
)
|
||||
assert explicit.cache_write_tokens == 100
|
||||
assert explicit.cache_creation_tokens == 100
|
||||
|
||||
non_int: Final = PromptTokensDetailsWrapper(cache_creation_input_tokens=None)
|
||||
assert not hasattr(non_int, "cache_write_tokens")
|
||||
|
||||
|
||||
def test_usage_server_tool_use_dict_is_coerced_and_round_trips():
|
||||
from litellm.types.utils import ServerToolUse, Usage
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue