Merge pull request #37377 from BerriAI/devin_ai_lit5757_dashscope_nested_cache_creation

fix(types): map nested prompt_tokens_details.cache_creation_input_tokens to cache_write_tokens
This commit is contained in:
Mateo Wang 2026-08-18 20:02:17 -07:00 • committed by GitHub
commit 822cd4c4ea
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 129 additions and 2 deletions

View file

@ -1631,8 +1631,20 @@ class PromptTokensDetailsWrapper(
def __init__(self, *args, **kwargs) -> None:
super().__init__(*args, **kwargs)
extra_fields: Final = self.model_extra
nested_cache_creation_input_tokens: Final = (
extra_fields.get("cache_creation_input_tokens") if extra_fields is not None else None
)
self.cache_write_tokens = (
self.cache_write_tokens if self.cache_write_tokens is not None else self.cache_creation_tokens
self.cache_write_tokens
if self.cache_write_tokens is not None
else (
self.cache_creation_tokens
if self.cache_creation_tokens is not None
else (
nested_cache_creation_input_tokens if isinstance(nested_cache_creation_input_tokens, int) else None
)
)
)
if self.character_count is None:
del self.character_count

View file

@ -271,6 +271,47 @@ class TestDashscopeCostCalculator:
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-10)
def test_dashscope_nested_cache_creation_input_tokens_bill_at_cache_write_rate(self):
"""
Regression (LIT-5757): DashScope nests cache_creation_input_tokens inside
prompt_tokens_details; those tokens must bill at the tier's cache-creation
rate instead of being folded into text tokens at the input rate.
"""
self._register_tiered_model(
"dashscope/qwen-nested-cache-write-test",
[
{
"range": [0, 128000],
"input_cost_per_token": 4e-07,
"cache_read_input_token_cost": 1.6e-07,
"cache_creation_input_token_cost": 5e-07,
"output_cost_per_token": 1.6e-06,
}
],
)
usage = Usage(
prompt_tokens=2059,
completion_tokens=201,
total_tokens=2260,
prompt_tokens_details={
"cached_tokens": 0,
"text_tokens": 2059,
"cache_type": "ephemeral",
"cache_creation_input_tokens": 2048,
"cache_creation": {"ephemeral_5m_input_tokens": 2048},
},
completion_tokens_details={"reasoning_tokens": 170},
)
prompt_cost, _ = dashscope_cost_per_token(
model="qwen-nested-cache-write-test", usage=usage
)
assert math.isclose(
prompt_cost, (2048 * 5e-07) + (11 * 4e-07), rel_tol=1e-10
)
def test_dashscope_tiered_cache_creation_falls_back_to_tier_input_rate(self):
"""
Tiers without a cache_creation_input_token_cost bill cache-creation tokens at

View file

@ -124,6 +124,30 @@ def test_get_logging_payload_maps_openai_cache_write_tokens_to_cache_creation_in
assert additional_usage_values["prompt_tokens_details"]["cache_write_tokens"] == 800
def test_get_logging_payload_maps_nested_cache_creation_input_tokens():
"""
Regression (LIT-5757): DashScope nests cache_creation_input_tokens inside
prompt_tokens_details; SpendLogs must record it as cache_creation_input_tokens.
"""
additional_usage_values: Final = _get_additional_usage_values_for_usage(
litellm.Usage(
prompt_tokens=2059,
completion_tokens=31,
total_tokens=2090,
prompt_tokens_details={
"cached_tokens": 0,
"text_tokens": 2059,
"cache_type": "ephemeral",
"cache_creation_input_tokens": 2048,
"cache_creation": {"ephemeral_5m_input_tokens": 2048},
},
)
)
assert additional_usage_values["cache_creation_input_tokens"] == 2048
assert additional_usage_values["prompt_tokens_details"]["cache_write_tokens"] == 2048
def test_get_logging_payload_preserves_anthropic_cache_creation_input_tokens():
additional_usage_values = _get_additional_usage_values_for_usage(
litellm.Usage(

View file

@ -1,6 +1,6 @@
import os
import sys
from typing import Optional
from typing import Final, Optional
from unittest.mock import Mock
import pytest
@ -192,6 +192,32 @@ def test_transform_usage_with_cached_tokens_only():
print("✓ Transformation works with cached_tokens only")
def test_transform_usage_maps_nested_cache_creation_input_tokens():
"""
Regression (LIT-5757): DashScope nests cache_creation_input_tokens inside
prompt_tokens_details; the bridge must surface it as cache_write_tokens.
"""
usage: Final = Usage(
prompt_tokens=2059,
completion_tokens=31,
total_tokens=2090,
prompt_tokens_details={
"cached_tokens": 0,
"text_tokens": 2059,
"cache_type": "ephemeral",
"cache_creation_input_tokens": 2048,
"cache_creation": {"ephemeral_5m_input_tokens": 2048},
},
)
responses_usage: Final = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
usage
)
assert responses_usage.input_tokens_details is not None
assert responses_usage.input_tokens_details.cache_write_tokens == 2048
def test_transform_usage_with_reasoning_tokens_only():
"""
Test transformation when only reasoning_tokens is provided (no cached_tokens).

View file

@ -1,5 +1,6 @@
import os
import sys
from typing import Final
import pytest
@ -75,6 +76,29 @@ def test_usage_dump():
assert new_usage.prompt_tokens_details.web_search_requests == 1
def test_prompt_tokens_details_maps_nested_cache_creation_input_tokens():
"""Regression (LIT-5757): DashScope nests the Anthropic-spelled
cache_creation_input_tokens inside prompt_tokens_details. It must populate
the canonical cache_write_tokens/cache_creation_tokens pair, without
overriding an explicitly provided canonical value."""
from litellm.types.utils import PromptTokensDetailsWrapper
nested: Final = PromptTokensDetailsWrapper(
cached_tokens=0, text_tokens=2059, cache_creation_input_tokens=2048
)
assert nested.cache_write_tokens == 2048
assert nested.cache_creation_tokens == 2048
explicit: Final = PromptTokensDetailsWrapper(
cache_write_tokens=100, cache_creation_input_tokens=2048
)
assert explicit.cache_write_tokens == 100
assert explicit.cache_creation_tokens == 100
non_int: Final = PromptTokensDetailsWrapper(cache_creation_input_tokens=None)
assert not hasattr(non_int, "cache_write_tokens")
def test_usage_server_tool_use_dict_is_coerced_and_round_trips():
from litellm.types.utils import ServerToolUse, Usage