mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-19 00:01:29 +00:00
fix(cost): carry image and video input tokens through the Responses usage bridge
Realtime cost is computed from *_tokens_details after the usage round-trips
through the Responses shape, and the input half of that shape carried audio
only, so image and video prompt tokens stopped being billable as themselves.
Vertex splits prompt tokens by modality, so a session sending camera frames
arrives with image_tokens set. Those were folded into text_tokens and lost
their attribution. The amount happens not to move today, because the
calculator falls back to input_cost_per_token when no per-modality rate is
set, but the tokens have to survive before any such rate can ever apply.
InputTokensDetails now declares image_tokens and video_tokens instead of
leaning on pydantic extras, the repeated per-field copying is a loop over the
modality names so adding a modality no longer adds a branch, and the read-back
in ResponseAPILoggingUtils picks up video_tokens, which
PromptTokensDetailsWrapper already declared.
The output half of the original change is dropped: 449c091391 landed the same
OutputTokensDetails.audio_tokens fix upstream, with its own coverage in
test_gemini_realtime_transformation.py, and it always sets
output_tokens_details rather than only when non-empty. That structure is kept
as upstream wrote it.
This commit is contained in:
parent
9496f16f12
commit
785c6cffc4
4 changed files with 38 additions and 0 deletions
|
|
@ -2851,6 +2851,8 @@ class LiteLLMCompletionResponsesConfig:
|
|||
cached_tokens=prompt_details.cached_tokens if prompt_details.cached_tokens is not None else 0,
|
||||
text_tokens=prompt_details.text_tokens,
|
||||
audio_tokens=prompt_details.audio_tokens,
|
||||
image_tokens=prompt_details.image_tokens,
|
||||
video_tokens=prompt_details.video_tokens,
|
||||
cached_tokens_details=(
|
||||
cached_tokens_details if isinstance(cached_tokens_details, CachedTokensDetails) else None
|
||||
),
|
||||
|
|
|
|||
|
|
@ -1182,6 +1182,7 @@ class ResponseAPILoggingUtils:
|
|||
cached_tokens_details=getattr(
|
||||
response_api_usage.input_tokens_details, "cached_tokens_details", None
|
||||
),
|
||||
video_tokens=getattr(response_api_usage.input_tokens_details, "video_tokens", None),
|
||||
cache_write_tokens=getattr(response_api_usage.input_tokens_details, "cache_write_tokens", None),
|
||||
web_search_requests=getattr(response_api_usage.input_tokens_details, "web_search_requests", None),
|
||||
google_maps_grounding_requests=getattr(
|
||||
|
|
|
|||
|
|
@ -1291,7 +1291,9 @@ class InputTokensDetails(BaseLiteLLMOpenAIResponseObject):
|
|||
audio_tokens: int | None = None
|
||||
cached_tokens: int = 0
|
||||
cached_tokens_details: CachedTokensDetails | None = None
|
||||
image_tokens: int | None = None
|
||||
text_tokens: int | None = None
|
||||
video_tokens: int | None = None
|
||||
|
||||
model_config = {"extra": "allow"}
|
||||
|
||||
|
|
|
|||
|
|
@ -2885,6 +2885,39 @@ class TestUsageTransformation:
|
|||
assert getattr(response_usage.input_tokens_details, "cache_write_tokens", None) == 800
|
||||
assert response_usage.input_tokens_details.model_dump()["cache_write_tokens"] == 800
|
||||
|
||||
def test_transform_usage_preserves_input_modality_tokens(self):
|
||||
"""Regression: the bridge dropped image and video input tokens.
|
||||
|
||||
Vertex reports prompt tokens split by modality, so a Live session that sends
|
||||
camera frames arrives with image_tokens set. InputTokensDetails declared only
|
||||
audio/cached/text, so those tokens were folded into text and lost their
|
||||
attribution, and any per-modality rate could never apply to them.
|
||||
"""
|
||||
usage = Usage(
|
||||
prompt_tokens=300,
|
||||
completion_tokens=10,
|
||||
total_tokens=310,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=20, audio_tokens=80, image_tokens=150, video_tokens=50, cached_tokens=0
|
||||
),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10),
|
||||
)
|
||||
|
||||
response_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
|
||||
chat_completion_response=usage
|
||||
)
|
||||
details = response_usage.input_tokens_details
|
||||
assert details is not None
|
||||
assert getattr(details, "image_tokens", None) == 150
|
||||
assert getattr(details, "video_tokens", None) == 50
|
||||
assert getattr(details, "audio_tokens", None) == 80
|
||||
|
||||
from litellm.responses.utils import ResponseAPILoggingUtils
|
||||
|
||||
back = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(response_usage.model_dump())
|
||||
assert back.prompt_tokens_details.image_tokens == 150
|
||||
assert back.prompt_tokens_details.video_tokens == 50
|
||||
|
||||
def test_transform_usage_with_reasoning_tokens_gemini(self):
|
||||
"""Test that reasoning_tokens from Gemini are properly transformed to output_tokens_details"""
|
||||
# Setup: Simulate Gemini usage with thoughtsTokenCount
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue