fix(cost): carry image and video input tokens through the Responses usage bridge

Realtime cost is computed from *_tokens_details after the usage round-trips
through the Responses shape, and the input half of that shape carried audio
only, so image and video prompt tokens stopped being billable as themselves.

Vertex splits prompt tokens by modality, so a session sending camera frames
arrives with image_tokens set. Those were folded into text_tokens and lost
their attribution. The amount happens not to move today, because the
calculator falls back to input_cost_per_token when no per-modality rate is
set, but the tokens have to survive before any such rate can ever apply.

InputTokensDetails now declares image_tokens and video_tokens instead of
leaning on pydantic extras, the repeated per-field copying is a loop over the
modality names so adding a modality no longer adds a branch, and the read-back
in ResponseAPILoggingUtils picks up video_tokens, which
PromptTokensDetailsWrapper already declared.

The output half of the original change is dropped: 449c091391 landed the same
OutputTokensDetails.audio_tokens fix upstream, with its own coverage in
test_gemini_realtime_transformation.py, and it always sets
output_tokens_details rather than only when non-empty. That structure is kept
as upstream wrote it.
This commit is contained in:
Marty Sullivan 2026-08-13 23:34:40 -04:00 committed by mateo-berri
parent 9496f16f12
commit 785c6cffc4
4 changed files with 38 additions and 0 deletions

View file

@ -2851,6 +2851,8 @@ class LiteLLMCompletionResponsesConfig:
cached_tokens=prompt_details.cached_tokens if prompt_details.cached_tokens is not None else 0,
text_tokens=prompt_details.text_tokens,
audio_tokens=prompt_details.audio_tokens,
image_tokens=prompt_details.image_tokens,
video_tokens=prompt_details.video_tokens,
cached_tokens_details=(
cached_tokens_details if isinstance(cached_tokens_details, CachedTokensDetails) else None
),

View file

@ -1182,6 +1182,7 @@ class ResponseAPILoggingUtils:
cached_tokens_details=getattr(
response_api_usage.input_tokens_details, "cached_tokens_details", None
),
video_tokens=getattr(response_api_usage.input_tokens_details, "video_tokens", None),
cache_write_tokens=getattr(response_api_usage.input_tokens_details, "cache_write_tokens", None),
web_search_requests=getattr(response_api_usage.input_tokens_details, "web_search_requests", None),
google_maps_grounding_requests=getattr(

View file

@ -1291,7 +1291,9 @@ class InputTokensDetails(BaseLiteLLMOpenAIResponseObject):
audio_tokens: int | None = None
cached_tokens: int = 0
cached_tokens_details: CachedTokensDetails | None = None
image_tokens: int | None = None
text_tokens: int | None = None
video_tokens: int | None = None
model_config = {"extra": "allow"}

View file

@ -2885,6 +2885,39 @@ class TestUsageTransformation:
assert getattr(response_usage.input_tokens_details, "cache_write_tokens", None) == 800
assert response_usage.input_tokens_details.model_dump()["cache_write_tokens"] == 800
def test_transform_usage_preserves_input_modality_tokens(self):
"""Regression: the bridge dropped image and video input tokens.
Vertex reports prompt tokens split by modality, so a Live session that sends
camera frames arrives with image_tokens set. InputTokensDetails declared only
audio/cached/text, so those tokens were folded into text and lost their
attribution, and any per-modality rate could never apply to them.
"""
usage = Usage(
prompt_tokens=300,
completion_tokens=10,
total_tokens=310,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=20, audio_tokens=80, image_tokens=150, video_tokens=50, cached_tokens=0
),
completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10),
)
response_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
chat_completion_response=usage
)
details = response_usage.input_tokens_details
assert details is not None
assert getattr(details, "image_tokens", None) == 150
assert getattr(details, "video_tokens", None) == 50
assert getattr(details, "audio_tokens", None) == 80
from litellm.responses.utils import ResponseAPILoggingUtils
back = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(response_usage.model_dump())
assert back.prompt_tokens_details.image_tokens == 150
assert back.prompt_tokens_details.video_tokens == 50
def test_transform_usage_with_reasoning_tokens_gemini(self):
"""Test that reasoning_tokens from Gemini are properly transformed to output_tokens_details"""
# Setup: Simulate Gemini usage with thoughtsTokenCount