fix(cost): carry image and video input tokens through the Responses usage bridge

Realtime cost is computed from *_tokens_details after the usage round-trips
through the Responses shape, and the input half of that shape carried audio
only, so image and video prompt tokens stopped being billable as themselves.

Vertex splits prompt tokens by modality, so a session sending camera frames
arrives with image_tokens set. Those were folded into text_tokens and lost
their attribution. The amount happens not to move today, because the
calculator falls back to input_cost_per_token when no per-modality rate is
set, but the tokens have to survive before any such rate can ever apply.

InputTokensDetails now declares image_tokens and video_tokens instead of
leaning on pydantic extras, the repeated per-field copying is a loop over the
modality names so adding a modality no longer adds a branch, and the read-back
in ResponseAPILoggingUtils picks up video_tokens, which
PromptTokensDetailsWrapper already declared.

The output half of the original change is dropped: 449c091391 landed the same
OutputTokensDetails.audio_tokens fix upstream, with its own coverage in
test_gemini_realtime_transformation.py, and it always sets
output_tokens_details rather than only when non-empty. That structure is kept
as upstream wrote it.
This commit is contained in:
Marty Sullivan 2026-08-13 23:34:40 -04:00
parent 168a0055a2
commit 6ab56b8fe5
4 changed files with 40 additions and 2 deletions

View file

@ -2664,8 +2664,10 @@ class LiteLLMCompletionResponsesConfig:
if hasattr(prompt_details, "text_tokens") and prompt_details.text_tokens is not None:
input_details_dict["text_tokens"] = prompt_details.text_tokens
if hasattr(prompt_details, "audio_tokens") and prompt_details.audio_tokens is not None:
input_details_dict["audio_tokens"] = prompt_details.audio_tokens
for modality in ("audio_tokens", "image_tokens", "video_tokens"):
value = getattr(prompt_details, modality, None)
if value is not None:
input_details_dict[modality] = value
cache_write_tokens = getattr(prompt_details, "cache_write_tokens", None) or getattr(
prompt_details, "cache_creation_tokens", None

View file

@ -1130,6 +1130,7 @@ class ResponseAPILoggingUtils:
audio_tokens=getattr(response_api_usage.input_tokens_details, "audio_tokens", None),
text_tokens=getattr(response_api_usage.input_tokens_details, "text_tokens", None),
image_tokens=getattr(response_api_usage.input_tokens_details, "image_tokens", None),
video_tokens=getattr(response_api_usage.input_tokens_details, "video_tokens", None),
cache_write_tokens=getattr(response_api_usage.input_tokens_details, "cache_write_tokens", None),
)
completion_tokens_details: CompletionTokensDetailsWrapper | None = None

View file

@ -1287,7 +1287,9 @@ class OutputTokensDetails(BaseLiteLLMOpenAIResponseObject):
class InputTokensDetails(BaseLiteLLMOpenAIResponseObject):
audio_tokens: int | None = None
cached_tokens: int = 0
image_tokens: int | None = None
text_tokens: int | None = None
video_tokens: int | None = None
model_config = {"extra": "allow"}

View file

@ -2585,6 +2585,39 @@ class TestUsageTransformation:
assert response_usage.input_tokens_details.cached_tokens == 100
assert getattr(response_usage.input_tokens_details, "cache_write_tokens", None) == 800
def test_transform_usage_preserves_input_modality_tokens(self):
"""Regression: the bridge dropped image and video input tokens.
Vertex reports prompt tokens split by modality, so a Live session that sends
camera frames arrives with image_tokens set. InputTokensDetails declared only
audio/cached/text, so those tokens were folded into text and lost their
attribution, and any per-modality rate could never apply to them.
"""
usage = Usage(
prompt_tokens=300,
completion_tokens=10,
total_tokens=310,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=20, audio_tokens=80, image_tokens=150, video_tokens=50, cached_tokens=0
),
completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=10),
)
response_usage = LiteLLMCompletionResponsesConfig._transform_chat_completion_usage_to_responses_usage(
chat_completion_response=usage
)
details = response_usage.input_tokens_details
assert details is not None
assert getattr(details, "image_tokens", None) == 150
assert getattr(details, "video_tokens", None) == 50
assert getattr(details, "audio_tokens", None) == 80
from litellm.responses.utils import ResponseAPILoggingUtils
back = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(response_usage.model_dump())
assert back.prompt_tokens_details.image_tokens == 150
assert back.prompt_tokens_details.video_tokens == 50
def test_transform_usage_with_reasoning_tokens_gemini(self):
"""Test that reasoning_tokens from Gemini are properly transformed to output_tokens_details"""
# Setup: Simulate Gemini usage with thoughtsTokenCount