diff --git a/litellm/constants.py b/litellm/constants.py index a292b654778..6f6469cff7c 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -85,6 +85,15 @@ DEFAULT_REPLICATE_POLLING_RETRIES: Final = int(os.getenv("DEFAULT_REPLICATE_POLL DEFAULT_REPLICATE_POLLING_DELAY_SECONDS: Final = int(os.getenv("DEFAULT_REPLICATE_POLLING_DELAY_SECONDS", 1)) DEFAULT_IMAGE_TOKEN_COUNT: Final = int(os.getenv("DEFAULT_IMAGE_TOKEN_COUNT", 250)) HF_CONFIG_FETCH_TIMEOUT_SECONDS: Final = 10.0 +# Token estimate for one `input_audio` content block (audio understanding). +# The real cost is the provider's server-side audio tokenization and cannot be +# derived exactly client-side. With a base64 payload the estimate comes from +# the decoded byte count at a conservative low bitrate -- 8 kHz mono PCM-16 +# (16 000 bytes/s) at 10 tokens/s -- so equal-duration higher-quality audio is +# never under-estimated; a payload-less (reference-only) block gets the flat +# per-block floor. parallel_request_limiter_v3 reserves with the same numbers. +DEFAULT_AUDIO_TOKEN_ESTIMATE: Final = 300 +AUDIO_BYTES_PER_TOKEN: Final = 1600 # Maximum wall-clock seconds a streaming response is allowed to run. # Streams exceeding this duration are terminated with a Timeout error. diff --git a/litellm/litellm_core_utils/token_counter.py b/litellm/litellm_core_utils/token_counter.py index cdd2d0654be..0b532270415 100644 --- a/litellm/litellm_core_utils/token_counter.py +++ b/litellm/litellm_core_utils/token_counter.py @@ -16,6 +16,8 @@ import litellm from litellm import verbose_logger from litellm._lazy_imports import _get_default_encoding from litellm.constants import ( + AUDIO_BYTES_PER_TOKEN, + DEFAULT_AUDIO_TOKEN_ESTIMATE, DEFAULT_IMAGE_HEIGHT, DEFAULT_IMAGE_TOKEN_COUNT, DEFAULT_IMAGE_WIDTH, @@ -866,6 +868,25 @@ def _count_anthropic_content( return tokens +def _count_input_audio_content_block(c: Mapping[str, object]) -> int: + """ + Estimate tokens for an OpenAI ``input_audio`` content block (audio + understanding), e.g. {"type": "input_audio", "input_audio": {"data": + "", "format": "wav"}}. The real token cost is the provider's + server-side audio tokenization and cannot be derived exactly client-side; + when the block carries a base64 payload, derive the estimate from the + decoded byte count at a conservative low bitrate, otherwise use the flat + per-block floor -- the same numbers ``parallel_request_limiter_v3`` + already uses for its audio reservations (issue #38459). + """ + input_audio: Final = c.get("input_audio") + b64_data: Final = input_audio.get("data") if isinstance(input_audio, dict) else None + if isinstance(b64_data, str) and b64_data: + decoded_bytes: Final = len(b64_data) * 3 // 4 + return max(decoded_bytes // AUDIO_BYTES_PER_TOKEN, DEFAULT_AUDIO_TOKEN_ESTIMATE) + return DEFAULT_AUDIO_TOKEN_ESTIMATE + + def _count_content_list( count_function: TokenCounterFunction, content_list: str @@ -921,8 +942,7 @@ def _count_content_list( # Claude extended thinking content block # Count the thinking text and skip the opaque blobs (signature, redacted data) thinking_text = str(c.get("thinking", "")) - if thinking_text: - num_tokens += count_function(thinking_text) + num_tokens += count_function(thinking_text) if thinking_text else 0 elif c["type"] == "tool_reference": # Anthropic tool-search reference block: a lightweight pointer to # a deferred tool, e.g. {"type": "tool_reference", "tool_name": ...}. @@ -934,13 +954,15 @@ def _count_content_list( tool_name = str(c.get("tool_name") or "") if tool_name: num_tokens += count_function(tool_name) + elif c["type"] == "input_audio": + num_tokens += _count_input_audio_content_block(c) else: content_type = c.get("type", type(c).__name__) if isinstance(c, dict) else type(c).__name__ raise ValueError( f"Invalid content item type: {content_type}. " f"Expected str or dict with 'type' field " f"(text, image_url, image, document, file, tool_use, tool_result, thinking, redacted_thinking, " - f"tool_reference)." + f"tool_reference, input_audio)." ) return num_tokens except Exception as e: diff --git a/tests/unit/litellm_core_utils/test_token_counter.py b/tests/unit/litellm_core_utils/test_token_counter.py index c71b1496bdd..dc0a9e223d5 100644 --- a/tests/unit/litellm_core_utils/test_token_counter.py +++ b/tests/unit/litellm_core_utils/test_token_counter.py @@ -1633,3 +1633,80 @@ def test_token_counter_uses_the_tokenizer_of_each_model_family_and_of_a_custom_t "custom": expected["Xenova/llama-3-tokenizer"], "requested": sorted(served), } + + +def test_token_counter_with_input_audio_content_block(): + """ + Regression test for issue #38459: a message containing an OpenAI + `input_audio` content block (audio understanding) must NOT raise from + token_counter. Before the fix the raise poisoned the whole message and + every caller that swallows counter errors failed open (router + context-window pre-call check, prompt-caching deployment check), while + /utils/token_counter returned HTTP 500. + + The estimate mirrors parallel_request_limiter_v3's audio reservation: + decoded-base64 byte count at AUDIO_BYTES_PER_TOKEN, floored at + DEFAULT_AUDIO_TOKEN_ESTIMATE per block. + """ + from litellm.constants import AUDIO_BYTES_PER_TOKEN, DEFAULT_AUDIO_TOKEN_ESTIMATE + + small_b64 = "UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YQAAAAA=" + messages = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "What does the audio say?"}, + {"type": "input_audio", "input_audio": {"data": small_b64, "format": "wav"}}, + ], + } + ] + + tokens = token_counter_new(model="gpt-4o-audio-preview", messages=messages) + assert tokens >= DEFAULT_AUDIO_TOKEN_ESTIMATE, f"Expected at least the per-block floor, got {tokens}" + + # a large payload must scale the estimate (decoded bytes / AUDIO_BYTES_PER_TOKEN) + big_b64 = "A" * (AUDIO_BYTES_PER_TOKEN * 4000) # decoded ~3000 * AUDIO_BYTES_PER_TOKEN bytes + tokens_big = token_counter_new( + model="gpt-4o-audio-preview", + messages=[ + { + "role": "user", + "content": [{"type": "input_audio", "input_audio": {"data": big_b64, "format": "wav"}}], + } + ], + ) + assert tokens_big >= 2900, f"large audio payload must scale the estimate, got {tokens_big}" + assert tokens_big > tokens + + # a payload-less (reference-only) block must not raise and gets the floor + tokens_bare = token_counter_new( + model="gpt-4o-audio-preview", + messages=[{"role": "user", "content": [{"type": "input_audio", "input_audio": {"format": "wav"}}]}], + ) + assert tokens_bare >= DEFAULT_AUDIO_TOKEN_ESTIMATE + + +def test_trim_messages_with_input_audio_content_block(): + """Companion to issue #38459, same shape as the `file`-block case + (#28409): trim_messages swallows the token_counter error and silently + returns an over-budget conversation UNTRIMMED. With the fix, trimming + must actually happen.""" + small_b64 = "UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YQAAAAA=" + messages = [ + {"role": "user", "content": "filler message " * 200}, + {"role": "user", "content": "filler message " * 200}, + { + "role": "user", + "content": [ + {"type": "text", "text": "What does the audio say?"}, + {"type": "input_audio", "input_audio": {"data": small_b64, "format": "wav"}}, + ], + }, + ] + + trimmed = litellm.utils.trim_messages(messages, model="gpt-4o-audio-preview", max_tokens=500) + + assert trimmed is not None + assert len(trimmed) < len(messages), ( + "trim_messages must actually trim an over-budget conversation containing an input_audio block" + )