fix(token_counter): support OpenAI 'input_audio' content blocks

token_counter raised 'Invalid content item type: input_audio' on audio
understanding payloads ({'type': 'input_audio', ...}). Callers that
swallow the error fail open: Router._pre_call_checks returns the
unfiltered deployment list (context-window guard skipped), the
prompt-caching deployment check skips (no cache-affinity pinning),
trim_messages returns over-budget conversations untrimmed, and
/utils/token_counter 500s outright. parallel_request_limiter_v3 already
documents the gap and strips/estimates audio blocks locally; every
other call site still hit the raise.

Estimate audio blocks the same way that limiter does -- decoded-base64
byte count at a conservative low bitrate, floored per block -- with the
constants promoted to litellm/constants.py (core cannot import from
proxy hooks).

Fixes #38459

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Mihidum Hettiyahandi 2026-08-27 14:53:23 +10:00
parent 74cad08997
commit 7b1c639bd7
3 changed files with 111 additions and 3 deletions

View file

@ -85,6 +85,15 @@ DEFAULT_REPLICATE_POLLING_RETRIES: Final = int(os.getenv("DEFAULT_REPLICATE_POLL
DEFAULT_REPLICATE_POLLING_DELAY_SECONDS: Final = int(os.getenv("DEFAULT_REPLICATE_POLLING_DELAY_SECONDS", 1))
DEFAULT_IMAGE_TOKEN_COUNT: Final = int(os.getenv("DEFAULT_IMAGE_TOKEN_COUNT", 250))
HF_CONFIG_FETCH_TIMEOUT_SECONDS: Final = 10.0
# Token estimate for one `input_audio` content block (audio understanding).
# The real cost is the provider's server-side audio tokenization and cannot be
# derived exactly client-side. With a base64 payload the estimate comes from
# the decoded byte count at a conservative low bitrate -- 8 kHz mono PCM-16
# (16 000 bytes/s) at 10 tokens/s -- so equal-duration higher-quality audio is
# never under-estimated; a payload-less (reference-only) block gets the flat
# per-block floor. parallel_request_limiter_v3 reserves with the same numbers.
DEFAULT_AUDIO_TOKEN_ESTIMATE: Final = 300
AUDIO_BYTES_PER_TOKEN: Final = 1600
# Maximum wall-clock seconds a streaming response is allowed to run.
# Streams exceeding this duration are terminated with a Timeout error.

View file

@ -16,6 +16,8 @@ import litellm
from litellm import verbose_logger
from litellm._lazy_imports import _get_default_encoding
from litellm.constants import (
AUDIO_BYTES_PER_TOKEN,
DEFAULT_AUDIO_TOKEN_ESTIMATE,
DEFAULT_IMAGE_HEIGHT,
DEFAULT_IMAGE_TOKEN_COUNT,
DEFAULT_IMAGE_WIDTH,
@ -866,6 +868,25 @@ def _count_anthropic_content(
return tokens
def _count_input_audio_content_block(c: Mapping[str, object]) -> int:
"""
Estimate tokens for an OpenAI ``input_audio`` content block (audio
understanding), e.g. {"type": "input_audio", "input_audio": {"data":
"<base64>", "format": "wav"}}. The real token cost is the provider's
server-side audio tokenization and cannot be derived exactly client-side;
when the block carries a base64 payload, derive the estimate from the
decoded byte count at a conservative low bitrate, otherwise use the flat
per-block floor -- the same numbers ``parallel_request_limiter_v3``
already uses for its audio reservations (issue #38459).
"""
input_audio: Final = c.get("input_audio")
b64_data: Final = input_audio.get("data") if isinstance(input_audio, dict) else None
if isinstance(b64_data, str) and b64_data:
decoded_bytes: Final = len(b64_data) * 3 // 4
return max(decoded_bytes // AUDIO_BYTES_PER_TOKEN, DEFAULT_AUDIO_TOKEN_ESTIMATE)
return DEFAULT_AUDIO_TOKEN_ESTIMATE
def _count_content_list(
count_function: TokenCounterFunction,
content_list: str
@ -921,8 +942,7 @@ def _count_content_list(
# Claude extended thinking content block
# Count the thinking text and skip the opaque blobs (signature, redacted data)
thinking_text = str(c.get("thinking", ""))
if thinking_text:
num_tokens += count_function(thinking_text)
num_tokens += count_function(thinking_text) if thinking_text else 0
elif c["type"] == "tool_reference":
# Anthropic tool-search reference block: a lightweight pointer to
# a deferred tool, e.g. {"type": "tool_reference", "tool_name": ...}.
@ -934,13 +954,15 @@ def _count_content_list(
tool_name = str(c.get("tool_name") or "")
if tool_name:
num_tokens += count_function(tool_name)
elif c["type"] == "input_audio":
num_tokens += _count_input_audio_content_block(c)
else:
content_type = c.get("type", type(c).__name__) if isinstance(c, dict) else type(c).__name__
raise ValueError(
f"Invalid content item type: {content_type}. "
f"Expected str or dict with 'type' field "
f"(text, image_url, image, document, file, tool_use, tool_result, thinking, redacted_thinking, "
f"tool_reference)."
f"tool_reference, input_audio)."
)
return num_tokens
except Exception as e:

View file

@ -1633,3 +1633,80 @@ def test_token_counter_uses_the_tokenizer_of_each_model_family_and_of_a_custom_t
"custom": expected["Xenova/llama-3-tokenizer"],
"requested": sorted(served),
}
def test_token_counter_with_input_audio_content_block():
"""
Regression test for issue #38459: a message containing an OpenAI
`input_audio` content block (audio understanding) must NOT raise from
token_counter. Before the fix the raise poisoned the whole message and
every caller that swallows counter errors failed open (router
context-window pre-call check, prompt-caching deployment check), while
/utils/token_counter returned HTTP 500.
The estimate mirrors parallel_request_limiter_v3's audio reservation:
decoded-base64 byte count at AUDIO_BYTES_PER_TOKEN, floored at
DEFAULT_AUDIO_TOKEN_ESTIMATE per block.
"""
from litellm.constants import AUDIO_BYTES_PER_TOKEN, DEFAULT_AUDIO_TOKEN_ESTIMATE
small_b64 = "UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YQAAAAA="
messages = [
{
"role": "user",
"content": [
{"type": "text", "text": "What does the audio say?"},
{"type": "input_audio", "input_audio": {"data": small_b64, "format": "wav"}},
],
}
]
tokens = token_counter_new(model="gpt-4o-audio-preview", messages=messages)
assert tokens >= DEFAULT_AUDIO_TOKEN_ESTIMATE, f"Expected at least the per-block floor, got {tokens}"
# a large payload must scale the estimate (decoded bytes / AUDIO_BYTES_PER_TOKEN)
big_b64 = "A" * (AUDIO_BYTES_PER_TOKEN * 4000) # decoded ~3000 * AUDIO_BYTES_PER_TOKEN bytes
tokens_big = token_counter_new(
model="gpt-4o-audio-preview",
messages=[
{
"role": "user",
"content": [{"type": "input_audio", "input_audio": {"data": big_b64, "format": "wav"}}],
}
],
)
assert tokens_big >= 2900, f"large audio payload must scale the estimate, got {tokens_big}"
assert tokens_big > tokens
# a payload-less (reference-only) block must not raise and gets the floor
tokens_bare = token_counter_new(
model="gpt-4o-audio-preview",
messages=[{"role": "user", "content": [{"type": "input_audio", "input_audio": {"format": "wav"}}]}],
)
assert tokens_bare >= DEFAULT_AUDIO_TOKEN_ESTIMATE
def test_trim_messages_with_input_audio_content_block():
"""Companion to issue #38459, same shape as the `file`-block case
(#28409): trim_messages swallows the token_counter error and silently
returns an over-budget conversation UNTRIMMED. With the fix, trimming
must actually happen."""
small_b64 = "UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YQAAAAA="
messages = [
{"role": "user", "content": "filler message " * 200},
{"role": "user", "content": "filler message " * 200},
{
"role": "user",
"content": [
{"type": "text", "text": "What does the audio say?"},
{"type": "input_audio", "input_audio": {"data": small_b64, "format": "wav"}},
],
},
]
trimmed = litellm.utils.trim_messages(messages, model="gpt-4o-audio-preview", max_tokens=500)
assert trimmed is not None
assert len(trimmed) < len(messages), (
"trim_messages must actually trim an over-budget conversation containing an input_audio block"
)