mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(token_counter): support OpenAI 'input_audio' content blocks
token_counter raised 'Invalid content item type: input_audio' on audio
understanding payloads ({'type': 'input_audio', ...}). Callers that
swallow the error fail open: Router._pre_call_checks returns the
unfiltered deployment list (context-window guard skipped), the
prompt-caching deployment check skips (no cache-affinity pinning),
trim_messages returns over-budget conversations untrimmed, and
/utils/token_counter 500s outright. parallel_request_limiter_v3 already
documents the gap and strips/estimates audio blocks locally; every
other call site still hit the raise.
Estimate audio blocks the same way that limiter does -- decoded-base64
byte count at a conservative low bitrate, floored per block -- with the
constants promoted to litellm/constants.py (core cannot import from
proxy hooks).
Fixes #38459
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
parent
74cad08997
commit
7b1c639bd7
3 changed files with 111 additions and 3 deletions
|
|
@ -85,6 +85,15 @@ DEFAULT_REPLICATE_POLLING_RETRIES: Final = int(os.getenv("DEFAULT_REPLICATE_POLL
|
|||
DEFAULT_REPLICATE_POLLING_DELAY_SECONDS: Final = int(os.getenv("DEFAULT_REPLICATE_POLLING_DELAY_SECONDS", 1))
|
||||
DEFAULT_IMAGE_TOKEN_COUNT: Final = int(os.getenv("DEFAULT_IMAGE_TOKEN_COUNT", 250))
|
||||
HF_CONFIG_FETCH_TIMEOUT_SECONDS: Final = 10.0
|
||||
# Token estimate for one `input_audio` content block (audio understanding).
|
||||
# The real cost is the provider's server-side audio tokenization and cannot be
|
||||
# derived exactly client-side. With a base64 payload the estimate comes from
|
||||
# the decoded byte count at a conservative low bitrate -- 8 kHz mono PCM-16
|
||||
# (16 000 bytes/s) at 10 tokens/s -- so equal-duration higher-quality audio is
|
||||
# never under-estimated; a payload-less (reference-only) block gets the flat
|
||||
# per-block floor. parallel_request_limiter_v3 reserves with the same numbers.
|
||||
DEFAULT_AUDIO_TOKEN_ESTIMATE: Final = 300
|
||||
AUDIO_BYTES_PER_TOKEN: Final = 1600
|
||||
|
||||
# Maximum wall-clock seconds a streaming response is allowed to run.
|
||||
# Streams exceeding this duration are terminated with a Timeout error.
|
||||
|
|
|
|||
|
|
@ -16,6 +16,8 @@ import litellm
|
|||
from litellm import verbose_logger
|
||||
from litellm._lazy_imports import _get_default_encoding
|
||||
from litellm.constants import (
|
||||
AUDIO_BYTES_PER_TOKEN,
|
||||
DEFAULT_AUDIO_TOKEN_ESTIMATE,
|
||||
DEFAULT_IMAGE_HEIGHT,
|
||||
DEFAULT_IMAGE_TOKEN_COUNT,
|
||||
DEFAULT_IMAGE_WIDTH,
|
||||
|
|
@ -866,6 +868,25 @@ def _count_anthropic_content(
|
|||
return tokens
|
||||
|
||||
|
||||
def _count_input_audio_content_block(c: Mapping[str, object]) -> int:
|
||||
"""
|
||||
Estimate tokens for an OpenAI ``input_audio`` content block (audio
|
||||
understanding), e.g. {"type": "input_audio", "input_audio": {"data":
|
||||
"<base64>", "format": "wav"}}. The real token cost is the provider's
|
||||
server-side audio tokenization and cannot be derived exactly client-side;
|
||||
when the block carries a base64 payload, derive the estimate from the
|
||||
decoded byte count at a conservative low bitrate, otherwise use the flat
|
||||
per-block floor -- the same numbers ``parallel_request_limiter_v3``
|
||||
already uses for its audio reservations (issue #38459).
|
||||
"""
|
||||
input_audio: Final = c.get("input_audio")
|
||||
b64_data: Final = input_audio.get("data") if isinstance(input_audio, dict) else None
|
||||
if isinstance(b64_data, str) and b64_data:
|
||||
decoded_bytes: Final = len(b64_data) * 3 // 4
|
||||
return max(decoded_bytes // AUDIO_BYTES_PER_TOKEN, DEFAULT_AUDIO_TOKEN_ESTIMATE)
|
||||
return DEFAULT_AUDIO_TOKEN_ESTIMATE
|
||||
|
||||
|
||||
def _count_content_list(
|
||||
count_function: TokenCounterFunction,
|
||||
content_list: str
|
||||
|
|
@ -921,8 +942,7 @@ def _count_content_list(
|
|||
# Claude extended thinking content block
|
||||
# Count the thinking text and skip the opaque blobs (signature, redacted data)
|
||||
thinking_text = str(c.get("thinking", ""))
|
||||
if thinking_text:
|
||||
num_tokens += count_function(thinking_text)
|
||||
num_tokens += count_function(thinking_text) if thinking_text else 0
|
||||
elif c["type"] == "tool_reference":
|
||||
# Anthropic tool-search reference block: a lightweight pointer to
|
||||
# a deferred tool, e.g. {"type": "tool_reference", "tool_name": ...}.
|
||||
|
|
@ -934,13 +954,15 @@ def _count_content_list(
|
|||
tool_name = str(c.get("tool_name") or "")
|
||||
if tool_name:
|
||||
num_tokens += count_function(tool_name)
|
||||
elif c["type"] == "input_audio":
|
||||
num_tokens += _count_input_audio_content_block(c)
|
||||
else:
|
||||
content_type = c.get("type", type(c).__name__) if isinstance(c, dict) else type(c).__name__
|
||||
raise ValueError(
|
||||
f"Invalid content item type: {content_type}. "
|
||||
f"Expected str or dict with 'type' field "
|
||||
f"(text, image_url, image, document, file, tool_use, tool_result, thinking, redacted_thinking, "
|
||||
f"tool_reference)."
|
||||
f"tool_reference, input_audio)."
|
||||
)
|
||||
return num_tokens
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -1633,3 +1633,80 @@ def test_token_counter_uses_the_tokenizer_of_each_model_family_and_of_a_custom_t
|
|||
"custom": expected["Xenova/llama-3-tokenizer"],
|
||||
"requested": sorted(served),
|
||||
}
|
||||
|
||||
|
||||
def test_token_counter_with_input_audio_content_block():
|
||||
"""
|
||||
Regression test for issue #38459: a message containing an OpenAI
|
||||
`input_audio` content block (audio understanding) must NOT raise from
|
||||
token_counter. Before the fix the raise poisoned the whole message and
|
||||
every caller that swallows counter errors failed open (router
|
||||
context-window pre-call check, prompt-caching deployment check), while
|
||||
/utils/token_counter returned HTTP 500.
|
||||
|
||||
The estimate mirrors parallel_request_limiter_v3's audio reservation:
|
||||
decoded-base64 byte count at AUDIO_BYTES_PER_TOKEN, floored at
|
||||
DEFAULT_AUDIO_TOKEN_ESTIMATE per block.
|
||||
"""
|
||||
from litellm.constants import AUDIO_BYTES_PER_TOKEN, DEFAULT_AUDIO_TOKEN_ESTIMATE
|
||||
|
||||
small_b64 = "UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YQAAAAA="
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "What does the audio say?"},
|
||||
{"type": "input_audio", "input_audio": {"data": small_b64, "format": "wav"}},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
tokens = token_counter_new(model="gpt-4o-audio-preview", messages=messages)
|
||||
assert tokens >= DEFAULT_AUDIO_TOKEN_ESTIMATE, f"Expected at least the per-block floor, got {tokens}"
|
||||
|
||||
# a large payload must scale the estimate (decoded bytes / AUDIO_BYTES_PER_TOKEN)
|
||||
big_b64 = "A" * (AUDIO_BYTES_PER_TOKEN * 4000) # decoded ~3000 * AUDIO_BYTES_PER_TOKEN bytes
|
||||
tokens_big = token_counter_new(
|
||||
model="gpt-4o-audio-preview",
|
||||
messages=[
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{"type": "input_audio", "input_audio": {"data": big_b64, "format": "wav"}}],
|
||||
}
|
||||
],
|
||||
)
|
||||
assert tokens_big >= 2900, f"large audio payload must scale the estimate, got {tokens_big}"
|
||||
assert tokens_big > tokens
|
||||
|
||||
# a payload-less (reference-only) block must not raise and gets the floor
|
||||
tokens_bare = token_counter_new(
|
||||
model="gpt-4o-audio-preview",
|
||||
messages=[{"role": "user", "content": [{"type": "input_audio", "input_audio": {"format": "wav"}}]}],
|
||||
)
|
||||
assert tokens_bare >= DEFAULT_AUDIO_TOKEN_ESTIMATE
|
||||
|
||||
|
||||
def test_trim_messages_with_input_audio_content_block():
|
||||
"""Companion to issue #38459, same shape as the `file`-block case
|
||||
(#28409): trim_messages swallows the token_counter error and silently
|
||||
returns an over-budget conversation UNTRIMMED. With the fix, trimming
|
||||
must actually happen."""
|
||||
small_b64 = "UklGRiQAAABXQVZFZm10IBAAAAABAAEAQB8AAIA+AAACABAAZGF0YQAAAAA="
|
||||
messages = [
|
||||
{"role": "user", "content": "filler message " * 200},
|
||||
{"role": "user", "content": "filler message " * 200},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "What does the audio say?"},
|
||||
{"type": "input_audio", "input_audio": {"data": small_b64, "format": "wav"}},
|
||||
],
|
||||
},
|
||||
]
|
||||
|
||||
trimmed = litellm.utils.trim_messages(messages, model="gpt-4o-audio-preview", max_tokens=500)
|
||||
|
||||
assert trimmed is not None
|
||||
assert len(trimmed) < len(messages), (
|
||||
"trim_messages must actually trim an over-budget conversation containing an input_audio block"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue