mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(token-counter): count Anthropic native image content blocks
`_count_content_list` accepted text, image_url, tool_use, tool_result,
thinking and tool_reference, and raised on anything else, so an
Anthropic-native `{"type": "image", "source": {...}}` block aborted the
whole count. That is the documented Anthropic image format and exactly
what /v1/messages receives.
Three user-visible effects. /v1/messages/count_tokens and
/utils/token_counter return 500, and the router's context-window
pre-call check swallows the ValueError and returns every deployment
unfiltered, so an oversized prompt carrying an image is dispatched to
the provider instead of being rejected locally with a 400.
Prices the block through the existing image path: a base64 source
becomes a data URI, a url source passes through, and a file source
falls back to the default image token count. Blocks nested inside
tool_result.content are covered too, because _count_anthropic_content
recurses back into _count_content_list.
Fixes #36604
This commit is contained in:
parent
ca0b951a43
commit
1cce589aa0
2 changed files with 147 additions and 1 deletions
|
|
@ -646,6 +646,24 @@ def _validate_anthropic_content(content: Mapping[str, Any]) -> type:
|
|||
return expected_cls
|
||||
|
||||
|
||||
def _anthropic_image_source_data(source: Mapping[str, str]) -> str:
|
||||
"""
|
||||
Resolve an Anthropic image `source` to the data string `calculate_img_tokens` prices.
|
||||
|
||||
Returns "" for a `file` source, whose bytes the proxy cannot resolve locally.
|
||||
"""
|
||||
source_type: Final = source.get("type")
|
||||
if source_type == "base64":
|
||||
data: Final = source.get("data")
|
||||
if not data:
|
||||
return ""
|
||||
media_type: Final = source.get("media_type") or "image/png"
|
||||
return f"data:{media_type};base64,{data}"
|
||||
if source_type == "url":
|
||||
return source.get("url") or ""
|
||||
return ""
|
||||
|
||||
|
||||
def _count_anthropic_content(
|
||||
content: Mapping[str, Any],
|
||||
count_function: TokenCounterFunction,
|
||||
|
|
@ -714,6 +732,13 @@ def _count_content_list(
|
|||
elif c["type"] == "image_url":
|
||||
image_url = c.get("image_url")
|
||||
num_tokens += _count_image_tokens(image_url, use_default_image_token_count)
|
||||
elif c["type"] == "image":
|
||||
source = c.get("source")
|
||||
num_tokens += calculate_img_tokens(
|
||||
data=_anthropic_image_source_data(source) if isinstance(source, dict) else "",
|
||||
mode="auto",
|
||||
use_default_image_token_count=use_default_image_token_count,
|
||||
)
|
||||
elif c["type"] in ("tool_use", "tool_result"):
|
||||
num_tokens += _count_anthropic_content(
|
||||
c,
|
||||
|
|
@ -742,7 +767,8 @@ def _count_content_list(
|
|||
content_type = c.get("type", type(c).__name__) if isinstance(c, dict) else type(c).__name__
|
||||
raise ValueError(
|
||||
f"Invalid content item type: {content_type}. "
|
||||
f"Expected str or dict with 'type' field (text, image_url, tool_use, tool_result, thinking, tool_reference)."
|
||||
f"Expected str or dict with 'type' field "
|
||||
f"(text, image_url, image, tool_use, tool_result, thinking, tool_reference)."
|
||||
)
|
||||
return num_tokens
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -1160,3 +1160,123 @@ def test_count_content_list_rejects_unknown_type():
|
|||
message = str(exc_info.value)
|
||||
assert "Invalid content item type: totally_unknown_block" in message
|
||||
assert "tool_reference" in message
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"source",
|
||||
[
|
||||
{"type": "base64", "media_type": "image/png", "data": "iVBORw0KGgo="},
|
||||
{"type": "url", "url": "https://example.com/image.png"},
|
||||
{"type": "file", "file_id": "file-abc123"},
|
||||
],
|
||||
ids=["base64", "url", "file"],
|
||||
)
|
||||
def test_token_counter_with_anthropic_image_block(source):
|
||||
"""
|
||||
Anthropic-native `image` blocks must NOT raise, for every source variant.
|
||||
|
||||
Before this fix `_count_content_list` raised
|
||||
`Invalid content item type: image`. That 500s /v1/messages/count_tokens and
|
||||
/utils/token_counter, and it makes the router's context-window pre-call
|
||||
check swallow the error and return every deployment unfiltered, so an
|
||||
oversized prompt carrying an image is dispatched upstream instead of being
|
||||
rejected locally.
|
||||
"""
|
||||
from litellm.constants import DEFAULT_IMAGE_TOKEN_COUNT
|
||||
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "What is in this image?"},
|
||||
{"type": "image", "source": source},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
tokens = token_counter(
|
||||
model="anthropic/claude-sonnet-4-5-20250929",
|
||||
messages=messages,
|
||||
use_default_image_token_count=True,
|
||||
)
|
||||
assert tokens > DEFAULT_IMAGE_TOKEN_COUNT, (
|
||||
f"Expected the image block to contribute tokens, got {tokens}"
|
||||
)
|
||||
|
||||
|
||||
def test_anthropic_image_block_matches_equivalent_image_url():
|
||||
"""
|
||||
An Anthropic `image` block must price identically to the OpenAI `image_url`
|
||||
block carrying the same bytes, so the count does not depend on which
|
||||
endpoint shape the caller used.
|
||||
"""
|
||||
anthropic_messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": "iVBORw0KGgo=",
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
openai_messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {"url": "data:image/png;base64,iVBORw0KGgo="},
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
anthropic_tokens = token_counter(
|
||||
model="anthropic/claude-sonnet-4-5-20250929", messages=anthropic_messages
|
||||
)
|
||||
openai_tokens = token_counter(
|
||||
model="anthropic/claude-sonnet-4-5-20250929", messages=openai_messages
|
||||
)
|
||||
assert anthropic_tokens == openai_tokens
|
||||
|
||||
|
||||
def test_anthropic_image_block_nested_in_tool_result():
|
||||
"""
|
||||
An `image` block nested inside a `tool_result.content` list must be counted
|
||||
too. `_count_anthropic_content` recurses back into `_count_content_list`, so
|
||||
the nested case failed for the same reason the top-level one did.
|
||||
"""
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "tool_result",
|
||||
"tool_use_id": "toolu_01",
|
||||
"content": [
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": "iVBORw0KGgo=",
|
||||
},
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
tokens = token_counter(
|
||||
model="anthropic/claude-sonnet-4-5-20250929",
|
||||
messages=messages,
|
||||
use_default_image_token_count=True,
|
||||
)
|
||||
assert tokens > 0
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue