mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(token_counter): stop undercounting image tokens for the default "auto" detail level
calculate_img_tokens() had "auto" grouped with "low" detail, so it
always returned the flat 85-token low-detail estimate no matter how
big the image actually was:
if mode == "low" or mode == "auto":
return base_tokens
elif mode == "high":
...tile-based sizing...
"auto" is the detail level every caller gets by default when they
don't pass detail explicitly, and OpenAI's real auto-detail sizing is
tile-based (the same math as "high"), not a flat low-res cost. So in
practice the most common code path (no explicit detail set) was
silently undercounting image tokens for anything but tiny images.
For a 4096x4096 image this is stark: low=85 tokens, high=765 tokens,
and auto was returning 85 instead of 765 - a ~680 token undercount for
what's likely a very common case.
Fix: "auto" now shares the "high" branch instead of "low".
Verified with calculate_img_tokens() directly (get_image_dimensions
mocked to avoid network):
before: low=85, high=765, auto=85 (auto == low, the bug)
after: low=85, high=765, auto=765 (auto == high, correct)
Added test_calculate_img_tokens_auto_matches_high_not_low to
tests/test_litellm/litellm_core_utils/test_token_counter.py to guard
against regressing back to the flat low-detail estimate. Full
token_counter test file still passes (86 passed, 1 skipped).
This commit is contained in:
parent
dc9297d36f
commit
14e116a1bb
2 changed files with 43 additions and 2 deletions
|
|
@ -289,9 +289,12 @@ def calculate_img_tokens(
|
|||
if use_default_image_token_count:
|
||||
verbose_logger.debug("Using default image token count: {}".format(DEFAULT_IMAGE_TOKEN_COUNT))
|
||||
return DEFAULT_IMAGE_TOKEN_COUNT
|
||||
if mode == "low" or mode == "auto":
|
||||
if mode == "low":
|
||||
return base_tokens
|
||||
elif mode == "high":
|
||||
elif mode == "high" or mode == "auto":
|
||||
# "auto" mirrors "high": actual auto-detail sizing is tile-based, so
|
||||
# treating it as a flat low-detail cost silently undercounts tokens
|
||||
# for the (default) case where callers never set an explicit detail.
|
||||
# Run the async function using the helper
|
||||
width, height = get_image_dimensions(
|
||||
data=data,
|
||||
|
|
|
|||
|
|
@ -526,6 +526,44 @@ def test_img_url_token_counter(img_url, monkeypatch):
|
|||
assert height is not None
|
||||
|
||||
|
||||
def test_calculate_img_tokens_auto_matches_high_not_low(monkeypatch):
|
||||
"""
|
||||
Regression test for the "auto" image detail mode silently being costed
|
||||
as "low" detail instead of using the real tile-based sizing.
|
||||
|
||||
"auto" is the *default* detail level whenever a caller doesn't set one
|
||||
explicitly, and OpenAI's actual auto-detail behavior is tile-based
|
||||
(same sizing math as "high"), not a flat low-res estimate. Before this
|
||||
fix, `calculate_img_tokens(mode="auto")` fell into the same branch as
|
||||
`mode="low"` and returned a flat 85-token estimate regardless of image
|
||||
size, which meant a large image sent with no explicit detail could be
|
||||
undercounted by hundreds of tokens relative to what the API actually
|
||||
bills for it.
|
||||
|
||||
For a 4096x4096 image this under-count is stark: "low" is a flat 85
|
||||
tokens, "high" (tile-based) is 765 tokens, and "auto" must match
|
||||
"high" here, not "low".
|
||||
"""
|
||||
from litellm.litellm_core_utils.token_counter import calculate_img_tokens
|
||||
|
||||
monkeypatch.setattr(
|
||||
"litellm.litellm_core_utils.token_counter.get_image_dimensions",
|
||||
lambda **kwargs: (4096, 4096),
|
||||
)
|
||||
|
||||
low_tokens = calculate_img_tokens(data=b"fake", mode="low")
|
||||
high_tokens = calculate_img_tokens(data=b"fake", mode="high")
|
||||
auto_tokens = calculate_img_tokens(data=b"fake", mode="auto")
|
||||
|
||||
assert low_tokens == 85
|
||||
assert high_tokens == 765
|
||||
assert auto_tokens == high_tokens, (
|
||||
"'auto' detail must use the same tile-based sizing as 'high', "
|
||||
f"got {auto_tokens} tokens vs high={high_tokens} tokens"
|
||||
)
|
||||
assert auto_tokens != low_tokens
|
||||
|
||||
|
||||
def test_token_encode_disallowed_special():
|
||||
encode(model="gpt-3.5-turbo", text="Hello, world! <|endoftext|>")
|
||||
token_counter(model="gpt-3.5-turbo", text="Hello, world! <|endoftext|>")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue