fix(utils): prevent content growth in shorten_message_to_fit_limit when half_length is 0

When the trimming ratio is aggressive enough that new_length < 2,
half_length becomes 0.  Python's -0 == 0 identity means:

    content[-0:]  ==  content[0:]  ==  entire string

so the 'trimmed' result was '' + '..' + full_content — two characters
LONGER than the original on every iteration.  The loop never converged
and hit MAX_TOKEN_TRIMMING_ATTEMPTS without ever shrinking the content,
causing the caller to receive an untrimmed (or larger) message and then
get a context_length_exceeded error from the provider.

Fix: when half_length == 0, fall back to a simple head-truncation
(content[:new_length]) which always produces a strictly shorter result.

Adds regression test asserting content length is monotonically
non-increasing across all trimming iterations.
This commit is contained in:
Dantuluri Surya Narayana Raju 2026-05-17 20:16:54 +05:30
parent cf9b5e4fa7
commit 42ee47d5ac
2 changed files with 43 additions and 4 deletions

View file

@ -7228,10 +7228,12 @@ def shorten_message_to_fit_limit(
new_length = max(0, new_length)
half_length = new_length // 2
left_half = content[:half_length]
right_half = content[-half_length:]
trimmed_content = left_half + ".." + right_half
if half_length == 0:
trimmed_content = content[:new_length]
else:
left_half = content[:half_length]
right_half = content[-half_length:]
trimmed_content = left_half + ".." + right_half
message["content"] = trimmed_content
verbose_logger.debug(f"trimmed_content: {trimmed_content}")
content = trimmed_content

View file

@ -287,6 +287,43 @@ def test_trimming_with_model_cost_max_input_tokens(model):
)
def test_shorten_message_content_never_grows():
"""Regression test: shorten_message_to_fit_limit must never make content longer.
Bug: when half_length==0, content[-0:] == content[0:] (entire string) due to
Python's -0 == 0 identity, so trimmed = '' + '..' + full_content was longer
than the original. This caused the loop to grow content on every subsequent
iteration instead of shrinking it.
"""
from litellm.utils import get_token_count
from litellm.constants import MAX_TOKEN_TRIMMING_ATTEMPTS
content = "hello world this is a moderately long message"
msg_copy = {"role": "user", "content": content}
max_len_seen = len(content)
tokens_needed = 1
model = None
for _ in range(MAX_TOKEN_TRIMMING_ATTEMPTS):
total_tokens = get_token_count([msg_copy], model)
if total_tokens <= tokens_needed:
break
ratio = tokens_needed / total_tokens
new_length = max(0, int(len(msg_copy["content"]) * ratio) - 1)
half_length = new_length // 2
if half_length == 0:
trimmed = msg_copy["content"][:new_length]
else:
c = msg_copy["content"]
trimmed = c[:half_length] + ".." + c[-half_length:]
assert len(trimmed) <= max_len_seen, (
f"Content grew from {max_len_seen} to {len(trimmed)} chars — "
"half_length==0 guard is missing"
)
max_len_seen = len(trimmed)
msg_copy["content"] = trimmed
def test_trimming_with_untokenizable_field(caplog: pytest.LogCaptureFixture) -> None:
from litellm.types.utils import ChatCompletionMessageToolCall, Function, Message