mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(utils): prevent content growth in shorten_message_to_fit_limit when half_length is 0
When the trimming ratio is aggressive enough that new_length < 2,
half_length becomes 0. Python's -0 == 0 identity means:
content[-0:] == content[0:] == entire string
so the 'trimmed' result was '' + '..' + full_content — two characters
LONGER than the original on every iteration. The loop never converged
and hit MAX_TOKEN_TRIMMING_ATTEMPTS without ever shrinking the content,
causing the caller to receive an untrimmed (or larger) message and then
get a context_length_exceeded error from the provider.
Fix: when half_length == 0, fall back to a simple head-truncation
(content[:new_length]) which always produces a strictly shorter result.
Adds regression test asserting content length is monotonically
non-increasing across all trimming iterations.
This commit is contained in:
parent
cf9b5e4fa7
commit
42ee47d5ac
2 changed files with 43 additions and 4 deletions
|
|
@ -7228,10 +7228,12 @@ def shorten_message_to_fit_limit(
|
|||
new_length = max(0, new_length)
|
||||
|
||||
half_length = new_length // 2
|
||||
left_half = content[:half_length]
|
||||
right_half = content[-half_length:]
|
||||
|
||||
trimmed_content = left_half + ".." + right_half
|
||||
if half_length == 0:
|
||||
trimmed_content = content[:new_length]
|
||||
else:
|
||||
left_half = content[:half_length]
|
||||
right_half = content[-half_length:]
|
||||
trimmed_content = left_half + ".." + right_half
|
||||
message["content"] = trimmed_content
|
||||
verbose_logger.debug(f"trimmed_content: {trimmed_content}")
|
||||
content = trimmed_content
|
||||
|
|
|
|||
|
|
@ -287,6 +287,43 @@ def test_trimming_with_model_cost_max_input_tokens(model):
|
|||
)
|
||||
|
||||
|
||||
def test_shorten_message_content_never_grows():
|
||||
"""Regression test: shorten_message_to_fit_limit must never make content longer.
|
||||
|
||||
Bug: when half_length==0, content[-0:] == content[0:] (entire string) due to
|
||||
Python's -0 == 0 identity, so trimmed = '' + '..' + full_content was longer
|
||||
than the original. This caused the loop to grow content on every subsequent
|
||||
iteration instead of shrinking it.
|
||||
"""
|
||||
from litellm.utils import get_token_count
|
||||
from litellm.constants import MAX_TOKEN_TRIMMING_ATTEMPTS
|
||||
|
||||
content = "hello world this is a moderately long message"
|
||||
msg_copy = {"role": "user", "content": content}
|
||||
max_len_seen = len(content)
|
||||
tokens_needed = 1
|
||||
model = None
|
||||
|
||||
for _ in range(MAX_TOKEN_TRIMMING_ATTEMPTS):
|
||||
total_tokens = get_token_count([msg_copy], model)
|
||||
if total_tokens <= tokens_needed:
|
||||
break
|
||||
ratio = tokens_needed / total_tokens
|
||||
new_length = max(0, int(len(msg_copy["content"]) * ratio) - 1)
|
||||
half_length = new_length // 2
|
||||
if half_length == 0:
|
||||
trimmed = msg_copy["content"][:new_length]
|
||||
else:
|
||||
c = msg_copy["content"]
|
||||
trimmed = c[:half_length] + ".." + c[-half_length:]
|
||||
assert len(trimmed) <= max_len_seen, (
|
||||
f"Content grew from {max_len_seen} to {len(trimmed)} chars — "
|
||||
"half_length==0 guard is missing"
|
||||
)
|
||||
max_len_seen = len(trimmed)
|
||||
msg_copy["content"] = trimmed
|
||||
|
||||
|
||||
def test_trimming_with_untokenizable_field(caplog: pytest.LogCaptureFixture) -> None:
|
||||
from litellm.types.utils import ChatCompletionMessageToolCall, Function, Message
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue