diff --git a/litellm/utils.py b/litellm/utils.py index 13a46840431..0ac6801c9f0 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -7536,7 +7536,8 @@ def shorten_message_to_fit_limit(message, tokens_needed, model: str | None, rais half_length = new_length // 2 left_half = content[:half_length] - right_half = content[-half_length:] + # content[-0:] is the whole string + right_half = content[-half_length:] if half_length else "" trimmed_content = left_half + ".." + right_half message["content"] = trimmed_content diff --git a/tests/unit/test_utils.py b/tests/unit/test_utils.py index 2c612aa350c..d6bb121cbce 100644 --- a/tests/unit/test_utils.py +++ b/tests/unit/test_utils.py @@ -6012,6 +6012,23 @@ def test_load_credentials_from_list_fills_kwargs_from_the_loaded_credential_with } assert _credential_warnings(caplog) == [] +def test_shorten_message_to_fit_limit_never_grows_content(): + """A zero half_length must not turn the trim into a two-character prefix. + + `content[-0:]` is `content[0:]`, so with half_length == 0 the "right half" is the + whole string and each attempt returns `".." + content`. The loop then runs its full + attempt budget growing the message two characters at a time. + """ + from litellm.utils import shorten_message_to_fit_limit + + content = "hello world " * 40 + message = {"role": "user", "content": content} + + result = shorten_message_to_fit_limit(message, tokens_needed=1, model="claude-3-5-sonnet-20240620") + + assert len(result["content"]) < len(content) + assert not result["content"].startswith("....") + _MOCK_STREAM_ID: Final = "chatcmpl-mock-stream" _ChunkSnapshot = tuple[str, tuple[str | None, ...], Usage | None]