From 60c2d9f590b4e72b25a9a6d60c6b4070f03a9637 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Fri, 25 Sep 2026 23:07:18 +0800 Subject: [PATCH 01/10] fix(llm_response_utils): expand every hallucinated multi_tool_use.parallel call _handle_invalid_parallel_tool_calls replaces one entry with len(replacement) entries, so the offsets of everything after it shift by len(replacement) - 1. Advancing by the full length skipped one entry per expansion: with two multi_tool_use.parallel calls in one message the second stayed in message.tool_calls (so agents tried to call a non-existent tool) and the call after it was overwritten. Adds a regression test that expands two such calls in one message. --- .../convert_dict_to_response.py | 6 +- .../test_convert_dict_to_chat_completion.py | 76 +++++++++++++++++++ 2 files changed, 81 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 9ea730a873f..ff624a3de8b 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -406,7 +406,11 @@ def _handle_invalid_parallel_tool_calls( shift = 0 for i, replacement in replacements.items(): tool_calls[:] = tool_calls[: i + shift] + replacement + tool_calls[i + shift + 1 :] - shift += len(replacement) + # One entry is replaced by ``len(replacement)`` entries, so the + # offsets of everything after it move by the difference - not by + # the full length, which would skip one entry per expansion and + # leave the next ``multi_tool_use.parallel`` call in place. + shift += len(replacement) - 1 return tool_calls except json.JSONDecodeError: diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py index 31c554985a7..1a5fbe95b5a 100644 --- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py +++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py @@ -2486,3 +2486,79 @@ class TestConvertToModelResponseObjectCompletion: }, model_response_object=None, ) + + +def test_convert_to_model_response_object_expands_every_parallel_tool_call(): + """ + Every hallucinated `multi_tool_use.parallel` entry in one message must be + expanded. Replacing one entry with its expansions moves the offsets of the + entries that follow it by `len(expansions) - 1`; advancing by the full + length skipped one entry per expansion, so with two such calls the second + one stayed in place and the call after it was overwritten. + """ + + def parallel(call_id, recipient_name, parameters): + return { + "id": call_id, + "type": "function", + "function": { + "name": "multi_tool_use.parallel", + "arguments": json.dumps( + { + "tool_uses": [ + { + "recipient_name": recipient_name, + "parameters": parameters, + } + ] + } + ), + }, + } + + def plain(call_id, name): + return { + "id": call_id, + "type": "function", + "function": {"name": name, "arguments": "{}"}, + } + + response_object = { + "id": "chatcmpl-parallel", + "object": "chat.completion", + "created": 1728933352, + "model": "gpt-4o-2024-08-06", + "choices": [ + { + "index": 0, + "finish_reason": "tool_calls", + "message": { + "role": "assistant", + "content": None, + "tool_calls": [ + plain("0", "get_weather"), + parallel("m1", "functions.get_time", {"tz": "UTC"}), + plain("2", "get_news"), + parallel("m2", "functions.get_quote", {"sym": "AAPL"}), + plain("4", "get_forecast"), + ], + }, + } + ], + } + + result = convert_to_model_response_object( + response_object=response_object, + model_response_object=ModelResponse(), + response_type="completion", + ) + + names = [tc.function.name for tc in result.choices[0].message.tool_calls] + assert names == [ + "get_weather", + "get_time", + "get_news", + "get_quote", + "get_forecast", + ] + assert "multi_tool_use.parallel" not in names From e2f83aadcf1298cb255b48dd76cf8d9a9680abc5 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Fri, 25 Sep 2026 23:45:49 +0800 Subject: [PATCH 02/10] test(llm_response_utils): expand one parallel call into two tool uses Addresses review feedback: with one tool use per parallel call the replacement never changed the list length, so the test did not pin the offset arithmetic. Now the first splice changes the length and the second has to land at the shifted offset. --- .../test_convert_dict_to_chat_completion.py | 20 ++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py index 1a5fbe95b5a..ff0bd628ab3 100644 --- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py +++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py @@ -2497,7 +2497,7 @@ def test_convert_to_model_response_object_expands_every_parallel_tool_call(): one stayed in place and the call after it was overwritten. """ - def parallel(call_id, recipient_name, parameters): + def parallel(call_id, *calls): return { "id": call_id, "type": "function", @@ -2506,10 +2506,8 @@ def test_convert_to_model_response_object_expands_every_parallel_tool_call(): "arguments": json.dumps( { "tool_uses": [ - { - "recipient_name": recipient_name, - "parameters": parameters, - } + {"recipient_name": name, "parameters": params} + for name, params in calls ] } ), @@ -2537,9 +2535,16 @@ def test_convert_to_model_response_object_expands_every_parallel_tool_call(): "content": None, "tool_calls": [ plain("0", "get_weather"), - parallel("m1", "functions.get_time", {"tz": "UTC"}), + # Two expansions here and one below, so the first splice + # changes the list length and the second splice has to land + # at the shifted offset. + parallel( + "m1", + ("functions.get_time", {"tz": "UTC"}), + ("functions.get_date", {"tz": "UTC"}), + ), plain("2", "get_news"), - parallel("m2", "functions.get_quote", {"sym": "AAPL"}), + parallel("m2", ("functions.get_quote", {"sym": "AAPL"})), plain("4", "get_forecast"), ], }, @@ -2557,6 +2562,7 @@ def test_convert_to_model_response_object_expands_every_parallel_tool_call(): assert names == [ "get_weather", "get_time", + "get_date", "get_news", "get_quote", "get_forecast", From 523c30dcf3c1e2b49b295a4dd330cd41c00d65fe Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 09:07:25 +0800 Subject: [PATCH 03/10] test(llm_response_utils): cover the malformed payload guard where unit coverage runs codecov/patch reported 0% of the diff hit: the new except-clause lines are in llm_response_utils, but this test lived under tests/llm_translation, which the unit coverage job does not run. Move it next to the other llm_response_utils tests and add a positive case so the guard is proven not to swallow valid payloads. --- ...test_handle_invalid_parallel_tool_calls.py | 59 +++++++++++++++++++ 1 file changed, 59 insertions(+) create mode 100644 tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py diff --git a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py new file mode 100644 index 00000000000..09ef7bca2d2 --- /dev/null +++ b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py @@ -0,0 +1,59 @@ +"""The hallucinated multi_tool_use.parallel expansion must not fail the response.""" + +import json + +import pytest + +from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( + _handle_invalid_parallel_tool_calls, +) +from litellm.types.utils import ChatCompletionMessageToolCall, Function + + +def _parallel_instance(arguments: str): + return [ + ChatCompletionMessageToolCall( + id="call_1", + type="function", + function=Function(name="multi_tool_use.parallel", arguments=arguments), + ) + ] + + +@pytest.mark.parametrize( + "arguments", + [ + '{"tool_uses": [{"recipient_name": "functions.get_weather"}]}', + '{"tool_uses": "nope"}', + '{"tool_uses": [42]}', + "{}", + ], +) +def test_malformed_tool_uses_returns_original_calls(arguments): + """A hallucinated payload we cannot expand must come back untouched.""" + tool_calls = _parallel_instance(arguments) + + result = _handle_invalid_parallel_tool_calls(tool_calls) + + assert len(result) == 1 + assert result[0].function.name == "multi_tool_use.parallel" + assert result[0].id == "call_1" + assert result[0].function.arguments == arguments + + +def test_valid_parallel_payload_still_expands(): + """The guard must not swallow well-formed payloads.""" + tool_calls = _parallel_instance( + json.dumps( + { + "tool_uses": [ + {"recipient_name": "functions.get_weather", "parameters": {"city": "NYC"}}, + {"recipient_name": "functions.get_time", "parameters": {"tz": "EST"}}, + ] + } + ) + ) + + result = _handle_invalid_parallel_tool_calls(tool_calls) + + assert [c.function.name for c in result] == ["get_weather", "get_time"] From 752e755da5b0e5b2404435b91f3f9af6d929e81b Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 09:08:03 +0800 Subject: [PATCH 04/10] fix(llm_response_utils): actually widen the except clause The earlier commit added the rationale comment but the clause still read json.JSONDecodeError alone, so the four malformed payloads kept escaping. --- .../llm_response_utils/convert_dict_to_response.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index ff624a3de8b..5395ba973d3 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -413,7 +413,7 @@ def _handle_invalid_parallel_tool_calls( shift += len(replacement) - 1 return tool_calls - except json.JSONDecodeError: + except (json.JSONDecodeError, KeyError, TypeError, AttributeError): # if there is a JSONDecodeError, return the original tool_calls return tool_calls From a99918e84598def918863f96632c8289cbbc84ec Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 09:18:11 +0800 Subject: [PATCH 05/10] test(llm_response_utils): cover the parallel expansion where unit coverage runs codecov/patch reported 0% of diff hit: the regression test lived under tests/llm_translation, which the unit coverage job does not run. Add it next to the other llm_response_utils unit tests, keeping the length-changing case (two tool uses in the first parallel call) that pins the shift arithmetic. --- .../test_parallel_tool_call_shift.py | 88 +++++++++++++++++++ 1 file changed, 88 insertions(+) create mode 100644 tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py diff --git a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py new file mode 100644 index 00000000000..52a6ae2cc07 --- /dev/null +++ b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py @@ -0,0 +1,88 @@ +"""Every hallucinated multi_tool_use.parallel call in one message must expand. + +Replacing one entry with len(expansions) entries moves everything after it by +len(expansions) - 1; advancing by the full length skipped one entry per +expansion, so with two such calls the second stayed in place (and the call +after it was overwritten). +""" + +import json + +from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( + convert_to_model_response_object, +) +from litellm.types.utils import ModelResponse + + +def _parallel(call_id: str, *calls): + return { + "id": call_id, + "type": "function", + "function": { + "name": "multi_tool_use.parallel", + "arguments": json.dumps( + { + "tool_uses": [ + {"recipient_name": name, "parameters": params} for name, params in calls + ] + } + ), + }, + } + + +def _plain(call_id: str, name: str): + return { + "id": call_id, + "type": "function", + "function": {"name": name, "arguments": "{}"}, + } + + +def test_every_parallel_tool_call_expands(): + response_object = { + "id": "chatcmpl-parallel", + "object": "chat.completion", + "created": 1728933352, + "model": "gpt-4o-2024-08-06", + "choices": [ + { + "index": 0, + "finish_reason": "tool_calls", + "message": { + "role": "assistant", + "content": None, + "tool_calls": [ + _plain("0", "get_weather"), + # Two expansions here and one below, so the first splice + # changes the list length and the second has to land at + # the shifted offset. + _parallel( + "m1", + ("functions.get_time", {"tz": "UTC"}), + ("functions.get_date", {"tz": "UTC"}), + ), + _plain("2", "get_news"), + _parallel("m2", ("functions.get_quote", {"sym": "AAPL"})), + _plain("4", "get_forecast"), + ], + }, + } + ], + } + + result = convert_to_model_response_object( + response_object=response_object, + model_response_object=ModelResponse(), + response_type="completion", + ) + + names = [tc.function.name for tc in result.choices[0].message.tool_calls] + assert names == [ + "get_weather", + "get_time", + "get_date", + "get_news", + "get_quote", + "get_forecast", + ] From c02f11dd7edb783a220ed6c49922b3b8de6dc9af Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sat, 26 Sep 2026 09:19:57 +0800 Subject: [PATCH 06/10] chore: keep this PR to the shift fix only Two commits that belong to the malformed-payload PR (#43173) had leaked onto this branch: the widened except clause and its test file. Restore the clause to main's version and drop those tests here, and keep the regression test only in tests/test_litellm/litellm_core_utils/llm_response_utils/ where the unit coverage job runs. --- .../convert_dict_to_response.py | 2 +- .../test_convert_dict_to_chat_completion.py | 82 ------------------- ...test_handle_invalid_parallel_tool_calls.py | 59 ------------- 3 files changed, 1 insertion(+), 142 deletions(-) delete mode 100644 tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index 5395ba973d3..ff624a3de8b 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -413,7 +413,7 @@ def _handle_invalid_parallel_tool_calls( shift += len(replacement) - 1 return tool_calls - except (json.JSONDecodeError, KeyError, TypeError, AttributeError): + except json.JSONDecodeError: # if there is a JSONDecodeError, return the original tool_calls return tool_calls diff --git a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py index ff0bd628ab3..31c554985a7 100644 --- a/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py +++ b/tests/llm_translation/test_llm_response_utils/test_convert_dict_to_chat_completion.py @@ -2486,85 +2486,3 @@ class TestConvertToModelResponseObjectCompletion: }, model_response_object=None, ) - - -def test_convert_to_model_response_object_expands_every_parallel_tool_call(): - """ - Every hallucinated `multi_tool_use.parallel` entry in one message must be - expanded. Replacing one entry with its expansions moves the offsets of the - entries that follow it by `len(expansions) - 1`; advancing by the full - length skipped one entry per expansion, so with two such calls the second - one stayed in place and the call after it was overwritten. - """ - - def parallel(call_id, *calls): - return { - "id": call_id, - "type": "function", - "function": { - "name": "multi_tool_use.parallel", - "arguments": json.dumps( - { - "tool_uses": [ - {"recipient_name": name, "parameters": params} - for name, params in calls - ] - } - ), - }, - } - - def plain(call_id, name): - return { - "id": call_id, - "type": "function", - "function": {"name": name, "arguments": "{}"}, - } - - response_object = { - "id": "chatcmpl-parallel", - "object": "chat.completion", - "created": 1728933352, - "model": "gpt-4o-2024-08-06", - "choices": [ - { - "index": 0, - "finish_reason": "tool_calls", - "message": { - "role": "assistant", - "content": None, - "tool_calls": [ - plain("0", "get_weather"), - # Two expansions here and one below, so the first splice - # changes the list length and the second splice has to land - # at the shifted offset. - parallel( - "m1", - ("functions.get_time", {"tz": "UTC"}), - ("functions.get_date", {"tz": "UTC"}), - ), - plain("2", "get_news"), - parallel("m2", ("functions.get_quote", {"sym": "AAPL"})), - plain("4", "get_forecast"), - ], - }, - } - ], - } - - result = convert_to_model_response_object( - response_object=response_object, - model_response_object=ModelResponse(), - response_type="completion", - ) - - names = [tc.function.name for tc in result.choices[0].message.tool_calls] - assert names == [ - "get_weather", - "get_time", - "get_date", - "get_news", - "get_quote", - "get_forecast", - ] - assert "multi_tool_use.parallel" not in names diff --git a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py deleted file mode 100644 index 09ef7bca2d2..00000000000 --- a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_handle_invalid_parallel_tool_calls.py +++ /dev/null @@ -1,59 +0,0 @@ -"""The hallucinated multi_tool_use.parallel expansion must not fail the response.""" - -import json - -import pytest - -from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( - _handle_invalid_parallel_tool_calls, -) -from litellm.types.utils import ChatCompletionMessageToolCall, Function - - -def _parallel_instance(arguments: str): - return [ - ChatCompletionMessageToolCall( - id="call_1", - type="function", - function=Function(name="multi_tool_use.parallel", arguments=arguments), - ) - ] - - -@pytest.mark.parametrize( - "arguments", - [ - '{"tool_uses": [{"recipient_name": "functions.get_weather"}]}', - '{"tool_uses": "nope"}', - '{"tool_uses": [42]}', - "{}", - ], -) -def test_malformed_tool_uses_returns_original_calls(arguments): - """A hallucinated payload we cannot expand must come back untouched.""" - tool_calls = _parallel_instance(arguments) - - result = _handle_invalid_parallel_tool_calls(tool_calls) - - assert len(result) == 1 - assert result[0].function.name == "multi_tool_use.parallel" - assert result[0].id == "call_1" - assert result[0].function.arguments == arguments - - -def test_valid_parallel_payload_still_expands(): - """The guard must not swallow well-formed payloads.""" - tool_calls = _parallel_instance( - json.dumps( - { - "tool_uses": [ - {"recipient_name": "functions.get_weather", "parameters": {"city": "NYC"}}, - {"recipient_name": "functions.get_time", "parameters": {"tz": "EST"}}, - ] - } - ) - ) - - result = _handle_invalid_parallel_tool_calls(tool_calls) - - assert [c.function.name for c in result] == ["get_weather", "get_time"] From 6066171c60cde82e2bcdd800719dfa9374e331d5 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sun, 27 Sep 2026 16:00:24 +0800 Subject: [PATCH 07/10] fix(router): keep retry breadcrumbs small enough to stay in budget RETRY_BREADCRUMB_LIMIT bounds how many breadcrumbs a request keeps, but each one stored str(e) whole, and litellm exceptions embed the upstream response body: a 200 KB error message produced an 800 KB retry record, and 300 failing requests grew the proxy RSS by 86.8 MB against the 48 MB budget of test_failing_requests_do_not_grow_rss_or_stored_request. Keep the diagnostic head of the message and append how much was dropped. --- litellm/router.py | 14 +++++- .../router/test_retry_breadcrumbs.py | 45 +++++++++++++++++++ 2 files changed, 58 insertions(+), 1 deletion(-) create mode 100644 tests/test_litellm/router/test_retry_breadcrumbs.py diff --git a/litellm/router.py b/litellm/router.py index 023b99cd64e..44e1665903d 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -694,6 +694,12 @@ set_live_deployment_replay(_replay_live_router_model_cost) RETRY_BREADCRUMB_LIMIT: Final = 4 +# Litellm exceptions embed the upstream response body, so a single retry record +# could hold hundreds of kilobytes: the release-gate test +# test_failing_requests_do_not_grow_rss_or_stored_request saw the proxy RSS grow +# 86.8 MB over 300 failing requests against a 48 MB budget. Only the diagnostic +# head of the message is kept. +RETRY_BREADCRUMB_EXCEPTION_LIMIT: Final = 500 class FallbackAwareStreamWrapper(CustomStreamWrapper): @@ -8296,11 +8302,17 @@ class Router: model_info: Final = request_metadata.get("model_info") deployment_id: Final = model_info.get("id") if isinstance(model_info, Mapping) else None attempted_retries: Final = request_metadata.get("attempted_retries") + exception_string: Final = str(e) attempt_record: Final[RetryAttemptRecord] = { "model_group": model_group if isinstance(model_group, str) else None, "deployment_id": deployment_id if isinstance(deployment_id, str) else None, "exception_type": type(e).__name__, - "exception_string": str(e), + "exception_string": ( + exception_string[:RETRY_BREADCRUMB_EXCEPTION_LIMIT] + + f"... [truncated, {len(exception_string)} chars]" + if len(exception_string) > RETRY_BREADCRUMB_EXCEPTION_LIMIT + else exception_string + ), "attempted_retries": attempted_retries if type(attempted_retries) is int else None, } earlier_breadcrumbs: Final = request_metadata.get("previous_models") diff --git a/tests/test_litellm/router/test_retry_breadcrumbs.py b/tests/test_litellm/router/test_retry_breadcrumbs.py new file mode 100644 index 00000000000..b5507467d5b --- /dev/null +++ b/tests/test_litellm/router/test_retry_breadcrumbs.py @@ -0,0 +1,45 @@ +"""Retry breadcrumbs must stay small: they are stored on the request record.""" + +from litellm.router import RETRY_BREADCRUMB_EXCEPTION_LIMIT, Router + + +def _record_for(error: Exception) -> dict: + kwargs = {"model": "gpt-4o", "metadata": {}} + Router.log_retry(object(), kwargs, error) + return kwargs["metadata"]["previous_models"][-1] + + +def test_large_exception_message_is_truncated_in_the_breadcrumb(): + """Litellm exceptions embed the upstream body, so str(e) can be hundreds of KB. + + The release-gate test test_failing_requests_do_not_grow_rss_or_stored_request + saw the proxy RSS grow 86.8 MB over 300 failing requests against a 48 MB budget. + """ + error = Exception("Error code: 500 - " + "x" * 200_000) + + record = _record_for(error) + + assert len(record["exception_string"]) < RETRY_BREADCRUMB_EXCEPTION_LIMIT + 100 + assert record["exception_string"].startswith("Error code: 500 - ") + assert "truncated" in record["exception_string"] + assert record["exception_type"] == "Exception" + + +def test_short_exception_message_is_kept_whole(): + record = _record_for(ValueError("bad request")) + + assert record["exception_string"] == "bad request" + assert record["exception_type"] == "ValueError" + + +def test_breadcrumb_metadata_stays_bounded_across_requests(): + error = Exception("Error code: 500 - " + "x" * 200_000) + + total = 0 + for _ in range(50): + kwargs = {"model": "gpt-4o", "metadata": {}} + for _ in range(6): + Router.log_retry(object(), kwargs, error) + total += len(repr(kwargs["metadata"])) + + assert total < 500_000 From 2d3b45c918536be7925c828bb98416bb33a5de4a Mon Sep 17 00:00:00 2001 From: sclfcz Date: Sun, 27 Sep 2026 21:36:33 +0800 Subject: [PATCH 08/10] ci: assign the new tests/test_litellm/router shard tests/unit/test_assert_ci_coverage.py requires every directory under tests/test_litellm to be named by a shard; the router tests this PR adds are not covered by the core-utils path, so misc/Run tests failed on the guard. --- .github/workflows/test-unit.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test-unit.yml b/.github/workflows/test-unit.yml index 126a6e26e6f..c6452d0a9f6 100644 --- a/.github/workflows/test-unit.yml +++ b/.github/workflows/test-unit.yml @@ -61,7 +61,7 @@ jobs: - shard: core-utils artifact-name: core-utils - test-path: "tests/test_litellm/litellm_core_utils" + test-path: "tests/test_litellm/litellm_core_utils tests/test_litellm/router" workers: 2 reruns: 1 timeout-minutes: 20 From 75896cb1f1b5599fa7e5b63e37193f46c5022e64 Mon Sep 17 00:00:00 2001 From: sclfcz Date: Mon, 28 Sep 2026 00:05:32 +0800 Subject: [PATCH 09/10] chore: drop the sibling-PR changes that leaked onto this branch Only router.py, its test and the shard entry belong here; the llm_response_utils changes are #43170/#43173. --- .../convert_dict_to_response.py | 6 +- .../test_parallel_tool_call_shift.py | 88 ------------------- 2 files changed, 1 insertion(+), 93 deletions(-) delete mode 100644 tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py diff --git a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py index ff624a3de8b..9ea730a873f 100644 --- a/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py +++ b/litellm/litellm_core_utils/llm_response_utils/convert_dict_to_response.py @@ -406,11 +406,7 @@ def _handle_invalid_parallel_tool_calls( shift = 0 for i, replacement in replacements.items(): tool_calls[:] = tool_calls[: i + shift] + replacement + tool_calls[i + shift + 1 :] - # One entry is replaced by ``len(replacement)`` entries, so the - # offsets of everything after it move by the difference - not by - # the full length, which would skip one entry per expansion and - # leave the next ``multi_tool_use.parallel`` call in place. - shift += len(replacement) - 1 + shift += len(replacement) return tool_calls except json.JSONDecodeError: diff --git a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py b/tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py deleted file mode 100644 index 52a6ae2cc07..00000000000 --- a/tests/test_litellm/litellm_core_utils/llm_response_utils/test_parallel_tool_call_shift.py +++ /dev/null @@ -1,88 +0,0 @@ -"""Every hallucinated multi_tool_use.parallel call in one message must expand. - -Replacing one entry with len(expansions) entries moves everything after it by -len(expansions) - 1; advancing by the full length skipped one entry per -expansion, so with two such calls the second stayed in place (and the call -after it was overwritten). -""" - -import json - -from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import ( - convert_to_model_response_object, -) -from litellm.types.utils import ModelResponse - - -def _parallel(call_id: str, *calls): - return { - "id": call_id, - "type": "function", - "function": { - "name": "multi_tool_use.parallel", - "arguments": json.dumps( - { - "tool_uses": [ - {"recipient_name": name, "parameters": params} for name, params in calls - ] - } - ), - }, - } - - -def _plain(call_id: str, name: str): - return { - "id": call_id, - "type": "function", - "function": {"name": name, "arguments": "{}"}, - } - - -def test_every_parallel_tool_call_expands(): - response_object = { - "id": "chatcmpl-parallel", - "object": "chat.completion", - "created": 1728933352, - "model": "gpt-4o-2024-08-06", - "choices": [ - { - "index": 0, - "finish_reason": "tool_calls", - "message": { - "role": "assistant", - "content": None, - "tool_calls": [ - _plain("0", "get_weather"), - # Two expansions here and one below, so the first splice - # changes the list length and the second has to land at - # the shifted offset. - _parallel( - "m1", - ("functions.get_time", {"tz": "UTC"}), - ("functions.get_date", {"tz": "UTC"}), - ), - _plain("2", "get_news"), - _parallel("m2", ("functions.get_quote", {"sym": "AAPL"})), - _plain("4", "get_forecast"), - ], - }, - } - ], - } - - result = convert_to_model_response_object( - response_object=response_object, - model_response_object=ModelResponse(), - response_type="completion", - ) - - names = [tc.function.name for tc in result.choices[0].message.tool_calls] - assert names == [ - "get_weather", - "get_time", - "get_date", - "get_news", - "get_quote", - "get_forecast", - ] From 0d46fe7c6ba1bd8dc5a6feec42e08bcb6ae91bdc Mon Sep 17 00:00:00 2001 From: sclfcz Date: Tue, 29 Sep 2026 10:41:37 +0800 Subject: [PATCH 10/10] chore: keep one copy of the retry breadcrumb test The move back and forth left the file at both paths. --- .../router_utils/test_retry_breadcrumbs.py | 45 ------------------- 1 file changed, 45 deletions(-) delete mode 100644 tests/test_litellm/router_utils/test_retry_breadcrumbs.py diff --git a/tests/test_litellm/router_utils/test_retry_breadcrumbs.py b/tests/test_litellm/router_utils/test_retry_breadcrumbs.py deleted file mode 100644 index b5507467d5b..00000000000 --- a/tests/test_litellm/router_utils/test_retry_breadcrumbs.py +++ /dev/null @@ -1,45 +0,0 @@ -"""Retry breadcrumbs must stay small: they are stored on the request record.""" - -from litellm.router import RETRY_BREADCRUMB_EXCEPTION_LIMIT, Router - - -def _record_for(error: Exception) -> dict: - kwargs = {"model": "gpt-4o", "metadata": {}} - Router.log_retry(object(), kwargs, error) - return kwargs["metadata"]["previous_models"][-1] - - -def test_large_exception_message_is_truncated_in_the_breadcrumb(): - """Litellm exceptions embed the upstream body, so str(e) can be hundreds of KB. - - The release-gate test test_failing_requests_do_not_grow_rss_or_stored_request - saw the proxy RSS grow 86.8 MB over 300 failing requests against a 48 MB budget. - """ - error = Exception("Error code: 500 - " + "x" * 200_000) - - record = _record_for(error) - - assert len(record["exception_string"]) < RETRY_BREADCRUMB_EXCEPTION_LIMIT + 100 - assert record["exception_string"].startswith("Error code: 500 - ") - assert "truncated" in record["exception_string"] - assert record["exception_type"] == "Exception" - - -def test_short_exception_message_is_kept_whole(): - record = _record_for(ValueError("bad request")) - - assert record["exception_string"] == "bad request" - assert record["exception_type"] == "ValueError" - - -def test_breadcrumb_metadata_stays_bounded_across_requests(): - error = Exception("Error code: 500 - " + "x" * 200_000) - - total = 0 - for _ in range(50): - kwargs = {"model": "gpt-4o", "metadata": {}} - for _ in range(6): - Router.log_retry(object(), kwargs, error) - total += len(repr(kwargs["metadata"])) - - assert total < 500_000