diff --git a/litellm/router.py b/litellm/router.py index 115faad000c..0f2f2d596ef 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -719,6 +719,12 @@ set_live_deployment_replay(_replay_live_router_model_cost) RETRY_BREADCRUMB_LIMIT: Final = 4 +# Litellm exceptions embed the upstream response body, so a single retry record +# could hold hundreds of kilobytes: the release-gate test +# test_failing_requests_do_not_grow_rss_or_stored_request saw the proxy RSS grow +# 86.8 MB over 300 failing requests against a 48 MB budget. Only the diagnostic +# head of the message is kept. +RETRY_BREADCRUMB_EXCEPTION_LIMIT: Final = 500 class FallbackAwareStreamWrapper(CustomStreamWrapper): @@ -8386,11 +8392,17 @@ class Router: model_info: Final = request_metadata.get("model_info") deployment_id: Final = model_info.get("id") if isinstance(model_info, Mapping) else None attempted_retries: Final = request_metadata.get("attempted_retries") + exception_string: Final = str(e) attempt_record: Final[RetryAttemptRecord] = { "model_group": model_group if isinstance(model_group, str) else None, "deployment_id": deployment_id if isinstance(deployment_id, str) else None, "exception_type": type(e).__name__, - "exception_string": str(e), + "exception_string": ( + exception_string[:RETRY_BREADCRUMB_EXCEPTION_LIMIT] + + f"... [truncated, {len(exception_string)} chars]" + if len(exception_string) > RETRY_BREADCRUMB_EXCEPTION_LIMIT + else exception_string + ), "attempted_retries": attempted_retries if type(attempted_retries) is int else None, } earlier_breadcrumbs: Final = request_metadata.get("previous_models") diff --git a/tests/test_litellm/router/test_retry_breadcrumbs.py b/tests/test_litellm/router/test_retry_breadcrumbs.py new file mode 100644 index 00000000000..b5507467d5b --- /dev/null +++ b/tests/test_litellm/router/test_retry_breadcrumbs.py @@ -0,0 +1,45 @@ +"""Retry breadcrumbs must stay small: they are stored on the request record.""" + +from litellm.router import RETRY_BREADCRUMB_EXCEPTION_LIMIT, Router + + +def _record_for(error: Exception) -> dict: + kwargs = {"model": "gpt-4o", "metadata": {}} + Router.log_retry(object(), kwargs, error) + return kwargs["metadata"]["previous_models"][-1] + + +def test_large_exception_message_is_truncated_in_the_breadcrumb(): + """Litellm exceptions embed the upstream body, so str(e) can be hundreds of KB. + + The release-gate test test_failing_requests_do_not_grow_rss_or_stored_request + saw the proxy RSS grow 86.8 MB over 300 failing requests against a 48 MB budget. + """ + error = Exception("Error code: 500 - " + "x" * 200_000) + + record = _record_for(error) + + assert len(record["exception_string"]) < RETRY_BREADCRUMB_EXCEPTION_LIMIT + 100 + assert record["exception_string"].startswith("Error code: 500 - ") + assert "truncated" in record["exception_string"] + assert record["exception_type"] == "Exception" + + +def test_short_exception_message_is_kept_whole(): + record = _record_for(ValueError("bad request")) + + assert record["exception_string"] == "bad request" + assert record["exception_type"] == "ValueError" + + +def test_breadcrumb_metadata_stays_bounded_across_requests(): + error = Exception("Error code: 500 - " + "x" * 200_000) + + total = 0 + for _ in range(50): + kwargs = {"model": "gpt-4o", "metadata": {}} + for _ in range(6): + Router.log_retry(object(), kwargs, error) + total += len(repr(kwargs["metadata"])) + + assert total < 500_000