fix(router): keep retry breadcrumbs small enough to stay in budget

RETRY_BREADCRUMB_LIMIT bounds how many breadcrumbs a request keeps, but each one
stored str(e) whole, and litellm exceptions embed the upstream response body: a
200 KB error message produced an 800 KB retry record, and 300 failing requests
grew the proxy RSS by 86.8 MB against the 48 MB budget of
test_failing_requests_do_not_grow_rss_or_stored_request.

Keep the diagnostic head of the message and append how much was dropped.
This commit is contained in:
sclfcz 2026-09-27 16:00:24 +08:00
parent c02f11dd7e
commit 6066171c60
2 changed files with 58 additions and 1 deletions

View file

@ -694,6 +694,12 @@ set_live_deployment_replay(_replay_live_router_model_cost)
RETRY_BREADCRUMB_LIMIT: Final = 4
# Litellm exceptions embed the upstream response body, so a single retry record
# could hold hundreds of kilobytes: the release-gate test
# test_failing_requests_do_not_grow_rss_or_stored_request saw the proxy RSS grow
# 86.8 MB over 300 failing requests against a 48 MB budget. Only the diagnostic
# head of the message is kept.
RETRY_BREADCRUMB_EXCEPTION_LIMIT: Final = 500
class FallbackAwareStreamWrapper(CustomStreamWrapper):
@ -8296,11 +8302,17 @@ class Router:
model_info: Final = request_metadata.get("model_info")
deployment_id: Final = model_info.get("id") if isinstance(model_info, Mapping) else None
attempted_retries: Final = request_metadata.get("attempted_retries")
exception_string: Final = str(e)
attempt_record: Final[RetryAttemptRecord] = {
"model_group": model_group if isinstance(model_group, str) else None,
"deployment_id": deployment_id if isinstance(deployment_id, str) else None,
"exception_type": type(e).__name__,
"exception_string": str(e),
"exception_string": (
exception_string[:RETRY_BREADCRUMB_EXCEPTION_LIMIT]
+ f"... [truncated, {len(exception_string)} chars]"
if len(exception_string) > RETRY_BREADCRUMB_EXCEPTION_LIMIT
else exception_string
),
"attempted_retries": attempted_retries if type(attempted_retries) is int else None,
}
earlier_breadcrumbs: Final = request_metadata.get("previous_models")

View file

@ -0,0 +1,45 @@
"""Retry breadcrumbs must stay small: they are stored on the request record."""
from litellm.router import RETRY_BREADCRUMB_EXCEPTION_LIMIT, Router
def _record_for(error: Exception) -> dict:
kwargs = {"model": "gpt-4o", "metadata": {}}
Router.log_retry(object(), kwargs, error)
return kwargs["metadata"]["previous_models"][-1]
def test_large_exception_message_is_truncated_in_the_breadcrumb():
"""Litellm exceptions embed the upstream body, so str(e) can be hundreds of KB.
The release-gate test test_failing_requests_do_not_grow_rss_or_stored_request
saw the proxy RSS grow 86.8 MB over 300 failing requests against a 48 MB budget.
"""
error = Exception("Error code: 500 - " + "x" * 200_000)
record = _record_for(error)
assert len(record["exception_string"]) < RETRY_BREADCRUMB_EXCEPTION_LIMIT + 100
assert record["exception_string"].startswith("Error code: 500 - ")
assert "truncated" in record["exception_string"]
assert record["exception_type"] == "Exception"
def test_short_exception_message_is_kept_whole():
record = _record_for(ValueError("bad request"))
assert record["exception_string"] == "bad request"
assert record["exception_type"] == "ValueError"
def test_breadcrumb_metadata_stays_bounded_across_requests():
error = Exception("Error code: 500 - " + "x" * 200_000)
total = 0
for _ in range(50):
kwargs = {"model": "gpt-4o", "metadata": {}}
for _ in range(6):
Router.log_retry(object(), kwargs, error)
total += len(repr(kwargs["metadata"]))
assert total < 500_000