test(router): expect the exact error per retry-cap case and drop explanatory docstrings

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yassin 2026-09-13 01:20:46 +00:00
parent 555e321cf1
commit 566da87771
3 changed files with 3 additions and 11 deletions

View file

@ -304,9 +304,6 @@ def get_metadata_variable_name_from_kwargs(
def max_retries_per_request_hit(kwargs: Mapping[str, object], num_retries_per_request: int | None) -> bool:
"""
Whether the Router retry about to run (``attempted_retries`` >= 1 in the metadata bucket) is past the cap
"""
if num_retries_per_request is None:
return False
metadata: Final = kwargs.get(get_metadata_variable_name_from_kwargs(kwargs))

View file

@ -10746,12 +10746,12 @@ def _always_failing_router(num_retries):
)
async def _fail_one_proxy_shaped_request(router, request_marker):
async def _fail_one_proxy_shaped_request(router, request_marker, expected_error=litellm.InternalServerError):
"""The proxy hands the router a metadata dict and a proxy_server_request whose body is a
shallow copy of the request, so body["metadata"] is the very same dict the router later
stamps previous_models onto."""
metadata = {"request_marker": request_marker}
with pytest.raises((litellm.InternalServerError, litellm.APIConnectionError)):
with pytest.raises(expected_error):
await router.acompletion(
model="broken-group",
messages=[{"role": "user", "content": "hi"}],
@ -10808,13 +10808,10 @@ async def test_retry_records_keep_only_the_last_four_attempts():
@pytest.mark.asyncio
async def test_num_retries_per_request_stops_retries_at_caps_above_four(monkeypatch):
"""The cap used to be read off len(previous_models), which never exceeds four, so any cap above
four was inert. Reading the Router's attempted_retries counter instead lets a cap of five refuse
retries five and six before they reach the deployment."""
monkeypatch.setattr(litellm, "num_retries_per_request", 5)
router = _always_failing_router(num_retries=6)
records = await _fail_one_proxy_shaped_request(router, "request-1")
records = await _fail_one_proxy_shaped_request(router, "request-1", expected_error=litellm.APIConnectionError)
assert [record["attempted_retries"] for record in records] == [3, 4, 5, 6]
assert ["Max retries per request hit!" in record["exception_string"] for record in records] == [

View file

@ -4084,8 +4084,6 @@ def _capped_completion_kwargs(metadata_key: str, metadata: object) -> dict[str,
@pytest.mark.parametrize("metadata_key", ["metadata", "litellm_metadata"])
@pytest.mark.parametrize("cap, metadata, refused", _RETRY_CAP_CASES)
def test_num_retries_per_request_reads_attempted_retries_sync(monkeypatch, metadata_key, cap, metadata, refused):
"""num_retries_per_request is enforced from the Router's attempted_retries counter in whichever
metadata bucket the call carries, so callers on litellm_metadata and caps above four both work"""
monkeypatch.setattr(litellm, "num_retries_per_request", cap)
kwargs: Final = _capped_completion_kwargs(metadata_key, metadata)
if refused: