mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
test(router): expect the exact error per retry-cap case and drop explanatory docstrings
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
555e321cf1
commit
566da87771
3 changed files with 3 additions and 11 deletions
|
|
@ -304,9 +304,6 @@ def get_metadata_variable_name_from_kwargs(
|
|||
|
||||
|
||||
def max_retries_per_request_hit(kwargs: Mapping[str, object], num_retries_per_request: int | None) -> bool:
|
||||
"""
|
||||
Whether the Router retry about to run (``attempted_retries`` >= 1 in the metadata bucket) is past the cap
|
||||
"""
|
||||
if num_retries_per_request is None:
|
||||
return False
|
||||
metadata: Final = kwargs.get(get_metadata_variable_name_from_kwargs(kwargs))
|
||||
|
|
|
|||
|
|
@ -10746,12 +10746,12 @@ def _always_failing_router(num_retries):
|
|||
)
|
||||
|
||||
|
||||
async def _fail_one_proxy_shaped_request(router, request_marker):
|
||||
async def _fail_one_proxy_shaped_request(router, request_marker, expected_error=litellm.InternalServerError):
|
||||
"""The proxy hands the router a metadata dict and a proxy_server_request whose body is a
|
||||
shallow copy of the request, so body["metadata"] is the very same dict the router later
|
||||
stamps previous_models onto."""
|
||||
metadata = {"request_marker": request_marker}
|
||||
with pytest.raises((litellm.InternalServerError, litellm.APIConnectionError)):
|
||||
with pytest.raises(expected_error):
|
||||
await router.acompletion(
|
||||
model="broken-group",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
|
|
@ -10808,13 +10808,10 @@ async def test_retry_records_keep_only_the_last_four_attempts():
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_num_retries_per_request_stops_retries_at_caps_above_four(monkeypatch):
|
||||
"""The cap used to be read off len(previous_models), which never exceeds four, so any cap above
|
||||
four was inert. Reading the Router's attempted_retries counter instead lets a cap of five refuse
|
||||
retries five and six before they reach the deployment."""
|
||||
monkeypatch.setattr(litellm, "num_retries_per_request", 5)
|
||||
router = _always_failing_router(num_retries=6)
|
||||
|
||||
records = await _fail_one_proxy_shaped_request(router, "request-1")
|
||||
records = await _fail_one_proxy_shaped_request(router, "request-1", expected_error=litellm.APIConnectionError)
|
||||
|
||||
assert [record["attempted_retries"] for record in records] == [3, 4, 5, 6]
|
||||
assert ["Max retries per request hit!" in record["exception_string"] for record in records] == [
|
||||
|
|
|
|||
|
|
@ -4084,8 +4084,6 @@ def _capped_completion_kwargs(metadata_key: str, metadata: object) -> dict[str,
|
|||
@pytest.mark.parametrize("metadata_key", ["metadata", "litellm_metadata"])
|
||||
@pytest.mark.parametrize("cap, metadata, refused", _RETRY_CAP_CASES)
|
||||
def test_num_retries_per_request_reads_attempted_retries_sync(monkeypatch, metadata_key, cap, metadata, refused):
|
||||
"""num_retries_per_request is enforced from the Router's attempted_retries counter in whichever
|
||||
metadata bucket the call carries, so callers on litellm_metadata and caps above four both work"""
|
||||
monkeypatch.setattr(litellm, "num_retries_per_request", cap)
|
||||
kwargs: Final = _capped_completion_kwargs(metadata_key, metadata)
|
||||
if refused:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue