From 077fc62e92d6565c6c63671414197c944513626a Mon Sep 17 00:00:00 2001 From: RosieOh <20172207@gm.hannam.ac.kr> Date: Tue, 22 Sep 2026 17:10:01 +0900 Subject: [PATCH 1/5] fix(router): skip retry backoff when a fallback can take over --- litellm/router.py | 18 ++++++- .../unit/test_router_retry_backoff_headers.py | 54 +++++++++++++++++++ 2 files changed, 70 insertions(+), 2 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 8aaf58d5a3e..93400951bac 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7776,6 +7776,9 @@ class Router: raise verbose_router_logger.debug("Retrying request with num_retries: %s", num_retries) + fallback_available: Final = self._regular_fallback_available( + fallbacks=fallbacks, model_group=model_group, kwargs=kwargs + ) # decides how long to sleep before retry retry_after: Final = self._time_to_sleep_before_retry( e=original_exception, @@ -7783,6 +7786,7 @@ class Router: num_retries=num_retries, healthy_deployments=_healthy_deployments, all_deployments=_all_deployments, + fallback_available=fallback_available, ) await asyncio.sleep(retry_after) @@ -7853,6 +7857,7 @@ class Router: num_retries=num_retries, healthy_deployments=_healthy_deployments, all_deployments=_all_deployments, + fallback_available=fallback_available, ) await asyncio.sleep(_timeout) @@ -8010,6 +8015,7 @@ class Router: num_retries: int, healthy_deployments: list | None = None, all_deployments: list | None = None, + fallback_available: bool = False, ) -> int | float: """ Calculate back-off, then retry @@ -8018,6 +8024,8 @@ class Router: 1. there are healthy deployments in the same model group 2. there are fallbacks for the completion call """ + if fallback_available: + return 0 ## base case - single deployment if all_deployments is not None and len(all_deployments) == 1: @@ -8482,8 +8490,14 @@ class Router: return self._has_content_policy_fallback(model_group, kwargs) if self._has_default_fallbacks(): return True - fallbacks: Final = kwargs.get("fallbacks", self.fallbacks) - if fallbacks is None: + return self._regular_fallback_available( + fallbacks=kwargs.get("fallbacks", self.fallbacks), model_group=model_group, kwargs=kwargs + ) + + def _regular_fallback_available( + self, fallbacks: list | None, model_group: str | None, kwargs: Mapping[str, Any] + ) -> bool: + if fallbacks is None or fallbacks_disabled_for_request(kwargs): return False resolved, _ = get_fallback_model_group_for_lookup_groups( fallbacks=fallbacks, diff --git a/tests/unit/test_router_retry_backoff_headers.py b/tests/unit/test_router_retry_backoff_headers.py index 03c3af692ce..e44e7855925 100644 --- a/tests/unit/test_router_retry_backoff_headers.py +++ b/tests/unit/test_router_retry_backoff_headers.py @@ -2,6 +2,8 @@ Tests for router retry backoff behavior. """ +import asyncio +from typing import Final from unittest.mock import patch import httpx @@ -9,6 +11,58 @@ import pytest import litellm from litellm import Router +from litellm.constants import MAX_RETRY_DELAY + +_BACKOFF_DETECTION_TIMEOUT: Final = MAX_RETRY_DELAY / 4 + + +def _router_with_single_failing_deployment(fallbacks: list[dict[str, list[str]]]) -> Router: + return Router( + model_list=[ + { + "model_name": "primary", + "litellm_params": { + "model": "openai/gpt-5.4-mini", + "api_key": "sk-test", + "mock_response": "litellm.InternalServerError", + }, + }, + { + "model_name": "backup", + "litellm_params": { + "model": "openai/gpt-5.4-mini", + "api_key": "sk-test", + "mock_response": "answered by backup", + }, + }, + ], + num_retries=2, + retry_after=int(MAX_RETRY_DELAY), + fallbacks=fallbacks, + ) + + +@pytest.mark.asyncio +async def test_single_deployment_group_with_fallback_does_not_back_off_before_falling_back(): + router: Final = _router_with_single_failing_deployment(fallbacks=[{"primary": ["backup"]}]) + + response: Final = await asyncio.wait_for( + router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]), + timeout=_BACKOFF_DETECTION_TIMEOUT, + ) + + assert response.choices[0].message.content == "answered by backup" + + +@pytest.mark.asyncio +async def test_fallback_configured_for_another_group_keeps_retry_backoff(): + router: Final = _router_with_single_failing_deployment(fallbacks=[{"backup": ["primary"]}]) + + with pytest.raises(asyncio.TimeoutError): + await asyncio.wait_for( + router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]), + timeout=_BACKOFF_DETECTION_TIMEOUT, + ) @pytest.mark.asyncio From e0c17839758c68808988aa8b9135693b36d43151 Mon Sep 17 00:00:00 2001 From: RosieOh <20172207@gm.hannam.ac.kr> Date: Tue, 22 Sep 2026 18:12:40 +0900 Subject: [PATCH 2/5] fix(router): check the fallback the dispatcher will use before skipping backoff --- litellm/router.py | 56 +++++++- .../unit/test_router_retry_backoff_headers.py | 121 +++++++++++++++++- 2 files changed, 166 insertions(+), 11 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 93400951bac..53301979b2f 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7776,9 +7776,6 @@ class Router: raise verbose_router_logger.debug("Retrying request with num_retries: %s", num_retries) - fallback_available: Final = self._regular_fallback_available( - fallbacks=fallbacks, model_group=model_group, kwargs=kwargs - ) # decides how long to sleep before retry retry_after: Final = self._time_to_sleep_before_retry( e=original_exception, @@ -7786,7 +7783,14 @@ class Router: num_retries=num_retries, healthy_deployments=_healthy_deployments, all_deployments=_all_deployments, - fallback_available=fallback_available, + fallback_available=self._fallback_available_for_error( + error=original_exception, + fallbacks=fallbacks, + context_window_fallbacks=context_window_fallbacks, + content_policy_fallbacks=content_policy_fallbacks, + model_group=model_group, + kwargs=kwargs, + ), ) await asyncio.sleep(retry_after) @@ -7857,7 +7861,14 @@ class Router: num_retries=num_retries, healthy_deployments=_healthy_deployments, all_deployments=_all_deployments, - fallback_available=fallback_available, + fallback_available=self._fallback_available_for_error( + error=e, + fallbacks=fallbacks, + context_window_fallbacks=context_window_fallbacks, + content_policy_fallbacks=content_policy_fallbacks, + model_group=model_group, + kwargs=kwargs, + ), ) await asyncio.sleep(_timeout) @@ -8555,6 +8566,41 @@ class Router: ) return has_unattempted_fallback_target(resolved, kwargs) + def _fallback_available_for_error( + self, + error: Exception, + fallbacks: list | None, + context_window_fallbacks: list | None, + content_policy_fallbacks: list | None, + model_group: str | None, + kwargs: Mapping[str, Any], + ) -> bool: + """ + Whether async_function_with_fallbacks_common_utils would hand this error to an untried + fallback, checked in the order it dispatches: client-side lists, then the dedicated + context-window or content-policy list (authoritative once set), then regular fallbacks + """ + if model_group is None or fallbacks_disabled_for_request(kwargs): + return False + if _check_non_standard_fallback_format(fallbacks=fallbacks): + return has_unattempted_fallback_target(fallbacks, kwargs) + dedicated_fallbacks: Final = ( + context_window_fallbacks + if isinstance(error, litellm.ContextWindowExceededError) + else content_policy_fallbacks + if isinstance(error, litellm.ContentPolicyViolationError) + else None + ) + if dedicated_fallbacks is not None: + return has_unattempted_fallback_target( + self._get_fallback_model_group_for_lookup_groups( + fallbacks=dedicated_fallbacks, + lookup_groups=fallback_lookup_groups(kwargs, model_group), + ), + kwargs, + ) + return self._regular_fallback_available(fallbacks=fallbacks, model_group=model_group, kwargs=kwargs) + def _should_raise_content_policy_error(self, model: str, response: ModelResponse, kwargs: dict) -> bool: """ Determines if a content policy error should be raised. diff --git a/tests/unit/test_router_retry_backoff_headers.py b/tests/unit/test_router_retry_backoff_headers.py index e44e7855925..7d1c416521a 100644 --- a/tests/unit/test_router_retry_backoff_headers.py +++ b/tests/unit/test_router_retry_backoff_headers.py @@ -12,20 +12,28 @@ import pytest import litellm from litellm import Router from litellm.constants import MAX_RETRY_DELAY +from litellm.router_utils.fallback_event_handlers import AttemptedFallbackTargets, record_disable_fallbacks +from litellm.types.router import RetryPolicy _BACKOFF_DETECTION_TIMEOUT: Final = MAX_RETRY_DELAY / 4 +_SERVER_ERROR: Final = litellm.InternalServerError(message="provider down", model="gpt-5.4-mini", llm_provider="openai") +_CONTENT_POLICY_ERROR: Final = litellm.ContentPolicyViolationError( + message="flagged", model="gpt-5.4-mini", llm_provider="openai" +) -def _router_with_single_failing_deployment(fallbacks: list[dict[str, list[str]]]) -> Router: +def _router_with_single_failing_deployment( + fallbacks: list[dict[str, list[str]]], + primary_error: str | None = "litellm.InternalServerError", + content_policy_fallbacks: list[dict[str, list[str]]] | None = None, + retry_policy: RetryPolicy | None = None, +) -> Router: return Router( model_list=[ { "model_name": "primary", - "litellm_params": { - "model": "openai/gpt-5.4-mini", - "api_key": "sk-test", - "mock_response": "litellm.InternalServerError", - }, + "litellm_params": {"model": "openai/gpt-5.4-mini", "api_key": "sk-test"} + | ({"mock_response": primary_error} if primary_error is not None else {}), }, { "model_name": "backup", @@ -39,6 +47,8 @@ def _router_with_single_failing_deployment(fallbacks: list[dict[str, list[str]]] num_retries=2, retry_after=int(MAX_RETRY_DELAY), fallbacks=fallbacks, + content_policy_fallbacks=content_policy_fallbacks, + retry_policy=retry_policy, ) @@ -65,6 +75,105 @@ async def test_fallback_configured_for_another_group_keeps_retry_backoff(): ) +@pytest.mark.asyncio +async def test_client_side_fallback_list_does_not_back_off_before_falling_back(): + router: Final = _router_with_single_failing_deployment(fallbacks=[]) + + response: Final = await asyncio.wait_for( + router.acompletion( + model="primary", + messages=[{"role": "user", "content": "Hello"}], + fallbacks=[{"model": "backup"}], + ), + timeout=_BACKOFF_DETECTION_TIMEOUT, + ) + + assert response.choices[0].message.content == "answered by backup" + + +@pytest.mark.asyncio +async def test_content_policy_error_keeps_backoff_when_its_dedicated_fallbacks_skip_the_group(): + router: Final = _router_with_single_failing_deployment( + fallbacks=[{"primary": ["backup"]}], + primary_error=None, + content_policy_fallbacks=[{"backup": ["primary"]}], + retry_policy=RetryPolicy(ContentPolicyViolationErrorRetries=2), + ) + + with pytest.raises(asyncio.TimeoutError): + await asyncio.wait_for( + router.acompletion( + model="primary", + messages=[{"role": "user", "content": "Hello"}], + mock_response=_CONTENT_POLICY_ERROR, + ), + timeout=_BACKOFF_DETECTION_TIMEOUT, + ) + + +@pytest.mark.parametrize( + ("error", "fallbacks", "content_policy_fallbacks", "expected"), + [ + pytest.param(_SERVER_ERROR, [{"primary": ["backup"]}], None, True, id="own-chain"), + pytest.param(_SERVER_ERROR, [{"backup": ["primary"]}], None, False, id="chain-for-another-group"), + pytest.param(_SERVER_ERROR, [{"*": ["backup"]}], None, True, id="generic-chain"), + pytest.param(_SERVER_ERROR, [{"model": "backup"}], None, True, id="client-side-list"), + pytest.param( + _CONTENT_POLICY_ERROR, + [{"primary": ["backup"]}], + [{"backup": ["primary"]}], + False, + id="dedicated-list-skips-group", + ), + pytest.param( + _CONTENT_POLICY_ERROR, + [{"backup": ["primary"]}], + [{"primary": ["backup"]}], + True, + id="dedicated-list-covers-group", + ), + pytest.param(_CONTENT_POLICY_ERROR, [{"primary": ["backup"]}], None, True, id="no-dedicated-list"), + ], +) +def test_fallback_available_for_error_follows_the_dispatch_order( + error: Exception, + fallbacks: list[dict[str, object]], + content_policy_fallbacks: list[dict[str, list[str]]] | None, + expected: bool, +): + router: Final = _router_with_single_failing_deployment(fallbacks=[]) + + available: Final = router._fallback_available_for_error( + error=error, + fallbacks=fallbacks, + context_window_fallbacks=None, + content_policy_fallbacks=content_policy_fallbacks, + model_group="primary", + kwargs={"model": "primary"}, + ) + + assert available is expected + + +def test_regular_fallback_available_is_false_once_the_chain_is_used_up_or_disabled(): + router: Final = _router_with_single_failing_deployment(fallbacks=[]) + chain: Final = [{"primary": ["backup"]}] + disabled_kwargs: Final = {"model": "primary", "metadata": {}} + record_disable_fallbacks(disabled_kwargs, True) + + fresh: Final = router._regular_fallback_available( + fallbacks=chain, model_group="primary", kwargs={"model": "primary"} + ) + used_up: Final = router._regular_fallback_available( + fallbacks=chain, + model_group="primary", + kwargs={"model": "primary", "attempted_targets": AttemptedFallbackTargets(keys=frozenset({"backup"}))}, + ) + disabled: Final = router._regular_fallback_available(fallbacks=chain, model_group="primary", kwargs=disabled_kwargs) + + assert (fresh, used_up, disabled) == (True, False, False) + + @pytest.mark.asyncio async def test_retry_backoff_uses_current_exception_headers(): """ From a9b0eaf44b6b428fe6aec0781e87b2decfc5c642 Mon Sep 17 00:00:00 2001 From: RosieOh <20172207@gm.hannam.ac.kr> Date: Tue, 22 Sep 2026 18:31:59 +0900 Subject: [PATCH 3/5] test(router): cover the no-model-group, disabled, and context-window fallback checks --- .../unit/test_router_retry_backoff_headers.py | 54 +++++++++++++++---- 1 file changed, 45 insertions(+), 9 deletions(-) diff --git a/tests/unit/test_router_retry_backoff_headers.py b/tests/unit/test_router_retry_backoff_headers.py index 7d1c416521a..75413ca633c 100644 --- a/tests/unit/test_router_retry_backoff_headers.py +++ b/tests/unit/test_router_retry_backoff_headers.py @@ -20,6 +20,9 @@ _SERVER_ERROR: Final = litellm.InternalServerError(message="provider down", mode _CONTENT_POLICY_ERROR: Final = litellm.ContentPolicyViolationError( message="flagged", model="gpt-5.4-mini", llm_provider="openai" ) +_CONTEXT_WINDOW_ERROR: Final = litellm.ContextWindowExceededError( + message="too long", model="gpt-5.4-mini", llm_provider="openai" +) def _router_with_single_failing_deployment( @@ -112,32 +115,43 @@ async def test_content_policy_error_keeps_backoff_when_its_dedicated_fallbacks_s @pytest.mark.parametrize( - ("error", "fallbacks", "content_policy_fallbacks", "expected"), + ("error", "fallbacks", "context_window_fallbacks", "content_policy_fallbacks", "expected"), [ - pytest.param(_SERVER_ERROR, [{"primary": ["backup"]}], None, True, id="own-chain"), - pytest.param(_SERVER_ERROR, [{"backup": ["primary"]}], None, False, id="chain-for-another-group"), - pytest.param(_SERVER_ERROR, [{"*": ["backup"]}], None, True, id="generic-chain"), - pytest.param(_SERVER_ERROR, [{"model": "backup"}], None, True, id="client-side-list"), + pytest.param(_SERVER_ERROR, [{"primary": ["backup"]}], None, None, True, id="own-chain"), + pytest.param(_SERVER_ERROR, [{"backup": ["primary"]}], None, None, False, id="chain-for-another-group"), + pytest.param(_SERVER_ERROR, [{"*": ["backup"]}], None, None, True, id="generic-chain"), + pytest.param(_SERVER_ERROR, [{"model": "backup"}], None, None, True, id="client-side-list"), pytest.param( _CONTENT_POLICY_ERROR, [{"primary": ["backup"]}], + None, [{"backup": ["primary"]}], False, - id="dedicated-list-skips-group", + id="content-policy-list-skips-group", ), pytest.param( _CONTENT_POLICY_ERROR, [{"backup": ["primary"]}], + None, [{"primary": ["backup"]}], True, - id="dedicated-list-covers-group", + id="content-policy-list-covers-group", ), - pytest.param(_CONTENT_POLICY_ERROR, [{"primary": ["backup"]}], None, True, id="no-dedicated-list"), + pytest.param( + _CONTEXT_WINDOW_ERROR, + [{"primary": ["backup"]}], + [{"backup": ["primary"]}], + None, + False, + id="context-window-list-skips-group", + ), + pytest.param(_CONTENT_POLICY_ERROR, [{"primary": ["backup"]}], None, None, True, id="no-dedicated-list"), ], ) def test_fallback_available_for_error_follows_the_dispatch_order( error: Exception, fallbacks: list[dict[str, object]], + context_window_fallbacks: list[dict[str, list[str]]] | None, content_policy_fallbacks: list[dict[str, list[str]]] | None, expected: bool, ): @@ -146,7 +160,7 @@ def test_fallback_available_for_error_follows_the_dispatch_order( available: Final = router._fallback_available_for_error( error=error, fallbacks=fallbacks, - context_window_fallbacks=None, + context_window_fallbacks=context_window_fallbacks, content_policy_fallbacks=content_policy_fallbacks, model_group="primary", kwargs={"model": "primary"}, @@ -155,6 +169,28 @@ def test_fallback_available_for_error_follows_the_dispatch_order( assert available is expected +def test_fallback_available_for_error_is_false_without_a_model_group_or_when_disabled(): + router: Final = _router_with_single_failing_deployment(fallbacks=[]) + disabled_kwargs: Final = {"model": "primary", "metadata": {}} + record_disable_fallbacks(disabled_kwargs, True) + + def available(model_group: str | None, kwargs: dict[str, object]) -> bool: + return router._fallback_available_for_error( + error=_SERVER_ERROR, + fallbacks=[{"model": "backup"}], + context_window_fallbacks=None, + content_policy_fallbacks=None, + model_group=model_group, + kwargs=kwargs, + ) + + assert (available("primary", {"model": "primary"}), available(None, {}), available("primary", disabled_kwargs)) == ( + True, + False, + False, + ) + + def test_regular_fallback_available_is_false_once_the_chain_is_used_up_or_disabled(): router: Final = _router_with_single_failing_deployment(fallbacks=[]) chain: Final = [{"primary": ["backup"]}] From f48bd7fd298a459806c967347da12dc72466497b Mon Sep 17 00:00:00 2001 From: RosieOh <20172207@gm.hannam.ac.kr> Date: Wed, 30 Sep 2026 13:01:48 +0900 Subject: [PATCH 4/5] fix(router): keep retry backoff when the resolved fallback chain is empty --- litellm/router.py | 15 +++++++++++--- .../unit/test_router_retry_backoff_headers.py | 20 +++++++++++++++++++ 2 files changed, 32 insertions(+), 3 deletions(-) diff --git a/litellm/router.py b/litellm/router.py index 53301979b2f..4b58629e38b 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -475,6 +475,15 @@ _SESSION_ADAPTER: Final = TypeAdapter(Mapping[str, object]) _SILENT_MODEL_ADAPTER: Final = TypeAdapter(str | list[str]) +def _untried_fallback_target_exists(chain: Sequence[object] | None, kwargs: Mapping[str, Any]) -> bool: + """ + Whether a resolved fallback chain still names a target this request can try. + `has_unattempted_fallback_target` reports an unattempted target for an empty chain + whenever the request carries no attempt record yet, so guard the empty case here. + """ + return bool(chain) and has_unattempted_fallback_target(chain, kwargs) + + def _as_retry_skipped_deployment_ids(value: object) -> tuple[str, ...]: return tuple(item for item in value if isinstance(item, str)) if isinstance(value, tuple) else () @@ -8514,7 +8523,7 @@ class Router: fallbacks=fallbacks, lookup_groups=fallback_lookup_groups(kwargs, model_group), ) - return has_unattempted_fallback_target(resolved, kwargs) + return _untried_fallback_target_exists(resolved, kwargs) def _anthropic_messages_order_levels(self, model_group: str, kwargs: Mapping[str, Any]) -> tuple[int, ...]: """ @@ -8583,7 +8592,7 @@ class Router: if model_group is None or fallbacks_disabled_for_request(kwargs): return False if _check_non_standard_fallback_format(fallbacks=fallbacks): - return has_unattempted_fallback_target(fallbacks, kwargs) + return _untried_fallback_target_exists(fallbacks, kwargs) dedicated_fallbacks: Final = ( context_window_fallbacks if isinstance(error, litellm.ContextWindowExceededError) @@ -8592,7 +8601,7 @@ class Router: else None ) if dedicated_fallbacks is not None: - return has_unattempted_fallback_target( + return _untried_fallback_target_exists( self._get_fallback_model_group_for_lookup_groups( fallbacks=dedicated_fallbacks, lookup_groups=fallback_lookup_groups(kwargs, model_group), diff --git a/tests/unit/test_router_retry_backoff_headers.py b/tests/unit/test_router_retry_backoff_headers.py index 75413ca633c..487fac08172 100644 --- a/tests/unit/test_router_retry_backoff_headers.py +++ b/tests/unit/test_router_retry_backoff_headers.py @@ -12,6 +12,7 @@ import pytest import litellm from litellm import Router from litellm.constants import MAX_RETRY_DELAY +from litellm.router import _untried_fallback_target_exists from litellm.router_utils.fallback_event_handlers import AttemptedFallbackTargets, record_disable_fallbacks from litellm.types.router import RetryPolicy @@ -78,6 +79,25 @@ async def test_fallback_configured_for_another_group_keeps_retry_backoff(): ) +@pytest.mark.asyncio +async def test_empty_fallback_chain_keeps_retry_backoff(): + """A chain configured as an empty list names no target, so the request has nothing to + fall back to and the retries must still space themselves out.""" + router: Final = _router_with_single_failing_deployment(fallbacks=[{"primary": []}]) + + with pytest.raises(asyncio.TimeoutError): + await asyncio.wait_for( + router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]), + timeout=_BACKOFF_DETECTION_TIMEOUT, + ) + + +def test_untried_fallback_target_exists_is_false_for_a_chain_with_no_entries(): + assert _untried_fallback_target_exists(["backup"], {}) is True + assert _untried_fallback_target_exists([], {}) is False + assert _untried_fallback_target_exists(None, {}) is False + + @pytest.mark.asyncio async def test_client_side_fallback_list_does_not_back_off_before_falling_back(): router: Final = _router_with_single_failing_deployment(fallbacks=[]) From fca7a3dce4c9234eeea169f75c8f6800c87805d0 Mon Sep 17 00:00:00 2001 From: RosieOh <20172207@gm.hannam.ac.kr> Date: Wed, 30 Sep 2026 14:16:28 +0900 Subject: [PATCH 5/5] test(router): assert the chosen retry delay instead of waiting on the clock --- .../unit/test_router_retry_backoff_headers.py | 67 +++++++++++-------- 1 file changed, 38 insertions(+), 29 deletions(-) diff --git a/tests/unit/test_router_retry_backoff_headers.py b/tests/unit/test_router_retry_backoff_headers.py index 487fac08172..6eca4ea5d12 100644 --- a/tests/unit/test_router_retry_backoff_headers.py +++ b/tests/unit/test_router_retry_backoff_headers.py @@ -4,7 +4,7 @@ Tests for router retry backoff behavior. import asyncio from typing import Final -from unittest.mock import patch +from unittest.mock import AsyncMock, patch import httpx import pytest @@ -16,7 +16,6 @@ from litellm.router import _untried_fallback_target_exists from litellm.router_utils.fallback_event_handlers import AttemptedFallbackTargets, record_disable_fallbacks from litellm.types.router import RetryPolicy -_BACKOFF_DETECTION_TIMEOUT: Final = MAX_RETRY_DELAY / 4 _SERVER_ERROR: Final = litellm.InternalServerError(message="provider down", model="gpt-5.4-mini", llm_provider="openai") _CONTENT_POLICY_ERROR: Final = litellm.ContentPolicyViolationError( message="flagged", model="gpt-5.4-mini", llm_provider="openai" @@ -56,27 +55,35 @@ def _router_with_single_failing_deployment( ) +def _backoff_delays(sleeper: AsyncMock) -> tuple[float, ...]: + """ + The delays the router asked to wait between retries. Reading its decision keeps these + tests off the clock, which tests/unit rules out, and off the CI scheduler's timing. + """ + return tuple(call.args[0] for call in sleeper.await_args_list if call.args) + + @pytest.mark.asyncio async def test_single_deployment_group_with_fallback_does_not_back_off_before_falling_back(): router: Final = _router_with_single_failing_deployment(fallbacks=[{"primary": ["backup"]}]) + sleeper: Final = AsyncMock() - response: Final = await asyncio.wait_for( - router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]), - timeout=_BACKOFF_DETECTION_TIMEOUT, - ) + with patch.object(asyncio, "sleep", sleeper): + response: Final = await router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]) assert response.choices[0].message.content == "answered by backup" + assert not any(delay > 0 for delay in _backoff_delays(sleeper)) @pytest.mark.asyncio async def test_fallback_configured_for_another_group_keeps_retry_backoff(): router: Final = _router_with_single_failing_deployment(fallbacks=[{"backup": ["primary"]}]) + sleeper: Final = AsyncMock() - with pytest.raises(asyncio.TimeoutError): - await asyncio.wait_for( - router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]), - timeout=_BACKOFF_DETECTION_TIMEOUT, - ) + with patch.object(asyncio, "sleep", sleeper), pytest.raises(litellm.InternalServerError): + await router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]) + + assert any(delay > 0 for delay in _backoff_delays(sleeper)) @pytest.mark.asyncio @@ -84,12 +91,12 @@ async def test_empty_fallback_chain_keeps_retry_backoff(): """A chain configured as an empty list names no target, so the request has nothing to fall back to and the retries must still space themselves out.""" router: Final = _router_with_single_failing_deployment(fallbacks=[{"primary": []}]) + sleeper: Final = AsyncMock() - with pytest.raises(asyncio.TimeoutError): - await asyncio.wait_for( - router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]), - timeout=_BACKOFF_DETECTION_TIMEOUT, - ) + with patch.object(asyncio, "sleep", sleeper), pytest.raises(litellm.InternalServerError): + await router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]) + + assert any(delay > 0 for delay in _backoff_delays(sleeper)) def test_untried_fallback_target_exists_is_false_for_a_chain_with_no_entries(): @@ -102,16 +109,17 @@ def test_untried_fallback_target_exists_is_false_for_a_chain_with_no_entries(): async def test_client_side_fallback_list_does_not_back_off_before_falling_back(): router: Final = _router_with_single_failing_deployment(fallbacks=[]) - response: Final = await asyncio.wait_for( - router.acompletion( + sleeper: Final = AsyncMock() + + with patch.object(asyncio, "sleep", sleeper): + response: Final = await router.acompletion( model="primary", messages=[{"role": "user", "content": "Hello"}], fallbacks=[{"model": "backup"}], - ), - timeout=_BACKOFF_DETECTION_TIMEOUT, - ) + ) assert response.choices[0].message.content == "answered by backup" + assert not any(delay > 0 for delay in _backoff_delays(sleeper)) @pytest.mark.asyncio @@ -123,16 +131,17 @@ async def test_content_policy_error_keeps_backoff_when_its_dedicated_fallbacks_s retry_policy=RetryPolicy(ContentPolicyViolationErrorRetries=2), ) - with pytest.raises(asyncio.TimeoutError): - await asyncio.wait_for( - router.acompletion( - model="primary", - messages=[{"role": "user", "content": "Hello"}], - mock_response=_CONTENT_POLICY_ERROR, - ), - timeout=_BACKOFF_DETECTION_TIMEOUT, + sleeper: Final = AsyncMock() + + with patch.object(asyncio, "sleep", sleeper), pytest.raises(litellm.ContentPolicyViolationError): + await router.acompletion( + model="primary", + messages=[{"role": "user", "content": "Hello"}], + mock_response=_CONTENT_POLICY_ERROR, ) + assert any(delay > 0 for delay in _backoff_delays(sleeper)) + @pytest.mark.parametrize( ("error", "fallbacks", "context_window_fallbacks", "content_policy_fallbacks", "expected"),