mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
test(router): assert the chosen retry delay instead of waiting on the clock
This commit is contained in:
parent
f48bd7fd29
commit
fca7a3dce4
1 changed files with 38 additions and 29 deletions
|
|
@ -4,7 +4,7 @@ Tests for router retry backoff behavior.
|
|||
|
||||
import asyncio
|
||||
from typing import Final
|
||||
from unittest.mock import patch
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
|
@ -16,7 +16,6 @@ from litellm.router import _untried_fallback_target_exists
|
|||
from litellm.router_utils.fallback_event_handlers import AttemptedFallbackTargets, record_disable_fallbacks
|
||||
from litellm.types.router import RetryPolicy
|
||||
|
||||
_BACKOFF_DETECTION_TIMEOUT: Final = MAX_RETRY_DELAY / 4
|
||||
_SERVER_ERROR: Final = litellm.InternalServerError(message="provider down", model="gpt-5.4-mini", llm_provider="openai")
|
||||
_CONTENT_POLICY_ERROR: Final = litellm.ContentPolicyViolationError(
|
||||
message="flagged", model="gpt-5.4-mini", llm_provider="openai"
|
||||
|
|
@ -56,27 +55,35 @@ def _router_with_single_failing_deployment(
|
|||
)
|
||||
|
||||
|
||||
def _backoff_delays(sleeper: AsyncMock) -> tuple[float, ...]:
|
||||
"""
|
||||
The delays the router asked to wait between retries. Reading its decision keeps these
|
||||
tests off the clock, which tests/unit rules out, and off the CI scheduler's timing.
|
||||
"""
|
||||
return tuple(call.args[0] for call in sleeper.await_args_list if call.args)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_single_deployment_group_with_fallback_does_not_back_off_before_falling_back():
|
||||
router: Final = _router_with_single_failing_deployment(fallbacks=[{"primary": ["backup"]}])
|
||||
sleeper: Final = AsyncMock()
|
||||
|
||||
response: Final = await asyncio.wait_for(
|
||||
router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]),
|
||||
timeout=_BACKOFF_DETECTION_TIMEOUT,
|
||||
)
|
||||
with patch.object(asyncio, "sleep", sleeper):
|
||||
response: Final = await router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}])
|
||||
|
||||
assert response.choices[0].message.content == "answered by backup"
|
||||
assert not any(delay > 0 for delay in _backoff_delays(sleeper))
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fallback_configured_for_another_group_keeps_retry_backoff():
|
||||
router: Final = _router_with_single_failing_deployment(fallbacks=[{"backup": ["primary"]}])
|
||||
sleeper: Final = AsyncMock()
|
||||
|
||||
with pytest.raises(asyncio.TimeoutError):
|
||||
await asyncio.wait_for(
|
||||
router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]),
|
||||
timeout=_BACKOFF_DETECTION_TIMEOUT,
|
||||
)
|
||||
with patch.object(asyncio, "sleep", sleeper), pytest.raises(litellm.InternalServerError):
|
||||
await router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}])
|
||||
|
||||
assert any(delay > 0 for delay in _backoff_delays(sleeper))
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -84,12 +91,12 @@ async def test_empty_fallback_chain_keeps_retry_backoff():
|
|||
"""A chain configured as an empty list names no target, so the request has nothing to
|
||||
fall back to and the retries must still space themselves out."""
|
||||
router: Final = _router_with_single_failing_deployment(fallbacks=[{"primary": []}])
|
||||
sleeper: Final = AsyncMock()
|
||||
|
||||
with pytest.raises(asyncio.TimeoutError):
|
||||
await asyncio.wait_for(
|
||||
router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}]),
|
||||
timeout=_BACKOFF_DETECTION_TIMEOUT,
|
||||
)
|
||||
with patch.object(asyncio, "sleep", sleeper), pytest.raises(litellm.InternalServerError):
|
||||
await router.acompletion(model="primary", messages=[{"role": "user", "content": "Hello"}])
|
||||
|
||||
assert any(delay > 0 for delay in _backoff_delays(sleeper))
|
||||
|
||||
|
||||
def test_untried_fallback_target_exists_is_false_for_a_chain_with_no_entries():
|
||||
|
|
@ -102,16 +109,17 @@ def test_untried_fallback_target_exists_is_false_for_a_chain_with_no_entries():
|
|||
async def test_client_side_fallback_list_does_not_back_off_before_falling_back():
|
||||
router: Final = _router_with_single_failing_deployment(fallbacks=[])
|
||||
|
||||
response: Final = await asyncio.wait_for(
|
||||
router.acompletion(
|
||||
sleeper: Final = AsyncMock()
|
||||
|
||||
with patch.object(asyncio, "sleep", sleeper):
|
||||
response: Final = await router.acompletion(
|
||||
model="primary",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
fallbacks=[{"model": "backup"}],
|
||||
),
|
||||
timeout=_BACKOFF_DETECTION_TIMEOUT,
|
||||
)
|
||||
)
|
||||
|
||||
assert response.choices[0].message.content == "answered by backup"
|
||||
assert not any(delay > 0 for delay in _backoff_delays(sleeper))
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -123,16 +131,17 @@ async def test_content_policy_error_keeps_backoff_when_its_dedicated_fallbacks_s
|
|||
retry_policy=RetryPolicy(ContentPolicyViolationErrorRetries=2),
|
||||
)
|
||||
|
||||
with pytest.raises(asyncio.TimeoutError):
|
||||
await asyncio.wait_for(
|
||||
router.acompletion(
|
||||
model="primary",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
mock_response=_CONTENT_POLICY_ERROR,
|
||||
),
|
||||
timeout=_BACKOFF_DETECTION_TIMEOUT,
|
||||
sleeper: Final = AsyncMock()
|
||||
|
||||
with patch.object(asyncio, "sleep", sleeper), pytest.raises(litellm.ContentPolicyViolationError):
|
||||
await router.acompletion(
|
||||
model="primary",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
mock_response=_CONTENT_POLICY_ERROR,
|
||||
)
|
||||
|
||||
assert any(delay > 0 for delay in _backoff_delays(sleeper))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("error", "fallbacks", "context_window_fallbacks", "content_policy_fallbacks", "expected"),
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue