From 4b1d9bf148e501f8f8037fb81fd420ccd3b0d760 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Thu, 1 Oct 2026 15:54:02 -0700 Subject: [PATCH] test(e2e): keep 1ms-timeout deployments off the provider cache (#44082) The timeout reliability tests rely on a 1ms deadline the real backend always misses. With E2E_PROVIDER_CACHE on, the deployment pointed at the cache edge, and its healthy sibling in the same test had already recorded a response for the same canonical request, so the edge answered from Redis inside the 1ms read window. Build 342 of litellm-e2e saw test_timeout_trips_cooldown_then_recovers get a 200 from the timing-out deployment itself, with a recording made about 12 hours earlier. Both timeout helpers now register on the live provider path, which PROVIDER_CACHE.md reserves for tests that need real provider timing --- tests/e2e/router/reliability_support.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/tests/e2e/router/reliability_support.py b/tests/e2e/router/reliability_support.py index 5984d5645d8..f600531e663 100644 --- a/tests/e2e/router/reliability_support.py +++ b/tests/e2e/router/reliability_support.py @@ -97,7 +97,9 @@ def create_never_benched_refusing_deployment(proxy: ProxyClient, name: str) -> s def create_timeout_deployment(proxy: ProxyClient, name: str) -> str: """Register a deployment with a 1ms deadline the real backend always exceeds.""" - return proxy.create_model(name, LiteLLMParamsBody(model=REAL_MODEL, api_key=REAL_KEY, timeout=0.001)) + return proxy.create_model( + name, LiteLLMParamsBody(model=REAL_MODEL, api_key=REAL_KEY, timeout=0.001), provider_live=True + ) def create_small_context_deployment(proxy: ProxyClient, name: str) -> str: @@ -149,7 +151,7 @@ def create_caching_deployment(proxy: ProxyClient, name: str) -> str: def _register_benched_on_first_failure( - proxy: ProxyClient, name: str, litellm_params: LiteLLMParamsBody, allowed_fails: str + proxy: ProxyClient, name: str, litellm_params: LiteLLMParamsBody, allowed_fails: str, *, provider_live: bool = False ) -> str: """The always-picked half of a failing pair: all of the group's shuffle weight, and a cooldown policy that benches it on its first failure of the given class, @@ -159,7 +161,8 @@ def _register_benched_on_first_failure( model_name=name, litellm_params=litellm_params, model_info=ModelInfoBody(allowed_fails_policy={allowed_fails: 0}), - ) + ), + provider_live=provider_live, ) @@ -170,6 +173,7 @@ def create_always_timing_out_deployment(proxy: ProxyClient, name: str, cooldown_ name, LiteLLMParamsBody(model=REAL_MODEL, api_key=REAL_KEY, timeout=0.001, weight=1, cooldown_time=cooldown_time), "TimeoutErrorAllowedFails", + provider_live=True, )