mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
test(e2e): keep 1ms-timeout deployments off the provider cache (#44082)
The timeout reliability tests rely on a 1ms deadline the real backend always misses. With E2E_PROVIDER_CACHE on, the deployment pointed at the cache edge, and its healthy sibling in the same test had already recorded a response for the same canonical request, so the edge answered from Redis inside the 1ms read window. Build 342 of litellm-e2e saw test_timeout_trips_cooldown_then_recovers get a 200 from the timing-out deployment itself, with a recording made about 12 hours earlier. Both timeout helpers now register on the live provider path, which PROVIDER_CACHE.md reserves for tests that need real provider timing
This commit is contained in:
parent
03743ae020
commit
4b1d9bf148
1 changed files with 7 additions and 3 deletions
|
|
@ -97,7 +97,9 @@ def create_never_benched_refusing_deployment(proxy: ProxyClient, name: str) -> s
|
|||
|
||||
def create_timeout_deployment(proxy: ProxyClient, name: str) -> str:
|
||||
"""Register a deployment with a 1ms deadline the real backend always exceeds."""
|
||||
return proxy.create_model(name, LiteLLMParamsBody(model=REAL_MODEL, api_key=REAL_KEY, timeout=0.001))
|
||||
return proxy.create_model(
|
||||
name, LiteLLMParamsBody(model=REAL_MODEL, api_key=REAL_KEY, timeout=0.001), provider_live=True
|
||||
)
|
||||
|
||||
|
||||
def create_small_context_deployment(proxy: ProxyClient, name: str) -> str:
|
||||
|
|
@ -149,7 +151,7 @@ def create_caching_deployment(proxy: ProxyClient, name: str) -> str:
|
|||
|
||||
|
||||
def _register_benched_on_first_failure(
|
||||
proxy: ProxyClient, name: str, litellm_params: LiteLLMParamsBody, allowed_fails: str
|
||||
proxy: ProxyClient, name: str, litellm_params: LiteLLMParamsBody, allowed_fails: str, *, provider_live: bool = False
|
||||
) -> str:
|
||||
"""The always-picked half of a failing pair: all of the group's shuffle weight,
|
||||
and a cooldown policy that benches it on its first failure of the given class,
|
||||
|
|
@ -159,7 +161,8 @@ def _register_benched_on_first_failure(
|
|||
model_name=name,
|
||||
litellm_params=litellm_params,
|
||||
model_info=ModelInfoBody(allowed_fails_policy={allowed_fails: 0}),
|
||||
)
|
||||
),
|
||||
provider_live=provider_live,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -170,6 +173,7 @@ def create_always_timing_out_deployment(proxy: ProxyClient, name: str, cooldown_
|
|||
name,
|
||||
LiteLLMParamsBody(model=REAL_MODEL, api_key=REAL_KEY, timeout=0.001, weight=1, cooldown_time=cooldown_time),
|
||||
"TimeoutErrorAllowedFails",
|
||||
provider_live=True,
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue