diff --git a/tests/e2e/coverage_registry/reliability.yaml b/tests/e2e/coverage_registry/reliability.yaml index 38427b647b9..3b378b56870 100644 --- a/tests/e2e/coverage_registry/reliability.yaml +++ b/tests/e2e/coverage_registry/reliability.yaml @@ -31,7 +31,7 @@ - {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"} - {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"} - {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"} -- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} +- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"} - {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"} - {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"} - {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"} diff --git a/tests/e2e/gateway/redis_timeout_ci_config.yml b/tests/e2e/gateway/redis_timeout_ci_config.yml index 69ab2ee14dc..735285adb36 100644 --- a/tests/e2e/gateway/redis_timeout_ci_config.yml +++ b/tests/e2e/gateway/redis_timeout_ci_config.yml @@ -21,7 +21,7 @@ model_list: litellm_params: model: openai/gpt-5-mini api_key: sk-redis-timeout-primary-not-used - mock_response: "litellm.InternalServerError" + api_base: http://127.0.0.1:1 - model_name: redis-timeout-backup litellm_params: model: openai/gpt-5-mini diff --git a/tests/e2e/router/test_redis_timeout_e2e.py b/tests/e2e/router/test_redis_timeout_e2e.py index eeb5224e051..d10f32b6482 100644 --- a/tests/e2e/router/test_redis_timeout_e2e.py +++ b/tests/e2e/router/test_redis_timeout_e2e.py @@ -1,25 +1,29 @@ """Live e2e: the proxy keeps answering while every Redis command times out. Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which -points cache_params at a real Redis with socket_timeout 0.001 so every command times out and -the circuit breaker opens. Each request fails its primary deployment, retries, falls back to the -backup and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend -counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0 -that string doubled per request until the worker hung (LIT-6780). Deselected unless -E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. +points cache_params at a real Redis with socket_timeout 0.001 so commands time out and the +circuit breaker opens. Each request fails its primary deployment, whose api_base is a closed +port, retries, falls back to the backup and succeeds, so it carries retry breadcrumbs; its cost +tracking then fails on the spend counter increment and stringifies the request metadata into a +failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung +(LIT-6780). Deselected unless E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy. """ from __future__ import annotations import time +from collections.abc import Callable +from dataclasses import dataclass from typing import Final import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker -from e2e_http import NoBody, Success +from e2e_http import NoBody, Result, Success from lifecycle import ResourceManager from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody +from proxy_client import ProxyClient +from pydantic import BaseModel pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout] @@ -31,42 +35,90 @@ MAX_LATENCY_GROWTH_RATIO: Final = 3.0 MAX_LIVELINESS_SECONDS: Final = 2.0 +class ResponsesBody(BaseModel): + model: str + input: str + max_output_tokens: int = 5 + + +class ResponsesObject(BaseModel): + id: str | None = None + status: str | None = None + output: list[object] = [] + + +@dataclass(frozen=True, slots=True) +class Endpoint: + name: str + send: Callable[[ProxyClient, str, str], Result[BaseModel]] + served: Callable[[BaseModel], bool] + + +def _send_chat(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]: + return proxy.transport.post( + "/chat/completions", + headers=proxy.transport.bearer(key), + json=ChatBody(model=PRIMARY_MODEL, messages=[ChatMessage(role="user", content=marker)], max_tokens=5), + response_type=ChatResponse, + timeout=MAX_SECONDS_PER_REQUEST, + ) + + +def _send_responses(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]: + return proxy.transport.post( + "/v1/responses", + headers=proxy.transport.bearer(key), + json=ResponsesBody(model=PRIMARY_MODEL, input=marker), + response_type=ResponsesObject, + timeout=MAX_SECONDS_PER_REQUEST, + ) + + +ENDPOINTS: Final = ( + Endpoint( + name="chat_completions", + send=_send_chat, + served=lambda data: isinstance(data, ChatResponse) and bool(data.choices), + ), + Endpoint( + name="responses", + send=_send_responses, + served=lambda data: isinstance(data, ResponsesObject) and bool(data.output), + ), +) + + class TestRedisTimeout: + @pytest.mark.parametrize("endpoint", ENDPOINTS, ids=[endpoint.name for endpoint in ENDPOINTS]) @pytest.mark.covers( "reliability.circuit_breaker.redis_timeout.stays_responsive", - exercised_on=["chat_completions"], + exercised_on=["chat_completions", "responses"], ) def test_retries_under_redis_timeouts_keep_answering( - self, client: ComplexityRouterClient, resources: ResourceManager + self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint ) -> None: proxy = client.proxy key = proxy.generate_key( - KeyGenerateBody(models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{unique_marker()}") + KeyGenerateBody( + models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}" + ) ) resources.defer(lambda: proxy.delete_key(key)) latencies: list[float] = [] for request_number in range(1, REQUESTS + 1): started = time.monotonic() - result = proxy.transport.post( - "/chat/completions", - headers=proxy.transport.bearer(key), - json=ChatBody( - model=PRIMARY_MODEL, - messages=[ChatMessage(role="user", content=f"redis timeout {unique_marker()} {request_number}")], - max_tokens=5, - ), - response_type=ChatResponse, - timeout=MAX_SECONDS_PER_REQUEST, - ) + result = endpoint.send(proxy, key, f"redis timeout {unique_marker()} {request_number}") elapsed = time.monotonic() - started assert isinstance(result, Success), ( - f"request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " + f"{endpoint.name} request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; " f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" ) - assert result.data.choices, f"request {request_number}: fallback to {BACKUP_MODEL} returned no choices" + assert endpoint.served(result.data), ( + f"{endpoint.name} request {request_number}: fallback to {BACKUP_MODEL} returned no output" + ) assert elapsed < MAX_SECONDS_PER_REQUEST, ( - f"request {request_number} took {elapsed:.1f}s with Redis timing out; " + f"{endpoint.name} request {request_number} took {elapsed:.1f}s with Redis timing out; " f"earlier requests took {[round(seconds, 2) for seconds in latencies]}" ) latencies.append(elapsed) @@ -75,7 +127,7 @@ class TestRedisTimeout: early = sum(latencies[:third]) / third late = sum(latencies[-third:]) / third assert late <= max(early * MAX_LATENCY_GROWTH_RATIO, 0.5), ( - f"per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests " + f"{endpoint.name} per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests " "while Redis timed out; the proxy is paying more for each failed request" ) @@ -89,5 +141,6 @@ class TestRedisTimeout: rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS) assert len(rows) >= REQUESTS, ( - f"only {len(rows)} of {REQUESTS} requests reached the spend log; a Redis outage must not lose spend rows" + f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; " + "a Redis outage must not lose spend rows" )