test(e2e): cover /v1/responses in the Redis timeout test and fail the primary for real

The Responses path returns a mock for any mock_response string, so the InternalServerError
sentinel never failed there. Point the primary deployment's api_base at a closed port instead,
which fails every endpoint the same way, then parametrize the test over /chat/completions and
/v1/responses and register both on the coverage cell.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T
This commit is contained in:
Kerry Lu 2026-09-09 16:03:36 -07:00
parent 02351da51f
commit 939ac04a93
3 changed files with 81 additions and 28 deletions

View file

@ -31,7 +31,7 @@
- {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"}
- {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"}
- {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"}
- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"}
- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"}
- {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"}
- {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"}
- {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"}

View file

@ -21,7 +21,7 @@ model_list:
litellm_params:
model: openai/gpt-5-mini
api_key: sk-redis-timeout-primary-not-used
mock_response: "litellm.InternalServerError"
api_base: http://127.0.0.1:1
- model_name: redis-timeout-backup
litellm_params:
model: openai/gpt-5-mini

View file

@ -1,25 +1,29 @@
"""Live e2e: the proxy keeps answering while every Redis command times out.
Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which
points cache_params at a real Redis with socket_timeout 0.001 so every command times out and
the circuit breaker opens. Each request fails its primary deployment, retries, falls back to the
backup and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend
counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0
that string doubled per request until the worker hung (LIT-6780). Deselected unless
E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy.
points cache_params at a real Redis with socket_timeout 0.001 so commands time out and the
circuit breaker opens. Each request fails its primary deployment, whose api_base is a closed
port, retries, falls back to the backup and succeeds, so it carries retry breadcrumbs; its cost
tracking then fails on the spend counter increment and stringifies the request metadata into a
failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung
(LIT-6780). Deselected unless E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy.
"""
from __future__ import annotations
import time
from collections.abc import Callable
from dataclasses import dataclass
from typing import Final
import pytest
from complexity_router_client import ComplexityRouterClient
from e2e_config import unique_marker
from e2e_http import NoBody, Success
from e2e_http import NoBody, Result, Success
from lifecycle import ResourceManager
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody
from proxy_client import ProxyClient
from pydantic import BaseModel
pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout]
@ -31,42 +35,90 @@ MAX_LATENCY_GROWTH_RATIO: Final = 3.0
MAX_LIVELINESS_SECONDS: Final = 2.0
class ResponsesBody(BaseModel):
model: str
input: str
max_output_tokens: int = 5
class ResponsesObject(BaseModel):
id: str | None = None
status: str | None = None
output: list[object] = []
@dataclass(frozen=True, slots=True)
class Endpoint:
name: str
send: Callable[[ProxyClient, str, str], Result[BaseModel]]
served: Callable[[BaseModel], bool]
def _send_chat(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]:
return proxy.transport.post(
"/chat/completions",
headers=proxy.transport.bearer(key),
json=ChatBody(model=PRIMARY_MODEL, messages=[ChatMessage(role="user", content=marker)], max_tokens=5),
response_type=ChatResponse,
timeout=MAX_SECONDS_PER_REQUEST,
)
def _send_responses(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]:
return proxy.transport.post(
"/v1/responses",
headers=proxy.transport.bearer(key),
json=ResponsesBody(model=PRIMARY_MODEL, input=marker),
response_type=ResponsesObject,
timeout=MAX_SECONDS_PER_REQUEST,
)
ENDPOINTS: Final = (
Endpoint(
name="chat_completions",
send=_send_chat,
served=lambda data: isinstance(data, ChatResponse) and bool(data.choices),
),
Endpoint(
name="responses",
send=_send_responses,
served=lambda data: isinstance(data, ResponsesObject) and bool(data.output),
),
)
class TestRedisTimeout:
@pytest.mark.parametrize("endpoint", ENDPOINTS, ids=[endpoint.name for endpoint in ENDPOINTS])
@pytest.mark.covers(
"reliability.circuit_breaker.redis_timeout.stays_responsive",
exercised_on=["chat_completions"],
exercised_on=["chat_completions", "responses"],
)
def test_retries_under_redis_timeouts_keep_answering(
self, client: ComplexityRouterClient, resources: ResourceManager
self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint
) -> None:
proxy = client.proxy
key = proxy.generate_key(
KeyGenerateBody(models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{unique_marker()}")
KeyGenerateBody(
models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}"
)
)
resources.defer(lambda: proxy.delete_key(key))
latencies: list[float] = []
for request_number in range(1, REQUESTS + 1):
started = time.monotonic()
result = proxy.transport.post(
"/chat/completions",
headers=proxy.transport.bearer(key),
json=ChatBody(
model=PRIMARY_MODEL,
messages=[ChatMessage(role="user", content=f"redis timeout {unique_marker()} {request_number}")],
max_tokens=5,
),
response_type=ChatResponse,
timeout=MAX_SECONDS_PER_REQUEST,
)
result = endpoint.send(proxy, key, f"redis timeout {unique_marker()} {request_number}")
elapsed = time.monotonic() - started
assert isinstance(result, Success), (
f"request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; "
f"{endpoint.name} request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; "
f"earlier requests took {[round(seconds, 2) for seconds in latencies]}"
)
assert result.data.choices, f"request {request_number}: fallback to {BACKUP_MODEL} returned no choices"
assert endpoint.served(result.data), (
f"{endpoint.name} request {request_number}: fallback to {BACKUP_MODEL} returned no output"
)
assert elapsed < MAX_SECONDS_PER_REQUEST, (
f"request {request_number} took {elapsed:.1f}s with Redis timing out; "
f"{endpoint.name} request {request_number} took {elapsed:.1f}s with Redis timing out; "
f"earlier requests took {[round(seconds, 2) for seconds in latencies]}"
)
latencies.append(elapsed)
@ -75,7 +127,7 @@ class TestRedisTimeout:
early = sum(latencies[:third]) / third
late = sum(latencies[-third:]) / third
assert late <= max(early * MAX_LATENCY_GROWTH_RATIO, 0.5), (
f"per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests "
f"{endpoint.name} per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests "
"while Redis timed out; the proxy is paying more for each failed request"
)
@ -89,5 +141,6 @@ class TestRedisTimeout:
rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS)
assert len(rows) >= REQUESTS, (
f"only {len(rows)} of {REQUESTS} requests reached the spend log; a Redis outage must not lose spend rows"
f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; "
"a Redis outage must not lose spend rows"
)