mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-20 00:11:50 +00:00
test(e2e): cover /v1/responses in the Redis timeout test and fail the primary for real
The Responses path returns a mock for any mock_response string, so the InternalServerError sentinel never failed there. Point the primary deployment's api_base at a closed port instead, which fails every endpoint the same way, then parametrize the test over /chat/completions and /v1/responses and register both on the coverage cell. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014ZDULyJPp17ZFiJenRxs2T
This commit is contained in:
parent
02351da51f
commit
939ac04a93
3 changed files with 81 additions and 28 deletions
|
|
@ -31,7 +31,7 @@
|
|||
- {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"}
|
||||
- {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"}
|
||||
- {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"}
|
||||
- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"}
|
||||
- {id: reliability.circuit_breaker.redis_timeout.stays_responsive, module: reliability, tier: P1, behavior: circuit_breaker, variant: redis_timeout, assertions: [stays_responsive], exercised_on: [chat_completions, responses], source: "litellm/proxy/hooks/proxy_track_cost_callback.py:386", fail_before_fix: proven, rationale: "With every Redis command timing out and every request retrying then falling back, per-request latency stays flat, liveliness stays fast, and spend rows still land; on v1.100.0 the failed-tracking alert body doubled per request until the worker OOMed (LIT-6780)"}
|
||||
- {id: reliability.timeout.request_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: request_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions, messages], source: "litellm/router.py:545-551", rationale: "Per-request timeout raises Timeout"}
|
||||
- {id: reliability.timeout.stream_timeout.exceeds_deadline, module: reliability, tier: P1, behavior: timeout, variant: stream_timeout, assertions: [exceeds_deadline], exercised_on: [chat_completions], source: "litellm/router.py:551", rationale: "Streaming chunk-delivery timeout"}
|
||||
- {id: reliability.perf.throughput.under_slo, module: reliability, tier: P1, behavior: perf, variant: throughput, assertions: [under_slo], exercised_on: [chat_completions, messages], source: grammar, rationale: "Throughput SLO under load"}
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ model_list:
|
|||
litellm_params:
|
||||
model: openai/gpt-5-mini
|
||||
api_key: sk-redis-timeout-primary-not-used
|
||||
mock_response: "litellm.InternalServerError"
|
||||
api_base: http://127.0.0.1:1
|
||||
- model_name: redis-timeout-backup
|
||||
litellm_params:
|
||||
model: openai/gpt-5-mini
|
||||
|
|
|
|||
|
|
@ -1,25 +1,29 @@
|
|||
"""Live e2e: the proxy keeps answering while every Redis command times out.
|
||||
|
||||
Runs only against a proxy booted from tests/e2e/gateway/redis_timeout_ci_config.yml, which
|
||||
points cache_params at a real Redis with socket_timeout 0.001 so every command times out and
|
||||
the circuit breaker opens. Each request fails its primary deployment, retries, falls back to the
|
||||
backup and succeeds, so it carries retry breadcrumbs; its cost tracking then fails on the spend
|
||||
counter increment and stringifies the request metadata into a failed-tracking alert. On v1.100.0
|
||||
that string doubled per request until the worker hung (LIT-6780). Deselected unless
|
||||
E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy.
|
||||
points cache_params at a real Redis with socket_timeout 0.001 so commands time out and the
|
||||
circuit breaker opens. Each request fails its primary deployment, whose api_base is a closed
|
||||
port, retries, falls back to the backup and succeeds, so it carries retry breadcrumbs; its cost
|
||||
tracking then fails on the spend counter increment and stringifies the request metadata into a
|
||||
failed-tracking alert. On v1.100.0 that string doubled per request until the worker hung
|
||||
(LIT-6780). Deselected unless E2E_REDIS_TIMEOUT is set, since it needs that dedicated proxy.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from complexity_router_client import ComplexityRouterClient
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import NoBody, Success
|
||||
from e2e_http import NoBody, Result, Success
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, ChatResponse, KeyGenerateBody
|
||||
from proxy_client import ProxyClient
|
||||
from pydantic import BaseModel
|
||||
|
||||
pytestmark = [pytest.mark.e2e, pytest.mark.redis_timeout]
|
||||
|
||||
|
|
@ -31,42 +35,90 @@ MAX_LATENCY_GROWTH_RATIO: Final = 3.0
|
|||
MAX_LIVELINESS_SECONDS: Final = 2.0
|
||||
|
||||
|
||||
class ResponsesBody(BaseModel):
|
||||
model: str
|
||||
input: str
|
||||
max_output_tokens: int = 5
|
||||
|
||||
|
||||
class ResponsesObject(BaseModel):
|
||||
id: str | None = None
|
||||
status: str | None = None
|
||||
output: list[object] = []
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class Endpoint:
|
||||
name: str
|
||||
send: Callable[[ProxyClient, str, str], Result[BaseModel]]
|
||||
served: Callable[[BaseModel], bool]
|
||||
|
||||
|
||||
def _send_chat(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]:
|
||||
return proxy.transport.post(
|
||||
"/chat/completions",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=ChatBody(model=PRIMARY_MODEL, messages=[ChatMessage(role="user", content=marker)], max_tokens=5),
|
||||
response_type=ChatResponse,
|
||||
timeout=MAX_SECONDS_PER_REQUEST,
|
||||
)
|
||||
|
||||
|
||||
def _send_responses(proxy: ProxyClient, key: str, marker: str) -> Result[BaseModel]:
|
||||
return proxy.transport.post(
|
||||
"/v1/responses",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=ResponsesBody(model=PRIMARY_MODEL, input=marker),
|
||||
response_type=ResponsesObject,
|
||||
timeout=MAX_SECONDS_PER_REQUEST,
|
||||
)
|
||||
|
||||
|
||||
ENDPOINTS: Final = (
|
||||
Endpoint(
|
||||
name="chat_completions",
|
||||
send=_send_chat,
|
||||
served=lambda data: isinstance(data, ChatResponse) and bool(data.choices),
|
||||
),
|
||||
Endpoint(
|
||||
name="responses",
|
||||
send=_send_responses,
|
||||
served=lambda data: isinstance(data, ResponsesObject) and bool(data.output),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class TestRedisTimeout:
|
||||
@pytest.mark.parametrize("endpoint", ENDPOINTS, ids=[endpoint.name for endpoint in ENDPOINTS])
|
||||
@pytest.mark.covers(
|
||||
"reliability.circuit_breaker.redis_timeout.stays_responsive",
|
||||
exercised_on=["chat_completions"],
|
||||
exercised_on=["chat_completions", "responses"],
|
||||
)
|
||||
def test_retries_under_redis_timeouts_keep_answering(
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager, endpoint: Endpoint
|
||||
) -> None:
|
||||
proxy = client.proxy
|
||||
key = proxy.generate_key(
|
||||
KeyGenerateBody(models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{unique_marker()}")
|
||||
KeyGenerateBody(
|
||||
models=[PRIMARY_MODEL, BACKUP_MODEL], key_alias=f"e2e-redis-timeout-{endpoint.name}-{unique_marker()}"
|
||||
)
|
||||
)
|
||||
resources.defer(lambda: proxy.delete_key(key))
|
||||
|
||||
latencies: list[float] = []
|
||||
for request_number in range(1, REQUESTS + 1):
|
||||
started = time.monotonic()
|
||||
result = proxy.transport.post(
|
||||
"/chat/completions",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=ChatBody(
|
||||
model=PRIMARY_MODEL,
|
||||
messages=[ChatMessage(role="user", content=f"redis timeout {unique_marker()} {request_number}")],
|
||||
max_tokens=5,
|
||||
),
|
||||
response_type=ChatResponse,
|
||||
timeout=MAX_SECONDS_PER_REQUEST,
|
||||
)
|
||||
result = endpoint.send(proxy, key, f"redis timeout {unique_marker()} {request_number}")
|
||||
elapsed = time.monotonic() - started
|
||||
assert isinstance(result, Success), (
|
||||
f"request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; "
|
||||
f"{endpoint.name} request {request_number} failed after {elapsed:.1f}s with Redis timing out: {result}; "
|
||||
f"earlier requests took {[round(seconds, 2) for seconds in latencies]}"
|
||||
)
|
||||
assert result.data.choices, f"request {request_number}: fallback to {BACKUP_MODEL} returned no choices"
|
||||
assert endpoint.served(result.data), (
|
||||
f"{endpoint.name} request {request_number}: fallback to {BACKUP_MODEL} returned no output"
|
||||
)
|
||||
assert elapsed < MAX_SECONDS_PER_REQUEST, (
|
||||
f"request {request_number} took {elapsed:.1f}s with Redis timing out; "
|
||||
f"{endpoint.name} request {request_number} took {elapsed:.1f}s with Redis timing out; "
|
||||
f"earlier requests took {[round(seconds, 2) for seconds in latencies]}"
|
||||
)
|
||||
latencies.append(elapsed)
|
||||
|
|
@ -75,7 +127,7 @@ class TestRedisTimeout:
|
|||
early = sum(latencies[:third]) / third
|
||||
late = sum(latencies[-third:]) / third
|
||||
assert late <= max(early * MAX_LATENCY_GROWTH_RATIO, 0.5), (
|
||||
f"per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests "
|
||||
f"{endpoint.name} per-request latency grew from {early:.2f}s to {late:.2f}s across {REQUESTS} requests "
|
||||
"while Redis timed out; the proxy is paying more for each failed request"
|
||||
)
|
||||
|
||||
|
|
@ -89,5 +141,6 @@ class TestRedisTimeout:
|
|||
|
||||
rows = proxy.poll_logs_for_key(key, min_rows=REQUESTS)
|
||||
assert len(rows) >= REQUESTS, (
|
||||
f"only {len(rows)} of {REQUESTS} requests reached the spend log; a Redis outage must not lose spend rows"
|
||||
f"only {len(rows)} of {REQUESTS} {endpoint.name} requests reached the spend log; "
|
||||
"a Redis outage must not lose spend rows"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue