mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-26 01:12:21 +00:00
* test(e2e): tolerate provider-side flakes on five full-suite cells Mistral OCR retries a provider-relayed 429 with backoff, the Vertex vision probe turns reasoning off so its 32 tokens go to the answer, the Vertex cache cell spaces eight never-seen prefixes 15s apart around Google's nondeterministic minimum-token rejection and prices the cached tokens instead of prompt_tokens, and the Azure content-policy cell resends the jailbreak prompt while Azure skips its filter * test(e2e): shorten the new helper docstrings * test(e2e): accept a relayed provider 429 on the rust OCR cells The gateway already retries a provider 429 three times per call and the Mistral key is shared across pipelines, so a throttle can hold across all four attempts of the OCR cell. After the bounded retries the cell now accepts the gateway's faithful relay of the provider's 429 (throttling_error, code 429) as its second expected outcome; the gateway's own 429 and any other error still fail the cell at once. * test(e2e): drop the harness unit tests, the live cells cover the helpers --------- Co-authored-by: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
173 lines
7.3 KiB
Python
173 lines
7.3 KiB
Python
"""Live e2e: per-request fallbacks reroute a failing deployment's traffic to a
|
|
healthy one.
|
|
|
|
Each test registers a primary deployment that fails (an unreachable base URL, or
|
|
a 1ms deadline) and calls it with a `router_settings_override` mapping it to the
|
|
real `gpt-5.5`. The proof the fallback fired is twofold: the response is a
|
|
completion from `gpt-5.5`, and the proxy reports at least one attempted fallback
|
|
in the x-litellm-attempted-fallbacks header. Empty content is accepted only when
|
|
`finish_reason == "length"` and the response billed completion tokens, since
|
|
gpt-5.5 counts reasoning against max_tokens and can consume the whole budget
|
|
before emitting any text; a fallback that produced nothing at all still fails.
|
|
|
|
The context-window and content-policy cases are different reroutes from a plain
|
|
failure: the provider refuses the prompt itself, on length or on policy, and
|
|
`context_window_fallbacks` / `content_policy_fallbacks` are the settings that
|
|
reroute those, not `fallbacks`. The policy refusal is a real one, from an Azure
|
|
OpenAI content filter rejecting a jailbreak prompt, and a control call first
|
|
proves the refusal reaches the customer as a 400 when no reroute is configured.
|
|
Azure intermittently answers without running its prompt filter at all (the
|
|
body's prompt_filter_results carry no verdict), which is not a pass, so both
|
|
calls send the prompt again, a bounded number of times, until the filter ran.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Callable
|
|
from typing import Final
|
|
|
|
import pytest
|
|
|
|
from complexity_router_client import ComplexityRouterClient
|
|
from e2e_config import unique_marker
|
|
from e2e_http import StreamingResponse
|
|
from lifecycle import ResourceManager
|
|
from models import RouterSettingsOverride
|
|
from reliability_support import (
|
|
CONTENT_POLICY_PROMPT,
|
|
azure_prompt_filter_skipped,
|
|
chat_override,
|
|
completion_tokens_of,
|
|
content_of,
|
|
create_bad_base_deployment,
|
|
create_content_filtered_deployment,
|
|
create_small_context_deployment,
|
|
create_timeout_deployment,
|
|
finish_reason_of,
|
|
oversized_prompt,
|
|
reasoning_tokens_of,
|
|
)
|
|
|
|
pytestmark = pytest.mark.e2e
|
|
|
|
|
|
def _assert_served_by_fallback(resp: StreamingResponse) -> None:
|
|
assert resp.status_code == 200, f"expected 200 after fallback, got {resp.status_code}: {resp.body[:300]}"
|
|
content = content_of(resp)
|
|
finish_reason = finish_reason_of(resp)
|
|
completion_tokens = completion_tokens_of(resp) or 0
|
|
reasoning_tokens = reasoning_tokens_of(resp) or 0
|
|
assert isinstance(content, str), (
|
|
f"the gpt-5.5 fallback should have returned a completion body, got content {content!r} (body={resp.body[:300]})"
|
|
)
|
|
assert content or (finish_reason == "length" and completion_tokens > 0), (
|
|
f"the gpt-5.5 fallback returned empty content with finish_reason={finish_reason!r}, "
|
|
f"completion_tokens={completion_tokens}, reasoning_tokens={reasoning_tokens}; empty "
|
|
f"content is only acceptable when the budget was spent on non-visible reasoning "
|
|
f"(body={resp.body[:300]})"
|
|
)
|
|
attempted = resp.headers.get("x-litellm-attempted-fallbacks")
|
|
assert attempted is not None, "response is missing the x-litellm-attempted-fallbacks header"
|
|
assert int(attempted) >= 1, f"x-litellm-attempted-fallbacks should be >= 1, got {attempted!r}"
|
|
|
|
|
|
AZURE_FILTER_ATTEMPTS: Final = 3
|
|
|
|
|
|
def _chat_once_azure_runs_its_filter(send: Callable[[], StreamingResponse]) -> StreamingResponse:
|
|
for attempt in range(1, AZURE_FILTER_ATTEMPTS):
|
|
resp = send()
|
|
if not azure_prompt_filter_skipped(resp):
|
|
return resp
|
|
print(
|
|
"e2e: azure answered without running its prompt filter; sending the jailbreak prompt again "
|
|
f"({attempt}/{AZURE_FILTER_ATTEMPTS - 1})",
|
|
flush=True,
|
|
)
|
|
return send()
|
|
|
|
|
|
def _filter_verdict(resp: StreamingResponse) -> str:
|
|
if azure_prompt_filter_skipped(resp):
|
|
return "azure skipped its prompt filter on every attempt"
|
|
return "the filter ran and let the prompt through"
|
|
|
|
|
|
class TestReliabilityFallbacks:
|
|
@pytest.mark.covers("reliability.fallback.5xx.routes_to_fallback")
|
|
def test_5xx_routes_to_fallback(
|
|
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
|
) -> None:
|
|
primary = f"reliability-fail-{unique_marker()}"
|
|
model_id = create_bad_base_deployment(client.proxy, primary)
|
|
resources.defer(lambda: client.proxy.delete_model(model_id))
|
|
|
|
resp = chat_override(
|
|
client.proxy,
|
|
scoped_key,
|
|
primary,
|
|
f"say hi {unique_marker()}",
|
|
override=RouterSettingsOverride(fallbacks=[{primary: ["gpt-5.5"]}]),
|
|
)
|
|
_assert_served_by_fallback(resp)
|
|
|
|
@pytest.mark.covers("reliability.fallback.timeout.routes_to_fallback")
|
|
def test_timeout_routes_to_fallback(
|
|
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
|
) -> None:
|
|
primary = f"reliability-tofail-{unique_marker()}"
|
|
model_id = create_timeout_deployment(client.proxy, primary)
|
|
resources.defer(lambda: client.proxy.delete_model(model_id))
|
|
|
|
resp = chat_override(
|
|
client.proxy,
|
|
scoped_key,
|
|
primary,
|
|
f"say hi {unique_marker()}",
|
|
override=RouterSettingsOverride(fallbacks=[{primary: ["gpt-5.5"]}]),
|
|
)
|
|
_assert_served_by_fallback(resp)
|
|
|
|
@pytest.mark.covers("reliability.fallback.context_window.routes_to_fallback")
|
|
def test_context_window_routes_to_fallback(
|
|
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
|
) -> None:
|
|
primary = f"reliability-ctxfail-{unique_marker()}"
|
|
model_id = create_small_context_deployment(client.proxy, primary)
|
|
resources.defer(lambda: client.proxy.delete_model(model_id))
|
|
|
|
resp = chat_override(
|
|
client.proxy,
|
|
scoped_key,
|
|
primary,
|
|
oversized_prompt(unique_marker()),
|
|
override=RouterSettingsOverride(context_window_fallbacks=[{primary: ["gpt-5.5"]}]),
|
|
)
|
|
_assert_served_by_fallback(resp)
|
|
|
|
@pytest.mark.covers("reliability.fallback.content_policy.routes_to_fallback")
|
|
def test_content_policy_routes_to_fallback(
|
|
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
|
) -> None:
|
|
primary = f"reliability-policyfail-{unique_marker()}"
|
|
model_id = create_content_filtered_deployment(client.proxy, primary)
|
|
resources.defer(lambda: client.proxy.delete_model(model_id))
|
|
|
|
refused = _chat_once_azure_runs_its_filter(
|
|
lambda: chat_override(client.proxy, scoped_key, primary, f"{CONTENT_POLICY_PROMPT} {unique_marker()}")
|
|
)
|
|
assert refused.status_code == 400, (
|
|
f"the content filter should have refused the jailbreak prompt with a 400, got {refused.status_code} "
|
|
f"({_filter_verdict(refused)}): {refused.body[:300]}"
|
|
)
|
|
|
|
resp = _chat_once_azure_runs_its_filter(
|
|
lambda: chat_override(
|
|
client.proxy,
|
|
scoped_key,
|
|
primary,
|
|
f"{CONTENT_POLICY_PROMPT} {unique_marker()}",
|
|
override=RouterSettingsOverride(content_policy_fallbacks=[{primary: ["gpt-5.5"]}]),
|
|
)
|
|
)
|
|
_assert_served_by_fallback(resp)
|