"""Live e2e: per-request fallbacks reroute a failing deployment's traffic to a healthy one. Each test registers a primary deployment that fails (an unreachable base URL, or a 1ms deadline) and calls it with a `router_settings_override` mapping it to the real `gpt-5.5`. The proof the fallback fired is twofold: the response is a completion from `gpt-5.5`, and the proxy reports at least one attempted fallback in the x-litellm-attempted-fallbacks header. Empty content is accepted only when `finish_reason == "length"` and the response billed completion tokens, since gpt-5.5 counts reasoning against max_tokens and can consume the whole budget before emitting any text; a fallback that produced nothing at all still fails. The context-window and content-policy cases are different reroutes from a plain failure: the provider refuses the prompt itself, on length or on policy, and `context_window_fallbacks` / `content_policy_fallbacks` are the settings that reroute those, not `fallbacks`. The policy refusal is a real one, from an Azure OpenAI content filter rejecting a jailbreak prompt, and a control call first proves the refusal reaches the customer as a 400 when no reroute is configured. Azure intermittently answers without running its prompt filter at all (the body's prompt_filter_results carry no verdict), which is not a pass, so both calls send the prompt again, a bounded number of times, until the filter ran. """ from __future__ import annotations from collections.abc import Callable from typing import Final import pytest from complexity_router_client import ComplexityRouterClient from e2e_config import unique_marker from e2e_http import StreamingResponse from lifecycle import ResourceManager from models import RouterSettingsOverride from reliability_support import ( CONTENT_POLICY_PROMPT, azure_prompt_filter_skipped, chat_override, completion_tokens_of, content_of, create_bad_base_deployment, create_content_filtered_deployment, create_small_context_deployment, create_timeout_deployment, finish_reason_of, oversized_prompt, reasoning_tokens_of, ) pytestmark = pytest.mark.e2e def _assert_served_by_fallback(resp: StreamingResponse) -> None: assert resp.status_code == 200, f"expected 200 after fallback, got {resp.status_code}: {resp.body[:300]}" content = content_of(resp) finish_reason = finish_reason_of(resp) completion_tokens = completion_tokens_of(resp) or 0 reasoning_tokens = reasoning_tokens_of(resp) or 0 assert isinstance(content, str), ( f"the gpt-5.5 fallback should have returned a completion body, got content {content!r} (body={resp.body[:300]})" ) assert content or (finish_reason == "length" and completion_tokens > 0), ( f"the gpt-5.5 fallback returned empty content with finish_reason={finish_reason!r}, " f"completion_tokens={completion_tokens}, reasoning_tokens={reasoning_tokens}; empty " f"content is only acceptable when the budget was spent on non-visible reasoning " f"(body={resp.body[:300]})" ) attempted = resp.headers.get("x-litellm-attempted-fallbacks") assert attempted is not None, "response is missing the x-litellm-attempted-fallbacks header" assert int(attempted) >= 1, f"x-litellm-attempted-fallbacks should be >= 1, got {attempted!r}" AZURE_FILTER_ATTEMPTS: Final = 3 def _chat_once_azure_runs_its_filter(send: Callable[[], StreamingResponse]) -> StreamingResponse: for attempt in range(1, AZURE_FILTER_ATTEMPTS): resp = send() if not azure_prompt_filter_skipped(resp): return resp print( "e2e: azure answered without running its prompt filter; sending the jailbreak prompt again " f"({attempt}/{AZURE_FILTER_ATTEMPTS - 1})", flush=True, ) return send() def _filter_verdict(resp: StreamingResponse) -> str: if azure_prompt_filter_skipped(resp): return "azure skipped its prompt filter on every attempt" return "the filter ran and let the prompt through" class TestReliabilityFallbacks: @pytest.mark.covers("reliability.fallback.5xx.routes_to_fallback") def test_5xx_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: primary = f"reliability-fail-{unique_marker()}" model_id = create_bad_base_deployment(client.proxy, primary) resources.defer(lambda: client.proxy.delete_model(model_id)) resp = chat_override( client.proxy, scoped_key, primary, f"say hi {unique_marker()}", override=RouterSettingsOverride(fallbacks=[{primary: ["gpt-5.5"]}]), ) _assert_served_by_fallback(resp) @pytest.mark.covers("reliability.fallback.timeout.routes_to_fallback") def test_timeout_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: primary = f"reliability-tofail-{unique_marker()}" model_id = create_timeout_deployment(client.proxy, primary) resources.defer(lambda: client.proxy.delete_model(model_id)) resp = chat_override( client.proxy, scoped_key, primary, f"say hi {unique_marker()}", override=RouterSettingsOverride(fallbacks=[{primary: ["gpt-5.5"]}]), ) _assert_served_by_fallback(resp) @pytest.mark.covers("reliability.fallback.context_window.routes_to_fallback") def test_context_window_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: primary = f"reliability-ctxfail-{unique_marker()}" model_id = create_small_context_deployment(client.proxy, primary) resources.defer(lambda: client.proxy.delete_model(model_id)) resp = chat_override( client.proxy, scoped_key, primary, oversized_prompt(unique_marker()), override=RouterSettingsOverride(context_window_fallbacks=[{primary: ["gpt-5.5"]}]), ) _assert_served_by_fallback(resp) @pytest.mark.covers("reliability.fallback.content_policy.routes_to_fallback") def test_content_policy_routes_to_fallback( self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str ) -> None: primary = f"reliability-policyfail-{unique_marker()}" model_id = create_content_filtered_deployment(client.proxy, primary) resources.defer(lambda: client.proxy.delete_model(model_id)) refused = _chat_once_azure_runs_its_filter( lambda: chat_override(client.proxy, scoped_key, primary, f"{CONTENT_POLICY_PROMPT} {unique_marker()}") ) assert refused.status_code == 400, ( f"the content filter should have refused the jailbreak prompt with a 400, got {refused.status_code} " f"({_filter_verdict(refused)}): {refused.body[:300]}" ) resp = _chat_once_azure_runs_its_filter( lambda: chat_override( client.proxy, scoped_key, primary, f"{CONTENT_POLICY_PROMPT} {unique_marker()}", override=RouterSettingsOverride(content_policy_fallbacks=[{primary: ["gpt-5.5"]}]), ) ) _assert_served_by_fallback(resp)