mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
test(e2e): cover router strategy selection and context-window fallback
Adds live coverage for four routing-strategy registry cells (cost, tpm, simple-shuffle weight, latency) and for context-window fallback rerouting. Each routing test builds a model group whose deployments differ in only the dimension the strategy under test reads, drives the strategy per request through router_settings_override, and asserts which deployment served the call from the x-litellm-model-id header. The fallback test overflows a real gpt-4 deployment's context window and asserts the same prompt is served by the gpt-5.5 fallback least_busy.picks_lowest_traffic is deliberately left uncovered. Proving it needs the router's in-flight tally for a deployment to be observable at the moment of the pick, and the proxy exposes no such read: the tally is absent from the in-memory cache dump, prometheus is off, and spend rows only land once a call finishes. Every barrier available instead observes an earlier layer, which would make the assertion a probability bet rather than a proof
This commit is contained in:
parent
bb6bb664b1
commit
53a2138d05
2 changed files with 409 additions and 0 deletions
107
tests/e2e/router/test_reliability_context_window_fallback_e2e.py
Normal file
107
tests/e2e/router/test_reliability_context_window_fallback_e2e.py
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
"""Live e2e: a prompt too large for a deployment's context window is rerouted to
|
||||
the configured context-window fallback.
|
||||
|
||||
The primary is a real `openai/gpt-4` deployment, chosen for the smallest context
|
||||
window OpenAI still serves (8k), so a prompt can overflow it for a few cents;
|
||||
`gpt-5.5`, the model the rest of this suite runs on, has a million-token window
|
||||
that no affordable prompt reaches. The overflow is therefore real: OpenAI itself
|
||||
rejects the call with a context-length error, which is the trigger
|
||||
`context_window_fallbacks` exists for.
|
||||
|
||||
The same oversized prompt is sent twice. The first call carries no fallback and
|
||||
must come back as a context-window rejection, which is what proves the prompt
|
||||
genuinely overflows the primary instead of the deployment being broken some other
|
||||
way. The second names `gpt-5.5` as the context-window fallback and must come back
|
||||
200, served by a different deployment than the primary, with the proxy reporting
|
||||
the attempted fallback.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from complexity_router_client import ComplexityRouterClient
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import StreamingResponse
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatMessage, LiteLLMParamsBody, ReliabilityChatBody, RouterSettingsOverride
|
||||
from proxy_client import ProxyClient
|
||||
from reliability_support import REAL_KEY, REAL_MODEL
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
SMALL_CONTEXT_MODEL = "openai/gpt-4"
|
||||
LARGE_CONTEXT_GROUP = "gpt-5.5"
|
||||
FILLER_SENTENCE = "The quick brown fox jumps over the lazy dog. "
|
||||
FILLER_REPEATS = 900
|
||||
ANSWER_TOKENS = 256
|
||||
|
||||
|
||||
def _oversized_prompt() -> str:
|
||||
"""Roughly ten thousand tokens of filler: past gpt-4's 8k window, and unique per
|
||||
call so the proxy's response cache never answers it from an earlier run."""
|
||||
return f"{FILLER_SENTENCE * FILLER_REPEATS} summarize the text above [{unique_marker()}]"
|
||||
|
||||
|
||||
def _summarize(
|
||||
proxy: ProxyClient, key: str, group: str, override: RouterSettingsOverride | None = None
|
||||
) -> StreamingResponse:
|
||||
"""Ask `group` to summarize an oversized prompt, optionally with a fallback wired
|
||||
for this one request. The answer budget is generous on purpose: gpt-5.5 spends
|
||||
tokens on reasoning before it writes anything, and a budget it cannot finish in
|
||||
is its own error, which would muddy the context-window one under test."""
|
||||
return proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=ReliabilityChatBody(
|
||||
model=group,
|
||||
messages=[ChatMessage(role="user", content=_oversized_prompt())],
|
||||
max_tokens=ANSWER_TOKENS,
|
||||
router_settings_override=override,
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
class TestReliabilityContextWindowFallback:
|
||||
@pytest.mark.covers("reliability.fallback.context_window.routes_to_fallback")
|
||||
def test_context_window_overflow_routes_to_fallback(
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
primary = f"reliability-ctx-{unique_marker()}"
|
||||
model_id = client.proxy.create_model(
|
||||
primary, LiteLLMParamsBody(model=SMALL_CONTEXT_MODEL, api_key=REAL_KEY)
|
||||
)
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
|
||||
rejected = _summarize(client.proxy, scoped_key, primary)
|
||||
assert rejected.status_code == 400, (
|
||||
f"the oversized prompt should be rejected by {SMALL_CONTEXT_MODEL} with a 400, got "
|
||||
f"{rejected.status_code}: {rejected.body[:300]}"
|
||||
)
|
||||
assert "context length" in rejected.body.lower(), (
|
||||
f"the rejection should name the context length, got: {rejected.body[:300]}"
|
||||
)
|
||||
|
||||
rerouted = _summarize(
|
||||
client.proxy,
|
||||
scoped_key,
|
||||
primary,
|
||||
RouterSettingsOverride(context_window_fallbacks=[{primary: [LARGE_CONTEXT_GROUP]}]),
|
||||
)
|
||||
assert rerouted.status_code == 200, (
|
||||
f"the same prompt should be served by the context-window fallback, got "
|
||||
f"{rerouted.status_code}: {rerouted.body[:300]}"
|
||||
)
|
||||
served = rerouted.headers.get("x-litellm-model-id")
|
||||
assert served is not None and served != model_id, (
|
||||
f"the fallback answer came from the deployment that could not fit the prompt "
|
||||
f"({model_id}); x-litellm-model-id={served!r}"
|
||||
)
|
||||
assert rerouted.headers.get("x-litellm-model-name") == REAL_MODEL, (
|
||||
f"the fallback should have been served by a {REAL_MODEL} deployment, but "
|
||||
f"x-litellm-model-name is {rerouted.headers.get('x-litellm-model-name')!r}"
|
||||
)
|
||||
attempted = rerouted.headers.get("x-litellm-attempted-fallbacks")
|
||||
assert attempted is not None and int(attempted) >= 1, (
|
||||
f"the proxy should report at least one attempted fallback, got {attempted!r}"
|
||||
)
|
||||
302
tests/e2e/router/test_reliability_routing_e2e.py
Normal file
302
tests/e2e/router/test_reliability_routing_e2e.py
Normal file
|
|
@ -0,0 +1,302 @@
|
|||
"""Live e2e: each routing strategy picks the deployment that strategy implies.
|
||||
|
||||
Every test builds its own model group of real `openai/gpt-5.5` deployments that
|
||||
differ in exactly one dimension the strategy under test reads - configured price,
|
||||
tpm ceiling, shuffle weight, or recorded latency - so only one
|
||||
deployment can win and the pick is a statement about the strategy rather than
|
||||
about luck. The strategy is selected per request through a
|
||||
`router_settings_override` on the /chat/completions body, so one long-lived proxy
|
||||
serves every strategy, and the deployment that actually served a call is read
|
||||
back from the x-litellm-model-id response header.
|
||||
|
||||
Every prompt carries a unique marker: the proxy under test has its response cache
|
||||
on, and a repeated prompt would be answered from cache before the router ever
|
||||
routed it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from typing import Literal
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel, ConfigDict
|
||||
|
||||
from complexity_router_client import ComplexityRouterClient
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import StreamingResponse, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatBody, ChatMessage, LiteLLMParamsBody, ModelInfoBody, ModelNewResponse
|
||||
from proxy_client import ProxyClient
|
||||
from reliability_support import REAL_KEY, REAL_MODEL
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
StrategyName = Literal[
|
||||
"simple-shuffle",
|
||||
"usage-based-routing-v2",
|
||||
"latency-based-routing",
|
||||
"cost-based-routing",
|
||||
]
|
||||
|
||||
PRICEY_PER_TOKEN = 1e-3
|
||||
CHEAP_PER_TOKEN = 1e-9
|
||||
EXHAUSTED_TPM = 1
|
||||
UNREACHABLE_BASE = "http://127.0.0.1:9/v1"
|
||||
UNMEETABLE_DEADLINE_SECONDS = 0.001
|
||||
|
||||
ANSWER_TOKENS = 256
|
||||
SHUFFLE_CALLS = 4
|
||||
|
||||
GROUP_READY_TIMEOUT_SECONDS = 30.0
|
||||
GROUP_READY_POLL_SECONDS = 0.5
|
||||
|
||||
SHORT_PROMPT = "answer in one word: what colour is a clear midday sky?"
|
||||
|
||||
|
||||
class RoutingParams(LiteLLMParamsBody):
|
||||
"""/model/new litellm_params plus the two per-deployment routing knobs the
|
||||
shared body does not model: the tpm ceiling usage-based routing reads, and the
|
||||
pick weight simple-shuffle reads."""
|
||||
|
||||
tpm: int | None = None
|
||||
weight: int | None = None
|
||||
|
||||
|
||||
class RoutingModelNewBody(BaseModel):
|
||||
"""POST /model/new carrying RoutingParams.
|
||||
|
||||
The shared ModelNewBody types `litellm_params` as LiteLLMParamsBody, and pydantic
|
||||
serializes a field by its declared type, so handing it a RoutingParams would drop
|
||||
tpm and weight from the request body without a word - every routing strategy
|
||||
would then see two identical deployments."""
|
||||
|
||||
model_config = ConfigDict(protected_namespaces=())
|
||||
model_name: str
|
||||
litellm_params: RoutingParams
|
||||
model_info: ModelInfoBody = ModelInfoBody()
|
||||
|
||||
|
||||
class StrategyOverride(BaseModel):
|
||||
"""The `router_settings_override` that picks a routing strategy for one call,
|
||||
and optionally the deadline that call is held to."""
|
||||
|
||||
routing_strategy: StrategyName
|
||||
timeout: float | None = None
|
||||
|
||||
|
||||
class StrategyChatBody(ChatBody):
|
||||
"""A /chat/completions body routed by one strategy. Composes ChatBody."""
|
||||
|
||||
router_settings_override: StrategyOverride
|
||||
|
||||
|
||||
def _group_size(client: ComplexityRouterClient, group: str) -> int:
|
||||
return sum(1 for entry in client.proxy.model_info() if entry.model_name == group)
|
||||
|
||||
|
||||
def _await_group_size(client: ComplexityRouterClient, group: str, size: int) -> None:
|
||||
"""Block until the proxy serves `size` deployments under `group`.
|
||||
|
||||
/model/new answers before the new deployment is necessarily on the router that
|
||||
serves the next call, and a group name shows up on /v1/models as soon as its
|
||||
first deployment lands, so routing without this gate can credit a strategy for a
|
||||
pick it never had a choice in."""
|
||||
deadline = time.monotonic() + GROUP_READY_TIMEOUT_SECONDS
|
||||
while time.monotonic() < deadline:
|
||||
if _group_size(client, group) >= size:
|
||||
return
|
||||
time.sleep(GROUP_READY_POLL_SECONDS)
|
||||
raise AssertionError(
|
||||
f"the proxy never reported {size} deployments under {group!r} within "
|
||||
f"{GROUP_READY_TIMEOUT_SECONDS}s of registering them"
|
||||
)
|
||||
|
||||
|
||||
def _deploy(
|
||||
client: ComplexityRouterClient,
|
||||
resources: ResourceManager,
|
||||
group: str,
|
||||
*,
|
||||
tpm: int | None = None,
|
||||
weight: int | None = None,
|
||||
input_cost_per_token: float | None = None,
|
||||
output_cost_per_token: float | None = None,
|
||||
api_base: str | None = None,
|
||||
) -> str:
|
||||
"""Register one real gpt-5.5 deployment under `group`, wait for the proxy to
|
||||
serve it alongside the group's existing deployments, delete it on teardown, and
|
||||
return the model_id the router reports as the server of a call."""
|
||||
expected = _group_size(client, group) + 1
|
||||
model_id = unwrap(
|
||||
client.proxy.transport.post(
|
||||
"/model/new",
|
||||
headers=client.proxy.transport.master,
|
||||
json=RoutingModelNewBody(
|
||||
model_name=group,
|
||||
litellm_params=RoutingParams(
|
||||
model=REAL_MODEL,
|
||||
api_key=REAL_KEY,
|
||||
tpm=tpm,
|
||||
weight=weight,
|
||||
input_cost_per_token=input_cost_per_token,
|
||||
output_cost_per_token=output_cost_per_token,
|
||||
api_base=api_base,
|
||||
),
|
||||
),
|
||||
response_type=ModelNewResponse,
|
||||
)
|
||||
).model_id
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
_await_group_size(client, group, expected)
|
||||
return model_id
|
||||
|
||||
|
||||
def _route(
|
||||
proxy: ProxyClient,
|
||||
key: str,
|
||||
group: str,
|
||||
strategy: StrategyName,
|
||||
*,
|
||||
prompt: str = SHORT_PROMPT,
|
||||
max_tokens: int = ANSWER_TOKENS,
|
||||
timeout: float | None = None,
|
||||
) -> StreamingResponse:
|
||||
"""Drive one real completion through `group` under `strategy`, returning the raw
|
||||
outcome so the caller can read the deployment header off it."""
|
||||
return proxy.transport.send(
|
||||
"/chat/completions",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=StrategyChatBody(
|
||||
model=group,
|
||||
messages=[ChatMessage(role="user", content=f"{prompt} [{unique_marker()}]")],
|
||||
max_tokens=max_tokens,
|
||||
router_settings_override=StrategyOverride(routing_strategy=strategy, timeout=timeout),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _served_by(resp: StreamingResponse) -> str:
|
||||
"""The model_id of the deployment that served a successful call."""
|
||||
assert resp.status_code == 200, f"routed call failed with {resp.status_code}: {resp.body[:300]}"
|
||||
model_id = resp.headers.get("x-litellm-model-id")
|
||||
assert model_id, (
|
||||
f"response carries no x-litellm-model-id, so the served deployment is unknowable; "
|
||||
f"headers={sorted(resp.headers)}"
|
||||
)
|
||||
return model_id
|
||||
|
||||
|
||||
class TestReliabilityRoutingStrategies:
|
||||
@pytest.mark.covers("reliability.routing.cost_based.picks_lowest_cost")
|
||||
def test_cost_based_picks_lowest_cost(
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
"""Two deployments of the same model priced a million-fold apart: cost-based
|
||||
routing must spend the request on the cheap one."""
|
||||
group = f"routing-cost-{unique_marker()}"
|
||||
pricey = _deploy(
|
||||
client,
|
||||
resources,
|
||||
group,
|
||||
input_cost_per_token=PRICEY_PER_TOKEN,
|
||||
output_cost_per_token=PRICEY_PER_TOKEN,
|
||||
)
|
||||
cheap = _deploy(
|
||||
client,
|
||||
resources,
|
||||
group,
|
||||
input_cost_per_token=CHEAP_PER_TOKEN,
|
||||
output_cost_per_token=CHEAP_PER_TOKEN,
|
||||
)
|
||||
|
||||
served = _served_by(_route(client.proxy, scoped_key, group, "cost-based-routing"))
|
||||
assert served == cheap, (
|
||||
f"cost-based routing served {served}, but the cheap deployment is {cheap} "
|
||||
f"(the other, {pricey}, costs {PRICEY_PER_TOKEN / CHEAP_PER_TOKEN:.0e}x more per token)"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("reliability.routing.usage_based.picks_under_tpm")
|
||||
def test_usage_based_picks_deployment_under_tpm(
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
"""One deployment's tpm ceiling is a single token, so no real prompt fits under
|
||||
it; usage-based routing must place the request on the deployment that has
|
||||
token budget left rather than over-allocating the capped one."""
|
||||
group = f"routing-tpm-{unique_marker()}"
|
||||
capped = _deploy(client, resources, group, tpm=EXHAUSTED_TPM)
|
||||
uncapped = _deploy(client, resources, group)
|
||||
|
||||
served = _served_by(_route(client.proxy, scoped_key, group, "usage-based-routing-v2"))
|
||||
assert served == uncapped, (
|
||||
f"usage-based routing served {served}, but only {uncapped} had tpm budget for the "
|
||||
f"request; {capped} is capped at {EXHAUSTED_TPM} tpm and cannot fit any prompt"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("reliability.routing.simple_shuffle.picks_healthy_deployment")
|
||||
def test_simple_shuffle_picks_healthy_deployment(
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
"""A zero-weight deployment pointed at an unreachable base sits next to the
|
||||
healthy one: simple-shuffle's weighted pick must land every call on the
|
||||
healthy deployment, so every call is answered instead of dying on a dead
|
||||
connection."""
|
||||
group = f"routing-shuffle-{unique_marker()}"
|
||||
unreachable = _deploy(client, resources, group, weight=0, api_base=UNREACHABLE_BASE)
|
||||
healthy = _deploy(client, resources, group, weight=1)
|
||||
|
||||
served = tuple(
|
||||
_served_by(_route(client.proxy, scoped_key, group, "simple-shuffle"))
|
||||
for _ in range(SHUFFLE_CALLS)
|
||||
)
|
||||
assert set(served) == {healthy}, (
|
||||
f"simple-shuffle served {served}, but every call had to land on the healthy "
|
||||
f"deployment {healthy}; {unreachable} carries weight 0 and an unreachable base"
|
||||
)
|
||||
|
||||
@pytest.mark.covers("reliability.routing.latency_based.picks_lowest_latency")
|
||||
def test_latency_based_picks_lowest_latency(
|
||||
self, client: ComplexityRouterClient, resources: ResourceManager, scoped_key: str
|
||||
) -> None:
|
||||
"""The first deployment answers a request held to a deadline it cannot meet,
|
||||
which is the router's worst latency observation; the second then joins the
|
||||
group and answers normally. Both deployments are configured identically and
|
||||
stay healthy - the deadline lived on the request, not on the deployment - so
|
||||
the only thing separating them is the latency the router measured, and the
|
||||
assertion call has to go to the deployment that was quick.
|
||||
|
||||
The simple-shuffle control in the middle is there to rule out the other reason
|
||||
a router skips a deployment: it forces the pick onto the timed-out deployment
|
||||
by weight and gets an answer, which no cooled-down deployment would give."""
|
||||
group = f"routing-latency-{unique_marker()}"
|
||||
timed_out = _deploy(client, resources, group, weight=1)
|
||||
|
||||
missed_deadline = _route(
|
||||
client.proxy,
|
||||
scoped_key,
|
||||
group,
|
||||
"latency-based-routing",
|
||||
timeout=UNMEETABLE_DEADLINE_SECONDS,
|
||||
)
|
||||
assert missed_deadline.status_code == 408, (
|
||||
f"the seeding call was meant to exceed its {UNMEETABLE_DEADLINE_SECONDS}s deadline on "
|
||||
f"{timed_out}, got {missed_deadline.status_code}: {missed_deadline.body[:300]}"
|
||||
)
|
||||
|
||||
quick = _deploy(client, resources, group, weight=0)
|
||||
assert _served_by(_route(client.proxy, scoped_key, group, "latency-based-routing")) == quick, (
|
||||
f"a deployment with no latency history is the router's lowest known latency, so this "
|
||||
f"call should have gone to {quick}"
|
||||
)
|
||||
|
||||
control = _route(client.proxy, scoped_key, group, "simple-shuffle")
|
||||
assert _served_by(control) == timed_out, (
|
||||
f"the weighted control call should have been served by {timed_out}, proving it is "
|
||||
f"still healthy and selectable after its timeout"
|
||||
)
|
||||
|
||||
served = _served_by(_route(client.proxy, scoped_key, group, "latency-based-routing"))
|
||||
assert served == quick, (
|
||||
f"latency-based routing served {served}, but {quick} answered in the time {timed_out} "
|
||||
f"blew a deadline in, so it is the lower-latency deployment"
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue