mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
* test(e2e): harness fixes for long_context, complexity router, UI, and unit coverage Point long_context_1m at 1M-capable models, harden complexity-smart-router registration and spend-log assertions, fix key models dropdown selectors, and add gateway/lifecycle/transport and claude_code unit tests * test(e2e): harden remaining stage failures in harness Register complexity-smart-router via create_model + callable probe, fix create-key UI navigation race, retry management writes and budget ALB 502s, mark Vertex count_tokens N/A when unsupported, and tighten tool_search model lists for Azure/Bedrock capability gaps * test(e2e): drop claude_code and harness unit tests from this PR Keep management, router, budget, and shared conftest harness fixes only * test(e2e): restore E2E_RESULT pytest_runtest_makereport hook Accidentally dropped in an earlier harness commit; Grafana status history depends on these structured log lines * test(e2e): drop management control-plane write retries Transient 500 retries do not fix the underlying control plane failures * test(e2e): skip stage-red claude_code cells; fix multi-window budget latency Mark the twelve failing claude_code matrix cells skip until product/config lands. Multi-window budget polls gpt-5.5 with max_tokens=1 instead of Claude so the reset wait stays under ALB target idle timeout rather than masking awselb 502s * test(e2e): require exactly one LLM-tier spend row for complexity router Keep alias membership for compose vs stage model names, but assert len(served) == 1 so a leaked classifier sub-call cannot pass. Also pin LIT-4521 skip and align LIT-4522/23/24 skip reasons * test(e2e): harden router callable probe and multi-window budget exhaustion _router_is_callable treated any non-success chat whose body lacked "Invalid model name" as callable, so an unpropagated probe key (401), a generic 502, or a connection reset let the session proceed and hit real "Invalid model name" failures inside the tests. Require a Success outcome instead; the reload-race 400 and every infra/auth error now correctly read as not-callable. The multi-window budget test capped the tight window at 3e-6, which gpt-5.5 exhausts on the first call but a cheaper CHEAP_OPENAI_MODEL might not within the 20-call loop, turning a reset test into a spurious "window never enforced" failure. Drop the tight cap to 1e-9 so the first billed call exhausts it regardless of model price; the roomy 1m window stays at 1.0 and never blocks. * test(e2e): use a tradeoff-decision prompt for the complexity router classifier "Is P equal to NP?" reads to the LLM classifier as a short yes/no question, so gpt-5.5 classified it SIMPLE and the request routed to the openai backend, which made the test fail even though the classifier was running. The tier definitions key on what the request demands, not how hard the answer is, and a short direct question maps to SIMPLE regardless of subject. Swap in "Should I pay off my mortgage early or invest the extra money instead?". It carries none of the heuristic scorer's reasoning/technical/code keywords and stays short, so heuristic scoring still lands SIMPLE (openai), but the LLM reads it as a decision that has to weigh tradeoffs and lands it above SIMPLE, which the config routes to anthropic. Any non-SIMPLE tier serves anthropic, so the classifier only has to avoid SIMPLE for the test to distinguish a real classifier run from the heuristic fallback.
127 lines
4.4 KiB
Python
127 lines
4.4 KiB
Python
"""Router suite's `client` fixture.
|
|
|
|
The shared lifecycle (resources/scoped_key), proxy liveness skip, and e2e marker
|
|
live in the parent tests/e2e/conftest.py. ComplexityRouterClient holds the shared
|
|
Gateway, so the `resources` fixture cleans up keys this suite creates.
|
|
|
|
Also registers `complexity-smart-router` via management /model/new when the
|
|
proxy does not already list it (compose has it in static config; stage does not).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections.abc import Iterator
|
|
|
|
import pytest
|
|
from requests import RequestException
|
|
|
|
from complexity_router_client import ComplexityRouterClient, build_client
|
|
from e2e_gateway import Gateway
|
|
from e2e_http import NoBody, Success
|
|
from lifecycle import ResourceManager
|
|
from models import (
|
|
ChatBody,
|
|
ChatMessage,
|
|
KeyGenerateBody,
|
|
LiteLLMParamsBody,
|
|
ModelsListResponse,
|
|
)
|
|
|
|
ROUTER_MODEL = "complexity-smart-router"
|
|
ROUTER_PARAMS = LiteLLMParamsBody(
|
|
model="auto_router/complexity_router",
|
|
complexity_router_config={
|
|
"classifier_type": "llm",
|
|
"classifier_llm_config": {"model": "gpt-5.5"},
|
|
"tiers": {
|
|
"SIMPLE": "gpt-5.5",
|
|
"MEDIUM": "claude-haiku-4-5",
|
|
"COMPLEX": "claude-haiku-4-5",
|
|
"REASONING": "claude-haiku-4-5",
|
|
},
|
|
},
|
|
)
|
|
# Key must be allowed to call the virtual router and both tier backends.
|
|
ROUTER_KEY_MODELS = [ROUTER_MODEL, "gpt-5.5", "claude-haiku-4-5"]
|
|
|
|
|
|
@pytest.fixture(scope="session")
|
|
def client() -> ComplexityRouterClient:
|
|
return build_client()
|
|
|
|
|
|
def _model_is_servable(gateway: Gateway, model_name: str) -> bool:
|
|
result = gateway.transport.get(
|
|
"/v1/models",
|
|
headers=gateway.transport.master,
|
|
params=NoBody(),
|
|
response_type=ModelsListResponse,
|
|
)
|
|
return isinstance(result, Success) and any(entry.id == model_name for entry in result.data.data)
|
|
|
|
|
|
def _router_is_callable(gateway: Gateway) -> bool:
|
|
"""True only when a short chat against the virtual router succeeds; every error
|
|
(the Invalid-model-name reload race, but also 401, 5xx, and network) counts as
|
|
not-callable so infra/auth blips can't be mistaken for a working router."""
|
|
key = gateway.generate_key(KeyGenerateBody(models=ROUTER_KEY_MODELS, user_id="e2e-complexity-probe"))
|
|
try:
|
|
result = gateway.chat(
|
|
key,
|
|
ChatBody(
|
|
model=ROUTER_MODEL,
|
|
messages=[ChatMessage(role="user", content="hi")],
|
|
max_tokens=1,
|
|
),
|
|
)
|
|
finally:
|
|
gateway.delete_key(key)
|
|
return isinstance(result, Success)
|
|
|
|
|
|
@pytest.fixture(scope="session", autouse=True)
|
|
def _ensure_complexity_smart_router( # pyright: ignore[reportUnusedFunction] # pytest autouse session fixture, wired by name
|
|
client: ComplexityRouterClient,
|
|
) -> Iterator[None]:
|
|
"""Ensure the complexity router virtual model exists for this session.
|
|
|
|
Compose already declares it in docker-compose.yml; stage does not. Register
|
|
via Gateway.create_model (waits for data-plane /v1/models) when missing, then
|
|
probe a real chat so a list-only false positive cannot pass the fixture.
|
|
"""
|
|
gateway = client.gateway
|
|
if _model_is_servable(gateway, ROUTER_MODEL) and _router_is_callable(gateway):
|
|
yield
|
|
return
|
|
|
|
try:
|
|
model_id = gateway.create_model(ROUTER_MODEL, ROUTER_PARAMS)
|
|
except (AssertionError, RequestException) as exc:
|
|
if _model_is_servable(gateway, ROUTER_MODEL) and _router_is_callable(gateway):
|
|
yield
|
|
return
|
|
raise AssertionError(
|
|
f"failed to register {ROUTER_MODEL!r} for the complexity router e2e "
|
|
f"(not listed/callable on the data plane and /model/new failed): {exc}"
|
|
) from exc
|
|
|
|
try:
|
|
if not _router_is_callable(gateway):
|
|
raise AssertionError(
|
|
f"{ROUTER_MODEL!r} registered as {model_id!r} and listed on "
|
|
f"/v1/models but chat still returns Invalid model name; "
|
|
f"data-plane router reload incomplete"
|
|
)
|
|
yield
|
|
finally:
|
|
gateway.delete_model(model_id)
|
|
|
|
|
|
@pytest.fixture
|
|
def complexity_key(resources: ResourceManager, client: ComplexityRouterClient) -> str:
|
|
"""Per-test key allowed to call the complexity router and its tier backends."""
|
|
key = client.gateway.generate_key(
|
|
KeyGenerateBody(models=ROUTER_KEY_MODELS, user_id="e2e-complexity-router")
|
|
)
|
|
resources.defer(lambda: client.gateway.delete_key(key))
|
|
return key
|