mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
test(e2e): pin auto-router tag-split, alias pricing, heuristic scope, and Responses routing regressions
This commit is contained in:
parent
035bd76669
commit
96cee087be
3 changed files with 645 additions and 0 deletions
|
|
@ -18,6 +18,15 @@
|
|||
- {id: reliability.routing.usage_based.picks_under_tpm, module: reliability, tier: P0, behavior: routing, variant: usage_based, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_tpm_rpm_v2.py", rationale: "Routes to lowest-TPM deployment; prevents over-allocation"}
|
||||
- {id: reliability.routing.least_busy.picks_lowest_traffic, module: reliability, tier: P1, behavior: routing, variant: least_busy, assertions: [picks_lowest_traffic], exercised_on: [chat_completions, messages], source: "router_strategy/least_busy.py", rationale: "Fewest in-flight requests"}
|
||||
- {id: reliability.routing.complexity_llm_classifier.routes_by_llm_tier, module: reliability, tier: P1, behavior: routing, variant: complexity_llm_classifier, assertions: [routes_by_llm_tier], exercised_on: [chat_completions], source: "router_strategy/complexity_router/complexity_router.py", fail_before_fix: proven, rationale: "v2 auto-router LLM complexity classifier runs over the proxy and routes by semantic tier instead of silently crashing on absent litellm_metadata and falling back to heuristic scoring"}
|
||||
- {id: reliability.routing.tagged_marker.request_tag_selects_marker, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [request_tag_selects_marker], exercised_on: [chat_completions], source: "litellm/router.py:11445", rationale: "Tagged request selects the tagged strategy marker under a shared model_name instead of the plain deployment registered first (GitHub issue #36619)"}
|
||||
- {id: reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [untagged_request_served_by_plain_deployment], exercised_on: [chat_completions, messages, responses], source: "litellm/router.py:11445", rationale: "Untagged requests to a shared model_name are served by the plain deployment on every call, never captured or errored by the tagged marker (GitHub issue #36620)"}
|
||||
- {id: reliability.routing.tagged_marker.header_tag_selects_marker, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [header_tag_selects_marker], exercised_on: [messages], source: "litellm/router.py:11445", rationale: "A request tagged only via the x-litellm-tags header selects the tagged marker on Anthropic-native /v1/messages (GitHub issue #36621)"}
|
||||
- {id: reliability.routing.tagged_marker.untagged_tier_deployments_still_served, module: reliability, tier: P1, behavior: routing, variant: tagged_marker, assertions: [untagged_tier_deployments_still_served], exercised_on: [chat_completions, messages], source: "litellm/router_strategy/tag_based_routing.py:433", rationale: "Routing tags the marker consumed no longer constrain deployment selection inside the routed tier group, so untagged tier deployments serve the rewrite (GitHub issue #36621)"}
|
||||
- {id: reliability.routing.tagged_marker.tag_semantics_stay_strict, module: reliability, tier: P1, behavior: routing, variant: tagged_marker, assertions: [tag_semantics_stay_strict], exercised_on: [chat_completions], source: "litellm/router_strategy/tag_based_routing.py:299", rationale: "Tag consumption must not loosen strict semantics: a tagged call aimed straight at an untagged deployment still gets the 401 tags-configuration denial"}
|
||||
- {id: reliability.routing.tagged_marker.responses_input_routes_through_marker, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [responses_input_routes_through_marker], exercised_on: [responses], source: "litellm/router.py:11489", rationale: "Tagged /v1/responses (header or litellm_metadata.tags, string or list input) routes through the marker to its tier, extending the GitHub issues #36620/#36621 tag split to the Responses surface"}
|
||||
- {id: reliability.routing.semantic_auto_router.responses_input_routed, module: reliability, tier: P0, behavior: routing, variant: semantic_auto_router, assertions: [responses_input_routed], exercised_on: [responses], source: "litellm/router_strategy/auto_router/auto_router.py:131", fail_before_fix: proven, rationale: "/v1/responses input is resolved into messages for the semantic auto-router pre-routing hook instead of failing 400 Unmapped LLM provider auto_router (GitHub PR #37333)"}
|
||||
- {id: reliability.routing.strategy_alias.custom_pricing_ignored, module: reliability, tier: P1, behavior: routing, variant: strategy_alias, assertions: [custom_pricing_ignored], exercised_on: [chat_completions], source: "litellm/router.py:11489", rationale: "Custom pricing on a strategy-router alias never prices the routed request; spend logs at the routed tier deployment's own rate (GitHub PR #36691)"}
|
||||
- {id: reliability.routing.complexity_heuristic.scores_current_ask_only, module: reliability, tier: P1, behavior: routing, variant: complexity_heuristic, assertions: [scores_current_ask_only], exercised_on: [chat_completions], source: "router_strategy/complexity_router/complexity_router.py:942", rationale: "The heuristic complexity classifier scores the caller's current ask only, so a keyword-heavy agent system prompt cannot inflate the tier (GitHub PR #36721)"}
|
||||
- {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"}
|
||||
- {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"}
|
||||
- {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"}
|
||||
|
|
|
|||
|
|
@ -744,6 +744,10 @@ class LiteLLMParamsBody(BaseModel):
|
|||
extra_headers: dict[str, str] | None = None
|
||||
use_in_pass_through: bool | None = None
|
||||
complexity_router_config: dict[str, object] | None = None
|
||||
auto_router_config: str | None = None
|
||||
auto_router_default_model: str | None = None
|
||||
auto_router_embedding_model: str | None = None
|
||||
tags: list[str] | None = None
|
||||
mock_response: str | None = None
|
||||
timeout: float | None = None
|
||||
tpm: int | None = None
|
||||
|
|
|
|||
632
tests/e2e/router/test_auto_router_regressions_e2e.py
Normal file
632
tests/e2e/router/test_auto_router_regressions_e2e.py
Normal file
|
|
@ -0,0 +1,632 @@
|
|||
"""Live e2e regression pins for strategy-router (auto-router) routing.
|
||||
|
||||
A strategy marker (an ``auto_router/complexity_router`` deployment) and a plain
|
||||
deployment can share one ``model_name``, split by tags once
|
||||
``enable_tag_filtering`` is on: tagged requests route through the marker to its
|
||||
tier models, untagged requests go to the plain deployment. That split, and the
|
||||
strategy-router alias behaviors around it, regressed repeatedly; each test here
|
||||
pins one fixed behavior:
|
||||
|
||||
- GitHub issue #36619: a tagged request selects the tagged marker under a
|
||||
shared name even when a plain deployment was registered first.
|
||||
- GitHub issue #36620: untagged requests keep being served by the plain
|
||||
deployment on every call, never captured or 400'd by the tagged marker.
|
||||
- GitHub issue #36621: a request tagged via the ``x-litellm-tags`` header
|
||||
routes through the marker even when the tier deployments carry no tags
|
||||
(the marker consumes the routing tags before deployment selection), while a
|
||||
tagged call aimed straight at an untagged deployment stays denied.
|
||||
- GitHub issues #36620/#36621 on /v1/responses: the same tag split holds for
|
||||
string and list input, whether the tag arrives in litellm_metadata or the
|
||||
x-litellm-tags header.
|
||||
- GitHub PR #37333: /v1/responses input is resolved into messages for a
|
||||
semantic ``auto_router`` deployment's pre-routing hook; such requests used
|
||||
to fail with 400 "Unmapped LLM provider auto_router" because only chat
|
||||
messages fed the route matcher.
|
||||
- GitHub PR #36691: custom pricing on the marker alias never prices the routed
|
||||
request; spend logs at the routed tier deployment's own rate.
|
||||
- GitHub PR #36721: the heuristic complexity classifier scores the caller's
|
||||
current ask only, so a large agent system prompt cannot inflate the tier.
|
||||
|
||||
Every deployment is registered via /model/new (stage has no static config for
|
||||
these) and ``enable_tag_filtering`` is flipped through /config/update and
|
||||
restored on teardown, mirroring TestRouterSettings in the management suite.
|
||||
The served deployment is always read back from the spend log's ``model``,
|
||||
which stores either the registered alias or the provider-prefixed form.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from collections.abc import Iterator
|
||||
from dataclasses import dataclass
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import AnthropicHeaders, AuthHeaders, NoBody, UnauthorizedError, unwrap
|
||||
from lifecycle import ResourceManager
|
||||
from models import (
|
||||
AnthropicMessagesBody,
|
||||
AnthropicMessagesResponse,
|
||||
ChatBody,
|
||||
ChatMessage,
|
||||
ChatMetadata,
|
||||
KeyGenerateBody,
|
||||
LiteLLMParamsBody,
|
||||
SpendLogRow,
|
||||
)
|
||||
from proxy_client import ProxyClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
PLAIN_MODEL = "anthropic/claude-sonnet-5"
|
||||
CHEAP_MODEL = "anthropic/claude-haiku-4-5"
|
||||
STRONG_MODEL = "openai/gpt-5.6"
|
||||
MAX_TOKENS = 16
|
||||
PLAIN_SERVED = frozenset({PLAIN_MODEL, "claude-sonnet-5"})
|
||||
CHEAP_SERVED = frozenset({CHEAP_MODEL, "claude-haiku-4-5"})
|
||||
EMBEDDING_MODEL = "openai/text-embedding-3-small"
|
||||
SEMANTIC_ROUTE_UTTERANCE = "summarize this quarterly revenue report into three bullet points"
|
||||
|
||||
KEYWORD_HEAVY_SYSTEM_PROMPT = (
|
||||
"You are the principal architecture assistant for a distributed systems platform. "
|
||||
"Analyze every request step by step: design the algorithm, prove its correctness, "
|
||||
"evaluate time and space complexity, and reason about concurrency, consistency, and "
|
||||
"fault tolerance tradeoffs. When asked, refactor and debug multi-threaded code, "
|
||||
"optimize database query plans, derive mathematical proofs, and explain the theorem "
|
||||
"or lemma behind each optimization. Think through edge cases rigorously before answering. "
|
||||
) * 4
|
||||
|
||||
|
||||
class TaggedAuthHeaders(AuthHeaders):
|
||||
x_litellm_tags: str | None = Field(default=None, serialization_alias="x-litellm-tags")
|
||||
|
||||
|
||||
class TaggedAnthropicHeaders(AnthropicHeaders):
|
||||
x_litellm_tags: str | None = Field(default=None, serialization_alias="x-litellm-tags")
|
||||
|
||||
|
||||
class ResponsesTagMetadata(BaseModel):
|
||||
tags: list[str]
|
||||
|
||||
|
||||
class ResponsesInputItem(BaseModel):
|
||||
role: str
|
||||
content: str
|
||||
|
||||
|
||||
class ResponsesBody(BaseModel):
|
||||
model: str
|
||||
input: str | list[ResponsesInputItem]
|
||||
max_output_tokens: int | None = None
|
||||
litellm_metadata: ResponsesTagMetadata | None = None
|
||||
|
||||
|
||||
class ResponsesApiResponse(BaseModel):
|
||||
"""Minimal /v1/responses answer shape; routing is proven from spend logs,
|
||||
so only the fields the assertions read are modeled."""
|
||||
|
||||
model_config = ConfigDict(extra="allow")
|
||||
id: str | None = None
|
||||
status: str | None = None
|
||||
model: str | None = None
|
||||
|
||||
|
||||
class RouterSettingsPatch(BaseModel):
|
||||
enable_tag_filtering: bool
|
||||
|
||||
|
||||
class ConfigUpdateBody(BaseModel):
|
||||
router_settings: RouterSettingsPatch
|
||||
|
||||
|
||||
class ConfigUpdateResponse(BaseModel):
|
||||
message: str
|
||||
|
||||
|
||||
class RouterCurrentValues(BaseModel):
|
||||
enable_tag_filtering: bool | None = None
|
||||
|
||||
|
||||
class RouterSettingsResponse(BaseModel):
|
||||
current_values: RouterCurrentValues
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class TagSplitDeployments:
|
||||
"""Scenario A mirrors the customer-shaped config from GitHub issue #36619:
|
||||
plain deployment registered first, tier deployment and marker both tagged.
|
||||
Scenario B flips both axes for GitHub issue #36621: marker registered first
|
||||
and its tier deployment left untagged, so routing depends neither on
|
||||
registration order nor on tier deployments carrying tags."""
|
||||
|
||||
tag_a: str
|
||||
shared_a: str
|
||||
tier_a: str
|
||||
tag_b: str
|
||||
shared_b: str
|
||||
tier_b: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ZeroPricedAlias:
|
||||
alias: str
|
||||
tier: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class HeuristicSplit:
|
||||
alias: str
|
||||
cheap: str
|
||||
strong: str
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class SemanticAutoRouter:
|
||||
marker: str
|
||||
target: str
|
||||
fallback: str
|
||||
embedding: str
|
||||
|
||||
|
||||
def _provider_key(env_var: str) -> str:
|
||||
return os.environ.get(env_var) or f"os.environ/{env_var}"
|
||||
|
||||
|
||||
def _uniform_tier_config(tier_model: str) -> dict[str, object]:
|
||||
return {
|
||||
"classifier_type": "heuristic",
|
||||
"tiers": {"SIMPLE": tier_model, "MEDIUM": tier_model, "COMPLEX": tier_model, "REASONING": tier_model},
|
||||
}
|
||||
|
||||
|
||||
def _read_tag_filtering(proxy: ProxyClient) -> bool | None:
|
||||
return unwrap(
|
||||
proxy.transport.get(
|
||||
"/router/settings",
|
||||
headers=proxy.transport.master,
|
||||
params=NoBody(),
|
||||
response_type=RouterSettingsResponse,
|
||||
)
|
||||
).current_values.enable_tag_filtering
|
||||
|
||||
|
||||
def _write_tag_filtering(proxy: ProxyClient, enabled: bool) -> None:
|
||||
response: Final = unwrap(
|
||||
proxy.transport.post(
|
||||
"/config/update",
|
||||
headers=proxy.transport.master,
|
||||
json=ConfigUpdateBody(router_settings=RouterSettingsPatch(enable_tag_filtering=enabled)),
|
||||
response_type=ConfigUpdateResponse,
|
||||
)
|
||||
)
|
||||
assert "success" in response.message.lower(), (
|
||||
f"/config/update reported {response.message!r}, expected a success message"
|
||||
)
|
||||
|
||||
|
||||
def _await_tag_filtering(proxy: ProxyClient, expected: bool) -> None:
|
||||
deadline: Final = time.monotonic() + proxy.poll_timeout
|
||||
while time.monotonic() < deadline:
|
||||
if _read_tag_filtering(proxy) is expected:
|
||||
return
|
||||
time.sleep(proxy.poll_interval)
|
||||
raise AssertionError(
|
||||
f"GET /router/settings never reported enable_tag_filtering={expected} after /config/update"
|
||||
)
|
||||
|
||||
|
||||
def _key_for(proxy: ProxyClient, resources: ResourceManager, models: list[str]) -> str:
|
||||
key: Final = proxy.generate_key(KeyGenerateBody(models=models, user_id="e2e-auto-router-regressions"))
|
||||
resources.defer(lambda: proxy.delete_key(key))
|
||||
return key
|
||||
|
||||
|
||||
def _hello_chat_body(model: str, tags: list[str] | None = None) -> ChatBody:
|
||||
return ChatBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=f"say hello {unique_marker()}")],
|
||||
max_tokens=MAX_TOKENS,
|
||||
metadata=ChatMetadata(tags=tags) if tags is not None else None,
|
||||
)
|
||||
|
||||
|
||||
def _hello_messages_body(model: str) -> AnthropicMessagesBody:
|
||||
return AnthropicMessagesBody(
|
||||
model=model,
|
||||
messages=[ChatMessage(role="user", content=f"say hello {unique_marker()}")],
|
||||
max_tokens=MAX_TOKENS,
|
||||
)
|
||||
|
||||
|
||||
def _assert_served_only_by(rows: list[SpendLogRow], allowed: frozenset[str], context: str) -> None:
|
||||
served: Final = tuple(row.model for row in rows)
|
||||
assert served and all(model in allowed for model in served), (
|
||||
f"{context}: expected every request to be served by one of {sorted(allowed)}, spend logs show {served}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def tag_filtering(proxy: ProxyClient) -> Iterator[None]:
|
||||
"""enable_tag_filtering is what splits tagged from untagged traffic in every
|
||||
scenario here. /config/update is the only write path for router_settings;
|
||||
the original value is restored on teardown so the shared proxy keeps its
|
||||
configuration for the rest of the run."""
|
||||
original: Final = bool(_read_tag_filtering(proxy))
|
||||
_write_tag_filtering(proxy, True)
|
||||
_await_tag_filtering(proxy, True)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_write_tag_filtering(proxy, original)
|
||||
_await_tag_filtering(proxy, original)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def split(proxy: ProxyClient, tag_filtering: None) -> Iterator[TagSplitDeployments]:
|
||||
marker: Final = unique_marker()
|
||||
deployments: Final = TagSplitDeployments(
|
||||
tag_a=f"e2e-split-a-{marker}",
|
||||
shared_a=f"e2e-autoroute-a-{marker}",
|
||||
tier_a=f"e2e-tier-a-{marker}",
|
||||
tag_b=f"e2e-split-b-{marker}",
|
||||
shared_b=f"e2e-autoroute-b-{marker}",
|
||||
tier_b=f"e2e-tier-b-{marker}",
|
||||
)
|
||||
anthropic_key: Final = _provider_key("ANTHROPIC_API_KEY")
|
||||
marker_params_a: Final = LiteLLMParamsBody(
|
||||
model="auto_router/complexity_router",
|
||||
complexity_router_config=_uniform_tier_config(deployments.tier_a),
|
||||
tags=[deployments.tag_a],
|
||||
)
|
||||
marker_params_b: Final = LiteLLMParamsBody(
|
||||
model="auto_router/complexity_router",
|
||||
complexity_router_config=_uniform_tier_config(deployments.tier_b),
|
||||
tags=[deployments.tag_b],
|
||||
)
|
||||
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
|
||||
(deployments.shared_a, LiteLLMParamsBody(model=PLAIN_MODEL, api_key=anthropic_key)),
|
||||
(deployments.tier_a, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=anthropic_key, tags=[deployments.tag_a])),
|
||||
(deployments.shared_a, marker_params_a),
|
||||
(deployments.shared_b, marker_params_b),
|
||||
(deployments.tier_b, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=anthropic_key)),
|
||||
(deployments.shared_b, LiteLLMParamsBody(model=PLAIN_MODEL, api_key=anthropic_key)),
|
||||
)
|
||||
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
|
||||
try:
|
||||
yield deployments
|
||||
finally:
|
||||
for model_id in created:
|
||||
proxy.delete_model(model_id)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def zero_priced_alias(proxy: ProxyClient) -> Iterator[ZeroPricedAlias]:
|
||||
marker: Final = unique_marker()
|
||||
named: Final = ZeroPricedAlias(alias=f"e2e-priced-alias-{marker}", tier=f"e2e-priced-tier-{marker}")
|
||||
alias_params: Final = LiteLLMParamsBody(
|
||||
model="auto_router/complexity_router",
|
||||
complexity_router_config=_uniform_tier_config(named.tier),
|
||||
input_cost_per_token=0.0,
|
||||
output_cost_per_token=0.0,
|
||||
)
|
||||
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
|
||||
(named.tier, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
|
||||
(named.alias, alias_params),
|
||||
)
|
||||
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
|
||||
try:
|
||||
yield named
|
||||
finally:
|
||||
for model_id in created:
|
||||
proxy.delete_model(model_id)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def heuristic_split(proxy: ProxyClient) -> Iterator[HeuristicSplit]:
|
||||
marker: Final = unique_marker()
|
||||
named: Final = HeuristicSplit(
|
||||
alias=f"e2e-heuristic-router-{marker}",
|
||||
cheap=f"e2e-heuristic-cheap-{marker}",
|
||||
strong=f"e2e-heuristic-strong-{marker}",
|
||||
)
|
||||
config: Final[dict[str, object]] = {
|
||||
"classifier_type": "heuristic",
|
||||
"token_thresholds": {"simple": 15, "complex": 400},
|
||||
"tiers": {"SIMPLE": named.cheap, "MEDIUM": named.strong, "COMPLEX": named.strong, "REASONING": named.strong},
|
||||
}
|
||||
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
|
||||
(named.cheap, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
|
||||
(named.strong, LiteLLMParamsBody(model=STRONG_MODEL, api_key=_provider_key("OPENAI_API_KEY"))),
|
||||
(named.alias, LiteLLMParamsBody(model="auto_router/complexity_router", complexity_router_config=config)),
|
||||
)
|
||||
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
|
||||
try:
|
||||
yield named
|
||||
finally:
|
||||
for model_id in created:
|
||||
proxy.delete_model(model_id)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def semantic_auto_router(proxy: ProxyClient) -> Iterator[SemanticAutoRouter]:
|
||||
marker: Final = unique_marker()
|
||||
named: Final = SemanticAutoRouter(
|
||||
marker=f"e2e-semantic-router-{marker}",
|
||||
target=f"e2e-semantic-target-{marker}",
|
||||
fallback=f"e2e-semantic-fallback-{marker}",
|
||||
embedding=f"e2e-semantic-embedding-{marker}",
|
||||
)
|
||||
router_config: Final = json.dumps(
|
||||
{"routes": [{"name": named.target, "utterances": [SEMANTIC_ROUTE_UTTERANCE], "score_threshold": 0.3}]}
|
||||
)
|
||||
marker_params: Final = LiteLLMParamsBody(
|
||||
model=f"auto_router/{named.marker}",
|
||||
auto_router_config=router_config,
|
||||
auto_router_default_model=named.fallback,
|
||||
auto_router_embedding_model=named.embedding,
|
||||
)
|
||||
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
|
||||
(named.embedding, LiteLLMParamsBody(model=EMBEDDING_MODEL, api_key=_provider_key("OPENAI_API_KEY"))),
|
||||
(named.target, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
|
||||
(named.fallback, LiteLLMParamsBody(model=PLAIN_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
|
||||
(named.marker, marker_params),
|
||||
)
|
||||
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
|
||||
try:
|
||||
yield named
|
||||
finally:
|
||||
for model_id in created:
|
||||
proxy.delete_model(model_id)
|
||||
|
||||
|
||||
class TestTagSplitRouting:
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.request_tag_selects_marker")
|
||||
def test_body_tagged_chat_routes_through_the_marker_to_its_tier(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins GitHub issue #36619: with tag filtering on, a chat request whose
|
||||
body metadata tags match the tagged marker under a shared model name is
|
||||
answered by the marker's tier deployment, not by the plain deployment
|
||||
that was registered under the name first."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
|
||||
chat: Final = unwrap(proxy.chat(key, _hello_chat_body(split.shared_a, tags=[split.tag_a])))
|
||||
assert chat.choices, "tagged chat through the shared name returned no choices"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_a}, "body-tagged chat on the shared name")
|
||||
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment")
|
||||
def test_untagged_chat_is_always_served_by_the_plain_deployment(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins GitHub issue #36620: untagged chat requests to the shared name
|
||||
succeed on every call and are all served by the plain deployment; the
|
||||
tagged marker never captures them, so no intermittent auto-router
|
||||
errors and no tier hijacking."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
|
||||
for _ in range(5):
|
||||
chat = unwrap(proxy.chat(key, _hello_chat_body(split.shared_a)))
|
||||
assert chat.choices, "untagged chat through the shared name returned no choices"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=5)
|
||||
_assert_served_only_by(rows, PLAIN_SERVED | {split.shared_a}, "untagged chat on the shared name")
|
||||
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment")
|
||||
def test_untagged_messages_is_served_by_the_plain_deployment(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins GitHub issue #36620 on the /v1/messages surface: an untagged
|
||||
Anthropic-native request to the shared name is served by the plain
|
||||
deployment, not captured by the tagged marker."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
|
||||
answer: Final = unwrap(proxy.messages(key, _hello_messages_body(split.shared_a)))
|
||||
assert answer.content or answer.choices, "untagged /v1/messages returned neither content nor choices"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, PLAIN_SERVED | {split.shared_a}, "untagged /v1/messages on the shared name")
|
||||
|
||||
|
||||
class TestUntaggedTierDeployments:
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.header_tag_selects_marker")
|
||||
def test_header_tagged_messages_routes_through_the_marker_to_an_untagged_tier(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins GitHub issue #36621: a /v1/messages request tagged only via the
|
||||
x-litellm-tags header selects the tagged marker, and the rewrite still
|
||||
lands on the tier deployment even though that deployment carries no
|
||||
tags, because the marker consumed the routing tags."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_b, split.tier_b])
|
||||
headers: Final = TaggedAnthropicHeaders(authorization=f"Bearer {key}", x_litellm_tags=split.tag_b)
|
||||
answer: Final = unwrap(
|
||||
proxy.transport.post(
|
||||
"/v1/messages",
|
||||
headers=headers,
|
||||
json=_hello_messages_body(split.shared_b),
|
||||
response_type=AnthropicMessagesResponse,
|
||||
)
|
||||
)
|
||||
assert answer.content or answer.choices, "header-tagged /v1/messages returned neither content nor choices"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_b}, "header-tagged /v1/messages on the shared name")
|
||||
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_tier_deployments_still_served")
|
||||
def test_body_tagged_chat_reaches_the_untagged_tier_after_marker_rewrite(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins the tag-consumption half of GitHub issue #36621: after the
|
||||
tagged marker rewrites the request to its tier model, the consumed
|
||||
routing tags no longer constrain deployment selection, so the untagged
|
||||
tier deployment serves the request instead of a strict-tag denial."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_b, split.tier_b])
|
||||
chat: Final = unwrap(proxy.chat(key, _hello_chat_body(split.shared_b, tags=[split.tag_b])))
|
||||
assert chat.choices, "body-tagged chat through the marker-first shared name returned no choices"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_b}, "body-tagged chat with untagged tier")
|
||||
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.tag_semantics_stay_strict")
|
||||
def test_tagged_call_straight_at_an_untagged_deployment_stays_denied(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""The tag-consumption fix must not loosen strict tag semantics: a
|
||||
tagged request aimed directly at an untagged deployment (no marker
|
||||
involved) is still rejected with the 401 tags-configuration error."""
|
||||
key: Final = _key_for(proxy, resources, [split.tier_b])
|
||||
result: Final = proxy.chat(key, _hello_chat_body(split.tier_b, tags=[split.tag_b]))
|
||||
assert isinstance(result, UnauthorizedError), (
|
||||
f"expected the tagged direct call to an untagged deployment to be denied with 401, got {result}"
|
||||
)
|
||||
|
||||
|
||||
class TestResponsesApiTagRouting:
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.responses_input_routes_through_marker")
|
||||
def test_header_tagged_responses_with_string_input_routes_to_the_tier(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins the /v1/responses surface of the tag split (GitHub issues
|
||||
#36620/#36621): a /v1/responses request with string input, tagged via
|
||||
the x-litellm-tags header, succeeds and routes through the tagged
|
||||
marker to its tier."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
|
||||
headers: Final = TaggedAuthHeaders(authorization=f"Bearer {key}", x_litellm_tags=split.tag_a)
|
||||
body: Final = ResponsesBody(
|
||||
model=split.shared_a, input=f"say hello {unique_marker()}", max_output_tokens=64
|
||||
)
|
||||
answer: Final = unwrap(
|
||||
proxy.transport.post("/v1/responses", headers=headers, json=body, response_type=ResponsesApiResponse)
|
||||
)
|
||||
assert answer.id, "header-tagged /v1/responses returned no response id"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_a}, "header-tagged /v1/responses string input")
|
||||
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.responses_input_routes_through_marker")
|
||||
def test_body_tagged_responses_with_list_input_routes_to_the_tier(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins the body-tag and list-input combination of the same split:
|
||||
/v1/responses with litellm_metadata.tags and structured input items
|
||||
routes through the tagged marker to its tier."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
|
||||
body: Final = ResponsesBody(
|
||||
model=split.shared_a,
|
||||
input=[ResponsesInputItem(role="user", content=f"say hello {unique_marker()}")],
|
||||
max_output_tokens=64,
|
||||
litellm_metadata=ResponsesTagMetadata(tags=[split.tag_a]),
|
||||
)
|
||||
answer: Final = unwrap(
|
||||
proxy.transport.post(
|
||||
"/v1/responses",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=body,
|
||||
response_type=ResponsesApiResponse,
|
||||
)
|
||||
)
|
||||
assert answer.id, "body-tagged /v1/responses returned no response id"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_a}, "body-tagged /v1/responses list input")
|
||||
|
||||
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment")
|
||||
def test_untagged_responses_is_served_by_the_plain_deployment(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
|
||||
) -> None:
|
||||
"""Pins the untagged half of the /v1/responses tag split: an untagged
|
||||
request to the shared name is served by the plain deployment, matching
|
||||
the chat and messages surfaces."""
|
||||
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
|
||||
body: Final = ResponsesBody(
|
||||
model=split.shared_a, input=f"say hello {unique_marker()}", max_output_tokens=64
|
||||
)
|
||||
answer: Final = unwrap(
|
||||
proxy.transport.post(
|
||||
"/v1/responses",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=body,
|
||||
response_type=ResponsesApiResponse,
|
||||
)
|
||||
)
|
||||
assert answer.id, "untagged /v1/responses returned no response id"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(rows, PLAIN_SERVED | {split.shared_a}, "untagged /v1/responses on the shared name")
|
||||
|
||||
|
||||
class TestStrategyAliasPricing:
|
||||
@pytest.mark.covers("reliability.routing.strategy_alias.custom_pricing_ignored")
|
||||
def test_zero_priced_alias_still_logs_spend_at_the_tier_rate(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, zero_priced_alias: ZeroPricedAlias
|
||||
) -> None:
|
||||
"""Pins GitHub PR #36691: custom pricing registered on a strategy-router
|
||||
alias never prices the routed request. The alias here carries explicit
|
||||
zero pricing, so any zero-spend row would prove the alias pricing was
|
||||
applied; the routed tier deployment's real rate must produce spend > 0."""
|
||||
key: Final = _key_for(proxy, resources, [zero_priced_alias.alias, zero_priced_alias.tier])
|
||||
chat: Final = unwrap(proxy.chat(key, _hello_chat_body(zero_priced_alias.alias)))
|
||||
assert chat.choices, "chat through the zero-priced alias returned no choices"
|
||||
rows: Final = proxy.poll_logs_for_key(
|
||||
key, min_rows=1, predicate=lambda logged: all((row.spend or 0.0) > 0.0 for row in logged)
|
||||
)
|
||||
_assert_served_only_by(rows, CHEAP_SERVED | {zero_priced_alias.tier}, "chat through the zero-priced alias")
|
||||
priced: Final = tuple((row.model, row.spend) for row in rows)
|
||||
assert all((row.spend or 0.0) > 0.0 for row in rows), (
|
||||
f"expected spend at the tier deployment's own rate, got zero-spend rows: {priced}"
|
||||
)
|
||||
|
||||
|
||||
class TestComplexityHeuristicScope:
|
||||
@pytest.mark.covers("reliability.routing.complexity_heuristic.scores_current_ask_only")
|
||||
def test_trivial_ask_behind_keyword_heavy_system_prompt_stays_on_the_cheap_tier(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, heuristic_split: HeuristicSplit
|
||||
) -> None:
|
||||
"""Pins GitHub PR #36721: the heuristic complexity classifier scores the
|
||||
caller's current ask alone. The trivial ask scores SIMPLE on its own,
|
||||
while the accompanying ~2KB agent system prompt is packed with enough
|
||||
reasoning and complexity keywords that scoring the combined text lands
|
||||
in REASONING; only ask-only scoring keeps this on the cheap tier."""
|
||||
key: Final = _key_for(
|
||||
proxy, resources, [heuristic_split.alias, heuristic_split.cheap, heuristic_split.strong]
|
||||
)
|
||||
body: Final = ChatBody(
|
||||
model=heuristic_split.alias,
|
||||
messages=[
|
||||
ChatMessage(role="system", content=KEYWORD_HEAVY_SYSTEM_PROMPT),
|
||||
ChatMessage(role="user", content=f"hi {unique_marker()}"),
|
||||
],
|
||||
max_tokens=MAX_TOKENS,
|
||||
)
|
||||
chat: Final = unwrap(proxy.chat(key, body))
|
||||
assert chat.choices, "chat through the heuristic router returned no choices"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(
|
||||
rows, CHEAP_SERVED | {heuristic_split.cheap}, "trivial ask behind a keyword-heavy system prompt"
|
||||
)
|
||||
|
||||
|
||||
class TestSemanticAutoRouterResponses:
|
||||
@pytest.mark.covers("reliability.routing.semantic_auto_router.responses_input_routed")
|
||||
def test_responses_input_reaches_the_semantic_auto_router(
|
||||
self, proxy: ProxyClient, resources: ResourceManager, semantic_auto_router: SemanticAutoRouter
|
||||
) -> None:
|
||||
"""Pins GitHub PR #37333: /v1/responses input is resolved into messages
|
||||
for the semantic auto-router's pre-routing hook, so the marker embeds
|
||||
the input, matches its route, and the target deployment serves the
|
||||
request; before the fix the hook saw no messages and the request
|
||||
failed with 400 "Unmapped LLM provider auto_router"."""
|
||||
key: Final = _key_for(
|
||||
proxy,
|
||||
resources,
|
||||
[semantic_auto_router.marker, semantic_auto_router.target, semantic_auto_router.fallback],
|
||||
)
|
||||
body: Final = ResponsesBody(
|
||||
model=semantic_auto_router.marker, input=SEMANTIC_ROUTE_UTTERANCE, max_output_tokens=64
|
||||
)
|
||||
answer: Final = unwrap(
|
||||
proxy.transport.post(
|
||||
"/v1/responses",
|
||||
headers=proxy.transport.bearer(key),
|
||||
json=body,
|
||||
response_type=ResponsesApiResponse,
|
||||
)
|
||||
)
|
||||
assert answer.id, "/v1/responses through the semantic auto-router returned no response id"
|
||||
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
|
||||
_assert_served_only_by(
|
||||
rows, CHEAP_SERVED | {semantic_auto_router.target}, "semantic auto-router /v1/responses string input"
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue