test(e2e): pin auto-router tag-split, alias pricing, heuristic scope, and Responses routing regressions

This commit is contained in:
mateo-berri 2026-08-18 14:49:12 -07:00
parent 035bd76669
commit 96cee087be
3 changed files with 645 additions and 0 deletions

View file

@ -18,6 +18,15 @@
- {id: reliability.routing.usage_based.picks_under_tpm, module: reliability, tier: P0, behavior: routing, variant: usage_based, assertions: [picks_under_tpm], exercised_on: [chat_completions, messages], source: "router_strategy/lowest_tpm_rpm_v2.py", rationale: "Routes to lowest-TPM deployment; prevents over-allocation"}
- {id: reliability.routing.least_busy.picks_lowest_traffic, module: reliability, tier: P1, behavior: routing, variant: least_busy, assertions: [picks_lowest_traffic], exercised_on: [chat_completions, messages], source: "router_strategy/least_busy.py", rationale: "Fewest in-flight requests"}
- {id: reliability.routing.complexity_llm_classifier.routes_by_llm_tier, module: reliability, tier: P1, behavior: routing, variant: complexity_llm_classifier, assertions: [routes_by_llm_tier], exercised_on: [chat_completions], source: "router_strategy/complexity_router/complexity_router.py", fail_before_fix: proven, rationale: "v2 auto-router LLM complexity classifier runs over the proxy and routes by semantic tier instead of silently crashing on absent litellm_metadata and falling back to heuristic scoring"}
- {id: reliability.routing.tagged_marker.request_tag_selects_marker, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [request_tag_selects_marker], exercised_on: [chat_completions], source: "litellm/router.py:11445", rationale: "Tagged request selects the tagged strategy marker under a shared model_name instead of the plain deployment registered first (GitHub issue #36619)"}
- {id: reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [untagged_request_served_by_plain_deployment], exercised_on: [chat_completions, messages, responses], source: "litellm/router.py:11445", rationale: "Untagged requests to a shared model_name are served by the plain deployment on every call, never captured or errored by the tagged marker (GitHub issue #36620)"}
- {id: reliability.routing.tagged_marker.header_tag_selects_marker, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [header_tag_selects_marker], exercised_on: [messages], source: "litellm/router.py:11445", rationale: "A request tagged only via the x-litellm-tags header selects the tagged marker on Anthropic-native /v1/messages (GitHub issue #36621)"}
- {id: reliability.routing.tagged_marker.untagged_tier_deployments_still_served, module: reliability, tier: P1, behavior: routing, variant: tagged_marker, assertions: [untagged_tier_deployments_still_served], exercised_on: [chat_completions, messages], source: "litellm/router_strategy/tag_based_routing.py:433", rationale: "Routing tags the marker consumed no longer constrain deployment selection inside the routed tier group, so untagged tier deployments serve the rewrite (GitHub issue #36621)"}
- {id: reliability.routing.tagged_marker.tag_semantics_stay_strict, module: reliability, tier: P1, behavior: routing, variant: tagged_marker, assertions: [tag_semantics_stay_strict], exercised_on: [chat_completions], source: "litellm/router_strategy/tag_based_routing.py:299", rationale: "Tag consumption must not loosen strict semantics: a tagged call aimed straight at an untagged deployment still gets the 401 tags-configuration denial"}
- {id: reliability.routing.tagged_marker.responses_input_routes_through_marker, module: reliability, tier: P0, behavior: routing, variant: tagged_marker, assertions: [responses_input_routes_through_marker], exercised_on: [responses], source: "litellm/router.py:11489", rationale: "Tagged /v1/responses (header or litellm_metadata.tags, string or list input) routes through the marker to its tier, extending the GitHub issues #36620/#36621 tag split to the Responses surface"}
- {id: reliability.routing.semantic_auto_router.responses_input_routed, module: reliability, tier: P0, behavior: routing, variant: semantic_auto_router, assertions: [responses_input_routed], exercised_on: [responses], source: "litellm/router_strategy/auto_router/auto_router.py:131", fail_before_fix: proven, rationale: "/v1/responses input is resolved into messages for the semantic auto-router pre-routing hook instead of failing 400 Unmapped LLM provider auto_router (GitHub PR #37333)"}
- {id: reliability.routing.strategy_alias.custom_pricing_ignored, module: reliability, tier: P1, behavior: routing, variant: strategy_alias, assertions: [custom_pricing_ignored], exercised_on: [chat_completions], source: "litellm/router.py:11489", rationale: "Custom pricing on a strategy-router alias never prices the routed request; spend logs at the routed tier deployment's own rate (GitHub PR #36691)"}
- {id: reliability.routing.complexity_heuristic.scores_current_ask_only, module: reliability, tier: P1, behavior: routing, variant: complexity_heuristic, assertions: [scores_current_ask_only], exercised_on: [chat_completions], source: "router_strategy/complexity_router/complexity_router.py:942", rationale: "The heuristic complexity classifier scores the caller's current ask only, so a keyword-heavy agent system prompt cannot inflate the tier (GitHub PR #36721)"}
- {id: reliability.cache.exact.returns_cached, module: reliability, tier: P1, behavior: cache, variant: exact, assertions: [returns_cached], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/caching.py", rationale: "Response cache returns cached on exact match"}
- {id: reliability.cache.prompt_caching_model_select.returns_cached, module: reliability, tier: P1, behavior: cache, variant: prompt_caching_model_select, assertions: [returns_cached], exercised_on: [chat_completions], source: "router_utils/prompt_caching_cache.py", rationale: "Selects model supporting prompt caching for cacheable prefix"}
- {id: reliability.circuit_breaker.redis.trips_then_recovers, module: reliability, tier: P0, behavior: circuit_breaker, variant: redis, assertions: [trips_then_recovers], exercised_on: [chat_completions, messages, embeddings], source: "litellm/caching/redis_cache.py:99", rationale: "Redis breaker CLOSED->OPEN->HALF_OPEN; guards all cache/rate-limit ops"}

View file

@ -744,6 +744,10 @@ class LiteLLMParamsBody(BaseModel):
extra_headers: dict[str, str] | None = None
use_in_pass_through: bool | None = None
complexity_router_config: dict[str, object] | None = None
auto_router_config: str | None = None
auto_router_default_model: str | None = None
auto_router_embedding_model: str | None = None
tags: list[str] | None = None
mock_response: str | None = None
timeout: float | None = None
tpm: int | None = None

View file

@ -0,0 +1,632 @@
"""Live e2e regression pins for strategy-router (auto-router) routing.
A strategy marker (an ``auto_router/complexity_router`` deployment) and a plain
deployment can share one ``model_name``, split by tags once
``enable_tag_filtering`` is on: tagged requests route through the marker to its
tier models, untagged requests go to the plain deployment. That split, and the
strategy-router alias behaviors around it, regressed repeatedly; each test here
pins one fixed behavior:
- GitHub issue #36619: a tagged request selects the tagged marker under a
shared name even when a plain deployment was registered first.
- GitHub issue #36620: untagged requests keep being served by the plain
deployment on every call, never captured or 400'd by the tagged marker.
- GitHub issue #36621: a request tagged via the ``x-litellm-tags`` header
routes through the marker even when the tier deployments carry no tags
(the marker consumes the routing tags before deployment selection), while a
tagged call aimed straight at an untagged deployment stays denied.
- GitHub issues #36620/#36621 on /v1/responses: the same tag split holds for
string and list input, whether the tag arrives in litellm_metadata or the
x-litellm-tags header.
- GitHub PR #37333: /v1/responses input is resolved into messages for a
semantic ``auto_router`` deployment's pre-routing hook; such requests used
to fail with 400 "Unmapped LLM provider auto_router" because only chat
messages fed the route matcher.
- GitHub PR #36691: custom pricing on the marker alias never prices the routed
request; spend logs at the routed tier deployment's own rate.
- GitHub PR #36721: the heuristic complexity classifier scores the caller's
current ask only, so a large agent system prompt cannot inflate the tier.
Every deployment is registered via /model/new (stage has no static config for
these) and ``enable_tag_filtering`` is flipped through /config/update and
restored on teardown, mirroring TestRouterSettings in the management suite.
The served deployment is always read back from the spend log's ``model``,
which stores either the registered alias or the provider-prefixed form.
"""
import json
import os
import time
from collections.abc import Iterator
from dataclasses import dataclass
from typing import Final
import pytest
from pydantic import BaseModel, ConfigDict, Field
from e2e_config import unique_marker
from e2e_http import AnthropicHeaders, AuthHeaders, NoBody, UnauthorizedError, unwrap
from lifecycle import ResourceManager
from models import (
AnthropicMessagesBody,
AnthropicMessagesResponse,
ChatBody,
ChatMessage,
ChatMetadata,
KeyGenerateBody,
LiteLLMParamsBody,
SpendLogRow,
)
from proxy_client import ProxyClient
pytestmark = pytest.mark.e2e
PLAIN_MODEL = "anthropic/claude-sonnet-5"
CHEAP_MODEL = "anthropic/claude-haiku-4-5"
STRONG_MODEL = "openai/gpt-5.6"
MAX_TOKENS = 16
PLAIN_SERVED = frozenset({PLAIN_MODEL, "claude-sonnet-5"})
CHEAP_SERVED = frozenset({CHEAP_MODEL, "claude-haiku-4-5"})
EMBEDDING_MODEL = "openai/text-embedding-3-small"
SEMANTIC_ROUTE_UTTERANCE = "summarize this quarterly revenue report into three bullet points"
KEYWORD_HEAVY_SYSTEM_PROMPT = (
"You are the principal architecture assistant for a distributed systems platform. "
"Analyze every request step by step: design the algorithm, prove its correctness, "
"evaluate time and space complexity, and reason about concurrency, consistency, and "
"fault tolerance tradeoffs. When asked, refactor and debug multi-threaded code, "
"optimize database query plans, derive mathematical proofs, and explain the theorem "
"or lemma behind each optimization. Think through edge cases rigorously before answering. "
) * 4
class TaggedAuthHeaders(AuthHeaders):
x_litellm_tags: str | None = Field(default=None, serialization_alias="x-litellm-tags")
class TaggedAnthropicHeaders(AnthropicHeaders):
x_litellm_tags: str | None = Field(default=None, serialization_alias="x-litellm-tags")
class ResponsesTagMetadata(BaseModel):
tags: list[str]
class ResponsesInputItem(BaseModel):
role: str
content: str
class ResponsesBody(BaseModel):
model: str
input: str | list[ResponsesInputItem]
max_output_tokens: int | None = None
litellm_metadata: ResponsesTagMetadata | None = None
class ResponsesApiResponse(BaseModel):
"""Minimal /v1/responses answer shape; routing is proven from spend logs,
so only the fields the assertions read are modeled."""
model_config = ConfigDict(extra="allow")
id: str | None = None
status: str | None = None
model: str | None = None
class RouterSettingsPatch(BaseModel):
enable_tag_filtering: bool
class ConfigUpdateBody(BaseModel):
router_settings: RouterSettingsPatch
class ConfigUpdateResponse(BaseModel):
message: str
class RouterCurrentValues(BaseModel):
enable_tag_filtering: bool | None = None
class RouterSettingsResponse(BaseModel):
current_values: RouterCurrentValues
@dataclass(frozen=True, slots=True)
class TagSplitDeployments:
"""Scenario A mirrors the customer-shaped config from GitHub issue #36619:
plain deployment registered first, tier deployment and marker both tagged.
Scenario B flips both axes for GitHub issue #36621: marker registered first
and its tier deployment left untagged, so routing depends neither on
registration order nor on tier deployments carrying tags."""
tag_a: str
shared_a: str
tier_a: str
tag_b: str
shared_b: str
tier_b: str
@dataclass(frozen=True, slots=True)
class ZeroPricedAlias:
alias: str
tier: str
@dataclass(frozen=True, slots=True)
class HeuristicSplit:
alias: str
cheap: str
strong: str
@dataclass(frozen=True, slots=True)
class SemanticAutoRouter:
marker: str
target: str
fallback: str
embedding: str
def _provider_key(env_var: str) -> str:
return os.environ.get(env_var) or f"os.environ/{env_var}"
def _uniform_tier_config(tier_model: str) -> dict[str, object]:
return {
"classifier_type": "heuristic",
"tiers": {"SIMPLE": tier_model, "MEDIUM": tier_model, "COMPLEX": tier_model, "REASONING": tier_model},
}
def _read_tag_filtering(proxy: ProxyClient) -> bool | None:
return unwrap(
proxy.transport.get(
"/router/settings",
headers=proxy.transport.master,
params=NoBody(),
response_type=RouterSettingsResponse,
)
).current_values.enable_tag_filtering
def _write_tag_filtering(proxy: ProxyClient, enabled: bool) -> None:
response: Final = unwrap(
proxy.transport.post(
"/config/update",
headers=proxy.transport.master,
json=ConfigUpdateBody(router_settings=RouterSettingsPatch(enable_tag_filtering=enabled)),
response_type=ConfigUpdateResponse,
)
)
assert "success" in response.message.lower(), (
f"/config/update reported {response.message!r}, expected a success message"
)
def _await_tag_filtering(proxy: ProxyClient, expected: bool) -> None:
deadline: Final = time.monotonic() + proxy.poll_timeout
while time.monotonic() < deadline:
if _read_tag_filtering(proxy) is expected:
return
time.sleep(proxy.poll_interval)
raise AssertionError(
f"GET /router/settings never reported enable_tag_filtering={expected} after /config/update"
)
def _key_for(proxy: ProxyClient, resources: ResourceManager, models: list[str]) -> str:
key: Final = proxy.generate_key(KeyGenerateBody(models=models, user_id="e2e-auto-router-regressions"))
resources.defer(lambda: proxy.delete_key(key))
return key
def _hello_chat_body(model: str, tags: list[str] | None = None) -> ChatBody:
return ChatBody(
model=model,
messages=[ChatMessage(role="user", content=f"say hello {unique_marker()}")],
max_tokens=MAX_TOKENS,
metadata=ChatMetadata(tags=tags) if tags is not None else None,
)
def _hello_messages_body(model: str) -> AnthropicMessagesBody:
return AnthropicMessagesBody(
model=model,
messages=[ChatMessage(role="user", content=f"say hello {unique_marker()}")],
max_tokens=MAX_TOKENS,
)
def _assert_served_only_by(rows: list[SpendLogRow], allowed: frozenset[str], context: str) -> None:
served: Final = tuple(row.model for row in rows)
assert served and all(model in allowed for model in served), (
f"{context}: expected every request to be served by one of {sorted(allowed)}, spend logs show {served}"
)
@pytest.fixture(scope="module")
def tag_filtering(proxy: ProxyClient) -> Iterator[None]:
"""enable_tag_filtering is what splits tagged from untagged traffic in every
scenario here. /config/update is the only write path for router_settings;
the original value is restored on teardown so the shared proxy keeps its
configuration for the rest of the run."""
original: Final = bool(_read_tag_filtering(proxy))
_write_tag_filtering(proxy, True)
_await_tag_filtering(proxy, True)
try:
yield
finally:
_write_tag_filtering(proxy, original)
_await_tag_filtering(proxy, original)
@pytest.fixture(scope="module")
def split(proxy: ProxyClient, tag_filtering: None) -> Iterator[TagSplitDeployments]:
marker: Final = unique_marker()
deployments: Final = TagSplitDeployments(
tag_a=f"e2e-split-a-{marker}",
shared_a=f"e2e-autoroute-a-{marker}",
tier_a=f"e2e-tier-a-{marker}",
tag_b=f"e2e-split-b-{marker}",
shared_b=f"e2e-autoroute-b-{marker}",
tier_b=f"e2e-tier-b-{marker}",
)
anthropic_key: Final = _provider_key("ANTHROPIC_API_KEY")
marker_params_a: Final = LiteLLMParamsBody(
model="auto_router/complexity_router",
complexity_router_config=_uniform_tier_config(deployments.tier_a),
tags=[deployments.tag_a],
)
marker_params_b: Final = LiteLLMParamsBody(
model="auto_router/complexity_router",
complexity_router_config=_uniform_tier_config(deployments.tier_b),
tags=[deployments.tag_b],
)
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
(deployments.shared_a, LiteLLMParamsBody(model=PLAIN_MODEL, api_key=anthropic_key)),
(deployments.tier_a, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=anthropic_key, tags=[deployments.tag_a])),
(deployments.shared_a, marker_params_a),
(deployments.shared_b, marker_params_b),
(deployments.tier_b, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=anthropic_key)),
(deployments.shared_b, LiteLLMParamsBody(model=PLAIN_MODEL, api_key=anthropic_key)),
)
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
try:
yield deployments
finally:
for model_id in created:
proxy.delete_model(model_id)
@pytest.fixture(scope="module")
def zero_priced_alias(proxy: ProxyClient) -> Iterator[ZeroPricedAlias]:
marker: Final = unique_marker()
named: Final = ZeroPricedAlias(alias=f"e2e-priced-alias-{marker}", tier=f"e2e-priced-tier-{marker}")
alias_params: Final = LiteLLMParamsBody(
model="auto_router/complexity_router",
complexity_router_config=_uniform_tier_config(named.tier),
input_cost_per_token=0.0,
output_cost_per_token=0.0,
)
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
(named.tier, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
(named.alias, alias_params),
)
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
try:
yield named
finally:
for model_id in created:
proxy.delete_model(model_id)
@pytest.fixture(scope="module")
def heuristic_split(proxy: ProxyClient) -> Iterator[HeuristicSplit]:
marker: Final = unique_marker()
named: Final = HeuristicSplit(
alias=f"e2e-heuristic-router-{marker}",
cheap=f"e2e-heuristic-cheap-{marker}",
strong=f"e2e-heuristic-strong-{marker}",
)
config: Final[dict[str, object]] = {
"classifier_type": "heuristic",
"token_thresholds": {"simple": 15, "complex": 400},
"tiers": {"SIMPLE": named.cheap, "MEDIUM": named.strong, "COMPLEX": named.strong, "REASONING": named.strong},
}
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
(named.cheap, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
(named.strong, LiteLLMParamsBody(model=STRONG_MODEL, api_key=_provider_key("OPENAI_API_KEY"))),
(named.alias, LiteLLMParamsBody(model="auto_router/complexity_router", complexity_router_config=config)),
)
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
try:
yield named
finally:
for model_id in created:
proxy.delete_model(model_id)
@pytest.fixture(scope="module")
def semantic_auto_router(proxy: ProxyClient) -> Iterator[SemanticAutoRouter]:
marker: Final = unique_marker()
named: Final = SemanticAutoRouter(
marker=f"e2e-semantic-router-{marker}",
target=f"e2e-semantic-target-{marker}",
fallback=f"e2e-semantic-fallback-{marker}",
embedding=f"e2e-semantic-embedding-{marker}",
)
router_config: Final = json.dumps(
{"routes": [{"name": named.target, "utterances": [SEMANTIC_ROUTE_UTTERANCE], "score_threshold": 0.3}]}
)
marker_params: Final = LiteLLMParamsBody(
model=f"auto_router/{named.marker}",
auto_router_config=router_config,
auto_router_default_model=named.fallback,
auto_router_embedding_model=named.embedding,
)
registrations: Final[tuple[tuple[str, LiteLLMParamsBody], ...]] = (
(named.embedding, LiteLLMParamsBody(model=EMBEDDING_MODEL, api_key=_provider_key("OPENAI_API_KEY"))),
(named.target, LiteLLMParamsBody(model=CHEAP_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
(named.fallback, LiteLLMParamsBody(model=PLAIN_MODEL, api_key=_provider_key("ANTHROPIC_API_KEY"))),
(named.marker, marker_params),
)
created: Final = tuple(proxy.create_model(name, params) for name, params in registrations)
try:
yield named
finally:
for model_id in created:
proxy.delete_model(model_id)
class TestTagSplitRouting:
@pytest.mark.covers("reliability.routing.tagged_marker.request_tag_selects_marker")
def test_body_tagged_chat_routes_through_the_marker_to_its_tier(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins GitHub issue #36619: with tag filtering on, a chat request whose
body metadata tags match the tagged marker under a shared model name is
answered by the marker's tier deployment, not by the plain deployment
that was registered under the name first."""
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
chat: Final = unwrap(proxy.chat(key, _hello_chat_body(split.shared_a, tags=[split.tag_a])))
assert chat.choices, "tagged chat through the shared name returned no choices"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_a}, "body-tagged chat on the shared name")
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment")
def test_untagged_chat_is_always_served_by_the_plain_deployment(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins GitHub issue #36620: untagged chat requests to the shared name
succeed on every call and are all served by the plain deployment; the
tagged marker never captures them, so no intermittent auto-router
errors and no tier hijacking."""
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
for _ in range(5):
chat = unwrap(proxy.chat(key, _hello_chat_body(split.shared_a)))
assert chat.choices, "untagged chat through the shared name returned no choices"
rows: Final = proxy.poll_logs_for_key(key, min_rows=5)
_assert_served_only_by(rows, PLAIN_SERVED | {split.shared_a}, "untagged chat on the shared name")
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment")
def test_untagged_messages_is_served_by_the_plain_deployment(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins GitHub issue #36620 on the /v1/messages surface: an untagged
Anthropic-native request to the shared name is served by the plain
deployment, not captured by the tagged marker."""
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
answer: Final = unwrap(proxy.messages(key, _hello_messages_body(split.shared_a)))
assert answer.content or answer.choices, "untagged /v1/messages returned neither content nor choices"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, PLAIN_SERVED | {split.shared_a}, "untagged /v1/messages on the shared name")
class TestUntaggedTierDeployments:
@pytest.mark.covers("reliability.routing.tagged_marker.header_tag_selects_marker")
def test_header_tagged_messages_routes_through_the_marker_to_an_untagged_tier(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins GitHub issue #36621: a /v1/messages request tagged only via the
x-litellm-tags header selects the tagged marker, and the rewrite still
lands on the tier deployment even though that deployment carries no
tags, because the marker consumed the routing tags."""
key: Final = _key_for(proxy, resources, [split.shared_b, split.tier_b])
headers: Final = TaggedAnthropicHeaders(authorization=f"Bearer {key}", x_litellm_tags=split.tag_b)
answer: Final = unwrap(
proxy.transport.post(
"/v1/messages",
headers=headers,
json=_hello_messages_body(split.shared_b),
response_type=AnthropicMessagesResponse,
)
)
assert answer.content or answer.choices, "header-tagged /v1/messages returned neither content nor choices"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_b}, "header-tagged /v1/messages on the shared name")
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_tier_deployments_still_served")
def test_body_tagged_chat_reaches_the_untagged_tier_after_marker_rewrite(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins the tag-consumption half of GitHub issue #36621: after the
tagged marker rewrites the request to its tier model, the consumed
routing tags no longer constrain deployment selection, so the untagged
tier deployment serves the request instead of a strict-tag denial."""
key: Final = _key_for(proxy, resources, [split.shared_b, split.tier_b])
chat: Final = unwrap(proxy.chat(key, _hello_chat_body(split.shared_b, tags=[split.tag_b])))
assert chat.choices, "body-tagged chat through the marker-first shared name returned no choices"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_b}, "body-tagged chat with untagged tier")
@pytest.mark.covers("reliability.routing.tagged_marker.tag_semantics_stay_strict")
def test_tagged_call_straight_at_an_untagged_deployment_stays_denied(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""The tag-consumption fix must not loosen strict tag semantics: a
tagged request aimed directly at an untagged deployment (no marker
involved) is still rejected with the 401 tags-configuration error."""
key: Final = _key_for(proxy, resources, [split.tier_b])
result: Final = proxy.chat(key, _hello_chat_body(split.tier_b, tags=[split.tag_b]))
assert isinstance(result, UnauthorizedError), (
f"expected the tagged direct call to an untagged deployment to be denied with 401, got {result}"
)
class TestResponsesApiTagRouting:
@pytest.mark.covers("reliability.routing.tagged_marker.responses_input_routes_through_marker")
def test_header_tagged_responses_with_string_input_routes_to_the_tier(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins the /v1/responses surface of the tag split (GitHub issues
#36620/#36621): a /v1/responses request with string input, tagged via
the x-litellm-tags header, succeeds and routes through the tagged
marker to its tier."""
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
headers: Final = TaggedAuthHeaders(authorization=f"Bearer {key}", x_litellm_tags=split.tag_a)
body: Final = ResponsesBody(
model=split.shared_a, input=f"say hello {unique_marker()}", max_output_tokens=64
)
answer: Final = unwrap(
proxy.transport.post("/v1/responses", headers=headers, json=body, response_type=ResponsesApiResponse)
)
assert answer.id, "header-tagged /v1/responses returned no response id"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_a}, "header-tagged /v1/responses string input")
@pytest.mark.covers("reliability.routing.tagged_marker.responses_input_routes_through_marker")
def test_body_tagged_responses_with_list_input_routes_to_the_tier(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins the body-tag and list-input combination of the same split:
/v1/responses with litellm_metadata.tags and structured input items
routes through the tagged marker to its tier."""
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
body: Final = ResponsesBody(
model=split.shared_a,
input=[ResponsesInputItem(role="user", content=f"say hello {unique_marker()}")],
max_output_tokens=64,
litellm_metadata=ResponsesTagMetadata(tags=[split.tag_a]),
)
answer: Final = unwrap(
proxy.transport.post(
"/v1/responses",
headers=proxy.transport.bearer(key),
json=body,
response_type=ResponsesApiResponse,
)
)
assert answer.id, "body-tagged /v1/responses returned no response id"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, CHEAP_SERVED | {split.tier_a}, "body-tagged /v1/responses list input")
@pytest.mark.covers("reliability.routing.tagged_marker.untagged_request_served_by_plain_deployment")
def test_untagged_responses_is_served_by_the_plain_deployment(
self, proxy: ProxyClient, resources: ResourceManager, split: TagSplitDeployments
) -> None:
"""Pins the untagged half of the /v1/responses tag split: an untagged
request to the shared name is served by the plain deployment, matching
the chat and messages surfaces."""
key: Final = _key_for(proxy, resources, [split.shared_a, split.tier_a])
body: Final = ResponsesBody(
model=split.shared_a, input=f"say hello {unique_marker()}", max_output_tokens=64
)
answer: Final = unwrap(
proxy.transport.post(
"/v1/responses",
headers=proxy.transport.bearer(key),
json=body,
response_type=ResponsesApiResponse,
)
)
assert answer.id, "untagged /v1/responses returned no response id"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(rows, PLAIN_SERVED | {split.shared_a}, "untagged /v1/responses on the shared name")
class TestStrategyAliasPricing:
@pytest.mark.covers("reliability.routing.strategy_alias.custom_pricing_ignored")
def test_zero_priced_alias_still_logs_spend_at_the_tier_rate(
self, proxy: ProxyClient, resources: ResourceManager, zero_priced_alias: ZeroPricedAlias
) -> None:
"""Pins GitHub PR #36691: custom pricing registered on a strategy-router
alias never prices the routed request. The alias here carries explicit
zero pricing, so any zero-spend row would prove the alias pricing was
applied; the routed tier deployment's real rate must produce spend > 0."""
key: Final = _key_for(proxy, resources, [zero_priced_alias.alias, zero_priced_alias.tier])
chat: Final = unwrap(proxy.chat(key, _hello_chat_body(zero_priced_alias.alias)))
assert chat.choices, "chat through the zero-priced alias returned no choices"
rows: Final = proxy.poll_logs_for_key(
key, min_rows=1, predicate=lambda logged: all((row.spend or 0.0) > 0.0 for row in logged)
)
_assert_served_only_by(rows, CHEAP_SERVED | {zero_priced_alias.tier}, "chat through the zero-priced alias")
priced: Final = tuple((row.model, row.spend) for row in rows)
assert all((row.spend or 0.0) > 0.0 for row in rows), (
f"expected spend at the tier deployment's own rate, got zero-spend rows: {priced}"
)
class TestComplexityHeuristicScope:
@pytest.mark.covers("reliability.routing.complexity_heuristic.scores_current_ask_only")
def test_trivial_ask_behind_keyword_heavy_system_prompt_stays_on_the_cheap_tier(
self, proxy: ProxyClient, resources: ResourceManager, heuristic_split: HeuristicSplit
) -> None:
"""Pins GitHub PR #36721: the heuristic complexity classifier scores the
caller's current ask alone. The trivial ask scores SIMPLE on its own,
while the accompanying ~2KB agent system prompt is packed with enough
reasoning and complexity keywords that scoring the combined text lands
in REASONING; only ask-only scoring keeps this on the cheap tier."""
key: Final = _key_for(
proxy, resources, [heuristic_split.alias, heuristic_split.cheap, heuristic_split.strong]
)
body: Final = ChatBody(
model=heuristic_split.alias,
messages=[
ChatMessage(role="system", content=KEYWORD_HEAVY_SYSTEM_PROMPT),
ChatMessage(role="user", content=f"hi {unique_marker()}"),
],
max_tokens=MAX_TOKENS,
)
chat: Final = unwrap(proxy.chat(key, body))
assert chat.choices, "chat through the heuristic router returned no choices"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(
rows, CHEAP_SERVED | {heuristic_split.cheap}, "trivial ask behind a keyword-heavy system prompt"
)
class TestSemanticAutoRouterResponses:
@pytest.mark.covers("reliability.routing.semantic_auto_router.responses_input_routed")
def test_responses_input_reaches_the_semantic_auto_router(
self, proxy: ProxyClient, resources: ResourceManager, semantic_auto_router: SemanticAutoRouter
) -> None:
"""Pins GitHub PR #37333: /v1/responses input is resolved into messages
for the semantic auto-router's pre-routing hook, so the marker embeds
the input, matches its route, and the target deployment serves the
request; before the fix the hook saw no messages and the request
failed with 400 "Unmapped LLM provider auto_router"."""
key: Final = _key_for(
proxy,
resources,
[semantic_auto_router.marker, semantic_auto_router.target, semantic_auto_router.fallback],
)
body: Final = ResponsesBody(
model=semantic_auto_router.marker, input=SEMANTIC_ROUTE_UTTERANCE, max_output_tokens=64
)
answer: Final = unwrap(
proxy.transport.post(
"/v1/responses",
headers=proxy.transport.bearer(key),
json=body,
response_type=ResponsesApiResponse,
)
)
assert answer.id, "/v1/responses through the semantic auto-router returned no response id"
rows: Final = proxy.poll_logs_for_key(key, min_rows=1)
_assert_served_only_by(
rows, CHEAP_SERVED | {semantic_auto_router.target}, "semantic auto-router /v1/responses string input"
)