mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
docs(auto-router compression): cut the explanatory comments back
The module, its routing hook and its tests carried long prose rationale where the repository allows only concise comments for genuinely complex logic. Trimmed to the non-obvious reasons and dropped the rest; no logic or test behaviour changes.
This commit is contained in:
parent
723bc2140f
commit
a0b07b4791
3 changed files with 49 additions and 126 deletions
|
|
@ -1,15 +1,10 @@
|
|||
"""
|
||||
Decouples prompt compression between an auto router's routing decision and the
|
||||
model it routes to. An auto router marker deployment may set
|
||||
``auto_router_routing_compression`` and/or ``auto_router_model_compression`` in its
|
||||
``litellm_params`` to name the compression guardrail that hop should use, or
|
||||
``"none"`` to run no compression on that hop. Neither key set means the request's
|
||||
own compression guardrails (key/team/model-level, or an "Always on" guardrail)
|
||||
apply to both hops unchanged, exactly as before this feature existed.
|
||||
Decouples prompt compression between an auto router's routing decision and the model
|
||||
it routes to, via ``auto_router_routing_compression`` / ``auto_router_model_compression``
|
||||
on the marker deployment: a guardrail name, or ``"none"``.
|
||||
|
||||
Once either key is set, this auto router is authoritative: every other compression
|
||||
guardrail is suppressed for that request, and only these two settings decide what
|
||||
each hop sees.
|
||||
Neither key set inherits today's behaviour. Either key set makes the auto router
|
||||
authoritative and suppresses every other compression guardrail for that request.
|
||||
"""
|
||||
|
||||
import contextvars
|
||||
|
|
@ -29,12 +24,8 @@ if TYPE_CHECKING:
|
|||
COMPRESSION_GUARDRAIL_PROVIDERS: Final = frozenset({"headroom", "compresr"})
|
||||
_NO_COMPRESSION: Final = "none"
|
||||
|
||||
# Compression guardrails this request's auto router has switched off. Deliberately a
|
||||
# ContextVar rather than a metadata key: `refresh_proxy_server_request_body_snapshot`
|
||||
# copies metadata into `proxy_server_request.body`, which deployments persist to spend
|
||||
# logs. A suppression list that reaches a log the caller can read is a list the caller
|
||||
# can replay, which would let any request switch off a PII or content-filter guardrail.
|
||||
# Nothing here is caller-supplied, so there is no marker to forge in the first place.
|
||||
# A ContextVar, not metadata: metadata reaches spend logs the caller can read, and a
|
||||
# suppression list they can read is one they can replay to disable any guardrail.
|
||||
_suppressed_compression_guardrails: Final[contextvars.ContextVar[frozenset[str]]] = contextvars.ContextVar(
|
||||
"litellm_auto_router_suppressed_compression_guardrails", default=frozenset()
|
||||
)
|
||||
|
|
@ -45,9 +36,8 @@ def suppressed_compression_guardrails() -> frozenset[str]:
|
|||
return _suppressed_compression_guardrails.get()
|
||||
|
||||
|
||||
# Whether `arm_pre_call` actually armed a model-side compression guardrail for this
|
||||
# request. Only the proxy calls `arm_pre_call`, so on the SDK path nothing arms and
|
||||
# nothing compresses; the router must not assume the model hop already ran.
|
||||
# Only the proxy calls `arm_pre_call`, so on the SDK path nothing arms and nothing
|
||||
# compresses; the router must not assume the model hop already ran.
|
||||
_model_hop_armed: Final[contextvars.ContextVar[bool]] = contextvars.ContextVar(
|
||||
"litellm_auto_router_model_hop_armed", default=False
|
||||
)
|
||||
|
|
@ -95,10 +85,8 @@ def policy_for_model(
|
|||
) -> AutoRouterCompressionPolicy | None:
|
||||
"""The compression policy of the auto router marker `model_alias` resolves to.
|
||||
|
||||
Both the proxy's pre-call arming and the router's routing hook resolve the policy
|
||||
through here, with the same tag rule, so an alias carrying several tag-scoped
|
||||
markers can never suppress one marker's guardrail and then route under another
|
||||
marker's policy.
|
||||
Pre-call arming and the routing hook both resolve through here, so an alias with
|
||||
several tag-scoped markers cannot suppress under one and then route under another.
|
||||
"""
|
||||
if llm_router is None:
|
||||
return None
|
||||
|
|
@ -113,13 +101,9 @@ def policy_for_model(
|
|||
tag_matched: Final = tuple(
|
||||
params for params in markers if (tags := params.get("tags")) and requested.issuperset(frozenset(tags))
|
||||
)
|
||||
# Only untagged markers may serve as the fallback. A marker scoped to tags this
|
||||
# request does not carry describes a different slice of traffic, so falling back
|
||||
# to it would apply, say, an "eu" policy to a "us" request purely on config order.
|
||||
# Untagged only: a marker scoped to tags this request lacks describes other traffic.
|
||||
untagged: Final = tuple(params for params in markers if not params.get("tags"))
|
||||
# Lazily, so the first marker carrying a policy still wins and the rest are never
|
||||
# read. A generator rather than a loop-local: the name is bound once per item and
|
||||
# never rebound, which `: Final` cannot express inside a loop body.
|
||||
# Lazy, so the first marker carrying a policy wins and the rest are never read.
|
||||
candidates: Final = (policy_from_litellm_params(params) for params in (*tag_matched, *untagged))
|
||||
return next((policy for policy in candidates if policy is not None), None)
|
||||
|
||||
|
|
@ -145,12 +129,8 @@ def _compression_guardrail_classes() -> tuple[type, ...]:
|
|||
def is_compression_guardrail(guardrail: object) -> bool:
|
||||
"""Whether `guardrail` is an instance of a compression guardrail provider.
|
||||
|
||||
Both hops are validated through here. The two policy fields are operator-supplied
|
||||
names and nothing else constrains them, so without this a name that resolves to an
|
||||
ordinary guardrail would be handed the conversation and invoked: the routing hop
|
||||
calls `apply_guardrail` directly, which POSTs the content wherever that guardrail
|
||||
sends it, and the model hop is added to `metadata["guardrails"]`, which runs it even
|
||||
when it is not `default_on`.
|
||||
Both hops validate through here: the policy fields are operator-supplied names, and
|
||||
an unvalidated one would get handed the conversation and invoked.
|
||||
"""
|
||||
classes: Final = _compression_guardrail_classes()
|
||||
return bool(classes) and isinstance(guardrail, classes)
|
||||
|
|
@ -185,9 +165,6 @@ async def arm_pre_call(
|
|||
if not isinstance(model_alias, str) or not model_alias:
|
||||
return
|
||||
|
||||
# Read-only until a policy is confirmed: creating the metadata bucket for every
|
||||
# request, including the vast majority with no auto-router compression policy,
|
||||
# would be an unwanted side effect of merely checking for one.
|
||||
from litellm.router_strategy.tag_based_routing import (
|
||||
_get_tags_from_request_kwargs, # pyright: ignore[reportPrivateUsage] # used in router.py and budget_limiter.py too
|
||||
)
|
||||
|
|
@ -209,8 +186,7 @@ async def arm_pre_call(
|
|||
)
|
||||
)
|
||||
|
||||
# Only a name that resolves to a real compression guardrail may be armed: this adds
|
||||
# it to `metadata["guardrails"]`, which runs it even when it is not `default_on`.
|
||||
# Arming adds the name to `metadata["guardrails"]`, which runs it even if not default_on.
|
||||
armed_model_hop: Final = policy.model is not None and any(
|
||||
guardrail.guardrail_name == policy.model for guardrail in _active_compression_guardrails()
|
||||
)
|
||||
|
|
@ -226,8 +202,7 @@ async def arm_pre_call(
|
|||
requested: Final = metadata.get("guardrails")
|
||||
existing: Final = tuple(requested) if isinstance(requested, (list, tuple)) else ()
|
||||
if policy.model not in existing:
|
||||
# A list, not a tuple: litellm_pre_call_utils tests this key with
|
||||
# isinstance(..., list) and extends it, and would drop a tuple on the floor.
|
||||
# A list: litellm_pre_call_utils isinstance-checks this key and drops a tuple.
|
||||
metadata["guardrails"] = [*existing, policy.model] # mutable-ok: this key's contract is a list
|
||||
|
||||
|
||||
|
|
@ -240,25 +215,17 @@ def _as_routing_messages(
|
|||
|
||||
async def messages_for_routing(
|
||||
policy: AutoRouterCompressionPolicy | None,
|
||||
# list[dict], not Sequence[Mapping]: the async_pre_routing_hook protocol in
|
||||
# litellm/types/router.py types `messages` as list[dict[str, Any]].
|
||||
# list[dict], not Sequence[Mapping]: fixed by the async_pre_routing_hook protocol.
|
||||
messages: list[dict[str, object]] | None, # mutable-ok: shape fixed by the pre-routing hook protocol
|
||||
request_kwargs: Mapping[str, object],
|
||||
) -> list[dict[str, object]] | None: # mutable-ok: shape fixed by the pre-routing hook protocol
|
||||
"""Messages to use for a routing decision, per `policy.routing`.
|
||||
"""Messages to use for a routing decision, per `policy.routing`. None means the
|
||||
caller should route on whatever it already has.
|
||||
|
||||
Returns None when the caller should route on whatever messages it already has.
|
||||
|
||||
Always reads the live messages, never a pre-guardrail copy of them. The routing
|
||||
hop compresses through a real guardrail, which POSTs the text to an external
|
||||
compression service, so it must see what every other guardrail has already done
|
||||
to the request. Routing on a snapshot taken before the pre-call hook would send
|
||||
a masking guardrail's own input straight back out of the proxy.
|
||||
|
||||
The consequence, when the model hop compressed and the two hops differ: the
|
||||
messages in hand are that guardrail's output, and there is no un-compressed copy
|
||||
left to route on. The routing decision reads the compressed text in that one
|
||||
combination rather than leaking the original.
|
||||
Reads the live messages, never a pre-guardrail copy: this compresses through a real
|
||||
guardrail that POSTs the text out, so routing on a pre-masking snapshot would leak
|
||||
what the masking guardrail stripped. When the model hop already compressed and the
|
||||
hops differ, routing therefore reads the compressed text rather than the original.
|
||||
"""
|
||||
if policy is None or policy.routing is None:
|
||||
return None
|
||||
|
|
@ -277,9 +244,6 @@ async def messages_for_routing(
|
|||
)
|
||||
return _as_routing_messages(messages)
|
||||
|
||||
# apply_guardrail below hands this guardrail the conversation and it POSTs the
|
||||
# content to whatever service backs it, so the name has to be a compression
|
||||
# guardrail rather than any guardrail the operator happened to name.
|
||||
if not is_compression_guardrail(guardrail):
|
||||
verbose_proxy_logger.warning(
|
||||
"AutoRouter compression: guardrail '%s' is not a compression guardrail; routing on uncompressed messages",
|
||||
|
|
@ -291,9 +255,8 @@ async def messages_for_routing(
|
|||
"structured_messages": _as_routing_messages(messages) # pyright: ignore[reportAssignmentType] # plain dicts, not AllMessageValues; see headroom.py's own use of this shape
|
||||
}
|
||||
model: Final = request_kwargs.get("model")
|
||||
# A throwaway request_data: apply_guardrail writes its stats onto this dict, not the
|
||||
# real request's metadata, so routing-side compression never double-counts against
|
||||
# extract_compression_saved_tokens's model-savings accounting.
|
||||
# Throwaway: apply_guardrail writes stats here, so routing never double-counts into
|
||||
# extract_compression_saved_tokens.
|
||||
stats_sink: Final = {"messages": messages, "model": model} # mutable-ok: apply_guardrail writes its stats here
|
||||
result: Final = await guardrail.apply_guardrail(
|
||||
inputs=inputs,
|
||||
|
|
|
|||
|
|
@ -13047,25 +13047,17 @@ class Router:
|
|||
team_id_from_request,
|
||||
)
|
||||
|
||||
# Resolved through the same tag-aware lookup the proxy's pre-call arming used,
|
||||
# so an alias carrying several tag-scoped markers cannot suppress one marker's
|
||||
# guardrail and then route under a different marker's policy.
|
||||
# Same tag-aware lookup the proxy's pre-call arming used, so an alias with
|
||||
# several tag-scoped markers cannot suppress under one and route under another.
|
||||
compression_policy: Final = policy_for_model(
|
||||
llm_router=self,
|
||||
model_alias=registered_model_name,
|
||||
team_id=team_id_from_request(request_kwargs),
|
||||
request_tags=_get_tags_from_request_kwargs(request_kwargs),
|
||||
)
|
||||
# When both hops share the same compression, the model-side guardrail already
|
||||
# ran in the proxy's ordinary pre-call hook and compressed `messages` in place
|
||||
# (arm_pre_call armed it whether or not it is `default_on`); reuse that result
|
||||
# for routing too instead of paying for a second compression call against the
|
||||
# same content.
|
||||
#
|
||||
# Only the proxy calls arm_pre_call, so that reuse is conditional on it having
|
||||
# actually run: on the SDK path nothing arms the model hop and nothing has
|
||||
# compressed anything, and taking the shortcut there would skip both hops and
|
||||
# silently serve the request with no compression at all.
|
||||
# Shared compression already ran in the pre-call hook, so reuse it rather than
|
||||
# compressing twice. Conditional on arming having actually happened: only the
|
||||
# proxy arms, and on the SDK path the shortcut would skip both hops entirely.
|
||||
needs_independent_routing_compression: Final = compression_policy is not None and not (
|
||||
compression_policy.is_same and compression_policy.model is not None and model_hop_compression_armed()
|
||||
)
|
||||
|
|
@ -13082,12 +13074,9 @@ class Router:
|
|||
input=input,
|
||||
specific_deployment=specific_deployment,
|
||||
)
|
||||
# The strategy only echoes back whatever `messages` it was handed, so a
|
||||
# routing-only compression must not leak into the response: the model call
|
||||
# and downstream deployment-context filtering both key off this field.
|
||||
# Compared by value, not identity: PreRoutingHookResponse is a pydantic model,
|
||||
# and pydantic reconstructs a validated list field rather than keeping the
|
||||
# exact object passed in, even when nothing about it changed.
|
||||
# Routing-only compression must not leak into the response: the model call and
|
||||
# deployment-context filtering key off this field. Compared by value, since
|
||||
# pydantic rebuilds the list rather than keeping the object passed in.
|
||||
pre_routing_hook_response: Final = (
|
||||
routed.model_copy(update={"messages": messages}) # mutable-ok: pydantic's model_copy takes a dict
|
||||
if routed is not None and routing_messages is not None and routed.messages == routing_messages
|
||||
|
|
|
|||
|
|
@ -1,20 +1,4 @@
|
|||
"""
|
||||
Unit tests for litellm.proxy.guardrails.auto_router_compression.
|
||||
|
||||
Covers:
|
||||
- policy_from_litellm_params: absent keys mean no policy; the "none" sentinel
|
||||
normalizes to explicit no-compression within an active policy; is_same
|
||||
- policy_for_model: finds the auto-router marker deployment for an alias, picks the
|
||||
tag-scoped marker the request's tags actually match, and never falls back to a
|
||||
marker scoped to tags the request does not carry
|
||||
- arm_pre_call: no-op without a policy; suppresses active compression guardrails
|
||||
through request-scoped state rather than metadata, which reaches spend logs a
|
||||
caller can read; arms the model-side guardrail even when it isn't default_on
|
||||
- messages_for_routing: no-op without a policy; compresses the live messages every
|
||||
earlier guardrail has already rewritten, never a pre-guardrail copy of them;
|
||||
never writes stats onto the caller's own request_kwargs (regression for
|
||||
double-counted compression savings)
|
||||
"""
|
||||
"""Unit tests for litellm.proxy.guardrails.auto_router_compression."""
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
|
@ -137,8 +121,7 @@ class TestPolicyForModel:
|
|||
assert policy == AutoRouterCompressionPolicy(routing="headroom-a", model=None)
|
||||
|
||||
def test_a_marker_scoped_to_other_tags_is_never_the_fallback(self):
|
||||
"""Regression: an "eu" marker describes a different slice of traffic, so a "us"
|
||||
request must not fall back to its policy just because it is configured first."""
|
||||
"""Regression: a "us" request must not fall back to an "eu" marker's policy."""
|
||||
router = _FakeRouter(
|
||||
[
|
||||
_marker({"auto_router_routing_compression": "headroom-eu"}, tags=["eu"]),
|
||||
|
|
@ -149,8 +132,7 @@ class TestPolicyForModel:
|
|||
assert policy == AutoRouterCompressionPolicy(routing="headroom-default", model=None)
|
||||
|
||||
def test_no_untagged_fallback_means_no_policy(self):
|
||||
"""With only tag-scoped markers and none matching, there is no policy to apply:
|
||||
inheriting an unrelated slice's compression is worse than inheriting nothing."""
|
||||
"""No matching marker means no policy, not an unrelated slice's compression."""
|
||||
router = _FakeRouter([_marker({"auto_router_routing_compression": "headroom-eu"}, tags=["eu"])])
|
||||
assert (
|
||||
policy_for_model(llm_router=router, model_alias="smart-router", team_id=None, request_tags=("us",)) is None
|
||||
|
|
@ -268,9 +250,8 @@ class TestArmPreCall:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_suppression_state_never_enters_request_metadata(self):
|
||||
"""Regression (security): a suppression list written to metadata is copied into
|
||||
proxy_server_request.body and persisted to spend logs, so a caller could read it
|
||||
back and replay it to switch off a PII or content-filter guardrail."""
|
||||
"""Regression (security): metadata reaches spend logs, so a suppression list
|
||||
there is one a caller could read back and replay to disable a guardrail."""
|
||||
guardrail = _RecordingCompressionGuardrail(guardrail_name="always-on-compression")
|
||||
import litellm
|
||||
|
||||
|
|
@ -312,9 +293,8 @@ class TestArmPreCall:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_arm_pre_call_keeps_no_copy_of_the_prompt(self):
|
||||
"""Regression (security): arm_pre_call runs before the pre-call guardrails, so
|
||||
any copy of the messages it retained would be the pre-masking text. Routing-side
|
||||
compression POSTs its input to an external service, so that copy must not exist."""
|
||||
"""Regression (security): arm_pre_call runs before the guardrails, so any copy it
|
||||
kept would be pre-masking text that routing then POSTs to an external service."""
|
||||
router = _FakeRouter([_marker({"auto_router_routing_compression": "headroom-a"})])
|
||||
data = {"model": "smart-router", "messages": [{"role": "user", "content": "my ssn is 123-45-6789"}]}
|
||||
|
||||
|
|
@ -337,10 +317,8 @@ class TestMessagesForRouting:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_routing_none_never_reaches_for_a_pre_guardrail_copy(self):
|
||||
"""Routing asked for no compression while the model hop compressed, so the
|
||||
messages in hand are that guardrail's output and no uncompressed copy survives.
|
||||
Routing reads them as-is: the alternative is keeping a pre-guardrail copy, which
|
||||
is the text a masking guardrail exists to remove."""
|
||||
"""No uncompressed copy survives the model hop, and keeping one would mean
|
||||
retaining the pre-masking text. Routing reads what it has."""
|
||||
policy = AutoRouterCompressionPolicy(routing=None, model="headroom-a")
|
||||
model_compressed = [{"role": "user", "content": "[COMPRESSED] the full original conversation"}]
|
||||
|
||||
|
|
@ -362,10 +340,8 @@ class TestMessagesForRouting:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_routing_compresses_what_the_other_guardrails_left_behind(self, registered_guardrail):
|
||||
"""Regression (security): routing-side compression POSTs its input to an external
|
||||
service, so it must read the live messages every earlier guardrail has already
|
||||
rewritten. Routing on a pre-guardrail copy would send a masking guardrail's own
|
||||
input straight back out of the proxy."""
|
||||
"""Regression (security): routing POSTs its input out, so it must read what the
|
||||
earlier guardrails left behind, not a pre-masking copy."""
|
||||
policy = AutoRouterCompressionPolicy(routing="fake-compress", model="headroom-b")
|
||||
masked = [{"role": "user", "content": "my ssn is [REDACTED]"}]
|
||||
|
||||
|
|
@ -376,10 +352,8 @@ class TestMessagesForRouting:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_non_compression_guardrail_is_never_invoked_for_routing(self, monkeypatch):
|
||||
"""Regression (security): the policy fields are operator-supplied names that
|
||||
nothing else constrains. apply_guardrail hands the guardrail the conversation
|
||||
and it POSTs that content to whatever service backs it, so naming an ordinary
|
||||
guardrail must not turn the routing hop into a way to ship prompts there."""
|
||||
"""Regression (security): naming an ordinary guardrail must not turn the routing
|
||||
hop into a way to ship prompts to whatever service backs it."""
|
||||
import litellm
|
||||
|
||||
other = _NonCompressionGuardrail(guardrail_name="pii-filter")
|
||||
|
|
@ -397,11 +371,8 @@ class TestMessagesForRouting:
|
|||
|
||||
@pytest.mark.asyncio
|
||||
async def test_guardrail_receives_a_throwaway_request_data_not_the_real_request_kwargs(self, registered_guardrail):
|
||||
"""Regression: a real compression guardrail writes its stats onto whatever
|
||||
`request_data` dict it's given (`add_standard_logging_guardrail_information_to_
|
||||
request_data`). If that were the caller's own `request_kwargs`, routing-side
|
||||
compression would double-count into extract_compression_saved_tokens, which
|
||||
sums every guardrail_information entry on the real request's metadata."""
|
||||
"""Regression: a guardrail writes stats onto the request_data it is given, so
|
||||
passing the caller's own would double-count into extract_compression_saved_tokens."""
|
||||
policy = AutoRouterCompressionPolicy(routing="fake-compress", model=None)
|
||||
messages = [{"role": "user", "content": "hi"}]
|
||||
request_kwargs = {"metadata": {}}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue