Merge pull request #27703 from BerriAI/litellm_lit-2531-affinity-cross-group-fix
Some checks are pending
Unit Tests: Caching (Redis) / caching-redis (push) Waiting to run
Unit Tests: Proxy DB Operations / assert-shard-coverage (push) Waiting to run
Unit Tests: Proxy DB Operations / auth-checks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / budgets (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / custom-logging (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / db-and-spend (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / endpoints-and-responses (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / guardrails-hooks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / jwt-and-keys (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / key-generation (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / logging-misc (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-runtime (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-server-core (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / schema-migration (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-utils (push) Blocked by required conditions
Unit Tests: Security / security (push) Waiting to run

fix(router): pin Responses API affinity to Azure resource on model-group switch
This commit is contained in:
Sameer Kankute 2026-05-12 10:39:06 +05:30 • committed by GitHub
commit a47cf03838
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 458 additions and 4 deletions

View file

@ -1673,7 +1673,7 @@ class Router:
for cb in self.optional_callbacks
)
if not already_registered:
ec_callback = EncryptedContentAffinityCheck()
ec_callback = EncryptedContentAffinityCheck(router=self)
self.optional_callbacks.append(ec_callback)
litellm.logging_callback_manager.add_litellm_callback(ec_callback)

View file

@ -36,13 +36,16 @@ Safe to enable globally:
- No cache required.
"""
from typing import Any, List, Optional, cast
from typing import TYPE_CHECKING, Any, List, Optional, cast
from litellm._logging import verbose_router_logger
from litellm.integrations.custom_logger import CustomLogger, Span
from litellm.responses.utils import ResponsesAPIRequestUtils
from litellm.types.llms.openai import AllMessageValues
if TYPE_CHECKING:
from litellm.router import Router
class EncryptedContentAffinityCheck(CustomLogger):
"""
@ -55,8 +58,9 @@ class EncryptedContentAffinityCheck(CustomLogger):
Wired via ``Router(optional_pre_call_checks=["encrypted_content_affinity"])``.
"""
def __init__(self) -> None:
def __init__(self, router: Optional["Router"] = None) -> None:
super().__init__()
self.router = router
# ------------------------------------------------------------------
# Helpers
@ -119,6 +123,58 @@ class EncryptedContentAffinityCheck(CustomLogger):
return deployment
return None
@staticmethod
def _encryption_boundary_key(
litellm_params: Any,
) -> Optional[tuple]:
"""
``(api_base, api_key)`` pair identifying an Azure resource. Two
deployments sharing both are interchangeable for ``encrypted_content``
follow-ups; Azure rejects content produced by any other resource.
Accepts any object exposing dict-style ``.get(key, default)``: plain
dicts (the common case in ``healthy_deployments``) as well as
``LiteLLM_Params``-style Pydantic instances, which define a custom
``.get()``. A stricter ``isinstance(dict)`` guard would silently drop
the latter from boundary matching and fall back to the full pool —
i.e. trigger the exact ``invalid_encrypted_content`` failure this
check exists to prevent.
"""
getter = getattr(litellm_params, "get", None)
if not callable(getter):
return None
api_base = getter("api_base")
api_key = getter("api_key")
if not api_base or not api_key:
return None
return (api_base, api_key)
def _find_deployments_on_same_encryption_boundary(
self,
healthy_deployments: List[dict],
model_id: str,
) -> List[dict]:
"""
Deployments in ``healthy_deployments`` sharing the originating
deployment's ``(api_base, api_key)``. Returns ``[]`` if router isn't
wired in, the originating deployment was removed, or no boundary match.
"""
if self.router is None:
return []
originating = self.router.get_deployment(model_id=model_id)
if originating is None:
return []
boundary = self._encryption_boundary_key(
originating.litellm_params.model_dump(exclude_none=True)
)
if boundary is None:
return []
return [
d
for d in healthy_deployments
if self._encryption_boundary_key(d.get("litellm_params", {})) == boundary
]
# ------------------------------------------------------------------
# Request routing (pre-call filter)
# ------------------------------------------------------------------
@ -172,8 +228,25 @@ class EncryptedContentAffinityCheck(CustomLogger):
request_kwargs["_encrypted_content_affinity_pinned"] = True
return [deployment]
# Follow-up switched model_name (LIT-2531): pin by Azure resource instead.
boundary_matches = self._find_deployments_on_same_encryption_boundary(
healthy_deployments=typed_healthy_deployments,
model_id=model_id,
)
if boundary_matches:
verbose_router_logger.debug(
"EncryptedContentAffinityCheck: model_id=%s not in healthy_deployments; "
"pinning to %d deployment(s) on same encryption boundary",
model_id,
len(boundary_matches),
)
request_kwargs["_encrypted_content_affinity_pinned"] = True
return boundary_matches
verbose_router_logger.error(
"EncryptedContentAffinityCheck: decoded deployment=%s not found in healthy_deployments",
"EncryptedContentAffinityCheck: decoded deployment=%s not found in "
"healthy_deployments and no boundary match available; falling back to "
"full deployment pool (encrypted_content may be rejected upstream)",
model_id,
)
return typed_healthy_deployments

View file

@ -791,3 +791,384 @@ def test_encrypted_content_wrapping_empty_string():
assert extracted_model_id == model_id
assert unwrapped == original_content
# ---------------------------------------------------------------------------
# LIT-2531: cross-model-group fallback via encryption boundary (api_base + api_key)
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_affinity_falls_back_to_same_encryption_boundary_on_model_group_switch():
"""
LIT-2531: Client starts a session on gpt-5.3-codex, follow-up switches to
gpt-5.4 mid-chat (e.g. via Codex `model_migrations`). Affinity must pin to
the gpt-5.4 deployment on the SAME Azure resource as the originating
gpt-5.3-codex deployment -- otherwise Azure rejects the encrypted_content.
"""
first_resp = _build_mock_response(
output_items=[
{
"type": "reasoning",
"id": "rs_encrypted_xyz",
"status": "completed",
"encrypted_content": "gAAAAABpnW_yEYmSNEyOG...",
},
],
response_id="resp_first",
)
second_resp = _build_mock_response(
output_items=[
{
"type": "message",
"id": "msg_ok",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": "answer"}],
},
],
response_id="resp_second",
)
ACCOUNT_A_BASE = "https://account-a.openai.azure.com/"
ACCOUNT_A_KEY = "key-a"
ACCOUNT_B_BASE = "https://account-b.openai.azure.com/"
ACCOUNT_B_KEY = "key-b"
router = litellm.Router(
model_list=[
{
"model_name": "gpt-5.3-codex",
"litellm_params": {
"model": "azure/gpt-5.3-codex",
"api_base": ACCOUNT_A_BASE,
"api_key": ACCOUNT_A_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.3-codex-account-a"},
},
{
"model_name": "gpt-5.3-codex",
"litellm_params": {
"model": "azure/gpt-5.3-codex",
"api_base": ACCOUNT_B_BASE,
"api_key": ACCOUNT_B_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.3-codex-account-b"},
},
{
"model_name": "gpt-5.4",
"litellm_params": {
"model": "azure/gpt-5.4",
"api_base": ACCOUNT_A_BASE,
"api_key": ACCOUNT_A_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.4-account-a"},
},
{
"model_name": "gpt-5.4",
"litellm_params": {
"model": "azure/gpt-5.4",
"api_base": ACCOUNT_B_BASE,
"api_key": ACCOUNT_B_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.4-account-b"},
},
],
optional_pre_call_checks=["encrypted_content_affinity"],
num_retries=0,
)
def first_call_picks_account_a(seq):
for d in seq:
if d["model_info"]["id"] == "gpt-5.3-codex-account-a":
return d
return seq[0]
with (
patch(
"litellm.llms.custom_httpx.llm_http_handler.BaseLLMHTTPHandler.async_response_api_handler",
new_callable=AsyncMock,
return_value=first_resp,
),
patch(
"litellm.router_strategy.simple_shuffle.random.choice",
side_effect=first_call_picks_account_a,
),
):
r1 = await router.aresponses(model="gpt-5.3-codex", input="hi")
assert r1._hidden_params["model_id"] == "gpt-5.3-codex-account-a"
encoded_id = _extract_encoded_item_id(r1)
assert encoded_id.startswith("encitem_")
# simple_shuffle.random.choice NOT patched: prove affinity narrows the
# candidate pool to a single deployment regardless of which one shuffle picks.
with patch(
"litellm.llms.custom_httpx.llm_http_handler.BaseLLMHTTPHandler.async_response_api_handler",
new_callable=AsyncMock,
return_value=second_resp,
):
r2 = await router.aresponses(
model="gpt-5.4",
input=[
{
"type": "reasoning",
"id": encoded_id,
"encrypted_content": "gAAAAABpnW_yEYmSNEyOG...",
},
],
)
assert r2._hidden_params["model_id"] == "gpt-5.4-account-a"
@pytest.mark.asyncio
async def test_affinity_falls_back_to_same_boundary_on_alias_switch():
"""
LIT-2531 alias path: gpt-5.2-codex is a LiteLLM alias that points at the
same underlying Azure model as gpt-5.3-codex. Different model_name groups
in the router, so model_id-based pinning misses, but the encryption
boundary (api_base + api_key) is identical -> follow-up must still pin.
"""
first_resp = _build_mock_response(
output_items=[
{
"type": "reasoning",
"id": "rs_alias_xyz",
"status": "completed",
"encrypted_content": "gAAAAABpnW_yEYmSNEyOG...",
},
],
response_id="resp_alias_first",
)
second_resp = _build_mock_response(
output_items=[
{
"type": "message",
"id": "msg_alias_ok",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": "ok"}],
},
],
response_id="resp_alias_second",
)
ACCOUNT_A_BASE = "https://account-a.openai.azure.com/"
ACCOUNT_A_KEY = "key-a"
ACCOUNT_B_BASE = "https://account-b.openai.azure.com/"
ACCOUNT_B_KEY = "key-b"
router = litellm.Router(
model_list=[
{
"model_name": "gpt-5.3-codex",
"litellm_params": {
"model": "azure/gpt-5.3-codex",
"api_base": ACCOUNT_A_BASE,
"api_key": ACCOUNT_A_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.3-codex-account-a"},
},
{
"model_name": "gpt-5.3-codex",
"litellm_params": {
"model": "azure/gpt-5.3-codex",
"api_base": ACCOUNT_B_BASE,
"api_key": ACCOUNT_B_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.3-codex-account-b"},
},
{
"model_name": "gpt-5.2-codex",
"litellm_params": {
"model": "azure/gpt-5.3-codex",
"api_base": ACCOUNT_A_BASE,
"api_key": ACCOUNT_A_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.2-codex-account-a"},
},
{
"model_name": "gpt-5.2-codex",
"litellm_params": {
"model": "azure/gpt-5.3-codex",
"api_base": ACCOUNT_B_BASE,
"api_key": ACCOUNT_B_KEY,
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "gpt-5.2-codex-account-b"},
},
],
optional_pre_call_checks=["encrypted_content_affinity"],
num_retries=0,
)
def pick_account_a(seq):
for d in seq:
if d["model_info"]["id"] == "gpt-5.3-codex-account-a":
return d
return seq[0]
with (
patch(
"litellm.llms.custom_httpx.llm_http_handler.BaseLLMHTTPHandler.async_response_api_handler",
new_callable=AsyncMock,
return_value=first_resp,
),
patch(
"litellm.router_strategy.simple_shuffle.random.choice",
side_effect=pick_account_a,
),
):
r1 = await router.aresponses(model="gpt-5.3-codex", input="hi")
encoded_id = _extract_encoded_item_id(r1)
assert encoded_id.startswith("encitem_")
with patch(
"litellm.llms.custom_httpx.llm_http_handler.BaseLLMHTTPHandler.async_response_api_handler",
new_callable=AsyncMock,
return_value=second_resp,
):
r2 = await router.aresponses(
model="gpt-5.2-codex",
input=[
{
"type": "reasoning",
"id": encoded_id,
"encrypted_content": "gAAAAABpnW_yEYmSNEyOG...",
},
],
)
assert r2._hidden_params["model_id"] == "gpt-5.2-codex-account-a"
def test_boundary_fallback_no_router_ref_returns_empty():
"""
Standalone use (no router wired in) -> the boundary lookup short-circuits
to ``[]`` instead of crashing on ``None.get_deployment``.
"""
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
check = EncryptedContentAffinityCheck(router=None)
healthy = [
{
"model_info": {"id": "dep-1"},
"litellm_params": {"api_base": "https://x", "api_key": "k"},
}
]
matches = check._find_deployments_on_same_encryption_boundary(
healthy_deployments=healthy,
model_id="dep-2",
)
assert matches == []
def test_boundary_fallback_originating_deployment_removed_returns_empty():
"""
If the originating deployment has been removed from the router (e.g. via
/model/delete), ``router.get_deployment`` returns None and we return [] so
the caller falls back to the full healthy_deployments list.
"""
from unittest.mock import MagicMock
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
mock_router = MagicMock()
mock_router.get_deployment.return_value = None
check = EncryptedContentAffinityCheck(router=mock_router)
healthy = [
{
"model_info": {"id": "dep-1"},
"litellm_params": {"api_base": "https://x", "api_key": "k"},
}
]
matches = check._find_deployments_on_same_encryption_boundary(
healthy_deployments=healthy,
model_id="dep-removed",
)
assert matches == []
mock_router.get_deployment.assert_called_once_with(model_id="dep-removed")
def test_boundary_key_accepts_pydantic_litellm_params_instance():
"""
Regression: ``_encryption_boundary_key`` must accept any object exposing
dict-style ``.get()`` (incl. ``LiteLLM_Params`` Pydantic instances) — not
just plain dicts.
A stricter ``isinstance(dict)`` guard would silently return ``None`` for a
``LiteLLM_Params`` value, drop the deployment from boundary matching, and
fall back to the full pool — which is the exact ``invalid_encrypted_content``
failure this check exists to prevent.
"""
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
from litellm.types.router import LiteLLM_Params
pydantic_params = LiteLLM_Params(
model="azure/gpt-5.3-codex",
api_base="https://mateo-resource.openai.azure.com",
api_key="fake-azure-resource-key-a",
)
plain_params = {
"model": "azure/gpt-5.3-codex",
"api_base": "https://mateo-resource.openai.azure.com",
"api_key": "fake-azure-resource-key-a",
}
pydantic_key = EncryptedContentAffinityCheck._encryption_boundary_key(
pydantic_params
)
plain_key = EncryptedContentAffinityCheck._encryption_boundary_key(plain_params)
assert pydantic_key is not None
assert (
pydantic_key
== plain_key
== (
"https://mateo-resource.openai.azure.com",
"fake-azure-resource-key-a",
)
)
def test_boundary_key_rejects_non_dict_like_inputs():
"""
Inputs that don't expose ``.get()`` (None, lists, strings, ints) -> None.
Guards against accidentally treating a stray non-dict-like value as a
valid boundary.
"""
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
for bad in (None, [], "not a dict", 42, object()):
assert EncryptedContentAffinityCheck._encryption_boundary_key(bad) is None
assert (
EncryptedContentAffinityCheck._encryption_boundary_key(
{"api_base": "", "api_key": "k"}
)
is None
)
assert (
EncryptedContentAffinityCheck._encryption_boundary_key(
{"api_base": "https://x"}
)
is None
)