fix(router): strip encrypted reasoning the pinned deployment cannot decrypt (#43781)

Co-authored-by: yassin <yassin@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-09-30 09:56:27 -07:00 • committed by GitHub
parent 82d8b3797c
commit 2ed9761921
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 431 additions and 30 deletions

View file

@ -2015,7 +2015,11 @@ def is_unsignable_thinking_block(block: object) -> bool:
return not (isinstance(thinking_text, str) and len(thinking_text.strip()) > 0)
def strip_encrypted_reasoning_from_messages(messages: object) -> None:
def strip_encrypted_reasoning_from_messages(
messages: object,
*,
should_strip: Callable[[Mapping[str, object]], bool] | None = None,
) -> None:
"""Drop the bridge-tagged reasoning blocks a routed deployment cannot decrypt from
Anthropic-shaped history.
@ -2030,7 +2034,7 @@ def strip_encrypted_reasoning_from_messages(messages: object) -> None:
if not isinstance(messages, list):
return
for content in anthropic_content_lists(cast(list[object], messages)): # cast-ok: untyped client json
_strip_encrypted_reasoning_from_blocks(content)
_strip_encrypted_reasoning_from_blocks(content, should_strip=should_strip)
def anthropic_content_lists(messages: Sequence[object]) -> Iterator[object]:
@ -2043,9 +2047,18 @@ def anthropic_content_lists(messages: Sequence[object]) -> Iterator[object]:
)
def _strip_encrypted_reasoning_from_blocks(content: object) -> None:
def _strip_encrypted_reasoning_from_blocks(
content: object,
*,
should_strip: Callable[[Mapping[str, object]], bool] | None = None,
) -> None:
blocks: Final = cast(list[object], content) # cast-ok: narrowed by the caller's isinstance
kept: Final = tuple(block for block in blocks if not is_encrypted_reasoning_block(block))
kept: Final = tuple(
block
for block in blocks
if not is_encrypted_reasoning_block(block)
or (should_strip is not None and not should_strip(cast(Mapping[str, object], block)))
)
blocks[:] = kept

View file

@ -1,6 +1,6 @@
import base64
import re
from collections.abc import Iterable, Mapping, Sequence
from collections.abc import Callable, Iterable, Mapping, Sequence
from functools import reduce
from typing import Any, Final, Optional, TypeVar, Union, cast, get_type_hints, overload
@ -556,7 +556,11 @@ class ResponsesAPIRequestUtils:
return request_input
@staticmethod
def strip_encrypted_reasoning_from_input(request_input: object) -> None:
def strip_encrypted_reasoning_from_input(
request_input: object,
*,
should_strip: Callable[[Mapping[str, object]], bool] | None = None,
) -> None:
"""Drop reasoning items the routed deployment cannot decrypt, keeping their readable summary.
Mutates ``request_input`` in place: the router's fallback snapshot shares this
@ -565,7 +569,12 @@ class ResponsesAPIRequestUtils:
if not isinstance(request_input, list):
return
items: Final = cast(list[object], request_input) # cast-ok: untyped client json
stripped: Final = tuple(ResponsesAPIRequestUtils._without_encrypted_reasoning(item) for item in items)
stripped: Final = tuple(
ResponsesAPIRequestUtils._without_encrypted_reasoning(item)
if should_strip is None or (isinstance(item, Mapping) and should_strip(cast(Mapping[str, object], item)))
else item
for item in items
)
items[:] = (item for item in stripped if item is not None)
@staticmethod

View file

@ -36,7 +36,8 @@ Safe to enable globally:
- No cache required.
"""
from collections.abc import Iterator, Mapping
from collections.abc import Iterator, Mapping, Sequence
from functools import cache
from typing import TYPE_CHECKING, Final, Optional, cast
from litellm._logging import verbose_router_logger
@ -114,23 +115,31 @@ class EncryptedContentAffinityCheck(CustomLogger):
if not isinstance(request_input, list):
return None
for item in request_input:
if not isinstance(item, dict):
continue
return next(
(
model_id
for item in request_input
if (model_id := EncryptedContentAffinityCheck._model_id_of_input_item(item)) is not None
),
None,
)
# First, try to decode from item ID (if present)
item_id = item.get("id")
if item_id and isinstance(item_id, str):
decoded = ResponsesAPIRequestUtils._decode_encrypted_item_id(item_id)
if decoded:
return decoded.get("model_id")
@staticmethod
def _model_id_of_input_item(item: object) -> str | None:
if not isinstance(item, dict):
return None
# If no encoded ID, check if encrypted_content itself is wrapped
encrypted_content = item.get("encrypted_content")
if encrypted_content and isinstance(encrypted_content, str):
model_id = EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(encrypted_content)
if model_id:
return model_id
item_id: Final = item.get("id")
if item_id and isinstance(item_id, str):
decoded: Final = ResponsesAPIRequestUtils._decode_encrypted_item_id(item_id)
if decoded:
return decoded.get("model_id")
encrypted_content: Final = item.get("encrypted_content")
if encrypted_content and isinstance(encrypted_content, str):
model_id: Final = EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(encrypted_content)
if model_id:
return model_id
return None
@ -150,19 +159,20 @@ class EncryptedContentAffinityCheck(CustomLogger):
model_id, _ = ResponsesAPIRequestUtils._unwrap_encrypted_content_with_model_id(encrypted_content)
return model_id or None
@staticmethod
def _model_id_of_anthropic_block(block: Mapping[str, object]) -> str | None:
encrypted_content: Final = encrypted_content_of_block(block)
if encrypted_content is None:
return None
return EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(encrypted_content)
@staticmethod
def _extract_model_id_from_anthropic_messages(messages: object) -> str | None:
return next(
(
model_id
for block in EncryptedContentAffinityCheck._anthropic_content_blocks(messages)
if (encrypted_content := encrypted_content_of_block(block)) is not None
if (
model_id := EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(
encrypted_content
)
)
is not None
if (model_id := EncryptedContentAffinityCheck._model_id_of_anthropic_block(block)) is not None
),
None,
)
@ -243,6 +253,50 @@ class EncryptedContentAffinityCheck(CustomLogger):
]
return matches, originating
def _strip_reasoning_the_target_cannot_decrypt(
self,
request_input: object,
anthropic_messages: object,
target_deployments: Sequence[Mapping[str, object]],
) -> None:
target_ids: Final = frozenset(
str(model_info["id"])
for target in target_deployments
if isinstance((model_info := target.get("model_info")), Mapping) and model_info.get("id") is not None
)
target_boundaries: Final = frozenset(
boundary
for target in target_deployments
if (boundary := self._encryption_boundary_key(target.get("litellm_params"))) is not None
)
@cache
def target_can_decrypt(origin_model_id: str) -> bool:
if origin_model_id in target_ids:
return True
if self.router is None:
return False
origin: Final = self.router.get_deployment(model_id=origin_model_id)
origin_boundary: Final = (
self._encryption_boundary_key(origin.litellm_params.model_dump(exclude_none=True))
if origin is not None
else None
)
return origin_boundary is not None and origin_boundary in target_boundaries
def should_strip_input_item(item: Mapping[str, object]) -> bool:
origin_model_id: Final = self._model_id_of_input_item(item)
return origin_model_id is not None and not target_can_decrypt(origin_model_id)
def should_strip_anthropic_block(block: Mapping[str, object]) -> bool:
origin_model_id: Final = self._model_id_of_anthropic_block(block)
return origin_model_id is not None and not target_can_decrypt(origin_model_id)
ResponsesAPIRequestUtils.strip_encrypted_reasoning_from_input(
request_input, should_strip=should_strip_input_item
)
strip_encrypted_reasoning_from_messages(anthropic_messages, should_strip=should_strip_anthropic_block)
# ------------------------------------------------------------------
# Request routing (pre-call filter)
# ------------------------------------------------------------------
@ -303,6 +357,7 @@ class EncryptedContentAffinityCheck(CustomLogger):
model_id,
)
request_kwargs["_encrypted_content_affinity_pinned"] = True
self._strip_reasoning_the_target_cannot_decrypt(request_input, anthropic_messages, (deployment,))
return [deployment]
# Follow-up switched model_name (LIT-2531): pin by Azure resource instead.
@ -318,6 +373,7 @@ class EncryptedContentAffinityCheck(CustomLogger):
len(boundary_matches),
)
request_kwargs["_encrypted_content_affinity_pinned"] = True
self._strip_reasoning_the_target_cannot_decrypt(request_input, anthropic_messages, boundary_matches)
return boundary_matches
# The origin cannot serve this turn and no peer shares its encryption boundary, so its

View file

@ -1879,6 +1879,20 @@ class TestEncryptedReasoningReplay:
assert messages[0] == {"role": "user", "content": "question"}
assert messages[2] == {"role": "user", "content": [{"type": "text", "text": "follow-up"}]}
def test_strip_uses_predicate_to_keep_selected_encrypted_blocks(self):
kept_signature = encrypted_reasoning_signature("keep")
stripped_signature = encrypted_reasoning_signature("strip")
content = [
{"type": "thinking", "thinking": "keep", "signature": kept_signature},
{"type": "thinking", "thinking": "strip", "signature": stripped_signature},
]
messages = [{"role": "assistant", "content": content}]
strip_encrypted_reasoning_from_messages(messages, should_strip=lambda block: block.get("thinking") == "strip")
assert messages[0]["content"] is content
assert content == [{"type": "thinking", "thinking": "keep", "signature": kept_signature}]
@pytest.mark.parametrize(
"messages",
[

View file

@ -1961,6 +1961,315 @@ class TestStripEncryptedReasoningFromInput:
ResponsesAPIRequestUtils.strip_encrypted_reasoning_from_input(request_input)
assert request_input == before
def test_strips_only_items_selected_by_predicate(self):
wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("gAAAAA-blob", "deployment-a")
request_input = [
{"type": "reasoning", "id": "keep", "encrypted_content": wrapped, "summary": "keep"},
{"type": "reasoning", "id": "strip", "encrypted_content": wrapped, "summary": "strip"},
]
ResponsesAPIRequestUtils.strip_encrypted_reasoning_from_input(
request_input, should_strip=lambda item: item.get("id") == "strip"
)
assert request_input == [
{"type": "reasoning", "id": "keep", "encrypted_content": wrapped, "summary": "keep"},
{"type": "reasoning", "summary": "strip"},
]
@pytest.mark.asyncio
async def test_real_router_selection_keeps_origin_reasoning_and_strips_foreign_origin():
router = litellm.Router(
model_list=[
{
"model_name": "gpt-openai",
"litellm_params": {
"model": "openai/gpt-5.1-codex",
"api_base": "https://api.openai.com/v1",
"api_key": "key-openai",
},
"model_info": {"id": "dep-openai"},
},
{
"model_name": "gpt-azure",
"litellm_params": {
"model": "azure/gpt-5.1-codex",
"api_base": "https://res-b.openai.azure.com/",
"api_key": "key-azure",
"api_version": "2025-04-01-preview",
},
"model_info": {"id": "dep-azure"},
},
],
optional_pre_call_checks=["encrypted_content_affinity"],
num_retries=0,
)
openai_item_id = ResponsesAPIRequestUtils._build_encrypted_item_id("dep-openai", "rs-openai")
azure_item_id = ResponsesAPIRequestUtils._build_encrypted_item_id("dep-azure", "rs-azure")
openai_wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-openai", "dep-openai")
azure_wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-azure", "dep-azure")
request_input = [
{"type": "message", "role": "user", "content": "first question"},
{
"type": "reasoning",
"id": openai_item_id,
"encrypted_content": openai_wrapped,
"summary": [{"type": "summary_text", "text": "openai summary"}],
},
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "first answer"}]},
{"type": "message", "role": "user", "content": "second question"},
{
"type": "reasoning",
"id": azure_item_id,
"encrypted_content": azure_wrapped,
"summary": [{"type": "summary_text", "text": "azure summary"}],
},
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "second answer"}]},
{"type": "message", "role": "user", "content": "third question"},
]
request_kwargs = {"input": request_input, "store": False}
try:
deployment = await router.async_get_available_deployment(
model="gpt-openai", request_kwargs=request_kwargs, input=request_kwargs["input"]
)
assert deployment["model_info"]["id"] == "dep-openai"
assert deployment["litellm_params"]["model"] == "openai/gpt-5.1-codex"
assert deployment["litellm_params"]["api_base"] == "https://api.openai.com/v1"
assert request_kwargs["input"] == [
{"type": "message", "role": "user", "content": "first question"},
{
"type": "reasoning",
"id": openai_item_id,
"encrypted_content": openai_wrapped,
"summary": [{"type": "summary_text", "text": "openai summary"}],
},
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "first answer"}]},
{"type": "message", "role": "user", "content": "second question"},
{
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "azure summary"}],
},
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "second answer"}]},
{"type": "message", "role": "user", "content": "third question"},
]
finally:
router.discard()
@pytest.mark.asyncio
async def test_affinity_keeps_mixed_origins_on_the_same_encryption_boundary():
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
shared_api_base = "https://account-a.openai.azure.com/"
shared_api_key = "shared-key"
origin_d2 = _make_originating_mock(shared_api_base, shared_api_key)
mock_router = _make_router_mock_with_cooldown(origin_d2, cooldown_entries=[], routed_group_model_ids=["d1", "d2"])
deployment_d1 = {
"model_info": {"id": "d1"},
"litellm_params": {"api_base": shared_api_base, "api_key": shared_api_key},
}
deployment_d2 = {
"model_info": {"id": "d2"},
"litellm_params": {"api_base": shared_api_base, "api_key": shared_api_key},
}
d2_item = {
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-d2", "d2"),
"summary": [{"type": "summary_text", "text": "second origin"}],
}
request_kwargs = {
"input": [
{
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-d1", "d1"),
"summary": [{"type": "summary_text", "text": "first origin"}],
},
d2_item.copy(),
]
}
mock_router.get_deployment.side_effect = lambda model_id: origin_d2 if model_id == "d2" else None
check = EncryptedContentAffinityCheck(router=mock_router)
result = await check.async_filter_deployments(
model="gpt-5.4",
healthy_deployments=[deployment_d1, deployment_d2],
messages=None,
request_kwargs=request_kwargs,
)
assert result == [deployment_d1]
assert request_kwargs["input"][1] == d2_item
@pytest.mark.asyncio
async def test_boundary_pin_strips_reasoning_from_a_different_origin():
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
origin_a = _make_originating_mock("https://account-a.openai.azure.com/", "key-a")
origin_b = _make_originating_mock("https://account-b.openai.azure.com/", "key-b")
mock_router = _make_router_mock_with_cooldown(origin_a, cooldown_entries=[], routed_group_model_ids=["peer-a"])
mock_router.get_deployment.side_effect = lambda model_id: {"origin-a": origin_a, "origin-b": origin_b}.get(model_id)
peer_a = {
"model_info": {"id": "peer-a"},
"litellm_params": {"api_base": "https://account-a.openai.azure.com/", "api_key": "key-a"},
}
request_kwargs = {
"input": [
{
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
"blob-origin-a", "origin-a"
),
"summary": [{"type": "summary_text", "text": "origin A summary"}],
},
{
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
"blob-origin-b", "origin-b"
),
"summary": [{"type": "summary_text", "text": "origin B summary"}],
},
]
}
check = EncryptedContentAffinityCheck(router=mock_router)
result = await check.async_filter_deployments(
model="gpt-5.4",
healthy_deployments=[peer_a],
messages=None,
request_kwargs=request_kwargs,
)
assert result == [peer_a]
assert request_kwargs["input"] == [
{
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
"blob-origin-a", "origin-a"
),
"summary": [{"type": "summary_text", "text": "origin A summary"}],
},
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "origin B summary"}]},
]
@pytest.mark.asyncio
async def test_affinity_keeps_only_anthropic_reasoning_from_the_pinned_origin():
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
origin_a = _make_originating_mock("https://account-a.openai.azure.com/", "key-a")
origin_b = _make_originating_mock("https://account-b.openai.azure.com/", "key-b")
mock_router = _make_router_mock_with_cooldown(origin_b, cooldown_entries=[], routed_group_model_ids=["origin-a"])
mock_router.get_deployment.side_effect = lambda model_id: {"origin-a": origin_a, "origin-b": origin_b}.get(model_id)
deployment_a = {
"model_info": {"id": "origin-a"},
"litellm_params": {"api_base": "https://account-a.openai.azure.com/", "api_key": "key-a"},
}
deployment_b = {
"model_info": {"id": "origin-b"},
"litellm_params": {"api_base": "https://account-b.openai.azure.com/", "api_key": "key-b"},
}
messages = _bridge_replayed_anthropic_messages(minted_by="origin-a")
foreign_messages = _bridge_replayed_anthropic_messages(minted_by="origin-b")
assistant_content = messages[1]["content"]
assistant_content.insert(3, foreign_messages[1]["content"][1])
check = EncryptedContentAffinityCheck(router=mock_router)
result = await check.async_filter_deployments(
model="gpt-5.4",
healthy_deployments=[deployment_a, deployment_b],
messages=messages,
request_kwargs={"model": "gpt-5.4"},
)
assert result == [deployment_a]
assert messages[1]["content"] is assistant_content
assert assistant_content == [
{"type": "thinking", "thinking": "Anthropic minted this one", "signature": "ErcCCpIBCBEYAipA"},
{
"type": "redacted_thinking",
"data": (
"litellm_encrypted_reasoning:"
f"{ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id('gAAAAA_turn_one', 'origin-a')}"
),
},
{
"type": "thinking",
"thinking": "The bridge packed this one",
"signature": (
"litellm_encrypted_reasoning:"
f"{ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id('gAAAAA_turn_one', 'origin-a')}"
),
},
{"type": "text", "text": "The zebra owner lives in the green house."},
]
@pytest.mark.asyncio
async def test_affinity_strips_unknown_origins_but_leaves_unmarked_encrypted_content():
from unittest.mock import MagicMock
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
EncryptedContentAffinityCheck,
)
mock_router = MagicMock()
mock_router.get_deployment.return_value = None
deployment_a = {
"model_info": {"id": "origin-a"},
"litellm_params": {"api_base": "https://account-a.openai.azure.com/", "api_key": "key-a"},
}
openai_item = {
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-a", "origin-a"),
"summary": [{"type": "summary_text", "text": "origin A"}],
}
request_kwargs = {
"input": [
openai_item.copy(),
{
"type": "reasoning",
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
"blob-removed", "origin-removed"
),
"summary": [{"type": "summary_text", "text": "removed origin"}],
},
{
"type": "reasoning",
"encrypted_content": "raw-encrypted-content",
"summary": [{"type": "summary_text", "text": "unmarked content"}],
},
]
}
check = EncryptedContentAffinityCheck(router=mock_router)
result = await check.async_filter_deployments(
model="gpt-5.4",
healthy_deployments=[deployment_a],
messages=None,
request_kwargs=request_kwargs,
)
assert result == [deployment_a]
assert request_kwargs["input"] == [
openai_item,
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "removed origin"}]},
{
"type": "reasoning",
"encrypted_content": "raw-encrypted-content",
"summary": [{"type": "summary_text", "text": "unmarked content"}],
},
]
def _cross_group_request_kwargs():
wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("gAAAAA-blob", "deployment-a")