mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix(router): strip encrypted reasoning the pinned deployment cannot decrypt (#43781)
Co-authored-by: yassin <yassin@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
82d8b3797c
commit
2ed9761921
5 changed files with 431 additions and 30 deletions
|
|
@ -2015,7 +2015,11 @@ def is_unsignable_thinking_block(block: object) -> bool:
|
|||
return not (isinstance(thinking_text, str) and len(thinking_text.strip()) > 0)
|
||||
|
||||
|
||||
def strip_encrypted_reasoning_from_messages(messages: object) -> None:
|
||||
def strip_encrypted_reasoning_from_messages(
|
||||
messages: object,
|
||||
*,
|
||||
should_strip: Callable[[Mapping[str, object]], bool] | None = None,
|
||||
) -> None:
|
||||
"""Drop the bridge-tagged reasoning blocks a routed deployment cannot decrypt from
|
||||
Anthropic-shaped history.
|
||||
|
||||
|
|
@ -2030,7 +2034,7 @@ def strip_encrypted_reasoning_from_messages(messages: object) -> None:
|
|||
if not isinstance(messages, list):
|
||||
return
|
||||
for content in anthropic_content_lists(cast(list[object], messages)): # cast-ok: untyped client json
|
||||
_strip_encrypted_reasoning_from_blocks(content)
|
||||
_strip_encrypted_reasoning_from_blocks(content, should_strip=should_strip)
|
||||
|
||||
|
||||
def anthropic_content_lists(messages: Sequence[object]) -> Iterator[object]:
|
||||
|
|
@ -2043,9 +2047,18 @@ def anthropic_content_lists(messages: Sequence[object]) -> Iterator[object]:
|
|||
)
|
||||
|
||||
|
||||
def _strip_encrypted_reasoning_from_blocks(content: object) -> None:
|
||||
def _strip_encrypted_reasoning_from_blocks(
|
||||
content: object,
|
||||
*,
|
||||
should_strip: Callable[[Mapping[str, object]], bool] | None = None,
|
||||
) -> None:
|
||||
blocks: Final = cast(list[object], content) # cast-ok: narrowed by the caller's isinstance
|
||||
kept: Final = tuple(block for block in blocks if not is_encrypted_reasoning_block(block))
|
||||
kept: Final = tuple(
|
||||
block
|
||||
for block in blocks
|
||||
if not is_encrypted_reasoning_block(block)
|
||||
or (should_strip is not None and not should_strip(cast(Mapping[str, object], block)))
|
||||
)
|
||||
blocks[:] = kept
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
import base64
|
||||
import re
|
||||
from collections.abc import Iterable, Mapping, Sequence
|
||||
from collections.abc import Callable, Iterable, Mapping, Sequence
|
||||
from functools import reduce
|
||||
from typing import Any, Final, Optional, TypeVar, Union, cast, get_type_hints, overload
|
||||
|
||||
|
|
@ -556,7 +556,11 @@ class ResponsesAPIRequestUtils:
|
|||
return request_input
|
||||
|
||||
@staticmethod
|
||||
def strip_encrypted_reasoning_from_input(request_input: object) -> None:
|
||||
def strip_encrypted_reasoning_from_input(
|
||||
request_input: object,
|
||||
*,
|
||||
should_strip: Callable[[Mapping[str, object]], bool] | None = None,
|
||||
) -> None:
|
||||
"""Drop reasoning items the routed deployment cannot decrypt, keeping their readable summary.
|
||||
|
||||
Mutates ``request_input`` in place: the router's fallback snapshot shares this
|
||||
|
|
@ -565,7 +569,12 @@ class ResponsesAPIRequestUtils:
|
|||
if not isinstance(request_input, list):
|
||||
return
|
||||
items: Final = cast(list[object], request_input) # cast-ok: untyped client json
|
||||
stripped: Final = tuple(ResponsesAPIRequestUtils._without_encrypted_reasoning(item) for item in items)
|
||||
stripped: Final = tuple(
|
||||
ResponsesAPIRequestUtils._without_encrypted_reasoning(item)
|
||||
if should_strip is None or (isinstance(item, Mapping) and should_strip(cast(Mapping[str, object], item)))
|
||||
else item
|
||||
for item in items
|
||||
)
|
||||
items[:] = (item for item in stripped if item is not None)
|
||||
|
||||
@staticmethod
|
||||
|
|
|
|||
|
|
@ -36,7 +36,8 @@ Safe to enable globally:
|
|||
- No cache required.
|
||||
"""
|
||||
|
||||
from collections.abc import Iterator, Mapping
|
||||
from collections.abc import Iterator, Mapping, Sequence
|
||||
from functools import cache
|
||||
from typing import TYPE_CHECKING, Final, Optional, cast
|
||||
|
||||
from litellm._logging import verbose_router_logger
|
||||
|
|
@ -114,23 +115,31 @@ class EncryptedContentAffinityCheck(CustomLogger):
|
|||
if not isinstance(request_input, list):
|
||||
return None
|
||||
|
||||
for item in request_input:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
return next(
|
||||
(
|
||||
model_id
|
||||
for item in request_input
|
||||
if (model_id := EncryptedContentAffinityCheck._model_id_of_input_item(item)) is not None
|
||||
),
|
||||
None,
|
||||
)
|
||||
|
||||
# First, try to decode from item ID (if present)
|
||||
item_id = item.get("id")
|
||||
if item_id and isinstance(item_id, str):
|
||||
decoded = ResponsesAPIRequestUtils._decode_encrypted_item_id(item_id)
|
||||
if decoded:
|
||||
return decoded.get("model_id")
|
||||
@staticmethod
|
||||
def _model_id_of_input_item(item: object) -> str | None:
|
||||
if not isinstance(item, dict):
|
||||
return None
|
||||
|
||||
# If no encoded ID, check if encrypted_content itself is wrapped
|
||||
encrypted_content = item.get("encrypted_content")
|
||||
if encrypted_content and isinstance(encrypted_content, str):
|
||||
model_id = EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(encrypted_content)
|
||||
if model_id:
|
||||
return model_id
|
||||
item_id: Final = item.get("id")
|
||||
if item_id and isinstance(item_id, str):
|
||||
decoded: Final = ResponsesAPIRequestUtils._decode_encrypted_item_id(item_id)
|
||||
if decoded:
|
||||
return decoded.get("model_id")
|
||||
|
||||
encrypted_content: Final = item.get("encrypted_content")
|
||||
if encrypted_content and isinstance(encrypted_content, str):
|
||||
model_id: Final = EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(encrypted_content)
|
||||
if model_id:
|
||||
return model_id
|
||||
|
||||
return None
|
||||
|
||||
|
|
@ -150,19 +159,20 @@ class EncryptedContentAffinityCheck(CustomLogger):
|
|||
model_id, _ = ResponsesAPIRequestUtils._unwrap_encrypted_content_with_model_id(encrypted_content)
|
||||
return model_id or None
|
||||
|
||||
@staticmethod
|
||||
def _model_id_of_anthropic_block(block: Mapping[str, object]) -> str | None:
|
||||
encrypted_content: Final = encrypted_content_of_block(block)
|
||||
if encrypted_content is None:
|
||||
return None
|
||||
return EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(encrypted_content)
|
||||
|
||||
@staticmethod
|
||||
def _extract_model_id_from_anthropic_messages(messages: object) -> str | None:
|
||||
return next(
|
||||
(
|
||||
model_id
|
||||
for block in EncryptedContentAffinityCheck._anthropic_content_blocks(messages)
|
||||
if (encrypted_content := encrypted_content_of_block(block)) is not None
|
||||
if (
|
||||
model_id := EncryptedContentAffinityCheck._model_id_from_wrapped_encrypted_content(
|
||||
encrypted_content
|
||||
)
|
||||
)
|
||||
is not None
|
||||
if (model_id := EncryptedContentAffinityCheck._model_id_of_anthropic_block(block)) is not None
|
||||
),
|
||||
None,
|
||||
)
|
||||
|
|
@ -243,6 +253,50 @@ class EncryptedContentAffinityCheck(CustomLogger):
|
|||
]
|
||||
return matches, originating
|
||||
|
||||
def _strip_reasoning_the_target_cannot_decrypt(
|
||||
self,
|
||||
request_input: object,
|
||||
anthropic_messages: object,
|
||||
target_deployments: Sequence[Mapping[str, object]],
|
||||
) -> None:
|
||||
target_ids: Final = frozenset(
|
||||
str(model_info["id"])
|
||||
for target in target_deployments
|
||||
if isinstance((model_info := target.get("model_info")), Mapping) and model_info.get("id") is not None
|
||||
)
|
||||
target_boundaries: Final = frozenset(
|
||||
boundary
|
||||
for target in target_deployments
|
||||
if (boundary := self._encryption_boundary_key(target.get("litellm_params"))) is not None
|
||||
)
|
||||
|
||||
@cache
|
||||
def target_can_decrypt(origin_model_id: str) -> bool:
|
||||
if origin_model_id in target_ids:
|
||||
return True
|
||||
if self.router is None:
|
||||
return False
|
||||
origin: Final = self.router.get_deployment(model_id=origin_model_id)
|
||||
origin_boundary: Final = (
|
||||
self._encryption_boundary_key(origin.litellm_params.model_dump(exclude_none=True))
|
||||
if origin is not None
|
||||
else None
|
||||
)
|
||||
return origin_boundary is not None and origin_boundary in target_boundaries
|
||||
|
||||
def should_strip_input_item(item: Mapping[str, object]) -> bool:
|
||||
origin_model_id: Final = self._model_id_of_input_item(item)
|
||||
return origin_model_id is not None and not target_can_decrypt(origin_model_id)
|
||||
|
||||
def should_strip_anthropic_block(block: Mapping[str, object]) -> bool:
|
||||
origin_model_id: Final = self._model_id_of_anthropic_block(block)
|
||||
return origin_model_id is not None and not target_can_decrypt(origin_model_id)
|
||||
|
||||
ResponsesAPIRequestUtils.strip_encrypted_reasoning_from_input(
|
||||
request_input, should_strip=should_strip_input_item
|
||||
)
|
||||
strip_encrypted_reasoning_from_messages(anthropic_messages, should_strip=should_strip_anthropic_block)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Request routing (pre-call filter)
|
||||
# ------------------------------------------------------------------
|
||||
|
|
@ -303,6 +357,7 @@ class EncryptedContentAffinityCheck(CustomLogger):
|
|||
model_id,
|
||||
)
|
||||
request_kwargs["_encrypted_content_affinity_pinned"] = True
|
||||
self._strip_reasoning_the_target_cannot_decrypt(request_input, anthropic_messages, (deployment,))
|
||||
return [deployment]
|
||||
|
||||
# Follow-up switched model_name (LIT-2531): pin by Azure resource instead.
|
||||
|
|
@ -318,6 +373,7 @@ class EncryptedContentAffinityCheck(CustomLogger):
|
|||
len(boundary_matches),
|
||||
)
|
||||
request_kwargs["_encrypted_content_affinity_pinned"] = True
|
||||
self._strip_reasoning_the_target_cannot_decrypt(request_input, anthropic_messages, boundary_matches)
|
||||
return boundary_matches
|
||||
|
||||
# The origin cannot serve this turn and no peer shares its encryption boundary, so its
|
||||
|
|
|
|||
|
|
@ -1879,6 +1879,20 @@ class TestEncryptedReasoningReplay:
|
|||
assert messages[0] == {"role": "user", "content": "question"}
|
||||
assert messages[2] == {"role": "user", "content": [{"type": "text", "text": "follow-up"}]}
|
||||
|
||||
def test_strip_uses_predicate_to_keep_selected_encrypted_blocks(self):
|
||||
kept_signature = encrypted_reasoning_signature("keep")
|
||||
stripped_signature = encrypted_reasoning_signature("strip")
|
||||
content = [
|
||||
{"type": "thinking", "thinking": "keep", "signature": kept_signature},
|
||||
{"type": "thinking", "thinking": "strip", "signature": stripped_signature},
|
||||
]
|
||||
messages = [{"role": "assistant", "content": content}]
|
||||
|
||||
strip_encrypted_reasoning_from_messages(messages, should_strip=lambda block: block.get("thinking") == "strip")
|
||||
|
||||
assert messages[0]["content"] is content
|
||||
assert content == [{"type": "thinking", "thinking": "keep", "signature": kept_signature}]
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"messages",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -1961,6 +1961,315 @@ class TestStripEncryptedReasoningFromInput:
|
|||
ResponsesAPIRequestUtils.strip_encrypted_reasoning_from_input(request_input)
|
||||
assert request_input == before
|
||||
|
||||
def test_strips_only_items_selected_by_predicate(self):
|
||||
wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("gAAAAA-blob", "deployment-a")
|
||||
request_input = [
|
||||
{"type": "reasoning", "id": "keep", "encrypted_content": wrapped, "summary": "keep"},
|
||||
{"type": "reasoning", "id": "strip", "encrypted_content": wrapped, "summary": "strip"},
|
||||
]
|
||||
|
||||
ResponsesAPIRequestUtils.strip_encrypted_reasoning_from_input(
|
||||
request_input, should_strip=lambda item: item.get("id") == "strip"
|
||||
)
|
||||
|
||||
assert request_input == [
|
||||
{"type": "reasoning", "id": "keep", "encrypted_content": wrapped, "summary": "keep"},
|
||||
{"type": "reasoning", "summary": "strip"},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_real_router_selection_keeps_origin_reasoning_and_strips_foreign_origin():
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-openai",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-5.1-codex",
|
||||
"api_base": "https://api.openai.com/v1",
|
||||
"api_key": "key-openai",
|
||||
},
|
||||
"model_info": {"id": "dep-openai"},
|
||||
},
|
||||
{
|
||||
"model_name": "gpt-azure",
|
||||
"litellm_params": {
|
||||
"model": "azure/gpt-5.1-codex",
|
||||
"api_base": "https://res-b.openai.azure.com/",
|
||||
"api_key": "key-azure",
|
||||
"api_version": "2025-04-01-preview",
|
||||
},
|
||||
"model_info": {"id": "dep-azure"},
|
||||
},
|
||||
],
|
||||
optional_pre_call_checks=["encrypted_content_affinity"],
|
||||
num_retries=0,
|
||||
)
|
||||
openai_item_id = ResponsesAPIRequestUtils._build_encrypted_item_id("dep-openai", "rs-openai")
|
||||
azure_item_id = ResponsesAPIRequestUtils._build_encrypted_item_id("dep-azure", "rs-azure")
|
||||
openai_wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-openai", "dep-openai")
|
||||
azure_wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-azure", "dep-azure")
|
||||
request_input = [
|
||||
{"type": "message", "role": "user", "content": "first question"},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"id": openai_item_id,
|
||||
"encrypted_content": openai_wrapped,
|
||||
"summary": [{"type": "summary_text", "text": "openai summary"}],
|
||||
},
|
||||
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "first answer"}]},
|
||||
{"type": "message", "role": "user", "content": "second question"},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"id": azure_item_id,
|
||||
"encrypted_content": azure_wrapped,
|
||||
"summary": [{"type": "summary_text", "text": "azure summary"}],
|
||||
},
|
||||
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "second answer"}]},
|
||||
{"type": "message", "role": "user", "content": "third question"},
|
||||
]
|
||||
|
||||
request_kwargs = {"input": request_input, "store": False}
|
||||
try:
|
||||
deployment = await router.async_get_available_deployment(
|
||||
model="gpt-openai", request_kwargs=request_kwargs, input=request_kwargs["input"]
|
||||
)
|
||||
|
||||
assert deployment["model_info"]["id"] == "dep-openai"
|
||||
assert deployment["litellm_params"]["model"] == "openai/gpt-5.1-codex"
|
||||
assert deployment["litellm_params"]["api_base"] == "https://api.openai.com/v1"
|
||||
assert request_kwargs["input"] == [
|
||||
{"type": "message", "role": "user", "content": "first question"},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"id": openai_item_id,
|
||||
"encrypted_content": openai_wrapped,
|
||||
"summary": [{"type": "summary_text", "text": "openai summary"}],
|
||||
},
|
||||
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "first answer"}]},
|
||||
{"type": "message", "role": "user", "content": "second question"},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"summary": [{"type": "summary_text", "text": "azure summary"}],
|
||||
},
|
||||
{"type": "message", "role": "assistant", "content": [{"type": "output_text", "text": "second answer"}]},
|
||||
{"type": "message", "role": "user", "content": "third question"},
|
||||
]
|
||||
finally:
|
||||
router.discard()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_affinity_keeps_mixed_origins_on_the_same_encryption_boundary():
|
||||
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
|
||||
EncryptedContentAffinityCheck,
|
||||
)
|
||||
|
||||
shared_api_base = "https://account-a.openai.azure.com/"
|
||||
shared_api_key = "shared-key"
|
||||
origin_d2 = _make_originating_mock(shared_api_base, shared_api_key)
|
||||
mock_router = _make_router_mock_with_cooldown(origin_d2, cooldown_entries=[], routed_group_model_ids=["d1", "d2"])
|
||||
deployment_d1 = {
|
||||
"model_info": {"id": "d1"},
|
||||
"litellm_params": {"api_base": shared_api_base, "api_key": shared_api_key},
|
||||
}
|
||||
deployment_d2 = {
|
||||
"model_info": {"id": "d2"},
|
||||
"litellm_params": {"api_base": shared_api_base, "api_key": shared_api_key},
|
||||
}
|
||||
d2_item = {
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-d2", "d2"),
|
||||
"summary": [{"type": "summary_text", "text": "second origin"}],
|
||||
}
|
||||
request_kwargs = {
|
||||
"input": [
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-d1", "d1"),
|
||||
"summary": [{"type": "summary_text", "text": "first origin"}],
|
||||
},
|
||||
d2_item.copy(),
|
||||
]
|
||||
}
|
||||
mock_router.get_deployment.side_effect = lambda model_id: origin_d2 if model_id == "d2" else None
|
||||
check = EncryptedContentAffinityCheck(router=mock_router)
|
||||
|
||||
result = await check.async_filter_deployments(
|
||||
model="gpt-5.4",
|
||||
healthy_deployments=[deployment_d1, deployment_d2],
|
||||
messages=None,
|
||||
request_kwargs=request_kwargs,
|
||||
)
|
||||
|
||||
assert result == [deployment_d1]
|
||||
assert request_kwargs["input"][1] == d2_item
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_boundary_pin_strips_reasoning_from_a_different_origin():
|
||||
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
|
||||
EncryptedContentAffinityCheck,
|
||||
)
|
||||
|
||||
origin_a = _make_originating_mock("https://account-a.openai.azure.com/", "key-a")
|
||||
origin_b = _make_originating_mock("https://account-b.openai.azure.com/", "key-b")
|
||||
mock_router = _make_router_mock_with_cooldown(origin_a, cooldown_entries=[], routed_group_model_ids=["peer-a"])
|
||||
mock_router.get_deployment.side_effect = lambda model_id: {"origin-a": origin_a, "origin-b": origin_b}.get(model_id)
|
||||
peer_a = {
|
||||
"model_info": {"id": "peer-a"},
|
||||
"litellm_params": {"api_base": "https://account-a.openai.azure.com/", "api_key": "key-a"},
|
||||
}
|
||||
request_kwargs = {
|
||||
"input": [
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
|
||||
"blob-origin-a", "origin-a"
|
||||
),
|
||||
"summary": [{"type": "summary_text", "text": "origin A summary"}],
|
||||
},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
|
||||
"blob-origin-b", "origin-b"
|
||||
),
|
||||
"summary": [{"type": "summary_text", "text": "origin B summary"}],
|
||||
},
|
||||
]
|
||||
}
|
||||
check = EncryptedContentAffinityCheck(router=mock_router)
|
||||
|
||||
result = await check.async_filter_deployments(
|
||||
model="gpt-5.4",
|
||||
healthy_deployments=[peer_a],
|
||||
messages=None,
|
||||
request_kwargs=request_kwargs,
|
||||
)
|
||||
|
||||
assert result == [peer_a]
|
||||
assert request_kwargs["input"] == [
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
|
||||
"blob-origin-a", "origin-a"
|
||||
),
|
||||
"summary": [{"type": "summary_text", "text": "origin A summary"}],
|
||||
},
|
||||
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "origin B summary"}]},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_affinity_keeps_only_anthropic_reasoning_from_the_pinned_origin():
|
||||
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
|
||||
EncryptedContentAffinityCheck,
|
||||
)
|
||||
|
||||
origin_a = _make_originating_mock("https://account-a.openai.azure.com/", "key-a")
|
||||
origin_b = _make_originating_mock("https://account-b.openai.azure.com/", "key-b")
|
||||
mock_router = _make_router_mock_with_cooldown(origin_b, cooldown_entries=[], routed_group_model_ids=["origin-a"])
|
||||
mock_router.get_deployment.side_effect = lambda model_id: {"origin-a": origin_a, "origin-b": origin_b}.get(model_id)
|
||||
deployment_a = {
|
||||
"model_info": {"id": "origin-a"},
|
||||
"litellm_params": {"api_base": "https://account-a.openai.azure.com/", "api_key": "key-a"},
|
||||
}
|
||||
deployment_b = {
|
||||
"model_info": {"id": "origin-b"},
|
||||
"litellm_params": {"api_base": "https://account-b.openai.azure.com/", "api_key": "key-b"},
|
||||
}
|
||||
messages = _bridge_replayed_anthropic_messages(minted_by="origin-a")
|
||||
foreign_messages = _bridge_replayed_anthropic_messages(minted_by="origin-b")
|
||||
assistant_content = messages[1]["content"]
|
||||
assistant_content.insert(3, foreign_messages[1]["content"][1])
|
||||
check = EncryptedContentAffinityCheck(router=mock_router)
|
||||
|
||||
result = await check.async_filter_deployments(
|
||||
model="gpt-5.4",
|
||||
healthy_deployments=[deployment_a, deployment_b],
|
||||
messages=messages,
|
||||
request_kwargs={"model": "gpt-5.4"},
|
||||
)
|
||||
|
||||
assert result == [deployment_a]
|
||||
assert messages[1]["content"] is assistant_content
|
||||
assert assistant_content == [
|
||||
{"type": "thinking", "thinking": "Anthropic minted this one", "signature": "ErcCCpIBCBEYAipA"},
|
||||
{
|
||||
"type": "redacted_thinking",
|
||||
"data": (
|
||||
"litellm_encrypted_reasoning:"
|
||||
f"{ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id('gAAAAA_turn_one', 'origin-a')}"
|
||||
),
|
||||
},
|
||||
{
|
||||
"type": "thinking",
|
||||
"thinking": "The bridge packed this one",
|
||||
"signature": (
|
||||
"litellm_encrypted_reasoning:"
|
||||
f"{ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id('gAAAAA_turn_one', 'origin-a')}"
|
||||
),
|
||||
},
|
||||
{"type": "text", "text": "The zebra owner lives in the green house."},
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_affinity_strips_unknown_origins_but_leaves_unmarked_encrypted_content():
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from litellm.router_utils.pre_call_checks.encrypted_content_affinity_check import (
|
||||
EncryptedContentAffinityCheck,
|
||||
)
|
||||
|
||||
mock_router = MagicMock()
|
||||
mock_router.get_deployment.return_value = None
|
||||
deployment_a = {
|
||||
"model_info": {"id": "origin-a"},
|
||||
"litellm_params": {"api_base": "https://account-a.openai.azure.com/", "api_key": "key-a"},
|
||||
}
|
||||
openai_item = {
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("blob-a", "origin-a"),
|
||||
"summary": [{"type": "summary_text", "text": "origin A"}],
|
||||
}
|
||||
request_kwargs = {
|
||||
"input": [
|
||||
openai_item.copy(),
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id(
|
||||
"blob-removed", "origin-removed"
|
||||
),
|
||||
"summary": [{"type": "summary_text", "text": "removed origin"}],
|
||||
},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": "raw-encrypted-content",
|
||||
"summary": [{"type": "summary_text", "text": "unmarked content"}],
|
||||
},
|
||||
]
|
||||
}
|
||||
check = EncryptedContentAffinityCheck(router=mock_router)
|
||||
|
||||
result = await check.async_filter_deployments(
|
||||
model="gpt-5.4",
|
||||
healthy_deployments=[deployment_a],
|
||||
messages=None,
|
||||
request_kwargs=request_kwargs,
|
||||
)
|
||||
|
||||
assert result == [deployment_a]
|
||||
assert request_kwargs["input"] == [
|
||||
openai_item,
|
||||
{"type": "reasoning", "summary": [{"type": "summary_text", "text": "removed origin"}]},
|
||||
{
|
||||
"type": "reasoning",
|
||||
"encrypted_content": "raw-encrypted-content",
|
||||
"summary": [{"type": "summary_text", "text": "unmarked content"}],
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _cross_group_request_kwargs():
|
||||
wrapped = ResponsesAPIRequestUtils._wrap_encrypted_content_with_model_id("gAAAAA-blob", "deployment-a")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue