fix(router): read the routed deployment's mode from its registration, not request metadata

Only the deployment id now comes from request metadata. The mode is read
from what the router registered under that id, so a caller who writes
model_info.mode into their own metadata can't move a chat deployment onto
the Responses API
This commit is contained in:
madhu19991 2026-09-29 11:59:12 -07:00
parent f0c57a5f1f
commit 9570f41081
4 changed files with 48 additions and 12 deletions

View file

@ -1284,12 +1284,15 @@ def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None:
def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None:
"""The ``model_info.mode`` of the deployment the router picked for this request, if any."""
for deployment_model_info in _router_deployment_model_infos(kwargs):
mode = deployment_model_info.get("mode")
if isinstance(mode, str):
return mode
return None
"""The ``mode`` the router registered for the deployment it picked for this request, if any.
Only the deployment id is taken from request metadata; the mode itself is read from the
router's own registration, so a caller can't hand-write one into their metadata.
"""
deployment_id: Final = _get_router_deployment_id(kwargs)
registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None
mode: Final = registered.get("mode") if isinstance(registered, dict) else None
return mode if isinstance(mode, str) else None
def _register_custom_pricing_for_request(

View file

@ -10,6 +10,7 @@ from typing import Final
import pytest
import litellm
from litellm.llms.anthropic.pass_through.adapters.handler import (
LiteLLMMessagesToCompletionTransformationHandler,
)
@ -122,11 +123,12 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge:
[("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")],
)
def test_the_routed_deployments_own_mode_decides_the_shape(
self, local_model_cost_map: None, deployment_mode: str, expected: object
self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object
) -> None:
"""https://github.com/BerriAI/litellm/issues/38543 - a deployment's mode rides with the
request instead of the model's shared cost-map entry, and this probe has to agree with
"""https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the
model's shared cost-map entry, decides the API, and this probe has to agree with
``litellm.completion`` about which API that request lands on"""
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs(
max_tokens=1024,
messages=MESSAGES,
@ -144,7 +146,7 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge:
output_format=None,
extra_kwargs={
"custom_llm_provider": "databricks",
"litellm_metadata": {"model_info": {"id": "routed-deployment", "mode": deployment_mode}},
"litellm_metadata": {"model_info": {"id": "routed-deployment"}},
},
)

View file

@ -2770,19 +2770,20 @@ class TestToolTransformation:
assert result["reasoning_summary"] == "auto"
@pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)])
def test_summary_follows_the_routed_deployments_own_mode(self, deployment_mode, expected_summary):
def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary):
"""
https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands
on the model's shared cost-map entry, so the probe has to read it off the routed
deployment the same way ``litellm.completion`` does, or it drops the summary of a
request that completion then bridges onto the Responses API
"""
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
model="gpt-4.1",
input="hi",
responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}},
custom_llm_provider="openai",
litellm_metadata={"model_info": {"id": "routed-deployment", "mode": deployment_mode}},
litellm_metadata={"model_info": {"id": "routed-deployment"}},
)
assert result["reasoning_effort"] == "medium"

View file

@ -854,6 +854,36 @@ def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_
_invalidate_model_cost_lowercase_map()
@pytest.mark.parametrize(
"caller_model_info",
({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}),
)
def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata(
respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str]
) -> None:
"""
The Responses bridge reads the mode the router registered for the routed deployment, never a
mode written into the request's own metadata
"""
_use_catalog_with_a_chat_model(monkeypatch)
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
)
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
)
router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)])
router.completion(
model="my-chat-model",
messages=[{"role": "user", "content": "hi"}],
litellm_metadata={"model_info": dict(caller_model_info)},
)
assert (chat_route.call_count, responses_route.call_count) == (1, 0)
_invalidate_model_cost_lowercase_map()
def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None:
"""
For a provider model the catalog doesn't know, proxy model_info is how its mode gets