diff --git a/litellm/main.py b/litellm/main.py index a0454b33736..df3fdef0ea1 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1284,12 +1284,15 @@ def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None: def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None: - """The ``model_info.mode`` of the deployment the router picked for this request, if any.""" - for deployment_model_info in _router_deployment_model_infos(kwargs): - mode = deployment_model_info.get("mode") - if isinstance(mode, str): - return mode - return None + """The ``mode`` the router registered for the deployment it picked for this request, if any. + + Only the deployment id is taken from request metadata; the mode itself is read from the + router's own registration, so a caller can't hand-write one into their metadata. + """ + deployment_id: Final = _get_router_deployment_id(kwargs) + registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None + mode: Final = registered.get("mode") if isinstance(registered, dict) else None + return mode if isinstance(mode, str) else None def _register_custom_pricing_for_request( diff --git a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py index 1d8ebb997c8..32b048e43a4 100644 --- a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py +++ b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py @@ -10,6 +10,7 @@ from typing import Final import pytest +import litellm from litellm.llms.anthropic.pass_through.adapters.handler import ( LiteLLMMessagesToCompletionTransformationHandler, ) @@ -122,11 +123,12 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge: [("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")], ) def test_the_routed_deployments_own_mode_decides_the_shape( - self, local_model_cost_map: None, deployment_mode: str, expected: object + self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object ) -> None: - """https://github.com/BerriAI/litellm/issues/38543 - a deployment's mode rides with the - request instead of the model's shared cost-map entry, and this probe has to agree with + """https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the + model's shared cost-map entry, decides the API, and this probe has to agree with ``litellm.completion`` about which API that request lands on""" + monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode}) completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( max_tokens=1024, messages=MESSAGES, @@ -144,7 +146,7 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge: output_format=None, extra_kwargs={ "custom_llm_provider": "databricks", - "litellm_metadata": {"model_info": {"id": "routed-deployment", "mode": deployment_mode}}, + "litellm_metadata": {"model_info": {"id": "routed-deployment"}}, }, ) diff --git a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 0ea38077e99..3030620f26a 100644 --- a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -2770,19 +2770,20 @@ class TestToolTransformation: assert result["reasoning_summary"] == "auto" @pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)]) - def test_summary_follows_the_routed_deployments_own_mode(self, deployment_mode, expected_summary): + def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary): """ https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands on the model's shared cost-map entry, so the probe has to read it off the routed deployment the same way ``litellm.completion`` does, or it drops the summary of a request that completion then bridges onto the Responses API """ + monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode}) result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request( model="gpt-4.1", input="hi", responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}}, custom_llm_provider="openai", - litellm_metadata={"model_info": {"id": "routed-deployment", "mode": deployment_mode}}, + litellm_metadata={"model_info": {"id": "routed-deployment"}}, ) assert result["reasoning_effort"] == "medium" diff --git a/tests/unit/test_router_model_cost_isolation.py b/tests/unit/test_router_model_cost_isolation.py index 67152cf7c92..a4c4d8ddf18 100644 --- a/tests/unit/test_router_model_cost_isolation.py +++ b/tests/unit/test_router_model_cost_isolation.py @@ -854,6 +854,36 @@ def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_ _invalidate_model_cost_lowercase_map() +@pytest.mark.parametrize( + "caller_model_info", + ({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}), +) +def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata( + respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str] +) -> None: + """ + The Responses bridge reads the mode the router registered for the routed deployment, never a + mode written into the request's own metadata + """ + _use_catalog_with_a_chat_model(monkeypatch) + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)]) + + router.completion( + model="my-chat-model", + messages=[{"role": "user", "content": "hi"}], + litellm_metadata={"model_info": dict(caller_model_info)}, + ) + + assert (chat_route.call_count, responses_route.call_count) == (1, 0) + _invalidate_model_cost_lowercase_map() + + def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None: """ For a provider model the catalog doesn't know, proxy model_info is how its mode gets