From c7c3ad64c6d8d9283840c2df7ae6840fccdfcea3 Mon Sep 17 00:00:00 2001 From: madhu19991 Date: Tue, 29 Sep 2026 01:39:30 -0700 Subject: [PATCH 1/3] fix(router): keep a deployment's mode off its siblings' shared cost-map entry A deployment registering model_info.mode wrote it onto the cost-map entry every deployment of the same provider model reads, so adding one `mode: responses` deployment sent all its siblings through the Responses API bridge. The existing guard read the provider-prefixed key, which register_model never writes, so it didn't protect the catalog entry. A deployment's mode now only fills an entry that has no mode yet, and the routed deployment's own mode reaches the Responses bridge with the request. Fixes #38543 --- .../pass_through/adapters/handler.py | 3 +- litellm/main.py | 39 ++++- .../transformation.py | 7 + litellm/router.py | 40 +---- litellm/types/utils.py | 12 +- litellm/utils.py | 17 ++ ..._handler_reasoning_effort_normalization.py | 33 ++++ .../test_litellm_completion_responses.py | 19 +++ .../unit/test_router_model_cost_isolation.py | 149 ++++++++++++++++++ 9 files changed, 273 insertions(+), 46 deletions(-) diff --git a/litellm/llms/anthropic/pass_through/adapters/handler.py b/litellm/llms/anthropic/pass_through/adapters/handler.py index 5bda3437a40..7ce62f036e5 100644 --- a/litellm/llms/anthropic/pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/pass_through/adapters/handler.py @@ -411,7 +411,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: return None if resolved_provider == "litellm_proxy": return None - from litellm.main import responses_api_bridge_check + from litellm.main import responses_api_bridge_check, router_deployment_mode web_search_options: Final = completion_kwargs.get("web_search_options") tools: Final = completion_kwargs.get("tools") @@ -424,6 +424,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: tools=cast("list[dict[str, object]]", tools) if isinstance(tools, list) else None, reasoning_effort=reasoning_effort, api_base=resolved_api_base, + deployment_mode=router_deployment_mode(completion_kwargs), ) return None if model_info.get("mode") == "responses" else effort diff --git a/litellm/main.py b/litellm/main.py index 6c85adf3ae8..a0454b33736 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -19,7 +19,7 @@ import random import sys import time import traceback -from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Mapping, Sequence +from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Iterator, Mapping, Sequence from concurrent import futures from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from copy import deepcopy @@ -168,6 +168,7 @@ from litellm.utils import ( get_secret, get_standard_openai_params, mock_completion_streaming_obj, + mode_register_model_would_merge_into, pre_process_non_default_params, read_config_args, should_run_mock_completion, @@ -1084,6 +1085,7 @@ def responses_api_bridge_check( reasoning_effort: str | Mapping[str, object] | None = None, reasoning_summary: object | None = None, api_base: str | None = None, + deployment_mode: str | None = None, ) -> tuple[dict, str]: model_info: dict[str, object] = {} @@ -1112,7 +1114,7 @@ def responses_api_bridge_check( mode = "responses" model_info["mode"] = mode - if web_search_options is not None and custom_llm_provider == "xai": + if deployment_mode == "responses" or (web_search_options is not None and custom_llm_provider == "xai"): model_info["mode"] = "responses" model = model.replace("responses/", "") @@ -1263,20 +1265,33 @@ def _build_custom_pricing_entry( return entry -def _get_router_deployment_id(kwargs: dict) -> str | None: +def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]: for metadata_key in ("litellm_metadata", "metadata"): - metadata = kwargs.get(metadata_key) or {} + metadata = kwargs.get(metadata_key) if not isinstance(metadata, dict): continue - deployment_model_info = metadata.get("model_info") or {} - if not isinstance(deployment_model_info, dict): - continue + deployment_model_info = metadata.get("model_info") + if isinstance(deployment_model_info, dict): + yield deployment_model_info + + +def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None: + for deployment_model_info in _router_deployment_model_infos(kwargs): deployment_id = deployment_model_info.get("id") if deployment_id is not None: return str(deployment_id) return None +def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None: + """The ``model_info.mode`` of the deployment the router picked for this request, if any.""" + for deployment_model_info in _router_deployment_model_infos(kwargs): + mode = deployment_model_info.get("mode") + if isinstance(mode, str): + return mode + return None + + def _register_custom_pricing_for_request( model: str, custom_llm_provider: str, @@ -1303,10 +1318,15 @@ def _register_custom_pricing_for_request( if deployment_id is None: litellm.register_model({shared_key: entry}, persist_across_reloads=False) return + shared_entry: Final = CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry) litellm.register_model( { deployment_id: entry, - shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry), + shared_key: ( + {k: v for k, v in shared_entry.items() if k != "mode"} + if mode_register_model_would_merge_into(shared_key, custom_llm_provider) is not None + else shared_entry + ), }, persist_across_reloads=False, warning_display_name=shared_key, @@ -5479,11 +5499,13 @@ def completion( ) ## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name + _deployment_mode: Final = router_deployment_mode(kwargs) responses_api_model_info, model = responses_api_bridge_check( model=model, custom_llm_provider=custom_llm_provider, web_search_options=web_search_options, api_base=api_base, + deployment_mode=_deployment_mode, ) if not _is_claude_tool_target(custom_llm_provider=custom_llm_provider, model=model): @@ -5763,6 +5785,7 @@ def completion( reasoning_effort=reasoning_effort, reasoning_summary=_reasoning_summary_for_bridge, api_base=api_base, + deployment_mode=_deployment_mode, ) # Use base_model (the true underlying model) for Azure model-type diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index bd239922fd3..3c828baa5d9 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -332,6 +332,7 @@ class LiteLLMCompletionResponsesConfig: reasoning_effort: str | None, reasoning_summary: str | None, api_base: str | None, + deployment_mode: str | None, ) -> bool: """ Whether ``litellm.completion`` will route this model back onto the Responses API. @@ -350,6 +351,7 @@ class LiteLLMCompletionResponsesConfig: reasoning_effort=reasoning_effort, reasoning_summary=reasoning_summary, api_base=api_base, + deployment_mode=deployment_mode, ) except Exception as e: # noqa: BLE001 # a capability probe must never fail the request it probes for verbose_logger.debug("responses bridge: reasoning effort mode check failed: %s", e) @@ -364,6 +366,7 @@ class LiteLLMCompletionResponsesConfig: tools: Sequence[ChatCompletionToolParam | OpenAIMcpServerTool] | None = None, web_search_options: OpenAIWebSearchOptions | None = None, api_base: str | None = None, + deployment_mode: str | None = None, ) -> ResponsesReasoningChatForm: """ Split the Responses ``reasoning`` object into the params Chat Completions understands. @@ -393,6 +396,7 @@ class LiteLLMCompletionResponsesConfig: reasoning_effort=effort, reasoning_summary=summary, api_base=api_base, + deployment_mode=deployment_mode, ) return ResponsesReasoningChatForm(effort=effort, summary=summary if bridges_back else None) @@ -409,6 +413,8 @@ class LiteLLMCompletionResponsesConfig: """ Transform a Responses API request into a Chat Completion request """ + from litellm.main import router_deployment_mode + ( tools, web_search_options, @@ -433,6 +439,7 @@ class LiteLLMCompletionResponsesConfig: tools=tools, web_search_options=web_search_options, api_base=kwargs.get("api_base"), + deployment_mode=router_deployment_mode(kwargs), ) litellm_completion_request: dict = { diff --git a/litellm/router.py b/litellm/router.py index cfed5dc81c0..efbcd012dd3 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -332,6 +332,7 @@ from litellm.utils import ( get_secret, get_utc_datetime, is_region_allowed, + mode_register_model_would_merge_into, provider_rejectable_params, set_live_deployment_replay, ) @@ -9973,41 +9974,10 @@ class Router: ## OLD MODEL REGISTRATION ## Kept to prevent breaking changes backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider) - backend_key: Final = backend_keys[0] - - # For the shared backend key, keep only cost-map schema fields - # (minus custom pricing) so that one deployment's pricing overrides - # or custom metadata (id, access_via_team_ids, arbitrary keys) - # don't pollute another deployment sharing the same backend model - # name. Each deployment's full model_info is already stored under - # its unique model_id above. - shared_model_info: Final = shared_backend_model_info(model_info) - existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode") - deployment_mode: Final = shared_model_info.get("mode") - # Keep the built-in bridge mode stable for shared backend keys. - # Multiple aliases can point at the same provider/model backend, - # but their deployment-level overrides should not downgrade the - # backend from responses -> chat via last-write-wins registration. - # Only preserve in that specific direction so legitimate upgrades - # (e.g. chat -> responses) and unrelated mode changes still apply, - # and so a missing deployment mode does not silently clear the - # existing shared backend mode. - is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat" - would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None - if is_responses_to_chat_downgrade or would_clear_existing_mode: - if deployment_mode is not None: - verbose_router_logger.warning( - "Router: preserving existing mode=%s for shared backend " - "key %s instead of the deployment-specified mode=%s " - "(prevents alias registration from downgrading the " - "shared backend mode).", - existing_shared_mode, - backend_key, - deployment_mode, - ) - shared_model_info["mode"] = existing_shared_mode - - # Always register the (possibly mode-preserved) shared backend info. + shared_model_info: Final = shared_backend_model_info( + model_info, + existing_mode=mode_register_model_would_merge_into(backend_keys[0], model_info.get("litellm_provider", "")), + ) litellm.register_model( model_cost={_key: shared_model_info for _key in backend_keys}, persist_across_reloads=False, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index dba15bc99a5..9e0b83486df 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3841,15 +3841,23 @@ SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = ( ) -def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]: +def shared_backend_model_info(model_info: Mapping[str, Any], existing_mode: object = None) -> dict[str, Any]: """Return only the fields safe to register under a shared ``{provider}/{model}`` key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus per-deployment pricing overrides and deployment-scoped pricing blocks such as ``off_peak_pricing``. Per-deployment metadata (``id``, ``access_via_team_ids``, arbitrary custom keys) never belongs on the shared key; it stays under the deployment's unique model id. + + ``mode`` only fills a shared key that has none yet (``existing_mode``). It is the API + surface one deployment picked, so it must not switch the surface its siblings use; + the routed deployment's own ``mode`` reaches the Responses bridge with the request. """ - return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS} + return { + k: v + for k, v in model_info.items() + if k in SHARED_BACKEND_MODEL_INFO_FIELDS and (k != "mode" or existing_mode is None) + } ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens$") diff --git a/litellm/utils.py b/litellm/utils.py index 13a46840431..1a7a0aa4c03 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -3217,6 +3217,23 @@ def _get_builtin_model_info_for_registration(model: str) -> ModelInfo | None: return None if is_generalized_model_info(info) else info +def mode_register_model_would_merge_into(key: str, provider: object) -> object: + """The ``mode`` already on the entry that ``register_model({key: ...})`` merges into. + + Mirrors ``register_model``'s key resolution: a provider-prefixed key such as + ``openai/gpt-5.6`` lands on the bare catalog entry ``gpt-5.6``, so reading + ``litellm.model_cost[key]`` alone would miss the mode that registration overwrites. + """ + skips_model_info_lookup: Final = provider in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO or any( + key.startswith(f"{p}/") for p in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO + ) + builtin_model_info: Final = None if skips_model_info_lookup else _get_builtin_model_info_for_registration(key) + if builtin_model_info is not None: + return builtin_model_info.get("mode") + entry: Final = litellm.model_cost.get(key) + return entry.get("mode") if isinstance(entry, dict) else None + + _runtime_registered_model_cost: Final[dict[str, dict[str, object]]] = {} # mutable-ok: replayed on reload diff --git a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py index 0dfe4a93649..1d8ebb997c8 100644 --- a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py +++ b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py @@ -117,6 +117,39 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge: assert sent == {"effort": "high", "summary": "auto"} + @pytest.mark.parametrize( + "deployment_mode, expected", + [("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")], + ) + def test_the_routed_deployments_own_mode_decides_the_shape( + self, local_model_cost_map: None, deployment_mode: str, expected: object + ) -> None: + """https://github.com/BerriAI/litellm/issues/38543 - a deployment's mode rides with the + request instead of the model's shared cost-map entry, and this probe has to agree with + ``litellm.completion`` about which API that request lands on""" + completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( + max_tokens=1024, + messages=MESSAGES, + model="databricks/databricks-qwen35-122b-a10b", + metadata=None, + stop_sequences=None, + stream=False, + system=None, + temperature=None, + thinking=SUMMARIZED_THINKING, + tool_choice=None, + tools=None, + top_k=None, + top_p=None, + output_format=None, + extra_kwargs={ + "custom_llm_provider": "databricks", + "litellm_metadata": {"model_info": {"id": "routed-deployment", "mode": deployment_mode}}, + }, + ) + + assert completion_kwargs.get("reasoning_effort") == expected + def test_auto_summary_still_reaches_a_bridged_target( self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 7b9de4644b4..0ea38077e99 100644 --- a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -2769,6 +2769,25 @@ class TestToolTransformation: assert result["reasoning_effort"] == "medium" assert result["reasoning_summary"] == "auto" + @pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)]) + def test_summary_follows_the_routed_deployments_own_mode(self, deployment_mode, expected_summary): + """ + https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands + on the model's shared cost-map entry, so the probe has to read it off the routed + deployment the same way ``litellm.completion`` does, or it drops the summary of a + request that completion then bridges onto the Responses API + """ + result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request( + model="gpt-4.1", + input="hi", + responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}}, + custom_llm_provider="openai", + litellm_metadata={"model_info": {"id": "routed-deployment", "mode": deployment_mode}}, + ) + + assert result["reasoning_effort"] == "medium" + assert result.get("reasoning_summary") == expected_summary + @pytest.mark.parametrize( "model, custom_llm_provider", [ diff --git a/tests/unit/test_router_model_cost_isolation.py b/tests/unit/test_router_model_cost_isolation.py index d73f5efa96b..867c8e8da16 100644 --- a/tests/unit/test_router_model_cost_isolation.py +++ b/tests/unit/test_router_model_cost_isolation.py @@ -12,6 +12,8 @@ import copy import logging import os import re +from collections.abc import Mapping +from types import MappingProxyType from typing import Final from unittest.mock import Mock, patch @@ -727,6 +729,153 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override(): _restore_model_cost_entries(model_keys) +_MODE_TEST_API_BASE: Final = "http://localhost:38543/v1" + +_CHAT_COMPLETION_REPLY: Final = { + "id": "chatcmpl-38543", + "object": "chat.completion", + "created": 1, + "model": "gpt-5.6", + "choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "ok"}}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, +} + +_RESPONSES_REPLY: Final = { + "id": "resp_38543", + "object": "response", + "created_at": 1, + "status": "completed", + "model": "gpt-5.6", + "output": [ + { + "type": "message", + "id": "msg_38543", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "ok", "annotations": []}], + } + ], + "parallel_tool_calls": True, + "tool_choice": "auto", + "tools": [], + "usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}, +} + + +def _mode_test_deployment( + model_name: str, mode: str | None, custom_pricing: Mapping[str, float] = MappingProxyType({}) +) -> dict[str, object]: + return { + "model_name": model_name, + "litellm_params": { + "model": "openai/gpt-5.6", + "api_key": "sk-fake", + "api_base": _MODE_TEST_API_BASE, + **custom_pricing, + }, + "model_info": {"id": f"{model_name}-id", **({"mode": mode} if mode is not None else {})}, + } + + +@pytest.mark.parametrize("sibling_mode", (None, "chat")) +@pytest.mark.parametrize("responses_deployment_first", (True, False)) +def test_a_responses_deployment_does_not_move_its_siblings_onto_the_responses_api( + respx_mock, monkeypatch: pytest.MonkeyPatch, sibling_mode: str | None, responses_deployment_first: bool +) -> None: + """ + https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model share + a litellm.model_cost key, so a `mode: responses` deployment used to send every sibling + through the Responses API bridge, whichever order they were registered in + """ + monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost)) + _invalidate_model_cost_lowercase_map() + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + sibling: Final = _mode_test_deployment("my-chat-model", sibling_mode) + responses_deployment: Final = _mode_test_deployment("my-responses-probe", "responses") + router: Final = Router( + model_list=[responses_deployment, sibling] if responses_deployment_first else [sibling, responses_deployment] + ) + messages: Final = [{"role": "user", "content": "hi"}] + + router.completion(model="my-chat-model", messages=messages) + assert (chat_route.call_count, responses_route.call_count) == (1, 0) + + router.completion(model="my-responses-probe", messages=messages) + assert (chat_route.call_count, responses_route.call_count) == (1, 1) + + _invalidate_model_cost_lowercase_map() + + +def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_a_request( + respx_mock, monkeypatch: pytest.MonkeyPatch +) -> None: + """ + A deployment with custom pricing re-registers its model_info on every request it serves, + so the shared key has to stay out of its mode on that path too, not just at router setup + """ + monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost)) + _invalidate_model_cost_lowercase_map() + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + router: Final = Router( + model_list=[ + _mode_test_deployment("my-chat-model", None), + _mode_test_deployment( + "my-responses-probe", + "responses", + custom_pricing=MappingProxyType({"input_cost_per_token": 1e-6, "output_cost_per_token": 2e-6}), + ), + ] + ) + messages: Final = [{"role": "user", "content": "hi"}] + + router.completion(model="my-responses-probe", messages=messages) + router.completion(model="my-chat-model", messages=messages) + + assert (chat_route.call_count, responses_route.call_count) == (1, 1) + _invalidate_model_cost_lowercase_map() + + +def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None: + """ + For a provider model the catalog doesn't know, proxy model_info is how its mode gets + onboarded (see mantle_supports_responses), so a deployment's mode must still fill a + shared key that has none; only an existing mode is left alone + """ + monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost)) + _invalidate_model_cost_lowercase_map() + backend_model: Final = "openai/unmapped-38543-model" + assert backend_model not in litellm.model_cost + + Router( + model_list=[ + { + "model_name": "unmapped-responses", + "litellm_params": {"model": backend_model, "api_key": "sk-fake"}, + "model_info": {"id": "unmapped-responses-id", "mode": "responses"}, + }, + { + "model_name": "unmapped-chat", + "litellm_params": {"model": backend_model, "api_key": "sk-fake"}, + "model_info": {"id": "unmapped-chat-id", "mode": "chat"}, + }, + ] + ) + + assert litellm.model_cost[backend_model]["mode"] == "responses" + assert litellm.model_cost["unmapped-chat-id"]["mode"] == "chat" + _invalidate_model_cost_lowercase_map() + + def test_partial_custom_pricing_inherits_builtin_cache_pricing(): """A deployment that overrides only input/output cost on a cache-supporting model must still bill cache_read and cache_creation tokens. Before the From f0c57a5f1f2cec53a5ce207cdf651a57c47e9f41 Mon Sep 17 00:00:00 2001 From: madhu19991 Date: Tue, 29 Sep 2026 02:07:25 -0700 Subject: [PATCH 2/3] test(router): seed the catalog entry the mode isolation tests rely on CI's cost map has no gpt-5.6, so the shared key had no catalog mode to protect and the tests measured the unmapped-model path instead. The tests now register their own catalog chat model --- .../unit/test_router_model_cost_isolation.py | 23 +++++++++++++------ 1 file changed, 16 insertions(+), 7 deletions(-) diff --git a/tests/unit/test_router_model_cost_isolation.py b/tests/unit/test_router_model_cost_isolation.py index 867c8e8da16..67152cf7c92 100644 --- a/tests/unit/test_router_model_cost_isolation.py +++ b/tests/unit/test_router_model_cost_isolation.py @@ -731,11 +731,13 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override(): _MODE_TEST_API_BASE: Final = "http://localhost:38543/v1" +_CATALOG_CHAT_MODEL: Final = "chat-model-38543" + _CHAT_COMPLETION_REPLY: Final = { "id": "chatcmpl-38543", "object": "chat.completion", "created": 1, - "model": "gpt-5.6", + "model": _CATALOG_CHAT_MODEL, "choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "ok"}}], "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, } @@ -745,7 +747,7 @@ _RESPONSES_REPLY: Final = { "object": "response", "created_at": 1, "status": "completed", - "model": "gpt-5.6", + "model": _CATALOG_CHAT_MODEL, "output": [ { "type": "message", @@ -762,13 +764,22 @@ _RESPONSES_REPLY: Final = { } +def _use_catalog_with_a_chat_model(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr( + litellm, + "model_cost", + {**copy.deepcopy(litellm.model_cost), _CATALOG_CHAT_MODEL: {"litellm_provider": "openai", "mode": "chat"}}, + ) + _invalidate_model_cost_lowercase_map() + + def _mode_test_deployment( model_name: str, mode: str | None, custom_pricing: Mapping[str, float] = MappingProxyType({}) ) -> dict[str, object]: return { "model_name": model_name, "litellm_params": { - "model": "openai/gpt-5.6", + "model": f"openai/{_CATALOG_CHAT_MODEL}", "api_key": "sk-fake", "api_base": _MODE_TEST_API_BASE, **custom_pricing, @@ -787,8 +798,7 @@ def test_a_responses_deployment_does_not_move_its_siblings_onto_the_responses_ap a litellm.model_cost key, so a `mode: responses` deployment used to send every sibling through the Responses API bridge, whichever order they were registered in """ - monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost)) - _invalidate_model_cost_lowercase_map() + _use_catalog_with_a_chat_model(monkeypatch) chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) ) @@ -818,8 +828,7 @@ def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_ A deployment with custom pricing re-registers its model_info on every request it serves, so the shared key has to stay out of its mode on that path too, not just at router setup """ - monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost)) - _invalidate_model_cost_lowercase_map() + _use_catalog_with_a_chat_model(monkeypatch) chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) ) From 9570f41081feb1fde15f7e3a4857380e35a06e0e Mon Sep 17 00:00:00 2001 From: madhu19991 Date: Tue, 29 Sep 2026 11:59:12 -0700 Subject: [PATCH 3/3] fix(router): read the routed deployment's mode from its registration, not request metadata Only the deployment id now comes from request metadata. The mode is read from what the router registered under that id, so a caller who writes model_info.mode into their own metadata can't move a chat deployment onto the Responses API --- litellm/main.py | 15 ++++++---- ..._handler_reasoning_effort_normalization.py | 10 ++++--- .../test_litellm_completion_responses.py | 5 ++-- .../unit/test_router_model_cost_isolation.py | 30 +++++++++++++++++++ 4 files changed, 48 insertions(+), 12 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index a0454b33736..df3fdef0ea1 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -1284,12 +1284,15 @@ def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None: def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None: - """The ``model_info.mode`` of the deployment the router picked for this request, if any.""" - for deployment_model_info in _router_deployment_model_infos(kwargs): - mode = deployment_model_info.get("mode") - if isinstance(mode, str): - return mode - return None + """The ``mode`` the router registered for the deployment it picked for this request, if any. + + Only the deployment id is taken from request metadata; the mode itself is read from the + router's own registration, so a caller can't hand-write one into their metadata. + """ + deployment_id: Final = _get_router_deployment_id(kwargs) + registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None + mode: Final = registered.get("mode") if isinstance(registered, dict) else None + return mode if isinstance(mode, str) else None def _register_custom_pricing_for_request( diff --git a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py index 1d8ebb997c8..32b048e43a4 100644 --- a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py +++ b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py @@ -10,6 +10,7 @@ from typing import Final import pytest +import litellm from litellm.llms.anthropic.pass_through.adapters.handler import ( LiteLLMMessagesToCompletionTransformationHandler, ) @@ -122,11 +123,12 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge: [("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")], ) def test_the_routed_deployments_own_mode_decides_the_shape( - self, local_model_cost_map: None, deployment_mode: str, expected: object + self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object ) -> None: - """https://github.com/BerriAI/litellm/issues/38543 - a deployment's mode rides with the - request instead of the model's shared cost-map entry, and this probe has to agree with + """https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the + model's shared cost-map entry, decides the API, and this probe has to agree with ``litellm.completion`` about which API that request lands on""" + monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode}) completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( max_tokens=1024, messages=MESSAGES, @@ -144,7 +146,7 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge: output_format=None, extra_kwargs={ "custom_llm_provider": "databricks", - "litellm_metadata": {"model_info": {"id": "routed-deployment", "mode": deployment_mode}}, + "litellm_metadata": {"model_info": {"id": "routed-deployment"}}, }, ) diff --git a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 0ea38077e99..3030620f26a 100644 --- a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -2770,19 +2770,20 @@ class TestToolTransformation: assert result["reasoning_summary"] == "auto" @pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)]) - def test_summary_follows_the_routed_deployments_own_mode(self, deployment_mode, expected_summary): + def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary): """ https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands on the model's shared cost-map entry, so the probe has to read it off the routed deployment the same way ``litellm.completion`` does, or it drops the summary of a request that completion then bridges onto the Responses API """ + monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode}) result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request( model="gpt-4.1", input="hi", responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}}, custom_llm_provider="openai", - litellm_metadata={"model_info": {"id": "routed-deployment", "mode": deployment_mode}}, + litellm_metadata={"model_info": {"id": "routed-deployment"}}, ) assert result["reasoning_effort"] == "medium" diff --git a/tests/unit/test_router_model_cost_isolation.py b/tests/unit/test_router_model_cost_isolation.py index 67152cf7c92..a4c4d8ddf18 100644 --- a/tests/unit/test_router_model_cost_isolation.py +++ b/tests/unit/test_router_model_cost_isolation.py @@ -854,6 +854,36 @@ def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_ _invalidate_model_cost_lowercase_map() +@pytest.mark.parametrize( + "caller_model_info", + ({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}), +) +def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata( + respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str] +) -> None: + """ + The Responses bridge reads the mode the router registered for the routed deployment, never a + mode written into the request's own metadata + """ + _use_catalog_with_a_chat_model(monkeypatch) + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)]) + + router.completion( + model="my-chat-model", + messages=[{"role": "user", "content": "hi"}], + litellm_metadata={"model_info": dict(caller_model_info)}, + ) + + assert (chat_route.call_count, responses_route.call_count) == (1, 0) + _invalidate_model_cost_lowercase_map() + + def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None: """ For a provider model the catalog doesn't know, proxy model_info is how its mode gets