From 8442b062fe67bd4a938485961b26ecdafc29c3d3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Sak=C4=B1p=20Han=20Dursun?= Date: Fri, 28 Aug 2026 01:51:48 +0300 Subject: [PATCH] fix(router): scope model_info.mode to its own deployment Deployments pointing at the same provider model share one litellm.model_cost entry, and `mode` was written into it. Whoever registered last decided which API every sibling got served on, so adding one `responses` deployment moved unrelated model groups onto the Responses API bridge, and deleting it again did not undo that until the cost map was rebuilt on a restart `mode` now stays under each deployment's own model id and travels with the request, so a deployment only ever speaks for itself. A backend the catalog marks responses-only still cannot be talked down to chat, which is what the removed downgrade guard was there for. That guard also read `openai/gpt-5.1` while registration writes `gpt-5.1`, so it never fired for prefixed models anyway Fixes #38543 --- litellm/main.py | 53 +++++-- litellm/router.py | 37 +---- litellm/types/utils.py | 26 +++- .../test_router_model_cost_isolation.py | 144 +++++++++++++++++- 4 files changed, 210 insertions(+), 50 deletions(-) diff --git a/litellm/main.py b/litellm/main.py index 0c8bff16f81..84d59609a48 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -19,7 +19,7 @@ import random import sys import time import traceback -from collections.abc import AsyncIterator, Coroutine, Iterable, Mapping, Sequence +from collections.abc import AsyncIterator, Coroutine, Iterable, Iterator, Mapping, Sequence from concurrent import futures from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from copy import deepcopy @@ -122,6 +122,7 @@ from litellm.types.utils import ( ModelResponseStream, RawRequestTypedDict, StreamingChoices, + strip_deployment_scoped_model_info, ) from litellm.utils import ( Choices, @@ -1009,6 +1010,7 @@ def responses_api_bridge_check( reasoning_effort: str | Mapping[str, object] | None = None, reasoning_summary: object | None = None, api_base: str | None = None, + deployment_mode: str | None = None, ) -> tuple[dict, str]: model_info: dict[str, object] = {} @@ -1041,6 +1043,14 @@ def responses_api_bridge_check( mode = "responses" model_info["mode"] = mode + # The cost-map entry is keyed by provider model, so every deployment of that model + # reads the same mode. A deployment asking for the Responses API only speaks for + # itself, hence the request-scoped override. It can only turn the bridge on: a + # backend the catalog marks responses-only (e.g. `chatgpt/*`) has no chat route to + # fall back to, so a deployment claiming `mode: chat` there would just fail. + if deployment_mode == "responses": + model_info["mode"] = "responses" + # OpenAI/Azure GPT-5 chat-completions that need Responses-only fields (e.g. # ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects # those keys. @@ -1174,20 +1184,39 @@ def _build_custom_pricing_entry( return entry -def _get_router_deployment_id(kwargs: dict) -> str | None: +def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]: + """The ``model_info`` the router stamped into this request's metadata, if any.""" for metadata_key in ("litellm_metadata", "metadata"): metadata = kwargs.get(metadata_key) or {} if not isinstance(metadata, dict): continue deployment_model_info = metadata.get("model_info") or {} - if not isinstance(deployment_model_info, dict): - continue + if isinstance(deployment_model_info, dict): + yield deployment_model_info + + +def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None: + for deployment_model_info in _router_deployment_model_infos(kwargs): deployment_id = deployment_model_info.get("id") if deployment_id is not None: return str(deployment_id) return None +def _get_router_deployment_mode(kwargs: Mapping[str, object]) -> str | None: + """The API surface the routed deployment configured for itself. + + Deployments of one provider model share a single ``litellm.model_cost`` entry, so + the mode a deployment sets has to travel with the request instead of being written + there, where it would answer for its siblings too. + """ + for deployment_model_info in _router_deployment_model_infos(kwargs): + mode = deployment_model_info.get("mode") + if isinstance(mode, str): + return mode + return None + + def _register_custom_pricing_for_request( model: str, custom_llm_provider: str, @@ -1199,10 +1228,11 @@ def _register_custom_pricing_for_request( Router-originated requests (identified by the deployment id the router puts in metadata) get their full pricing registered under that unique id only; the shared ``{provider}/{model}`` key receives the entry with pricing fields - stripped, mirroring Router._create_deployment. This keeps one deployment's - pricing overrides (e.g. a zero-cost wildcard) from clobbering built-in - pricing used by sibling deployments of the same backend model. Direct SDK - calls keep the legacy behavior of registering the shared key with pricing. + and ``mode`` stripped, mirroring Router._create_deployment. This keeps one + deployment's pricing overrides (e.g. a zero-cost wildcard) or its choice of + API surface from clobbering the built-in entry that sibling deployments of + the same backend model read. Direct SDK calls keep the legacy behavior of + registering the shared key with pricing. """ entry: Final = _build_custom_pricing_entry( custom_llm_provider=custom_llm_provider, @@ -1217,7 +1247,9 @@ def _register_custom_pricing_for_request( litellm.register_model( { deployment_id: entry, - shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry), + shared_key: strip_deployment_scoped_model_info( + CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry) + ), }, persist_across_reloads=False, warning_display_name=shared_key, @@ -5302,11 +5334,13 @@ def completion( ) ## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name + deployment_mode: Final = _get_router_deployment_mode(kwargs) responses_api_model_info, model = responses_api_bridge_check( model=model, custom_llm_provider=custom_llm_provider, web_search_options=web_search_options, api_base=api_base, + deployment_mode=deployment_mode, ) if not _should_allow_input_examples(custom_llm_provider=custom_llm_provider, model=model): @@ -5552,6 +5586,7 @@ def completion( reasoning_effort=reasoning_effort, reasoning_summary=_reasoning_summary_for_bridge, api_base=api_base, + deployment_mode=deployment_mode, ) # Use base_model (the true underlying model) for Azure model-type diff --git a/litellm/router.py b/litellm/router.py index edff8294c3e..7acbb2b90d2 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -9298,41 +9298,14 @@ class Router: ## OLD MODEL REGISTRATION ## Kept to prevent breaking changes backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider) - backend_key: Final = backend_keys[0] # For the shared backend key, keep only cost-map schema fields - # (minus custom pricing) so that one deployment's pricing overrides - # or custom metadata (id, access_via_team_ids, arbitrary keys) - # don't pollute another deployment sharing the same backend model - # name. Each deployment's full model_info is already stored under - # its unique model_id above. + # (minus custom pricing and minus `mode`) so that one deployment's + # pricing overrides, API surface or custom metadata (id, + # access_via_team_ids, arbitrary keys) don't pollute another + # deployment sharing the same backend model name. Each deployment's + # full model_info is already stored under its unique model_id above. shared_model_info: Final = shared_backend_model_info(model_info) - existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode") - deployment_mode: Final = shared_model_info.get("mode") - # Keep the built-in bridge mode stable for shared backend keys. - # Multiple aliases can point at the same provider/model backend, - # but their deployment-level overrides should not downgrade the - # backend from responses -> chat via last-write-wins registration. - # Only preserve in that specific direction so legitimate upgrades - # (e.g. chat -> responses) and unrelated mode changes still apply, - # and so a missing deployment mode does not silently clear the - # existing shared backend mode. - is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat" - would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None - if is_responses_to_chat_downgrade or would_clear_existing_mode: - if deployment_mode is not None: - verbose_router_logger.warning( - "Router: preserving existing mode=%s for shared backend " - "key %s instead of the deployment-specified mode=%s " - "(prevents alias registration from downgrading the " - "shared backend mode).", - existing_shared_mode, - backend_key, - deployment_mode, - ) - shared_model_info["mode"] = existing_shared_mode - - # Always register the (possibly mode-preserved) shared backend info. litellm.register_model( model_cost={_key: shared_model_info for _key in backend_keys}, persist_across_reloads=False, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 55a32989b1c..2a53cd076d1 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3486,17 +3486,31 @@ class CustomPricingLiteLLMParams(MirroredPricingParams): return {k: v for k, v in model_info.items() if k not in cls.model_fields} -SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = frozenset( - ModelInfoBase.__required_keys__ | ModelInfoBase.__optional_keys__ -) - frozenset(CustomPricingLiteLLMParams.model_fields) +# ``mode`` picks the API surface a request is served on, and two deployments of one +# provider model are allowed to disagree about it. The shared key holds a single value, +# so whichever deployment registered last would decide for all of them; it stays under +# each deployment's own model id instead. +DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS: Final[frozenset[str]] = frozenset({"mode"}) + +SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = ( + frozenset(ModelInfoBase.__required_keys__ | ModelInfoBase.__optional_keys__) + - frozenset(CustomPricingLiteLLMParams.model_fields) + - DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS +) + + +def strip_deployment_scoped_model_info(model_info: Mapping[str, Any]) -> Mapping[str, Any]: + """Return a copy of ``model_info`` without the fields that describe a single + deployment rather than the backend model, leaving every other key intact.""" + return {k: v for k, v in model_info.items() if k not in DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS} def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]: """Return only the fields safe to register under a shared ``{provider}/{model}`` key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus - per-deployment pricing overrides. Per-deployment metadata (``id``, - ``access_via_team_ids``, arbitrary custom keys) never belongs on the shared key; - it stays under the deployment's unique model id. + per-deployment pricing overrides and minus the deployment-scoped fields above. + Per-deployment metadata (``id``, ``access_via_team_ids``, arbitrary custom keys) + never belongs on the shared key; it stays under the deployment's unique model id. """ return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS} diff --git a/tests/test_litellm/test_router_model_cost_isolation.py b/tests/test_litellm/test_router_model_cost_isolation.py index b580b03574e..4c5e291fd02 100644 --- a/tests/test_litellm/test_router_model_cost_isolation.py +++ b/tests/test_litellm/test_router_model_cost_isolation.py @@ -13,6 +13,7 @@ import os import re from unittest.mock import patch +import httpx import pytest @@ -399,6 +400,144 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override(): ) assert bridge_model == "gpt-5.4" assert bridge_model_info["mode"] == "responses" + + # A responses-only backend has no chat route, so the alias asking for chat + # still gets bridged rather than being taken at its word. + downgraded_info, _ = responses_api_bridge_check( + model="gpt-5.4", + custom_llm_provider="chatgpt", + deployment_mode="chat", + ) + assert downgraded_info["mode"] == "responses" + finally: + _restore_model_cost_entries(model_keys) + + +def test_a_responses_deployment_does_not_move_its_siblings_onto_the_bridge(): + """ + https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model + share a single litellm.model_cost entry, so registering a `mode: responses` + deployment used to switch every other deployment of that model onto the Responses + API, and deleting it again did not undo that. + """ + from litellm.main import responses_api_bridge_check + + backend_model = "openai/gpt-5.1" + model_keys = { + key: copy.deepcopy(litellm.model_cost.get(key)) + for key in (backend_model, "gpt-5.1", "chat-38543", "responses-probe-38543") + } + + try: + router = Router( + model_list=[ + { + "model_name": "my-chat-model", + "litellm_params": {"model": backend_model, "api_key": "sk-fake"}, + "model_info": {"id": "chat-38543"}, + } + ] + ) + catalog_mode = litellm.get_model_info(model="gpt-5.1", custom_llm_provider="openai").get("mode") + assert catalog_mode == "chat", "test needs a backend the catalog serves over chat completions" + + router.add_deployment( + deployment=Deployment( + model_name="my-responses-probe", + litellm_params=LiteLLM_Params(model=backend_model, api_key="sk-fake"), + model_info=ModelInfo(id="responses-probe-38543", mode="responses"), + ) + ) + + # The probe speaks for itself... + probe_info, _ = responses_api_bridge_check( + model="gpt-5.1", + custom_llm_provider="openai", + deployment_mode="responses", + ) + assert probe_info["mode"] == "responses" + + # ...and for nobody else, whether or not it is still registered. + for step in ("with the probe registered", "after deleting the probe"): + sibling_info, _ = responses_api_bridge_check(model="gpt-5.1", custom_llm_provider="openai") + assert sibling_info.get("mode") == "chat", f"sibling moved onto the responses bridge {step}" + assert litellm.model_cost["gpt-5.1"].get("mode") == "chat", f"shared key rewritten {step}" + router.delete_deployment(id="responses-probe-38543") + finally: + _restore_model_cost_entries(model_keys) + + +_OPENAI_CHAT_COMPLETION_BODY = { + "id": "chatcmpl-38543", + "object": "chat.completion", + "created": 0, + "model": "gpt-5.1", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, +} + +_OPENAI_RESPONSES_BODY = { + "id": "resp_38543", + "object": "response", + "created_at": 0, + "model": "gpt-5.1", + "status": "completed", + "output": [ + { + "type": "message", + "id": "msg_38543", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "ok", "annotations": []}], + } + ], + "usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}, +} + + +@pytest.mark.parametrize( + "model_name, expected_endpoint", + [("my-chat-model", "chat"), ("my-responses-probe", "responses")], +) +def test_each_deployment_is_served_on_the_api_its_own_model_info_asks_for(model_name, expected_endpoint): + """End to end for https://github.com/BerriAI/litellm/issues/38543: the plain + deployment keeps hitting chat completions while its `mode: responses` sibling, + registered against the same provider model, goes to the Responses API. + """ + import respx + + model_keys = { + key: copy.deepcopy(litellm.model_cost.get(key)) + for key in ("openai/gpt-5.1", "gpt-5.1", "chat-38543-routed", "responses-probe-38543-routed") + } + + try: + router = Router( + model_list=[ + { + "model_name": "my-chat-model", + "litellm_params": {"model": "openai/gpt-5.1", "api_key": "sk-fake"}, + "model_info": {"id": "chat-38543-routed"}, + }, + { + "model_name": "my-responses-probe", + "litellm_params": {"model": "openai/gpt-5.1", "api_key": "sk-fake"}, + "model_info": {"id": "responses-probe-38543-routed", "mode": "responses"}, + }, + ] + ) + + with respx.mock(assert_all_called=False) as openai_api: + chat = openai_api.post("https://api.openai.com/v1/chat/completions").mock( + return_value=httpx.Response(200, json=_OPENAI_CHAT_COMPLETION_BODY) + ) + responses = openai_api.post("https://api.openai.com/v1/responses").mock( + return_value=httpx.Response(200, json=_OPENAI_RESPONSES_BODY) + ) + router.completion(model=model_name, messages=[{"role": "user", "content": "hi"}]) + + called = {"chat": chat.called, "responses": responses.called} + assert called == {"chat": expected_endpoint == "chat", "responses": expected_endpoint == "responses"} finally: _restore_model_cost_entries(model_keys) @@ -775,8 +914,8 @@ def test_add_deployment_does_not_leak_custom_metadata_to_shared_backend_key(): def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest(): - """Unit test of the whitelist helper: cost-map schema fields survive, - custom pricing overrides and per-deployment metadata do not. + """Unit test of the whitelist helper: cost-map schema fields survive, custom + pricing overrides, per-deployment metadata and `mode` do not. """ from litellm.types.utils import shared_backend_model_info @@ -799,7 +938,6 @@ def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest(): ) assert filtered == { - "mode": "chat", "litellm_provider": "openai", "max_tokens": 128000, "supports_vision": True,