fix(router): scope model_info.mode to its own deployment

Deployments pointing at the same provider model share one litellm.model_cost
entry, and `mode` was written into it. Whoever registered last decided which
API every sibling got served on, so adding one `responses` deployment moved
unrelated model groups onto the Responses API bridge, and deleting it again
did not undo that until the cost map was rebuilt on a restart

`mode` now stays under each deployment's own model id and travels with the
request, so a deployment only ever speaks for itself. A backend the catalog
marks responses-only still cannot be talked down to chat, which is what the
removed downgrade guard was there for. That guard also read `openai/gpt-5.1`
while registration writes `gpt-5.1`, so it never fired for prefixed models
anyway

Fixes #38543
This commit is contained in:
Sakıp Han Dursun 2026-08-28 01:51:48 +03:00
parent ec3f8183c3
commit 8442b062fe
No known key found for this signature in database
GPG key ID: E0F8AC218793784A
4 changed files with 210 additions and 50 deletions

View file

@ -19,7 +19,7 @@ import random
import sys
import time
import traceback
from collections.abc import AsyncIterator, Coroutine, Iterable, Mapping, Sequence
from collections.abc import AsyncIterator, Coroutine, Iterable, Iterator, Mapping, Sequence
from concurrent import futures
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
from copy import deepcopy
@ -122,6 +122,7 @@ from litellm.types.utils import (
ModelResponseStream,
RawRequestTypedDict,
StreamingChoices,
strip_deployment_scoped_model_info,
)
from litellm.utils import (
Choices,
@ -1009,6 +1010,7 @@ def responses_api_bridge_check(
reasoning_effort: str | Mapping[str, object] | None = None,
reasoning_summary: object | None = None,
api_base: str | None = None,
deployment_mode: str | None = None,
) -> tuple[dict, str]:
model_info: dict[str, object] = {}
@ -1041,6 +1043,14 @@ def responses_api_bridge_check(
mode = "responses"
model_info["mode"] = mode
# The cost-map entry is keyed by provider model, so every deployment of that model
# reads the same mode. A deployment asking for the Responses API only speaks for
# itself, hence the request-scoped override. It can only turn the bridge on: a
# backend the catalog marks responses-only (e.g. `chatgpt/*`) has no chat route to
# fall back to, so a deployment claiming `mode: chat` there would just fail.
if deployment_mode == "responses":
model_info["mode"] = "responses"
# OpenAI/Azure GPT-5 chat-completions that need Responses-only fields (e.g.
# ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects
# those keys.
@ -1174,20 +1184,39 @@ def _build_custom_pricing_entry(
return entry
def _get_router_deployment_id(kwargs: dict) -> str | None:
def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]:
"""The ``model_info`` the router stamped into this request's metadata, if any."""
for metadata_key in ("litellm_metadata", "metadata"):
metadata = kwargs.get(metadata_key) or {}
if not isinstance(metadata, dict):
continue
deployment_model_info = metadata.get("model_info") or {}
if not isinstance(deployment_model_info, dict):
continue
if isinstance(deployment_model_info, dict):
yield deployment_model_info
def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None:
for deployment_model_info in _router_deployment_model_infos(kwargs):
deployment_id = deployment_model_info.get("id")
if deployment_id is not None:
return str(deployment_id)
return None
def _get_router_deployment_mode(kwargs: Mapping[str, object]) -> str | None:
"""The API surface the routed deployment configured for itself.
Deployments of one provider model share a single ``litellm.model_cost`` entry, so
the mode a deployment sets has to travel with the request instead of being written
there, where it would answer for its siblings too.
"""
for deployment_model_info in _router_deployment_model_infos(kwargs):
mode = deployment_model_info.get("mode")
if isinstance(mode, str):
return mode
return None
def _register_custom_pricing_for_request(
model: str,
custom_llm_provider: str,
@ -1199,10 +1228,11 @@ def _register_custom_pricing_for_request(
Router-originated requests (identified by the deployment id the router puts
in metadata) get their full pricing registered under that unique id only;
the shared ``{provider}/{model}`` key receives the entry with pricing fields
stripped, mirroring Router._create_deployment. This keeps one deployment's
pricing overrides (e.g. a zero-cost wildcard) from clobbering built-in
pricing used by sibling deployments of the same backend model. Direct SDK
calls keep the legacy behavior of registering the shared key with pricing.
and ``mode`` stripped, mirroring Router._create_deployment. This keeps one
deployment's pricing overrides (e.g. a zero-cost wildcard) or its choice of
API surface from clobbering the built-in entry that sibling deployments of
the same backend model read. Direct SDK calls keep the legacy behavior of
registering the shared key with pricing.
"""
entry: Final = _build_custom_pricing_entry(
custom_llm_provider=custom_llm_provider,
@ -1217,7 +1247,9 @@ def _register_custom_pricing_for_request(
litellm.register_model(
{
deployment_id: entry,
shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry),
shared_key: strip_deployment_scoped_model_info(
CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry)
),
},
persist_across_reloads=False,
warning_display_name=shared_key,
@ -5302,11 +5334,13 @@ def completion(
)
## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name
deployment_mode: Final = _get_router_deployment_mode(kwargs)
responses_api_model_info, model = responses_api_bridge_check(
model=model,
custom_llm_provider=custom_llm_provider,
web_search_options=web_search_options,
api_base=api_base,
deployment_mode=deployment_mode,
)
if not _should_allow_input_examples(custom_llm_provider=custom_llm_provider, model=model):
@ -5552,6 +5586,7 @@ def completion(
reasoning_effort=reasoning_effort,
reasoning_summary=_reasoning_summary_for_bridge,
api_base=api_base,
deployment_mode=deployment_mode,
)
# Use base_model (the true underlying model) for Azure model-type

View file

@ -9298,41 +9298,14 @@ class Router:
## OLD MODEL REGISTRATION ## Kept to prevent breaking changes
backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider)
backend_key: Final = backend_keys[0]
# For the shared backend key, keep only cost-map schema fields
# (minus custom pricing) so that one deployment's pricing overrides
# or custom metadata (id, access_via_team_ids, arbitrary keys)
# don't pollute another deployment sharing the same backend model
# name. Each deployment's full model_info is already stored under
# its unique model_id above.
# (minus custom pricing and minus `mode`) so that one deployment's
# pricing overrides, API surface or custom metadata (id,
# access_via_team_ids, arbitrary keys) don't pollute another
# deployment sharing the same backend model name. Each deployment's
# full model_info is already stored under its unique model_id above.
shared_model_info: Final = shared_backend_model_info(model_info)
existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode")
deployment_mode: Final = shared_model_info.get("mode")
# Keep the built-in bridge mode stable for shared backend keys.
# Multiple aliases can point at the same provider/model backend,
# but their deployment-level overrides should not downgrade the
# backend from responses -> chat via last-write-wins registration.
# Only preserve in that specific direction so legitimate upgrades
# (e.g. chat -> responses) and unrelated mode changes still apply,
# and so a missing deployment mode does not silently clear the
# existing shared backend mode.
is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat"
would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None
if is_responses_to_chat_downgrade or would_clear_existing_mode:
if deployment_mode is not None:
verbose_router_logger.warning(
"Router: preserving existing mode=%s for shared backend "
"key %s instead of the deployment-specified mode=%s "
"(prevents alias registration from downgrading the "
"shared backend mode).",
existing_shared_mode,
backend_key,
deployment_mode,
)
shared_model_info["mode"] = existing_shared_mode
# Always register the (possibly mode-preserved) shared backend info.
litellm.register_model(
model_cost={_key: shared_model_info for _key in backend_keys},
persist_across_reloads=False,

View file

@ -3486,17 +3486,31 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
return {k: v for k, v in model_info.items() if k not in cls.model_fields}
SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = frozenset(
ModelInfoBase.__required_keys__ | ModelInfoBase.__optional_keys__
) - frozenset(CustomPricingLiteLLMParams.model_fields)
# ``mode`` picks the API surface a request is served on, and two deployments of one
# provider model are allowed to disagree about it. The shared key holds a single value,
# so whichever deployment registered last would decide for all of them; it stays under
# each deployment's own model id instead.
DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS: Final[frozenset[str]] = frozenset({"mode"})
SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = (
frozenset(ModelInfoBase.__required_keys__ | ModelInfoBase.__optional_keys__)
- frozenset(CustomPricingLiteLLMParams.model_fields)
- DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS
)
def strip_deployment_scoped_model_info(model_info: Mapping[str, Any]) -> Mapping[str, Any]:
"""Return a copy of ``model_info`` without the fields that describe a single
deployment rather than the backend model, leaving every other key intact."""
return {k: v for k, v in model_info.items() if k not in DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS}
def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]:
"""Return only the fields safe to register under a shared ``{provider}/{model}``
key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus
per-deployment pricing overrides. Per-deployment metadata (``id``,
``access_via_team_ids``, arbitrary custom keys) never belongs on the shared key;
it stays under the deployment's unique model id.
per-deployment pricing overrides and minus the deployment-scoped fields above.
Per-deployment metadata (``id``, ``access_via_team_ids``, arbitrary custom keys)
never belongs on the shared key; it stays under the deployment's unique model id.
"""
return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS}

View file

@ -13,6 +13,7 @@ import os
import re
from unittest.mock import patch
import httpx
import pytest
@ -399,6 +400,144 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override():
)
assert bridge_model == "gpt-5.4"
assert bridge_model_info["mode"] == "responses"
# A responses-only backend has no chat route, so the alias asking for chat
# still gets bridged rather than being taken at its word.
downgraded_info, _ = responses_api_bridge_check(
model="gpt-5.4",
custom_llm_provider="chatgpt",
deployment_mode="chat",
)
assert downgraded_info["mode"] == "responses"
finally:
_restore_model_cost_entries(model_keys)
def test_a_responses_deployment_does_not_move_its_siblings_onto_the_bridge():
"""
https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model
share a single litellm.model_cost entry, so registering a `mode: responses`
deployment used to switch every other deployment of that model onto the Responses
API, and deleting it again did not undo that.
"""
from litellm.main import responses_api_bridge_check
backend_model = "openai/gpt-5.1"
model_keys = {
key: copy.deepcopy(litellm.model_cost.get(key))
for key in (backend_model, "gpt-5.1", "chat-38543", "responses-probe-38543")
}
try:
router = Router(
model_list=[
{
"model_name": "my-chat-model",
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
"model_info": {"id": "chat-38543"},
}
]
)
catalog_mode = litellm.get_model_info(model="gpt-5.1", custom_llm_provider="openai").get("mode")
assert catalog_mode == "chat", "test needs a backend the catalog serves over chat completions"
router.add_deployment(
deployment=Deployment(
model_name="my-responses-probe",
litellm_params=LiteLLM_Params(model=backend_model, api_key="sk-fake"),
model_info=ModelInfo(id="responses-probe-38543", mode="responses"),
)
)
# The probe speaks for itself...
probe_info, _ = responses_api_bridge_check(
model="gpt-5.1",
custom_llm_provider="openai",
deployment_mode="responses",
)
assert probe_info["mode"] == "responses"
# ...and for nobody else, whether or not it is still registered.
for step in ("with the probe registered", "after deleting the probe"):
sibling_info, _ = responses_api_bridge_check(model="gpt-5.1", custom_llm_provider="openai")
assert sibling_info.get("mode") == "chat", f"sibling moved onto the responses bridge {step}"
assert litellm.model_cost["gpt-5.1"].get("mode") == "chat", f"shared key rewritten {step}"
router.delete_deployment(id="responses-probe-38543")
finally:
_restore_model_cost_entries(model_keys)
_OPENAI_CHAT_COMPLETION_BODY = {
"id": "chatcmpl-38543",
"object": "chat.completion",
"created": 0,
"model": "gpt-5.1",
"choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}],
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
}
_OPENAI_RESPONSES_BODY = {
"id": "resp_38543",
"object": "response",
"created_at": 0,
"model": "gpt-5.1",
"status": "completed",
"output": [
{
"type": "message",
"id": "msg_38543",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": "ok", "annotations": []}],
}
],
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2},
}
@pytest.mark.parametrize(
"model_name, expected_endpoint",
[("my-chat-model", "chat"), ("my-responses-probe", "responses")],
)
def test_each_deployment_is_served_on_the_api_its_own_model_info_asks_for(model_name, expected_endpoint):
"""End to end for https://github.com/BerriAI/litellm/issues/38543: the plain
deployment keeps hitting chat completions while its `mode: responses` sibling,
registered against the same provider model, goes to the Responses API.
"""
import respx
model_keys = {
key: copy.deepcopy(litellm.model_cost.get(key))
for key in ("openai/gpt-5.1", "gpt-5.1", "chat-38543-routed", "responses-probe-38543-routed")
}
try:
router = Router(
model_list=[
{
"model_name": "my-chat-model",
"litellm_params": {"model": "openai/gpt-5.1", "api_key": "sk-fake"},
"model_info": {"id": "chat-38543-routed"},
},
{
"model_name": "my-responses-probe",
"litellm_params": {"model": "openai/gpt-5.1", "api_key": "sk-fake"},
"model_info": {"id": "responses-probe-38543-routed", "mode": "responses"},
},
]
)
with respx.mock(assert_all_called=False) as openai_api:
chat = openai_api.post("https://api.openai.com/v1/chat/completions").mock(
return_value=httpx.Response(200, json=_OPENAI_CHAT_COMPLETION_BODY)
)
responses = openai_api.post("https://api.openai.com/v1/responses").mock(
return_value=httpx.Response(200, json=_OPENAI_RESPONSES_BODY)
)
router.completion(model=model_name, messages=[{"role": "user", "content": "hi"}])
called = {"chat": chat.called, "responses": responses.called}
assert called == {"chat": expected_endpoint == "chat", "responses": expected_endpoint == "responses"}
finally:
_restore_model_cost_entries(model_keys)
@ -775,8 +914,8 @@ def test_add_deployment_does_not_leak_custom_metadata_to_shared_backend_key():
def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest():
"""Unit test of the whitelist helper: cost-map schema fields survive,
custom pricing overrides and per-deployment metadata do not.
"""Unit test of the whitelist helper: cost-map schema fields survive, custom
pricing overrides, per-deployment metadata and `mode` do not.
"""
from litellm.types.utils import shared_backend_model_info
@ -799,7 +938,6 @@ def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest():
)
assert filtered == {
"mode": "chat",
"litellm_provider": "openai",
"max_tokens": 128000,
"supports_vision": True,