mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(router): scope model_info.mode to its own deployment
Deployments pointing at the same provider model share one litellm.model_cost entry, and `mode` was written into it. Whoever registered last decided which API every sibling got served on, so adding one `responses` deployment moved unrelated model groups onto the Responses API bridge, and deleting it again did not undo that until the cost map was rebuilt on a restart `mode` now stays under each deployment's own model id and travels with the request, so a deployment only ever speaks for itself. A backend the catalog marks responses-only still cannot be talked down to chat, which is what the removed downgrade guard was there for. That guard also read `openai/gpt-5.1` while registration writes `gpt-5.1`, so it never fired for prefixed models anyway Fixes #38543
This commit is contained in:
parent
ec3f8183c3
commit
8442b062fe
4 changed files with 210 additions and 50 deletions
|
|
@ -19,7 +19,7 @@ import random
|
|||
import sys
|
||||
import time
|
||||
import traceback
|
||||
from collections.abc import AsyncIterator, Coroutine, Iterable, Mapping, Sequence
|
||||
from collections.abc import AsyncIterator, Coroutine, Iterable, Iterator, Mapping, Sequence
|
||||
from concurrent import futures
|
||||
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
|
||||
from copy import deepcopy
|
||||
|
|
@ -122,6 +122,7 @@ from litellm.types.utils import (
|
|||
ModelResponseStream,
|
||||
RawRequestTypedDict,
|
||||
StreamingChoices,
|
||||
strip_deployment_scoped_model_info,
|
||||
)
|
||||
from litellm.utils import (
|
||||
Choices,
|
||||
|
|
@ -1009,6 +1010,7 @@ def responses_api_bridge_check(
|
|||
reasoning_effort: str | Mapping[str, object] | None = None,
|
||||
reasoning_summary: object | None = None,
|
||||
api_base: str | None = None,
|
||||
deployment_mode: str | None = None,
|
||||
) -> tuple[dict, str]:
|
||||
model_info: dict[str, object] = {}
|
||||
|
||||
|
|
@ -1041,6 +1043,14 @@ def responses_api_bridge_check(
|
|||
mode = "responses"
|
||||
model_info["mode"] = mode
|
||||
|
||||
# The cost-map entry is keyed by provider model, so every deployment of that model
|
||||
# reads the same mode. A deployment asking for the Responses API only speaks for
|
||||
# itself, hence the request-scoped override. It can only turn the bridge on: a
|
||||
# backend the catalog marks responses-only (e.g. `chatgpt/*`) has no chat route to
|
||||
# fall back to, so a deployment claiming `mode: chat` there would just fail.
|
||||
if deployment_mode == "responses":
|
||||
model_info["mode"] = "responses"
|
||||
|
||||
# OpenAI/Azure GPT-5 chat-completions that need Responses-only fields (e.g.
|
||||
# ``reasoningSummary`` in ``extra_body``) must be bridged; Chat Completions rejects
|
||||
# those keys.
|
||||
|
|
@ -1174,20 +1184,39 @@ def _build_custom_pricing_entry(
|
|||
return entry
|
||||
|
||||
|
||||
def _get_router_deployment_id(kwargs: dict) -> str | None:
|
||||
def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]:
|
||||
"""The ``model_info`` the router stamped into this request's metadata, if any."""
|
||||
for metadata_key in ("litellm_metadata", "metadata"):
|
||||
metadata = kwargs.get(metadata_key) or {}
|
||||
if not isinstance(metadata, dict):
|
||||
continue
|
||||
deployment_model_info = metadata.get("model_info") or {}
|
||||
if not isinstance(deployment_model_info, dict):
|
||||
continue
|
||||
if isinstance(deployment_model_info, dict):
|
||||
yield deployment_model_info
|
||||
|
||||
|
||||
def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None:
|
||||
for deployment_model_info in _router_deployment_model_infos(kwargs):
|
||||
deployment_id = deployment_model_info.get("id")
|
||||
if deployment_id is not None:
|
||||
return str(deployment_id)
|
||||
return None
|
||||
|
||||
|
||||
def _get_router_deployment_mode(kwargs: Mapping[str, object]) -> str | None:
|
||||
"""The API surface the routed deployment configured for itself.
|
||||
|
||||
Deployments of one provider model share a single ``litellm.model_cost`` entry, so
|
||||
the mode a deployment sets has to travel with the request instead of being written
|
||||
there, where it would answer for its siblings too.
|
||||
"""
|
||||
for deployment_model_info in _router_deployment_model_infos(kwargs):
|
||||
mode = deployment_model_info.get("mode")
|
||||
if isinstance(mode, str):
|
||||
return mode
|
||||
return None
|
||||
|
||||
|
||||
def _register_custom_pricing_for_request(
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
|
|
@ -1199,10 +1228,11 @@ def _register_custom_pricing_for_request(
|
|||
Router-originated requests (identified by the deployment id the router puts
|
||||
in metadata) get their full pricing registered under that unique id only;
|
||||
the shared ``{provider}/{model}`` key receives the entry with pricing fields
|
||||
stripped, mirroring Router._create_deployment. This keeps one deployment's
|
||||
pricing overrides (e.g. a zero-cost wildcard) from clobbering built-in
|
||||
pricing used by sibling deployments of the same backend model. Direct SDK
|
||||
calls keep the legacy behavior of registering the shared key with pricing.
|
||||
and ``mode`` stripped, mirroring Router._create_deployment. This keeps one
|
||||
deployment's pricing overrides (e.g. a zero-cost wildcard) or its choice of
|
||||
API surface from clobbering the built-in entry that sibling deployments of
|
||||
the same backend model read. Direct SDK calls keep the legacy behavior of
|
||||
registering the shared key with pricing.
|
||||
"""
|
||||
entry: Final = _build_custom_pricing_entry(
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
|
|
@ -1217,7 +1247,9 @@ def _register_custom_pricing_for_request(
|
|||
litellm.register_model(
|
||||
{
|
||||
deployment_id: entry,
|
||||
shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry),
|
||||
shared_key: strip_deployment_scoped_model_info(
|
||||
CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry)
|
||||
),
|
||||
},
|
||||
persist_across_reloads=False,
|
||||
warning_display_name=shared_key,
|
||||
|
|
@ -5302,11 +5334,13 @@ def completion(
|
|||
)
|
||||
|
||||
## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name
|
||||
deployment_mode: Final = _get_router_deployment_mode(kwargs)
|
||||
responses_api_model_info, model = responses_api_bridge_check(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
web_search_options=web_search_options,
|
||||
api_base=api_base,
|
||||
deployment_mode=deployment_mode,
|
||||
)
|
||||
|
||||
if not _should_allow_input_examples(custom_llm_provider=custom_llm_provider, model=model):
|
||||
|
|
@ -5552,6 +5586,7 @@ def completion(
|
|||
reasoning_effort=reasoning_effort,
|
||||
reasoning_summary=_reasoning_summary_for_bridge,
|
||||
api_base=api_base,
|
||||
deployment_mode=deployment_mode,
|
||||
)
|
||||
|
||||
# Use base_model (the true underlying model) for Azure model-type
|
||||
|
|
|
|||
|
|
@ -9298,41 +9298,14 @@ class Router:
|
|||
|
||||
## OLD MODEL REGISTRATION ## Kept to prevent breaking changes
|
||||
backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider)
|
||||
backend_key: Final = backend_keys[0]
|
||||
|
||||
# For the shared backend key, keep only cost-map schema fields
|
||||
# (minus custom pricing) so that one deployment's pricing overrides
|
||||
# or custom metadata (id, access_via_team_ids, arbitrary keys)
|
||||
# don't pollute another deployment sharing the same backend model
|
||||
# name. Each deployment's full model_info is already stored under
|
||||
# its unique model_id above.
|
||||
# (minus custom pricing and minus `mode`) so that one deployment's
|
||||
# pricing overrides, API surface or custom metadata (id,
|
||||
# access_via_team_ids, arbitrary keys) don't pollute another
|
||||
# deployment sharing the same backend model name. Each deployment's
|
||||
# full model_info is already stored under its unique model_id above.
|
||||
shared_model_info: Final = shared_backend_model_info(model_info)
|
||||
existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode")
|
||||
deployment_mode: Final = shared_model_info.get("mode")
|
||||
# Keep the built-in bridge mode stable for shared backend keys.
|
||||
# Multiple aliases can point at the same provider/model backend,
|
||||
# but their deployment-level overrides should not downgrade the
|
||||
# backend from responses -> chat via last-write-wins registration.
|
||||
# Only preserve in that specific direction so legitimate upgrades
|
||||
# (e.g. chat -> responses) and unrelated mode changes still apply,
|
||||
# and so a missing deployment mode does not silently clear the
|
||||
# existing shared backend mode.
|
||||
is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat"
|
||||
would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None
|
||||
if is_responses_to_chat_downgrade or would_clear_existing_mode:
|
||||
if deployment_mode is not None:
|
||||
verbose_router_logger.warning(
|
||||
"Router: preserving existing mode=%s for shared backend "
|
||||
"key %s instead of the deployment-specified mode=%s "
|
||||
"(prevents alias registration from downgrading the "
|
||||
"shared backend mode).",
|
||||
existing_shared_mode,
|
||||
backend_key,
|
||||
deployment_mode,
|
||||
)
|
||||
shared_model_info["mode"] = existing_shared_mode
|
||||
|
||||
# Always register the (possibly mode-preserved) shared backend info.
|
||||
litellm.register_model(
|
||||
model_cost={_key: shared_model_info for _key in backend_keys},
|
||||
persist_across_reloads=False,
|
||||
|
|
|
|||
|
|
@ -3486,17 +3486,31 @@ class CustomPricingLiteLLMParams(MirroredPricingParams):
|
|||
return {k: v for k, v in model_info.items() if k not in cls.model_fields}
|
||||
|
||||
|
||||
SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = frozenset(
|
||||
ModelInfoBase.__required_keys__ | ModelInfoBase.__optional_keys__
|
||||
) - frozenset(CustomPricingLiteLLMParams.model_fields)
|
||||
# ``mode`` picks the API surface a request is served on, and two deployments of one
|
||||
# provider model are allowed to disagree about it. The shared key holds a single value,
|
||||
# so whichever deployment registered last would decide for all of them; it stays under
|
||||
# each deployment's own model id instead.
|
||||
DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS: Final[frozenset[str]] = frozenset({"mode"})
|
||||
|
||||
SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = (
|
||||
frozenset(ModelInfoBase.__required_keys__ | ModelInfoBase.__optional_keys__)
|
||||
- frozenset(CustomPricingLiteLLMParams.model_fields)
|
||||
- DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS
|
||||
)
|
||||
|
||||
|
||||
def strip_deployment_scoped_model_info(model_info: Mapping[str, Any]) -> Mapping[str, Any]:
|
||||
"""Return a copy of ``model_info`` without the fields that describe a single
|
||||
deployment rather than the backend model, leaving every other key intact."""
|
||||
return {k: v for k, v in model_info.items() if k not in DEPLOYMENT_SCOPED_MODEL_INFO_FIELDS}
|
||||
|
||||
|
||||
def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Return only the fields safe to register under a shared ``{provider}/{model}``
|
||||
key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus
|
||||
per-deployment pricing overrides. Per-deployment metadata (``id``,
|
||||
``access_via_team_ids``, arbitrary custom keys) never belongs on the shared key;
|
||||
it stays under the deployment's unique model id.
|
||||
per-deployment pricing overrides and minus the deployment-scoped fields above.
|
||||
Per-deployment metadata (``id``, ``access_via_team_ids``, arbitrary custom keys)
|
||||
never belongs on the shared key; it stays under the deployment's unique model id.
|
||||
"""
|
||||
return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS}
|
||||
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ import os
|
|||
import re
|
||||
from unittest.mock import patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
|
||||
|
|
@ -399,6 +400,144 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override():
|
|||
)
|
||||
assert bridge_model == "gpt-5.4"
|
||||
assert bridge_model_info["mode"] == "responses"
|
||||
|
||||
# A responses-only backend has no chat route, so the alias asking for chat
|
||||
# still gets bridged rather than being taken at its word.
|
||||
downgraded_info, _ = responses_api_bridge_check(
|
||||
model="gpt-5.4",
|
||||
custom_llm_provider="chatgpt",
|
||||
deployment_mode="chat",
|
||||
)
|
||||
assert downgraded_info["mode"] == "responses"
|
||||
finally:
|
||||
_restore_model_cost_entries(model_keys)
|
||||
|
||||
|
||||
def test_a_responses_deployment_does_not_move_its_siblings_onto_the_bridge():
|
||||
"""
|
||||
https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model
|
||||
share a single litellm.model_cost entry, so registering a `mode: responses`
|
||||
deployment used to switch every other deployment of that model onto the Responses
|
||||
API, and deleting it again did not undo that.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
backend_model = "openai/gpt-5.1"
|
||||
model_keys = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in (backend_model, "gpt-5.1", "chat-38543", "responses-probe-38543")
|
||||
}
|
||||
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "my-chat-model",
|
||||
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
|
||||
"model_info": {"id": "chat-38543"},
|
||||
}
|
||||
]
|
||||
)
|
||||
catalog_mode = litellm.get_model_info(model="gpt-5.1", custom_llm_provider="openai").get("mode")
|
||||
assert catalog_mode == "chat", "test needs a backend the catalog serves over chat completions"
|
||||
|
||||
router.add_deployment(
|
||||
deployment=Deployment(
|
||||
model_name="my-responses-probe",
|
||||
litellm_params=LiteLLM_Params(model=backend_model, api_key="sk-fake"),
|
||||
model_info=ModelInfo(id="responses-probe-38543", mode="responses"),
|
||||
)
|
||||
)
|
||||
|
||||
# The probe speaks for itself...
|
||||
probe_info, _ = responses_api_bridge_check(
|
||||
model="gpt-5.1",
|
||||
custom_llm_provider="openai",
|
||||
deployment_mode="responses",
|
||||
)
|
||||
assert probe_info["mode"] == "responses"
|
||||
|
||||
# ...and for nobody else, whether or not it is still registered.
|
||||
for step in ("with the probe registered", "after deleting the probe"):
|
||||
sibling_info, _ = responses_api_bridge_check(model="gpt-5.1", custom_llm_provider="openai")
|
||||
assert sibling_info.get("mode") == "chat", f"sibling moved onto the responses bridge {step}"
|
||||
assert litellm.model_cost["gpt-5.1"].get("mode") == "chat", f"shared key rewritten {step}"
|
||||
router.delete_deployment(id="responses-probe-38543")
|
||||
finally:
|
||||
_restore_model_cost_entries(model_keys)
|
||||
|
||||
|
||||
_OPENAI_CHAT_COMPLETION_BODY = {
|
||||
"id": "chatcmpl-38543",
|
||||
"object": "chat.completion",
|
||||
"created": 0,
|
||||
"model": "gpt-5.1",
|
||||
"choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}],
|
||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
|
||||
}
|
||||
|
||||
_OPENAI_RESPONSES_BODY = {
|
||||
"id": "resp_38543",
|
||||
"object": "response",
|
||||
"created_at": 0,
|
||||
"model": "gpt-5.1",
|
||||
"status": "completed",
|
||||
"output": [
|
||||
{
|
||||
"type": "message",
|
||||
"id": "msg_38543",
|
||||
"status": "completed",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "output_text", "text": "ok", "annotations": []}],
|
||||
}
|
||||
],
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2},
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name, expected_endpoint",
|
||||
[("my-chat-model", "chat"), ("my-responses-probe", "responses")],
|
||||
)
|
||||
def test_each_deployment_is_served_on_the_api_its_own_model_info_asks_for(model_name, expected_endpoint):
|
||||
"""End to end for https://github.com/BerriAI/litellm/issues/38543: the plain
|
||||
deployment keeps hitting chat completions while its `mode: responses` sibling,
|
||||
registered against the same provider model, goes to the Responses API.
|
||||
"""
|
||||
import respx
|
||||
|
||||
model_keys = {
|
||||
key: copy.deepcopy(litellm.model_cost.get(key))
|
||||
for key in ("openai/gpt-5.1", "gpt-5.1", "chat-38543-routed", "responses-probe-38543-routed")
|
||||
}
|
||||
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "my-chat-model",
|
||||
"litellm_params": {"model": "openai/gpt-5.1", "api_key": "sk-fake"},
|
||||
"model_info": {"id": "chat-38543-routed"},
|
||||
},
|
||||
{
|
||||
"model_name": "my-responses-probe",
|
||||
"litellm_params": {"model": "openai/gpt-5.1", "api_key": "sk-fake"},
|
||||
"model_info": {"id": "responses-probe-38543-routed", "mode": "responses"},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
with respx.mock(assert_all_called=False) as openai_api:
|
||||
chat = openai_api.post("https://api.openai.com/v1/chat/completions").mock(
|
||||
return_value=httpx.Response(200, json=_OPENAI_CHAT_COMPLETION_BODY)
|
||||
)
|
||||
responses = openai_api.post("https://api.openai.com/v1/responses").mock(
|
||||
return_value=httpx.Response(200, json=_OPENAI_RESPONSES_BODY)
|
||||
)
|
||||
router.completion(model=model_name, messages=[{"role": "user", "content": "hi"}])
|
||||
|
||||
called = {"chat": chat.called, "responses": responses.called}
|
||||
assert called == {"chat": expected_endpoint == "chat", "responses": expected_endpoint == "responses"}
|
||||
finally:
|
||||
_restore_model_cost_entries(model_keys)
|
||||
|
||||
|
|
@ -775,8 +914,8 @@ def test_add_deployment_does_not_leak_custom_metadata_to_shared_backend_key():
|
|||
|
||||
|
||||
def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest():
|
||||
"""Unit test of the whitelist helper: cost-map schema fields survive,
|
||||
custom pricing overrides and per-deployment metadata do not.
|
||||
"""Unit test of the whitelist helper: cost-map schema fields survive, custom
|
||||
pricing overrides, per-deployment metadata and `mode` do not.
|
||||
"""
|
||||
from litellm.types.utils import shared_backend_model_info
|
||||
|
||||
|
|
@ -799,7 +938,6 @@ def test_shared_backend_model_info_keeps_schema_fields_and_drops_the_rest():
|
|||
)
|
||||
|
||||
assert filtered == {
|
||||
"mode": "chat",
|
||||
"litellm_provider": "openai",
|
||||
"max_tokens": 128000,
|
||||
"supports_vision": True,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue