This commit is contained in:
madhu19991 2026-09-30 16:55:07 -04:00 • committed by GitHub
commit c49bb69462
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 318 additions and 46 deletions

View file

@ -411,7 +411,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
return None
if resolved_provider == "litellm_proxy":
return None
from litellm.main import responses_api_bridge_check
from litellm.main import responses_api_bridge_check, router_deployment_mode
web_search_options: Final = completion_kwargs.get("web_search_options")
tools: Final = completion_kwargs.get("tools")
@ -424,6 +424,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
tools=cast("list[dict[str, object]]", tools) if isinstance(tools, list) else None,
reasoning_effort=reasoning_effort,
api_base=resolved_api_base,
deployment_mode=router_deployment_mode(completion_kwargs),
)
return None if model_info.get("mode") == "responses" else effort

View file

@ -19,7 +19,7 @@ import random
import sys
import time
import traceback
from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Mapping, Sequence
from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Iterator, Mapping, Sequence
from concurrent import futures
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
from copy import deepcopy
@ -168,6 +168,7 @@ from litellm.utils import (
get_secret,
get_standard_openai_params,
mock_completion_streaming_obj,
mode_register_model_would_merge_into,
pre_process_non_default_params,
read_config_args,
should_run_mock_completion,
@ -1084,6 +1085,7 @@ def responses_api_bridge_check(
reasoning_effort: str | Mapping[str, object] | None = None,
reasoning_summary: object | None = None,
api_base: str | None = None,
deployment_mode: str | None = None,
) -> tuple[dict, str]:
model_info: dict[str, object] = {}
@ -1112,7 +1114,7 @@ def responses_api_bridge_check(
mode = "responses"
model_info["mode"] = mode
if web_search_options is not None and custom_llm_provider == "xai":
if deployment_mode == "responses" or (web_search_options is not None and custom_llm_provider == "xai"):
model_info["mode"] = "responses"
model = model.replace("responses/", "")
@ -1263,20 +1265,36 @@ def _build_custom_pricing_entry(
return entry
def _get_router_deployment_id(kwargs: dict) -> str | None:
def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]:
for metadata_key in ("litellm_metadata", "metadata"):
metadata = kwargs.get(metadata_key) or {}
metadata = kwargs.get(metadata_key)
if not isinstance(metadata, dict):
continue
deployment_model_info = metadata.get("model_info") or {}
if not isinstance(deployment_model_info, dict):
continue
deployment_model_info = metadata.get("model_info")
if isinstance(deployment_model_info, dict):
yield deployment_model_info
def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None:
for deployment_model_info in _router_deployment_model_infos(kwargs):
deployment_id = deployment_model_info.get("id")
if deployment_id is not None:
return str(deployment_id)
return None
def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None:
"""The ``mode`` the router registered for the deployment it picked for this request, if any.
Only the deployment id is taken from request metadata; the mode itself is read from the
router's own registration, so a caller can't hand-write one into their metadata.
"""
deployment_id: Final = _get_router_deployment_id(kwargs)
registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None
mode: Final = registered.get("mode") if isinstance(registered, dict) else None
return mode if isinstance(mode, str) else None
def _register_custom_pricing_for_request(
model: str,
custom_llm_provider: str,
@ -1303,10 +1321,15 @@ def _register_custom_pricing_for_request(
if deployment_id is None:
litellm.register_model({shared_key: entry}, persist_across_reloads=False)
return
shared_entry: Final = CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry)
litellm.register_model(
{
deployment_id: entry,
shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry),
shared_key: (
{k: v for k, v in shared_entry.items() if k != "mode"}
if mode_register_model_would_merge_into(shared_key, custom_llm_provider) is not None
else shared_entry
),
},
persist_across_reloads=False,
warning_display_name=shared_key,
@ -5480,11 +5503,13 @@ def completion(
)
## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name
_deployment_mode: Final = router_deployment_mode(kwargs)
responses_api_model_info, model = responses_api_bridge_check(
model=model,
custom_llm_provider=custom_llm_provider,
web_search_options=web_search_options,
api_base=api_base,
deployment_mode=_deployment_mode,
)
if not _is_claude_tool_target(custom_llm_provider=custom_llm_provider, model=model):
@ -5768,6 +5793,7 @@ def completion(
reasoning_effort=reasoning_effort,
reasoning_summary=_reasoning_summary_for_bridge,
api_base=api_base,
deployment_mode=_deployment_mode,
)
# Use base_model (the true underlying model) for Azure model-type

View file

@ -332,6 +332,7 @@ class LiteLLMCompletionResponsesConfig:
reasoning_effort: str | None,
reasoning_summary: str | None,
api_base: str | None,
deployment_mode: str | None,
) -> bool:
"""
Whether ``litellm.completion`` will route this model back onto the Responses API.
@ -350,6 +351,7 @@ class LiteLLMCompletionResponsesConfig:
reasoning_effort=reasoning_effort,
reasoning_summary=reasoning_summary,
api_base=api_base,
deployment_mode=deployment_mode,
)
except Exception as e: # noqa: BLE001 # a capability probe must never fail the request it probes for
verbose_logger.debug("responses bridge: reasoning effort mode check failed: %s", e)
@ -364,6 +366,7 @@ class LiteLLMCompletionResponsesConfig:
tools: Sequence[ChatCompletionToolParam | OpenAIMcpServerTool] | None = None,
web_search_options: OpenAIWebSearchOptions | None = None,
api_base: str | None = None,
deployment_mode: str | None = None,
) -> ResponsesReasoningChatForm:
"""
Split the Responses ``reasoning`` object into the params Chat Completions understands.
@ -393,6 +396,7 @@ class LiteLLMCompletionResponsesConfig:
reasoning_effort=effort,
reasoning_summary=summary,
api_base=api_base,
deployment_mode=deployment_mode,
)
return ResponsesReasoningChatForm(effort=effort, summary=summary if bridges_back else None)
@ -409,6 +413,8 @@ class LiteLLMCompletionResponsesConfig:
"""
Transform a Responses API request into a Chat Completion request
"""
from litellm.main import router_deployment_mode
(
tools,
web_search_options,
@ -433,6 +439,7 @@ class LiteLLMCompletionResponsesConfig:
tools=tools,
web_search_options=web_search_options,
api_base=kwargs.get("api_base"),
deployment_mode=router_deployment_mode(kwargs),
)
litellm_completion_request: dict = {

View file

@ -333,6 +333,7 @@ from litellm.utils import (
get_secret,
get_utc_datetime,
is_region_allowed,
mode_register_model_would_merge_into,
provider_rejectable_params,
set_live_deployment_replay,
)
@ -10058,41 +10059,10 @@ class Router:
## OLD MODEL REGISTRATION ## Kept to prevent breaking changes
backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider)
backend_key: Final = backend_keys[0]
# For the shared backend key, keep only cost-map schema fields
# (minus custom pricing) so that one deployment's pricing overrides
# or custom metadata (id, access_via_team_ids, arbitrary keys)
# don't pollute another deployment sharing the same backend model
# name. Each deployment's full model_info is already stored under
# its unique model_id above.
shared_model_info: Final = shared_backend_model_info(model_info)
existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode")
deployment_mode: Final = shared_model_info.get("mode")
# Keep the built-in bridge mode stable for shared backend keys.
# Multiple aliases can point at the same provider/model backend,
# but their deployment-level overrides should not downgrade the
# backend from responses -> chat via last-write-wins registration.
# Only preserve in that specific direction so legitimate upgrades
# (e.g. chat -> responses) and unrelated mode changes still apply,
# and so a missing deployment mode does not silently clear the
# existing shared backend mode.
is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat"
would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None
if is_responses_to_chat_downgrade or would_clear_existing_mode:
if deployment_mode is not None:
verbose_router_logger.warning(
"Router: preserving existing mode=%s for shared backend "
"key %s instead of the deployment-specified mode=%s "
"(prevents alias registration from downgrading the "
"shared backend mode).",
existing_shared_mode,
backend_key,
deployment_mode,
)
shared_model_info["mode"] = existing_shared_mode
# Always register the (possibly mode-preserved) shared backend info.
shared_model_info: Final = shared_backend_model_info(
model_info,
existing_mode=mode_register_model_would_merge_into(backend_keys[0], model_info.get("litellm_provider", "")),
)
litellm.register_model(
model_cost={_key: shared_model_info for _key in backend_keys},
persist_across_reloads=False,

View file

@ -3852,15 +3852,23 @@ SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = (
)
def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]:
def shared_backend_model_info(model_info: Mapping[str, Any], existing_mode: object = None) -> dict[str, Any]:
"""Return only the fields safe to register under a shared ``{provider}/{model}``
key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus
per-deployment pricing overrides and deployment-scoped pricing blocks such as
``off_peak_pricing``. Per-deployment metadata (``id``, ``access_via_team_ids``,
arbitrary custom keys) never belongs on the shared key; it stays under the
deployment's unique model id.
``mode`` only fills a shared key that has none yet (``existing_mode``). It is the API
surface one deployment picked, so it must not switch the surface its siblings use;
the routed deployment's own ``mode`` reaches the Responses bridge with the request.
"""
return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS}
return {
k: v
for k, v in model_info.items()
if k in SHARED_BACKEND_MODEL_INFO_FIELDS and (k != "mode" or existing_mode is None)
}
ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens$")

View file

@ -3218,6 +3218,23 @@ def _get_builtin_model_info_for_registration(model: str) -> ModelInfo | None:
return None if is_generalized_model_info(info) else info
def mode_register_model_would_merge_into(key: str, provider: object) -> object:
"""The ``mode`` already on the entry that ``register_model({key: ...})`` merges into.
Mirrors ``register_model``'s key resolution: a provider-prefixed key such as
``openai/gpt-5.6`` lands on the bare catalog entry ``gpt-5.6``, so reading
``litellm.model_cost[key]`` alone would miss the mode that registration overwrites.
"""
skips_model_info_lookup: Final = provider in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO or any(
key.startswith(f"{p}/") for p in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO
)
builtin_model_info: Final = None if skips_model_info_lookup else _get_builtin_model_info_for_registration(key)
if builtin_model_info is not None:
return builtin_model_info.get("mode")
entry: Final = litellm.model_cost.get(key)
return entry.get("mode") if isinstance(entry, dict) else None
_runtime_registered_model_cost: Final[dict[str, dict[str, object]]] = {} # mutable-ok: replayed on reload

View file

@ -10,6 +10,7 @@ from typing import Final
import pytest
import litellm
from litellm.llms.anthropic.pass_through.adapters.handler import (
LiteLLMMessagesToCompletionTransformationHandler,
)
@ -117,6 +118,40 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge:
assert sent == {"effort": "high", "summary": "auto"}
@pytest.mark.parametrize(
"deployment_mode, expected",
[("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")],
)
def test_the_routed_deployments_own_mode_decides_the_shape(
self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object
) -> None:
"""https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the
model's shared cost-map entry, decides the API, and this probe has to agree with
``litellm.completion`` about which API that request lands on"""
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs(
max_tokens=1024,
messages=MESSAGES,
model="databricks/databricks-qwen35-122b-a10b",
metadata=None,
stop_sequences=None,
stream=False,
system=None,
temperature=None,
thinking=SUMMARIZED_THINKING,
tool_choice=None,
tools=None,
top_k=None,
top_p=None,
output_format=None,
extra_kwargs={
"custom_llm_provider": "databricks",
"litellm_metadata": {"model_info": {"id": "routed-deployment"}},
},
)
assert completion_kwargs.get("reasoning_effort") == expected
def test_auto_summary_still_reaches_a_bridged_target(
self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch
) -> None:

View file

@ -2769,6 +2769,26 @@ class TestToolTransformation:
assert result["reasoning_effort"] == "medium"
assert result["reasoning_summary"] == "auto"
@pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)])
def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary):
"""
https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands
on the model's shared cost-map entry, so the probe has to read it off the routed
deployment the same way ``litellm.completion`` does, or it drops the summary of a
request that completion then bridges onto the Responses API
"""
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
model="gpt-4.1",
input="hi",
responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}},
custom_llm_provider="openai",
litellm_metadata={"model_info": {"id": "routed-deployment"}},
)
assert result["reasoning_effort"] == "medium"
assert result.get("reasoning_summary") == expected_summary
@pytest.mark.parametrize(
"model, custom_llm_provider",
[

View file

@ -12,6 +12,8 @@ import copy
import logging
import os
import re
from collections.abc import Mapping
from types import MappingProxyType
from typing import Final
from unittest.mock import Mock, patch
@ -727,6 +729,192 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override():
_restore_model_cost_entries(model_keys)
_MODE_TEST_API_BASE: Final = "http://localhost:38543/v1"
_CATALOG_CHAT_MODEL: Final = "chat-model-38543"
_CHAT_COMPLETION_REPLY: Final = {
"id": "chatcmpl-38543",
"object": "chat.completion",
"created": 1,
"model": _CATALOG_CHAT_MODEL,
"choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "ok"}}],
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
}
_RESPONSES_REPLY: Final = {
"id": "resp_38543",
"object": "response",
"created_at": 1,
"status": "completed",
"model": _CATALOG_CHAT_MODEL,
"output": [
{
"type": "message",
"id": "msg_38543",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": "ok", "annotations": []}],
}
],
"parallel_tool_calls": True,
"tool_choice": "auto",
"tools": [],
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2},
}
def _use_catalog_with_a_chat_model(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(
litellm,
"model_cost",
{**copy.deepcopy(litellm.model_cost), _CATALOG_CHAT_MODEL: {"litellm_provider": "openai", "mode": "chat"}},
)
_invalidate_model_cost_lowercase_map()
def _mode_test_deployment(
model_name: str, mode: str | None, custom_pricing: Mapping[str, float] = MappingProxyType({})
) -> dict[str, object]:
return {
"model_name": model_name,
"litellm_params": {
"model": f"openai/{_CATALOG_CHAT_MODEL}",
"api_key": "sk-fake",
"api_base": _MODE_TEST_API_BASE,
**custom_pricing,
},
"model_info": {"id": f"{model_name}-id", **({"mode": mode} if mode is not None else {})},
}
@pytest.mark.parametrize("sibling_mode", (None, "chat"))
@pytest.mark.parametrize("responses_deployment_first", (True, False))
def test_a_responses_deployment_does_not_move_its_siblings_onto_the_responses_api(
respx_mock, monkeypatch: pytest.MonkeyPatch, sibling_mode: str | None, responses_deployment_first: bool
) -> None:
"""
https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model share
a litellm.model_cost key, so a `mode: responses` deployment used to send every sibling
through the Responses API bridge, whichever order they were registered in
"""
_use_catalog_with_a_chat_model(monkeypatch)
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
)
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
)
sibling: Final = _mode_test_deployment("my-chat-model", sibling_mode)
responses_deployment: Final = _mode_test_deployment("my-responses-probe", "responses")
router: Final = Router(
model_list=[responses_deployment, sibling] if responses_deployment_first else [sibling, responses_deployment]
)
messages: Final = [{"role": "user", "content": "hi"}]
router.completion(model="my-chat-model", messages=messages)
assert (chat_route.call_count, responses_route.call_count) == (1, 0)
router.completion(model="my-responses-probe", messages=messages)
assert (chat_route.call_count, responses_route.call_count) == (1, 1)
_invalidate_model_cost_lowercase_map()
def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_a_request(
respx_mock, monkeypatch: pytest.MonkeyPatch
) -> None:
"""
A deployment with custom pricing re-registers its model_info on every request it serves,
so the shared key has to stay out of its mode on that path too, not just at router setup
"""
_use_catalog_with_a_chat_model(monkeypatch)
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
)
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
)
router: Final = Router(
model_list=[
_mode_test_deployment("my-chat-model", None),
_mode_test_deployment(
"my-responses-probe",
"responses",
custom_pricing=MappingProxyType({"input_cost_per_token": 1e-6, "output_cost_per_token": 2e-6}),
),
]
)
messages: Final = [{"role": "user", "content": "hi"}]
router.completion(model="my-responses-probe", messages=messages)
router.completion(model="my-chat-model", messages=messages)
assert (chat_route.call_count, responses_route.call_count) == (1, 1)
_invalidate_model_cost_lowercase_map()
@pytest.mark.parametrize(
"caller_model_info",
({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}),
)
def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata(
respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str]
) -> None:
"""
The Responses bridge reads the mode the router registered for the routed deployment, never a
mode written into the request's own metadata
"""
_use_catalog_with_a_chat_model(monkeypatch)
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
)
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
)
router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)])
router.completion(
model="my-chat-model",
messages=[{"role": "user", "content": "hi"}],
litellm_metadata={"model_info": dict(caller_model_info)},
)
assert (chat_route.call_count, responses_route.call_count) == (1, 0)
_invalidate_model_cost_lowercase_map()
def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None:
"""
For a provider model the catalog doesn't know, proxy model_info is how its mode gets
onboarded (see mantle_supports_responses), so a deployment's mode must still fill a
shared key that has none; only an existing mode is left alone
"""
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
_invalidate_model_cost_lowercase_map()
backend_model: Final = "openai/unmapped-38543-model"
assert backend_model not in litellm.model_cost
Router(
model_list=[
{
"model_name": "unmapped-responses",
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
"model_info": {"id": "unmapped-responses-id", "mode": "responses"},
},
{
"model_name": "unmapped-chat",
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
"model_info": {"id": "unmapped-chat-id", "mode": "chat"},
},
]
)
assert litellm.model_cost[backend_model]["mode"] == "responses"
assert litellm.model_cost["unmapped-chat-id"]["mode"] == "chat"
_invalidate_model_cost_lowercase_map()
def test_partial_custom_pricing_inherits_builtin_cache_pricing():
"""A deployment that overrides only input/output cost on a cache-supporting
model must still bill cache_read and cache_creation tokens. Before the