mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(router): read the routed deployment's mode from its registration, not request metadata
Only the deployment id now comes from request metadata. The mode is read from what the router registered under that id, so a caller who writes model_info.mode into their own metadata can't move a chat deployment onto the Responses API
This commit is contained in:
parent
f0c57a5f1f
commit
9570f41081
4 changed files with 48 additions and 12 deletions
|
|
@ -1284,12 +1284,15 @@ def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None:
|
|||
|
||||
|
||||
def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None:
|
||||
"""The ``model_info.mode`` of the deployment the router picked for this request, if any."""
|
||||
for deployment_model_info in _router_deployment_model_infos(kwargs):
|
||||
mode = deployment_model_info.get("mode")
|
||||
if isinstance(mode, str):
|
||||
return mode
|
||||
return None
|
||||
"""The ``mode`` the router registered for the deployment it picked for this request, if any.
|
||||
|
||||
Only the deployment id is taken from request metadata; the mode itself is read from the
|
||||
router's own registration, so a caller can't hand-write one into their metadata.
|
||||
"""
|
||||
deployment_id: Final = _get_router_deployment_id(kwargs)
|
||||
registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None
|
||||
mode: Final = registered.get("mode") if isinstance(registered, dict) else None
|
||||
return mode if isinstance(mode, str) else None
|
||||
|
||||
|
||||
def _register_custom_pricing_for_request(
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ from typing import Final
|
|||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.anthropic.pass_through.adapters.handler import (
|
||||
LiteLLMMessagesToCompletionTransformationHandler,
|
||||
)
|
||||
|
|
@ -122,11 +123,12 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge:
|
|||
[("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")],
|
||||
)
|
||||
def test_the_routed_deployments_own_mode_decides_the_shape(
|
||||
self, local_model_cost_map: None, deployment_mode: str, expected: object
|
||||
self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object
|
||||
) -> None:
|
||||
"""https://github.com/BerriAI/litellm/issues/38543 - a deployment's mode rides with the
|
||||
request instead of the model's shared cost-map entry, and this probe has to agree with
|
||||
"""https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the
|
||||
model's shared cost-map entry, decides the API, and this probe has to agree with
|
||||
``litellm.completion`` about which API that request lands on"""
|
||||
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
|
||||
completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs(
|
||||
max_tokens=1024,
|
||||
messages=MESSAGES,
|
||||
|
|
@ -144,7 +146,7 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge:
|
|||
output_format=None,
|
||||
extra_kwargs={
|
||||
"custom_llm_provider": "databricks",
|
||||
"litellm_metadata": {"model_info": {"id": "routed-deployment", "mode": deployment_mode}},
|
||||
"litellm_metadata": {"model_info": {"id": "routed-deployment"}},
|
||||
},
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -2770,19 +2770,20 @@ class TestToolTransformation:
|
|||
assert result["reasoning_summary"] == "auto"
|
||||
|
||||
@pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)])
|
||||
def test_summary_follows_the_routed_deployments_own_mode(self, deployment_mode, expected_summary):
|
||||
def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary):
|
||||
"""
|
||||
https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands
|
||||
on the model's shared cost-map entry, so the probe has to read it off the routed
|
||||
deployment the same way ``litellm.completion`` does, or it drops the summary of a
|
||||
request that completion then bridges onto the Responses API
|
||||
"""
|
||||
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
|
||||
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="gpt-4.1",
|
||||
input="hi",
|
||||
responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}},
|
||||
custom_llm_provider="openai",
|
||||
litellm_metadata={"model_info": {"id": "routed-deployment", "mode": deployment_mode}},
|
||||
litellm_metadata={"model_info": {"id": "routed-deployment"}},
|
||||
)
|
||||
|
||||
assert result["reasoning_effort"] == "medium"
|
||||
|
|
|
|||
|
|
@ -854,6 +854,36 @@ def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_
|
|||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"caller_model_info",
|
||||
({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}),
|
||||
)
|
||||
def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata(
|
||||
respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str]
|
||||
) -> None:
|
||||
"""
|
||||
The Responses bridge reads the mode the router registered for the routed deployment, never a
|
||||
mode written into the request's own metadata
|
||||
"""
|
||||
_use_catalog_with_a_chat_model(monkeypatch)
|
||||
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
|
||||
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
|
||||
)
|
||||
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
|
||||
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
|
||||
)
|
||||
router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)])
|
||||
|
||||
router.completion(
|
||||
model="my-chat-model",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
litellm_metadata={"model_info": dict(caller_model_info)},
|
||||
)
|
||||
|
||||
assert (chat_route.call_count, responses_route.call_count) == (1, 0)
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""
|
||||
For a provider model the catalog doesn't know, proxy model_info is how its mode gets
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue