mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Merge 9570f41081 into f285229b51
This commit is contained in:
commit
c49bb69462
9 changed files with 318 additions and 46 deletions
|
|
@ -411,7 +411,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
return None
|
||||
if resolved_provider == "litellm_proxy":
|
||||
return None
|
||||
from litellm.main import responses_api_bridge_check
|
||||
from litellm.main import responses_api_bridge_check, router_deployment_mode
|
||||
|
||||
web_search_options: Final = completion_kwargs.get("web_search_options")
|
||||
tools: Final = completion_kwargs.get("tools")
|
||||
|
|
@ -424,6 +424,7 @@ class LiteLLMMessagesToCompletionTransformationHandler:
|
|||
tools=cast("list[dict[str, object]]", tools) if isinstance(tools, list) else None,
|
||||
reasoning_effort=reasoning_effort,
|
||||
api_base=resolved_api_base,
|
||||
deployment_mode=router_deployment_mode(completion_kwargs),
|
||||
)
|
||||
return None if model_info.get("mode") == "responses" else effort
|
||||
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@ import random
|
|||
import sys
|
||||
import time
|
||||
import traceback
|
||||
from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Mapping, Sequence
|
||||
from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Iterator, Mapping, Sequence
|
||||
from concurrent import futures
|
||||
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
|
||||
from copy import deepcopy
|
||||
|
|
@ -168,6 +168,7 @@ from litellm.utils import (
|
|||
get_secret,
|
||||
get_standard_openai_params,
|
||||
mock_completion_streaming_obj,
|
||||
mode_register_model_would_merge_into,
|
||||
pre_process_non_default_params,
|
||||
read_config_args,
|
||||
should_run_mock_completion,
|
||||
|
|
@ -1084,6 +1085,7 @@ def responses_api_bridge_check(
|
|||
reasoning_effort: str | Mapping[str, object] | None = None,
|
||||
reasoning_summary: object | None = None,
|
||||
api_base: str | None = None,
|
||||
deployment_mode: str | None = None,
|
||||
) -> tuple[dict, str]:
|
||||
model_info: dict[str, object] = {}
|
||||
|
||||
|
|
@ -1112,7 +1114,7 @@ def responses_api_bridge_check(
|
|||
mode = "responses"
|
||||
model_info["mode"] = mode
|
||||
|
||||
if web_search_options is not None and custom_llm_provider == "xai":
|
||||
if deployment_mode == "responses" or (web_search_options is not None and custom_llm_provider == "xai"):
|
||||
model_info["mode"] = "responses"
|
||||
model = model.replace("responses/", "")
|
||||
|
||||
|
|
@ -1263,20 +1265,36 @@ def _build_custom_pricing_entry(
|
|||
return entry
|
||||
|
||||
|
||||
def _get_router_deployment_id(kwargs: dict) -> str | None:
|
||||
def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]:
|
||||
for metadata_key in ("litellm_metadata", "metadata"):
|
||||
metadata = kwargs.get(metadata_key) or {}
|
||||
metadata = kwargs.get(metadata_key)
|
||||
if not isinstance(metadata, dict):
|
||||
continue
|
||||
deployment_model_info = metadata.get("model_info") or {}
|
||||
if not isinstance(deployment_model_info, dict):
|
||||
continue
|
||||
deployment_model_info = metadata.get("model_info")
|
||||
if isinstance(deployment_model_info, dict):
|
||||
yield deployment_model_info
|
||||
|
||||
|
||||
def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None:
|
||||
for deployment_model_info in _router_deployment_model_infos(kwargs):
|
||||
deployment_id = deployment_model_info.get("id")
|
||||
if deployment_id is not None:
|
||||
return str(deployment_id)
|
||||
return None
|
||||
|
||||
|
||||
def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None:
|
||||
"""The ``mode`` the router registered for the deployment it picked for this request, if any.
|
||||
|
||||
Only the deployment id is taken from request metadata; the mode itself is read from the
|
||||
router's own registration, so a caller can't hand-write one into their metadata.
|
||||
"""
|
||||
deployment_id: Final = _get_router_deployment_id(kwargs)
|
||||
registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None
|
||||
mode: Final = registered.get("mode") if isinstance(registered, dict) else None
|
||||
return mode if isinstance(mode, str) else None
|
||||
|
||||
|
||||
def _register_custom_pricing_for_request(
|
||||
model: str,
|
||||
custom_llm_provider: str,
|
||||
|
|
@ -1303,10 +1321,15 @@ def _register_custom_pricing_for_request(
|
|||
if deployment_id is None:
|
||||
litellm.register_model({shared_key: entry}, persist_across_reloads=False)
|
||||
return
|
||||
shared_entry: Final = CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry)
|
||||
litellm.register_model(
|
||||
{
|
||||
deployment_id: entry,
|
||||
shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry),
|
||||
shared_key: (
|
||||
{k: v for k, v in shared_entry.items() if k != "mode"}
|
||||
if mode_register_model_would_merge_into(shared_key, custom_llm_provider) is not None
|
||||
else shared_entry
|
||||
),
|
||||
},
|
||||
persist_across_reloads=False,
|
||||
warning_display_name=shared_key,
|
||||
|
|
@ -5480,11 +5503,13 @@ def completion(
|
|||
)
|
||||
|
||||
## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name
|
||||
_deployment_mode: Final = router_deployment_mode(kwargs)
|
||||
responses_api_model_info, model = responses_api_bridge_check(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
web_search_options=web_search_options,
|
||||
api_base=api_base,
|
||||
deployment_mode=_deployment_mode,
|
||||
)
|
||||
|
||||
if not _is_claude_tool_target(custom_llm_provider=custom_llm_provider, model=model):
|
||||
|
|
@ -5768,6 +5793,7 @@ def completion(
|
|||
reasoning_effort=reasoning_effort,
|
||||
reasoning_summary=_reasoning_summary_for_bridge,
|
||||
api_base=api_base,
|
||||
deployment_mode=_deployment_mode,
|
||||
)
|
||||
|
||||
# Use base_model (the true underlying model) for Azure model-type
|
||||
|
|
|
|||
|
|
@ -332,6 +332,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
reasoning_effort: str | None,
|
||||
reasoning_summary: str | None,
|
||||
api_base: str | None,
|
||||
deployment_mode: str | None,
|
||||
) -> bool:
|
||||
"""
|
||||
Whether ``litellm.completion`` will route this model back onto the Responses API.
|
||||
|
|
@ -350,6 +351,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
reasoning_effort=reasoning_effort,
|
||||
reasoning_summary=reasoning_summary,
|
||||
api_base=api_base,
|
||||
deployment_mode=deployment_mode,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 # a capability probe must never fail the request it probes for
|
||||
verbose_logger.debug("responses bridge: reasoning effort mode check failed: %s", e)
|
||||
|
|
@ -364,6 +366,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
tools: Sequence[ChatCompletionToolParam | OpenAIMcpServerTool] | None = None,
|
||||
web_search_options: OpenAIWebSearchOptions | None = None,
|
||||
api_base: str | None = None,
|
||||
deployment_mode: str | None = None,
|
||||
) -> ResponsesReasoningChatForm:
|
||||
"""
|
||||
Split the Responses ``reasoning`` object into the params Chat Completions understands.
|
||||
|
|
@ -393,6 +396,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
reasoning_effort=effort,
|
||||
reasoning_summary=summary,
|
||||
api_base=api_base,
|
||||
deployment_mode=deployment_mode,
|
||||
)
|
||||
return ResponsesReasoningChatForm(effort=effort, summary=summary if bridges_back else None)
|
||||
|
||||
|
|
@ -409,6 +413,8 @@ class LiteLLMCompletionResponsesConfig:
|
|||
"""
|
||||
Transform a Responses API request into a Chat Completion request
|
||||
"""
|
||||
from litellm.main import router_deployment_mode
|
||||
|
||||
(
|
||||
tools,
|
||||
web_search_options,
|
||||
|
|
@ -433,6 +439,7 @@ class LiteLLMCompletionResponsesConfig:
|
|||
tools=tools,
|
||||
web_search_options=web_search_options,
|
||||
api_base=kwargs.get("api_base"),
|
||||
deployment_mode=router_deployment_mode(kwargs),
|
||||
)
|
||||
|
||||
litellm_completion_request: dict = {
|
||||
|
|
|
|||
|
|
@ -333,6 +333,7 @@ from litellm.utils import (
|
|||
get_secret,
|
||||
get_utc_datetime,
|
||||
is_region_allowed,
|
||||
mode_register_model_would_merge_into,
|
||||
provider_rejectable_params,
|
||||
set_live_deployment_replay,
|
||||
)
|
||||
|
|
@ -10058,41 +10059,10 @@ class Router:
|
|||
|
||||
## OLD MODEL REGISTRATION ## Kept to prevent breaking changes
|
||||
backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider)
|
||||
backend_key: Final = backend_keys[0]
|
||||
|
||||
# For the shared backend key, keep only cost-map schema fields
|
||||
# (minus custom pricing) so that one deployment's pricing overrides
|
||||
# or custom metadata (id, access_via_team_ids, arbitrary keys)
|
||||
# don't pollute another deployment sharing the same backend model
|
||||
# name. Each deployment's full model_info is already stored under
|
||||
# its unique model_id above.
|
||||
shared_model_info: Final = shared_backend_model_info(model_info)
|
||||
existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode")
|
||||
deployment_mode: Final = shared_model_info.get("mode")
|
||||
# Keep the built-in bridge mode stable for shared backend keys.
|
||||
# Multiple aliases can point at the same provider/model backend,
|
||||
# but their deployment-level overrides should not downgrade the
|
||||
# backend from responses -> chat via last-write-wins registration.
|
||||
# Only preserve in that specific direction so legitimate upgrades
|
||||
# (e.g. chat -> responses) and unrelated mode changes still apply,
|
||||
# and so a missing deployment mode does not silently clear the
|
||||
# existing shared backend mode.
|
||||
is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat"
|
||||
would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None
|
||||
if is_responses_to_chat_downgrade or would_clear_existing_mode:
|
||||
if deployment_mode is not None:
|
||||
verbose_router_logger.warning(
|
||||
"Router: preserving existing mode=%s for shared backend "
|
||||
"key %s instead of the deployment-specified mode=%s "
|
||||
"(prevents alias registration from downgrading the "
|
||||
"shared backend mode).",
|
||||
existing_shared_mode,
|
||||
backend_key,
|
||||
deployment_mode,
|
||||
)
|
||||
shared_model_info["mode"] = existing_shared_mode
|
||||
|
||||
# Always register the (possibly mode-preserved) shared backend info.
|
||||
shared_model_info: Final = shared_backend_model_info(
|
||||
model_info,
|
||||
existing_mode=mode_register_model_would_merge_into(backend_keys[0], model_info.get("litellm_provider", "")),
|
||||
)
|
||||
litellm.register_model(
|
||||
model_cost={_key: shared_model_info for _key in backend_keys},
|
||||
persist_across_reloads=False,
|
||||
|
|
|
|||
|
|
@ -3852,15 +3852,23 @@ SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = (
|
|||
)
|
||||
|
||||
|
||||
def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]:
|
||||
def shared_backend_model_info(model_info: Mapping[str, Any], existing_mode: object = None) -> dict[str, Any]:
|
||||
"""Return only the fields safe to register under a shared ``{provider}/{model}``
|
||||
key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus
|
||||
per-deployment pricing overrides and deployment-scoped pricing blocks such as
|
||||
``off_peak_pricing``. Per-deployment metadata (``id``, ``access_via_team_ids``,
|
||||
arbitrary custom keys) never belongs on the shared key; it stays under the
|
||||
deployment's unique model id.
|
||||
|
||||
``mode`` only fills a shared key that has none yet (``existing_mode``). It is the API
|
||||
surface one deployment picked, so it must not switch the surface its siblings use;
|
||||
the routed deployment's own ``mode`` reaches the Responses bridge with the request.
|
||||
"""
|
||||
return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS}
|
||||
return {
|
||||
k: v
|
||||
for k, v in model_info.items()
|
||||
if k in SHARED_BACKEND_MODEL_INFO_FIELDS and (k != "mode" or existing_mode is None)
|
||||
}
|
||||
|
||||
|
||||
ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens$")
|
||||
|
|
|
|||
|
|
@ -3218,6 +3218,23 @@ def _get_builtin_model_info_for_registration(model: str) -> ModelInfo | None:
|
|||
return None if is_generalized_model_info(info) else info
|
||||
|
||||
|
||||
def mode_register_model_would_merge_into(key: str, provider: object) -> object:
|
||||
"""The ``mode`` already on the entry that ``register_model({key: ...})`` merges into.
|
||||
|
||||
Mirrors ``register_model``'s key resolution: a provider-prefixed key such as
|
||||
``openai/gpt-5.6`` lands on the bare catalog entry ``gpt-5.6``, so reading
|
||||
``litellm.model_cost[key]`` alone would miss the mode that registration overwrites.
|
||||
"""
|
||||
skips_model_info_lookup: Final = provider in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO or any(
|
||||
key.startswith(f"{p}/") for p in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO
|
||||
)
|
||||
builtin_model_info: Final = None if skips_model_info_lookup else _get_builtin_model_info_for_registration(key)
|
||||
if builtin_model_info is not None:
|
||||
return builtin_model_info.get("mode")
|
||||
entry: Final = litellm.model_cost.get(key)
|
||||
return entry.get("mode") if isinstance(entry, dict) else None
|
||||
|
||||
|
||||
_runtime_registered_model_cost: Final[dict[str, dict[str, object]]] = {} # mutable-ok: replayed on reload
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ from typing import Final
|
|||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.anthropic.pass_through.adapters.handler import (
|
||||
LiteLLMMessagesToCompletionTransformationHandler,
|
||||
)
|
||||
|
|
@ -117,6 +118,40 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge:
|
|||
|
||||
assert sent == {"effort": "high", "summary": "auto"}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"deployment_mode, expected",
|
||||
[("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")],
|
||||
)
|
||||
def test_the_routed_deployments_own_mode_decides_the_shape(
|
||||
self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object
|
||||
) -> None:
|
||||
"""https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the
|
||||
model's shared cost-map entry, decides the API, and this probe has to agree with
|
||||
``litellm.completion`` about which API that request lands on"""
|
||||
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
|
||||
completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs(
|
||||
max_tokens=1024,
|
||||
messages=MESSAGES,
|
||||
model="databricks/databricks-qwen35-122b-a10b",
|
||||
metadata=None,
|
||||
stop_sequences=None,
|
||||
stream=False,
|
||||
system=None,
|
||||
temperature=None,
|
||||
thinking=SUMMARIZED_THINKING,
|
||||
tool_choice=None,
|
||||
tools=None,
|
||||
top_k=None,
|
||||
top_p=None,
|
||||
output_format=None,
|
||||
extra_kwargs={
|
||||
"custom_llm_provider": "databricks",
|
||||
"litellm_metadata": {"model_info": {"id": "routed-deployment"}},
|
||||
},
|
||||
)
|
||||
|
||||
assert completion_kwargs.get("reasoning_effort") == expected
|
||||
|
||||
def test_auto_summary_still_reaches_a_bridged_target(
|
||||
self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -2769,6 +2769,26 @@ class TestToolTransformation:
|
|||
assert result["reasoning_effort"] == "medium"
|
||||
assert result["reasoning_summary"] == "auto"
|
||||
|
||||
@pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)])
|
||||
def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary):
|
||||
"""
|
||||
https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands
|
||||
on the model's shared cost-map entry, so the probe has to read it off the routed
|
||||
deployment the same way ``litellm.completion`` does, or it drops the summary of a
|
||||
request that completion then bridges onto the Responses API
|
||||
"""
|
||||
monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode})
|
||||
result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request(
|
||||
model="gpt-4.1",
|
||||
input="hi",
|
||||
responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}},
|
||||
custom_llm_provider="openai",
|
||||
litellm_metadata={"model_info": {"id": "routed-deployment"}},
|
||||
)
|
||||
|
||||
assert result["reasoning_effort"] == "medium"
|
||||
assert result.get("reasoning_summary") == expected_summary
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, custom_llm_provider",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -12,6 +12,8 @@ import copy
|
|||
import logging
|
||||
import os
|
||||
import re
|
||||
from collections.abc import Mapping
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
from unittest.mock import Mock, patch
|
||||
|
||||
|
|
@ -727,6 +729,192 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override():
|
|||
_restore_model_cost_entries(model_keys)
|
||||
|
||||
|
||||
_MODE_TEST_API_BASE: Final = "http://localhost:38543/v1"
|
||||
|
||||
_CATALOG_CHAT_MODEL: Final = "chat-model-38543"
|
||||
|
||||
_CHAT_COMPLETION_REPLY: Final = {
|
||||
"id": "chatcmpl-38543",
|
||||
"object": "chat.completion",
|
||||
"created": 1,
|
||||
"model": _CATALOG_CHAT_MODEL,
|
||||
"choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "ok"}}],
|
||||
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
|
||||
}
|
||||
|
||||
_RESPONSES_REPLY: Final = {
|
||||
"id": "resp_38543",
|
||||
"object": "response",
|
||||
"created_at": 1,
|
||||
"status": "completed",
|
||||
"model": _CATALOG_CHAT_MODEL,
|
||||
"output": [
|
||||
{
|
||||
"type": "message",
|
||||
"id": "msg_38543",
|
||||
"status": "completed",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "output_text", "text": "ok", "annotations": []}],
|
||||
}
|
||||
],
|
||||
"parallel_tool_calls": True,
|
||||
"tool_choice": "auto",
|
||||
"tools": [],
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2},
|
||||
}
|
||||
|
||||
|
||||
def _use_catalog_with_a_chat_model(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(
|
||||
litellm,
|
||||
"model_cost",
|
||||
{**copy.deepcopy(litellm.model_cost), _CATALOG_CHAT_MODEL: {"litellm_provider": "openai", "mode": "chat"}},
|
||||
)
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def _mode_test_deployment(
|
||||
model_name: str, mode: str | None, custom_pricing: Mapping[str, float] = MappingProxyType({})
|
||||
) -> dict[str, object]:
|
||||
return {
|
||||
"model_name": model_name,
|
||||
"litellm_params": {
|
||||
"model": f"openai/{_CATALOG_CHAT_MODEL}",
|
||||
"api_key": "sk-fake",
|
||||
"api_base": _MODE_TEST_API_BASE,
|
||||
**custom_pricing,
|
||||
},
|
||||
"model_info": {"id": f"{model_name}-id", **({"mode": mode} if mode is not None else {})},
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("sibling_mode", (None, "chat"))
|
||||
@pytest.mark.parametrize("responses_deployment_first", (True, False))
|
||||
def test_a_responses_deployment_does_not_move_its_siblings_onto_the_responses_api(
|
||||
respx_mock, monkeypatch: pytest.MonkeyPatch, sibling_mode: str | None, responses_deployment_first: bool
|
||||
) -> None:
|
||||
"""
|
||||
https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model share
|
||||
a litellm.model_cost key, so a `mode: responses` deployment used to send every sibling
|
||||
through the Responses API bridge, whichever order they were registered in
|
||||
"""
|
||||
_use_catalog_with_a_chat_model(monkeypatch)
|
||||
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
|
||||
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
|
||||
)
|
||||
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
|
||||
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
|
||||
)
|
||||
sibling: Final = _mode_test_deployment("my-chat-model", sibling_mode)
|
||||
responses_deployment: Final = _mode_test_deployment("my-responses-probe", "responses")
|
||||
router: Final = Router(
|
||||
model_list=[responses_deployment, sibling] if responses_deployment_first else [sibling, responses_deployment]
|
||||
)
|
||||
messages: Final = [{"role": "user", "content": "hi"}]
|
||||
|
||||
router.completion(model="my-chat-model", messages=messages)
|
||||
assert (chat_route.call_count, responses_route.call_count) == (1, 0)
|
||||
|
||||
router.completion(model="my-responses-probe", messages=messages)
|
||||
assert (chat_route.call_count, responses_route.call_count) == (1, 1)
|
||||
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_a_request(
|
||||
respx_mock, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""
|
||||
A deployment with custom pricing re-registers its model_info on every request it serves,
|
||||
so the shared key has to stay out of its mode on that path too, not just at router setup
|
||||
"""
|
||||
_use_catalog_with_a_chat_model(monkeypatch)
|
||||
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
|
||||
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
|
||||
)
|
||||
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
|
||||
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
|
||||
)
|
||||
router: Final = Router(
|
||||
model_list=[
|
||||
_mode_test_deployment("my-chat-model", None),
|
||||
_mode_test_deployment(
|
||||
"my-responses-probe",
|
||||
"responses",
|
||||
custom_pricing=MappingProxyType({"input_cost_per_token": 1e-6, "output_cost_per_token": 2e-6}),
|
||||
),
|
||||
]
|
||||
)
|
||||
messages: Final = [{"role": "user", "content": "hi"}]
|
||||
|
||||
router.completion(model="my-responses-probe", messages=messages)
|
||||
router.completion(model="my-chat-model", messages=messages)
|
||||
|
||||
assert (chat_route.call_count, responses_route.call_count) == (1, 1)
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"caller_model_info",
|
||||
({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}),
|
||||
)
|
||||
def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata(
|
||||
respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str]
|
||||
) -> None:
|
||||
"""
|
||||
The Responses bridge reads the mode the router registered for the routed deployment, never a
|
||||
mode written into the request's own metadata
|
||||
"""
|
||||
_use_catalog_with_a_chat_model(monkeypatch)
|
||||
chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock(
|
||||
return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY)
|
||||
)
|
||||
responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock(
|
||||
return_value=httpx.Response(200, json=_RESPONSES_REPLY)
|
||||
)
|
||||
router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)])
|
||||
|
||||
router.completion(
|
||||
model="my-chat-model",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
litellm_metadata={"model_info": dict(caller_model_info)},
|
||||
)
|
||||
|
||||
assert (chat_route.call_count, responses_route.call_count) == (1, 0)
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""
|
||||
For a provider model the catalog doesn't know, proxy model_info is how its mode gets
|
||||
onboarded (see mantle_supports_responses), so a deployment's mode must still fill a
|
||||
shared key that has none; only an existing mode is left alone
|
||||
"""
|
||||
monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost))
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
backend_model: Final = "openai/unmapped-38543-model"
|
||||
assert backend_model not in litellm.model_cost
|
||||
|
||||
Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "unmapped-responses",
|
||||
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
|
||||
"model_info": {"id": "unmapped-responses-id", "mode": "responses"},
|
||||
},
|
||||
{
|
||||
"model_name": "unmapped-chat",
|
||||
"litellm_params": {"model": backend_model, "api_key": "sk-fake"},
|
||||
"model_info": {"id": "unmapped-chat-id", "mode": "chat"},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
assert litellm.model_cost[backend_model]["mode"] == "responses"
|
||||
assert litellm.model_cost["unmapped-chat-id"]["mode"] == "chat"
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_partial_custom_pricing_inherits_builtin_cache_pricing():
|
||||
"""A deployment that overrides only input/output cost on a cache-supporting
|
||||
model must still bill cache_read and cache_creation tokens. Before the
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue