diff --git a/litellm/llms/anthropic/pass_through/adapters/handler.py b/litellm/llms/anthropic/pass_through/adapters/handler.py index 5bda3437a40..7ce62f036e5 100644 --- a/litellm/llms/anthropic/pass_through/adapters/handler.py +++ b/litellm/llms/anthropic/pass_through/adapters/handler.py @@ -411,7 +411,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: return None if resolved_provider == "litellm_proxy": return None - from litellm.main import responses_api_bridge_check + from litellm.main import responses_api_bridge_check, router_deployment_mode web_search_options: Final = completion_kwargs.get("web_search_options") tools: Final = completion_kwargs.get("tools") @@ -424,6 +424,7 @@ class LiteLLMMessagesToCompletionTransformationHandler: tools=cast("list[dict[str, object]]", tools) if isinstance(tools, list) else None, reasoning_effort=reasoning_effort, api_base=resolved_api_base, + deployment_mode=router_deployment_mode(completion_kwargs), ) return None if model_info.get("mode") == "responses" else effort diff --git a/litellm/main.py b/litellm/main.py index 769eac79488..2be30cd498e 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -19,7 +19,7 @@ import random import sys import time import traceback -from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Mapping, Sequence +from collections.abc import AsyncIterator, Callable, Coroutine, Iterable, Iterator, Mapping, Sequence from concurrent import futures from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from copy import deepcopy @@ -168,6 +168,7 @@ from litellm.utils import ( get_secret, get_standard_openai_params, mock_completion_streaming_obj, + mode_register_model_would_merge_into, pre_process_non_default_params, read_config_args, should_run_mock_completion, @@ -1084,6 +1085,7 @@ def responses_api_bridge_check( reasoning_effort: str | Mapping[str, object] | None = None, reasoning_summary: object | None = None, api_base: str | None = None, + deployment_mode: str | None = None, ) -> tuple[dict, str]: model_info: dict[str, object] = {} @@ -1112,7 +1114,7 @@ def responses_api_bridge_check( mode = "responses" model_info["mode"] = mode - if web_search_options is not None and custom_llm_provider == "xai": + if deployment_mode == "responses" or (web_search_options is not None and custom_llm_provider == "xai"): model_info["mode"] = "responses" model = model.replace("responses/", "") @@ -1263,20 +1265,36 @@ def _build_custom_pricing_entry( return entry -def _get_router_deployment_id(kwargs: dict) -> str | None: +def _router_deployment_model_infos(kwargs: Mapping[str, object]) -> Iterator[Mapping[str, object]]: for metadata_key in ("litellm_metadata", "metadata"): - metadata = kwargs.get(metadata_key) or {} + metadata = kwargs.get(metadata_key) if not isinstance(metadata, dict): continue - deployment_model_info = metadata.get("model_info") or {} - if not isinstance(deployment_model_info, dict): - continue + deployment_model_info = metadata.get("model_info") + if isinstance(deployment_model_info, dict): + yield deployment_model_info + + +def _get_router_deployment_id(kwargs: Mapping[str, object]) -> str | None: + for deployment_model_info in _router_deployment_model_infos(kwargs): deployment_id = deployment_model_info.get("id") if deployment_id is not None: return str(deployment_id) return None +def router_deployment_mode(kwargs: Mapping[str, object]) -> str | None: + """The ``mode`` the router registered for the deployment it picked for this request, if any. + + Only the deployment id is taken from request metadata; the mode itself is read from the + router's own registration, so a caller can't hand-write one into their metadata. + """ + deployment_id: Final = _get_router_deployment_id(kwargs) + registered: Final = litellm.model_cost.get(deployment_id) if deployment_id is not None else None + mode: Final = registered.get("mode") if isinstance(registered, dict) else None + return mode if isinstance(mode, str) else None + + def _register_custom_pricing_for_request( model: str, custom_llm_provider: str, @@ -1303,10 +1321,15 @@ def _register_custom_pricing_for_request( if deployment_id is None: litellm.register_model({shared_key: entry}, persist_across_reloads=False) return + shared_entry: Final = CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry) litellm.register_model( { deployment_id: entry, - shared_key: CustomPricingLiteLLMParams.strip_custom_pricing_fields(entry), + shared_key: ( + {k: v for k, v in shared_entry.items() if k != "mode"} + if mode_register_model_would_merge_into(shared_key, custom_llm_provider) is not None + else shared_entry + ), }, persist_across_reloads=False, warning_display_name=shared_key, @@ -5480,11 +5503,13 @@ def completion( ) ## RESPONSES API BRIDGE LOGIC ## - check early and normalize model name + _deployment_mode: Final = router_deployment_mode(kwargs) responses_api_model_info, model = responses_api_bridge_check( model=model, custom_llm_provider=custom_llm_provider, web_search_options=web_search_options, api_base=api_base, + deployment_mode=_deployment_mode, ) if not _is_claude_tool_target(custom_llm_provider=custom_llm_provider, model=model): @@ -5768,6 +5793,7 @@ def completion( reasoning_effort=reasoning_effort, reasoning_summary=_reasoning_summary_for_bridge, api_base=api_base, + deployment_mode=_deployment_mode, ) # Use base_model (the true underlying model) for Azure model-type diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index bd239922fd3..3c828baa5d9 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -332,6 +332,7 @@ class LiteLLMCompletionResponsesConfig: reasoning_effort: str | None, reasoning_summary: str | None, api_base: str | None, + deployment_mode: str | None, ) -> bool: """ Whether ``litellm.completion`` will route this model back onto the Responses API. @@ -350,6 +351,7 @@ class LiteLLMCompletionResponsesConfig: reasoning_effort=reasoning_effort, reasoning_summary=reasoning_summary, api_base=api_base, + deployment_mode=deployment_mode, ) except Exception as e: # noqa: BLE001 # a capability probe must never fail the request it probes for verbose_logger.debug("responses bridge: reasoning effort mode check failed: %s", e) @@ -364,6 +366,7 @@ class LiteLLMCompletionResponsesConfig: tools: Sequence[ChatCompletionToolParam | OpenAIMcpServerTool] | None = None, web_search_options: OpenAIWebSearchOptions | None = None, api_base: str | None = None, + deployment_mode: str | None = None, ) -> ResponsesReasoningChatForm: """ Split the Responses ``reasoning`` object into the params Chat Completions understands. @@ -393,6 +396,7 @@ class LiteLLMCompletionResponsesConfig: reasoning_effort=effort, reasoning_summary=summary, api_base=api_base, + deployment_mode=deployment_mode, ) return ResponsesReasoningChatForm(effort=effort, summary=summary if bridges_back else None) @@ -409,6 +413,8 @@ class LiteLLMCompletionResponsesConfig: """ Transform a Responses API request into a Chat Completion request """ + from litellm.main import router_deployment_mode + ( tools, web_search_options, @@ -433,6 +439,7 @@ class LiteLLMCompletionResponsesConfig: tools=tools, web_search_options=web_search_options, api_base=kwargs.get("api_base"), + deployment_mode=router_deployment_mode(kwargs), ) litellm_completion_request: dict = { diff --git a/litellm/router.py b/litellm/router.py index 115faad000c..ce66aa3ee0d 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -333,6 +333,7 @@ from litellm.utils import ( get_secret, get_utc_datetime, is_region_allowed, + mode_register_model_would_merge_into, provider_rejectable_params, set_live_deployment_replay, ) @@ -10058,41 +10059,10 @@ class Router: ## OLD MODEL REGISTRATION ## Kept to prevent breaking changes backend_keys: Final = Router._backend_cost_map_keys(model=model, custom_llm_provider=custom_llm_provider) - backend_key: Final = backend_keys[0] - - # For the shared backend key, keep only cost-map schema fields - # (minus custom pricing) so that one deployment's pricing overrides - # or custom metadata (id, access_via_team_ids, arbitrary keys) - # don't pollute another deployment sharing the same backend model - # name. Each deployment's full model_info is already stored under - # its unique model_id above. - shared_model_info: Final = shared_backend_model_info(model_info) - existing_shared_mode: Final = (cast(dict | None, litellm.model_cost.get(backend_key, {})) or {}).get("mode") - deployment_mode: Final = shared_model_info.get("mode") - # Keep the built-in bridge mode stable for shared backend keys. - # Multiple aliases can point at the same provider/model backend, - # but their deployment-level overrides should not downgrade the - # backend from responses -> chat via last-write-wins registration. - # Only preserve in that specific direction so legitimate upgrades - # (e.g. chat -> responses) and unrelated mode changes still apply, - # and so a missing deployment mode does not silently clear the - # existing shared backend mode. - is_responses_to_chat_downgrade: Final = existing_shared_mode == "responses" and deployment_mode == "chat" - would_clear_existing_mode: Final = existing_shared_mode is not None and deployment_mode is None - if is_responses_to_chat_downgrade or would_clear_existing_mode: - if deployment_mode is not None: - verbose_router_logger.warning( - "Router: preserving existing mode=%s for shared backend " - "key %s instead of the deployment-specified mode=%s " - "(prevents alias registration from downgrading the " - "shared backend mode).", - existing_shared_mode, - backend_key, - deployment_mode, - ) - shared_model_info["mode"] = existing_shared_mode - - # Always register the (possibly mode-preserved) shared backend info. + shared_model_info: Final = shared_backend_model_info( + model_info, + existing_mode=mode_register_model_would_merge_into(backend_keys[0], model_info.get("litellm_provider", "")), + ) litellm.register_model( model_cost={_key: shared_model_info for _key in backend_keys}, persist_across_reloads=False, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index c12a4def69a..7bf2d67051f 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -3852,15 +3852,23 @@ SHARED_BACKEND_MODEL_INFO_FIELDS: Final[frozenset[str]] = ( ) -def shared_backend_model_info(model_info: dict[str, Any]) -> dict[str, Any]: +def shared_backend_model_info(model_info: Mapping[str, Any], existing_mode: object = None) -> dict[str, Any]: """Return only the fields safe to register under a shared ``{provider}/{model}`` key in ``litellm.model_cost``: cost-map schema fields (``ModelInfoBase``) minus per-deployment pricing overrides and deployment-scoped pricing blocks such as ``off_peak_pricing``. Per-deployment metadata (``id``, ``access_via_team_ids``, arbitrary custom keys) never belongs on the shared key; it stays under the deployment's unique model id. + + ``mode`` only fills a shared key that has none yet (``existing_mode``). It is the API + surface one deployment picked, so it must not switch the surface its siblings use; + the routed deployment's own ``mode`` reaches the Responses bridge with the request. """ - return {k: v for k, v in model_info.items() if k in SHARED_BACKEND_MODEL_INFO_FIELDS} + return { + k: v + for k, v in model_info.items() + if k in SHARED_BACKEND_MODEL_INFO_FIELDS and (k != "mode" or existing_mode is None) + } ABOVE_THRESHOLD_COST_KEY_PATTERN: Final = re.compile(r"_above_\d+k?_tokens$") diff --git a/litellm/utils.py b/litellm/utils.py index 71186a28be4..ed5b0d29c07 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -3218,6 +3218,23 @@ def _get_builtin_model_info_for_registration(model: str) -> ModelInfo | None: return None if is_generalized_model_info(info) else info +def mode_register_model_would_merge_into(key: str, provider: object) -> object: + """The ``mode`` already on the entry that ``register_model({key: ...})`` merges into. + + Mirrors ``register_model``'s key resolution: a provider-prefixed key such as + ``openai/gpt-5.6`` lands on the bare catalog entry ``gpt-5.6``, so reading + ``litellm.model_cost[key]`` alone would miss the mode that registration overwrites. + """ + skips_model_info_lookup: Final = provider in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO or any( + key.startswith(f"{p}/") for p in PROVIDERS_THAT_AUTHENTICATE_ON_PROVIDER_INFO + ) + builtin_model_info: Final = None if skips_model_info_lookup else _get_builtin_model_info_for_registration(key) + if builtin_model_info is not None: + return builtin_model_info.get("mode") + entry: Final = litellm.model_cost.get(key) + return entry.get("mode") if isinstance(entry, dict) else None + + _runtime_registered_model_cost: Final[dict[str, dict[str, object]]] = {} # mutable-ok: replayed on reload diff --git a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py index 0dfe4a93649..32b048e43a4 100644 --- a/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py +++ b/tests/unit/llms/anthropic/pass_through/adapters/test_handler_reasoning_effort_normalization.py @@ -10,6 +10,7 @@ from typing import Final import pytest +import litellm from litellm.llms.anthropic.pass_through.adapters.handler import ( LiteLLMMessagesToCompletionTransformationHandler, ) @@ -117,6 +118,40 @@ class TestTheSummaryWrappingOnlyRidesTheResponsesBridge: assert sent == {"effort": "high", "summary": "auto"} + @pytest.mark.parametrize( + "deployment_mode, expected", + [("responses", {"effort": "high", "summary": "auto"}), ("chat", "high")], + ) + def test_the_routed_deployments_own_mode_decides_the_shape( + self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch, deployment_mode: str, expected: object + ) -> None: + """https://github.com/BerriAI/litellm/issues/38543 - the routed deployment's own mode, not the + model's shared cost-map entry, decides the API, and this probe has to agree with + ``litellm.completion`` about which API that request lands on""" + monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode}) + completion_kwargs, _ = LiteLLMMessagesToCompletionTransformationHandler._prepare_completion_kwargs( + max_tokens=1024, + messages=MESSAGES, + model="databricks/databricks-qwen35-122b-a10b", + metadata=None, + stop_sequences=None, + stream=False, + system=None, + temperature=None, + thinking=SUMMARIZED_THINKING, + tool_choice=None, + tools=None, + top_k=None, + top_p=None, + output_format=None, + extra_kwargs={ + "custom_llm_provider": "databricks", + "litellm_metadata": {"model_info": {"id": "routed-deployment"}}, + }, + ) + + assert completion_kwargs.get("reasoning_effort") == expected + def test_auto_summary_still_reaches_a_bridged_target( self, local_model_cost_map: None, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py index 7b9de4644b4..3030620f26a 100644 --- a/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py +++ b/tests/unit/responses/litellm_completion_transformation/test_litellm_completion_responses.py @@ -2769,6 +2769,26 @@ class TestToolTransformation: assert result["reasoning_effort"] == "medium" assert result["reasoning_summary"] == "auto" + @pytest.mark.parametrize("deployment_mode, expected_summary", [("responses", "auto"), ("chat", None)]) + def test_summary_follows_the_routed_deployments_own_mode(self, monkeypatch, deployment_mode, expected_summary): + """ + https://github.com/BerriAI/litellm/issues/38543 - a deployment's ``mode`` no longer lands + on the model's shared cost-map entry, so the probe has to read it off the routed + deployment the same way ``litellm.completion`` does, or it drops the summary of a + request that completion then bridges onto the Responses API + """ + monkeypatch.setitem(litellm.model_cost, "routed-deployment", {"mode": deployment_mode}) + result = LiteLLMCompletionResponsesConfig.transform_responses_api_request_to_chat_completion_request( + model="gpt-4.1", + input="hi", + responses_api_request={"reasoning": {"effort": "medium", "summary": "auto"}}, + custom_llm_provider="openai", + litellm_metadata={"model_info": {"id": "routed-deployment"}}, + ) + + assert result["reasoning_effort"] == "medium" + assert result.get("reasoning_summary") == expected_summary + @pytest.mark.parametrize( "model, custom_llm_provider", [ diff --git a/tests/unit/test_router_model_cost_isolation.py b/tests/unit/test_router_model_cost_isolation.py index 86206da16a1..5f31d8c19f5 100644 --- a/tests/unit/test_router_model_cost_isolation.py +++ b/tests/unit/test_router_model_cost_isolation.py @@ -12,6 +12,8 @@ import copy import logging import os import re +from collections.abc import Mapping +from types import MappingProxyType from typing import Final from unittest.mock import Mock, patch @@ -727,6 +729,192 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override(): _restore_model_cost_entries(model_keys) +_MODE_TEST_API_BASE: Final = "http://localhost:38543/v1" + +_CATALOG_CHAT_MODEL: Final = "chat-model-38543" + +_CHAT_COMPLETION_REPLY: Final = { + "id": "chatcmpl-38543", + "object": "chat.completion", + "created": 1, + "model": _CATALOG_CHAT_MODEL, + "choices": [{"index": 0, "finish_reason": "stop", "message": {"role": "assistant", "content": "ok"}}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, +} + +_RESPONSES_REPLY: Final = { + "id": "resp_38543", + "object": "response", + "created_at": 1, + "status": "completed", + "model": _CATALOG_CHAT_MODEL, + "output": [ + { + "type": "message", + "id": "msg_38543", + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": "ok", "annotations": []}], + } + ], + "parallel_tool_calls": True, + "tool_choice": "auto", + "tools": [], + "usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}, +} + + +def _use_catalog_with_a_chat_model(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr( + litellm, + "model_cost", + {**copy.deepcopy(litellm.model_cost), _CATALOG_CHAT_MODEL: {"litellm_provider": "openai", "mode": "chat"}}, + ) + _invalidate_model_cost_lowercase_map() + + +def _mode_test_deployment( + model_name: str, mode: str | None, custom_pricing: Mapping[str, float] = MappingProxyType({}) +) -> dict[str, object]: + return { + "model_name": model_name, + "litellm_params": { + "model": f"openai/{_CATALOG_CHAT_MODEL}", + "api_key": "sk-fake", + "api_base": _MODE_TEST_API_BASE, + **custom_pricing, + }, + "model_info": {"id": f"{model_name}-id", **({"mode": mode} if mode is not None else {})}, + } + + +@pytest.mark.parametrize("sibling_mode", (None, "chat")) +@pytest.mark.parametrize("responses_deployment_first", (True, False)) +def test_a_responses_deployment_does_not_move_its_siblings_onto_the_responses_api( + respx_mock, monkeypatch: pytest.MonkeyPatch, sibling_mode: str | None, responses_deployment_first: bool +) -> None: + """ + https://github.com/BerriAI/litellm/issues/38543 - deployments of one provider model share + a litellm.model_cost key, so a `mode: responses` deployment used to send every sibling + through the Responses API bridge, whichever order they were registered in + """ + _use_catalog_with_a_chat_model(monkeypatch) + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + sibling: Final = _mode_test_deployment("my-chat-model", sibling_mode) + responses_deployment: Final = _mode_test_deployment("my-responses-probe", "responses") + router: Final = Router( + model_list=[responses_deployment, sibling] if responses_deployment_first else [sibling, responses_deployment] + ) + messages: Final = [{"role": "user", "content": "hi"}] + + router.completion(model="my-chat-model", messages=messages) + assert (chat_route.call_count, responses_route.call_count) == (1, 0) + + router.completion(model="my-responses-probe", messages=messages) + assert (chat_route.call_count, responses_route.call_count) == (1, 1) + + _invalidate_model_cost_lowercase_map() + + +def test_a_priced_responses_deployment_does_not_move_its_siblings_after_serving_a_request( + respx_mock, monkeypatch: pytest.MonkeyPatch +) -> None: + """ + A deployment with custom pricing re-registers its model_info on every request it serves, + so the shared key has to stay out of its mode on that path too, not just at router setup + """ + _use_catalog_with_a_chat_model(monkeypatch) + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + router: Final = Router( + model_list=[ + _mode_test_deployment("my-chat-model", None), + _mode_test_deployment( + "my-responses-probe", + "responses", + custom_pricing=MappingProxyType({"input_cost_per_token": 1e-6, "output_cost_per_token": 2e-6}), + ), + ] + ) + messages: Final = [{"role": "user", "content": "hi"}] + + router.completion(model="my-responses-probe", messages=messages) + router.completion(model="my-chat-model", messages=messages) + + assert (chat_route.call_count, responses_route.call_count) == (1, 1) + _invalidate_model_cost_lowercase_map() + + +@pytest.mark.parametrize( + "caller_model_info", + ({"mode": "responses"}, {"id": "not-a-deployment", "mode": "responses"}), +) +def test_a_caller_cannot_move_a_chat_deployment_onto_the_responses_api_through_its_metadata( + respx_mock, monkeypatch: pytest.MonkeyPatch, caller_model_info: Mapping[str, str] +) -> None: + """ + The Responses bridge reads the mode the router registered for the routed deployment, never a + mode written into the request's own metadata + """ + _use_catalog_with_a_chat_model(monkeypatch) + chat_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/chat/completions").mock( + return_value=httpx.Response(200, json=_CHAT_COMPLETION_REPLY) + ) + responses_route: Final = respx_mock.post(f"{_MODE_TEST_API_BASE}/responses").mock( + return_value=httpx.Response(200, json=_RESPONSES_REPLY) + ) + router: Final = Router(model_list=[_mode_test_deployment("my-chat-model", None)]) + + router.completion( + model="my-chat-model", + messages=[{"role": "user", "content": "hi"}], + litellm_metadata={"model_info": dict(caller_model_info)}, + ) + + assert (chat_route.call_count, responses_route.call_count) == (1, 0) + _invalidate_model_cost_lowercase_map() + + +def test_deployment_mode_still_fills_a_shared_key_with_no_mode(monkeypatch: pytest.MonkeyPatch) -> None: + """ + For a provider model the catalog doesn't know, proxy model_info is how its mode gets + onboarded (see mantle_supports_responses), so a deployment's mode must still fill a + shared key that has none; only an existing mode is left alone + """ + monkeypatch.setattr(litellm, "model_cost", copy.deepcopy(litellm.model_cost)) + _invalidate_model_cost_lowercase_map() + backend_model: Final = "openai/unmapped-38543-model" + assert backend_model not in litellm.model_cost + + Router( + model_list=[ + { + "model_name": "unmapped-responses", + "litellm_params": {"model": backend_model, "api_key": "sk-fake"}, + "model_info": {"id": "unmapped-responses-id", "mode": "responses"}, + }, + { + "model_name": "unmapped-chat", + "litellm_params": {"model": backend_model, "api_key": "sk-fake"}, + "model_info": {"id": "unmapped-chat-id", "mode": "chat"}, + }, + ] + ) + + assert litellm.model_cost[backend_model]["mode"] == "responses" + assert litellm.model_cost["unmapped-chat-id"]["mode"] == "chat" + _invalidate_model_cost_lowercase_map() + + def test_partial_custom_pricing_inherits_builtin_cache_pricing(): """A deployment that overrides only input/output cost on a cache-supporting model must still bill cache_read and cache_creation tokens. Before the