diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 4114bda47c9..312fca2305a 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -6062,6 +6062,11 @@ class ProxyConfig: # Repopulate provider model sets (e.g. litellm.anthropic_models) so that # wildcard patterns like "anthropic/*" include any newly added models. litellm.add_known_models(model_cost_map=new_model_cost_map) + # Replacing model_cost wholesale drops per-deployment overrides + # (custom pricing, mode, ...); replay them so e.g. a configured + # mode=responses isn't reverted to the built-in mode=chat. + if llm_router is not None: + llm_router.re_register_deployments_in_model_cost() # Update pod's in-memory last reload time last_model_cost_map_reload = current_time.isoformat() @@ -15131,6 +15136,11 @@ async def reload_model_cost_map( # Repopulate provider model sets (e.g. litellm.anthropic_models) so that # wildcard patterns like "anthropic/*" include any newly added models. litellm.add_known_models(model_cost_map=new_model_cost_map) + # Replacing model_cost wholesale drops per-deployment overrides + # (custom pricing, mode, ...); replay them so e.g. a configured + # mode=responses isn't reverted to the built-in mode=chat. + if llm_router is not None: + llm_router.re_register_deployments_in_model_cost() # Update pod's in-memory last reload time global last_model_cost_map_reload diff --git a/litellm/router.py b/litellm/router.py index 5ffe60c2da0..ceff1bad446 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -7340,6 +7340,100 @@ class Router: if backend_value is not None: model_info[field] = backend_value + def _register_deployment_in_model_cost(self, deployment: Deployment, model_info: dict) -> None: + """Register a deployment's model_info into ``litellm.model_cost``. + + Registers under the deployment's unique ``model_id`` (full pricing) and + under the shared provider/model backend key (custom pricing stripped so + one alias can't pollute another). The shared registration preserves an + existing ``mode == "responses"`` backend so last-write-wins alias + registration can't silently downgrade it to ``chat``. + + Written to be idempotent so it can be replayed to restore deployment + overrides after ``litellm.model_cost`` is replaced wholesale (e.g. a + remote cost-map reload). + """ + for field in CustomPricingLiteLLMParams.model_fields.keys(): + if deployment.litellm_params.get(field) is not None: + model_info[field] = deployment.litellm_params[field] + + if model_info.get("input_cost_per_token") is not None: + Router._inherit_builtin_cache_pricing( + model_info=model_info, + backend_model=deployment.litellm_params.model, + custom_llm_provider=deployment.litellm_params.custom_llm_provider, + ) + + ## REGISTER MODEL INFO IN LITELLM MODEL COST MAP + model_id = deployment.model_info.id + if model_id is not None: + litellm.register_model(model_cost={model_id: model_info}) + + ## OLD MODEL REGISTRATION ## Kept to prevent breaking changes + backend_model_name = deployment.litellm_params.model + if deployment.litellm_params.custom_llm_provider is not None: + backend_model_name = deployment.litellm_params.custom_llm_provider + "/" + backend_model_name + + # For the shared backend key, strip custom pricing fields so that + # one deployment's pricing overrides don't pollute another + # deployment sharing the same backend model name. + # Each deployment's full pricing is already stored under its + # unique model_id above. + _shared_model_info = CustomPricingLiteLLMParams.strip_custom_pricing_fields(model_info) + _existing_shared_mode = (cast(Optional[dict], litellm.model_cost.get(backend_model_name, {})) or {}).get("mode") + _deployment_mode = _shared_model_info.get("mode") + # Keep the built-in bridge mode stable for shared backend keys. + # Multiple aliases can point at the same provider/model backend, + # but their deployment-level overrides should not downgrade the + # backend from responses -> chat via last-write-wins registration. + # Only preserve in that specific direction so legitimate upgrades + # (e.g. chat -> responses) and unrelated mode changes still apply, + # and so a missing deployment mode does not silently clear the + # existing shared backend mode. + _is_responses_to_chat_downgrade = _existing_shared_mode == "responses" and _deployment_mode == "chat" + _would_clear_existing_mode = _existing_shared_mode is not None and _deployment_mode is None + if _is_responses_to_chat_downgrade or _would_clear_existing_mode: + if _deployment_mode is not None: + verbose_router_logger.warning( + "Router: preserving existing mode=%s for shared backend " + "key %s instead of the deployment-specified mode=%s " + "(prevents alias registration from downgrading the " + "shared backend mode).", + _existing_shared_mode, + backend_model_name, + _deployment_mode, + ) + _shared_model_info["mode"] = _existing_shared_mode + + # Always register the (possibly mode-preserved) shared backend info. + _backend_alias_cost = {backend_model_name: _shared_model_info} + if "responses/" in backend_model_name: + _backend_alias_cost[backend_model_name.replace("responses/", "")] = _shared_model_info + litellm.register_model(model_cost=_backend_alias_cost) + + def re_register_deployments_in_model_cost(self) -> None: + """Replay per-deployment ``model_cost`` registration for all deployments. + + ``litellm.model_cost`` is a process-global dict that the proxy replaces + wholesale when it reloads the remote cost map. That drops the overrides + (custom pricing, ``mode``, ``supported_endpoints``, ...) previously + layered on top from the model_list, which e.g. reverts a configured + ``mode: responses`` back to the built-in ``mode: chat`` and breaks the + chat -> responses bridge on the next request. Replaying registration + restores the deployment overrides on top of the fresh cost map. + """ + for deployment_dict in self.model_list: + litellm_params = deployment_dict.get("litellm_params") or {} + deployment = Deployment( + model_name=deployment_dict["model_name"], + litellm_params=LiteLLM_Params(**litellm_params), + model_info=dict(deployment_dict.get("model_info") or {}), + ) + self._register_deployment_in_model_cost( + deployment=deployment, + model_info=deployment.model_info.model_dump(exclude_none=True), + ) + def _create_deployment( self, deployment_info: dict, @@ -7364,68 +7458,7 @@ class Router: litellm_params=litellm_params, model_info=_model_info, ) - for field in CustomPricingLiteLLMParams.model_fields.keys(): - if deployment.litellm_params.get(field) is not None: - _model_info[field] = deployment.litellm_params[field] - - if _model_info.get("input_cost_per_token") is not None: - Router._inherit_builtin_cache_pricing( - model_info=_model_info, - backend_model=deployment.litellm_params.model, - custom_llm_provider=deployment.litellm_params.custom_llm_provider, - ) - - ## REGISTER MODEL INFO IN LITELLM MODEL COST MAP - model_id = deployment.model_info.id - if model_id is not None: - litellm.register_model( - model_cost={ - model_id: _model_info, - } - ) - - ## OLD MODEL REGISTRATION ## Kept to prevent breaking changes - _model_name = deployment.litellm_params.model - if deployment.litellm_params.custom_llm_provider is not None: - _model_name = deployment.litellm_params.custom_llm_provider + "/" + _model_name - - # For the shared backend key, strip custom pricing fields so that - # one deployment's pricing overrides don't pollute another - # deployment sharing the same backend model name. - # Each deployment's full pricing is already stored under its - # unique model_id above. - _shared_model_info = CustomPricingLiteLLMParams.strip_custom_pricing_fields(_model_info) - _existing_shared_mode = (cast(Optional[dict], litellm.model_cost.get(_model_name, {})) or {}).get("mode") - _deployment_mode = _shared_model_info.get("mode") - # Keep the built-in bridge mode stable for shared backend keys. - # Multiple aliases can point at the same provider/model backend, - # but their deployment-level overrides should not downgrade the - # backend from responses -> chat via last-write-wins registration. - # Only preserve in that specific direction so legitimate upgrades - # (e.g. chat -> responses) and unrelated mode changes still apply, - # and so a missing deployment mode does not silently clear the - # existing shared backend mode. - _is_responses_to_chat_downgrade = _existing_shared_mode == "responses" and _deployment_mode == "chat" - _would_clear_existing_mode = _existing_shared_mode is not None and _deployment_mode is None - if _is_responses_to_chat_downgrade or _would_clear_existing_mode: - if _deployment_mode is not None: - verbose_router_logger.warning( - "Router: preserving existing mode=%s for shared backend " - "key %s instead of the deployment-specified mode=%s " - "(prevents alias registration from downgrading the " - "shared backend mode).", - _existing_shared_mode, - _model_name, - _deployment_mode, - ) - _shared_model_info["mode"] = _existing_shared_mode - - # Always register the (possibly mode-preserved) shared backend info. - _backend_alias_cost = {_model_name: _shared_model_info} - if "responses/" in _model_name: - _stripped_model_name = _model_name.replace("responses/", "") - _backend_alias_cost[_stripped_model_name] = _shared_model_info - litellm.register_model(model_cost=_backend_alias_cost) + self._register_deployment_in_model_cost(deployment=deployment, model_info=_model_info) ## Check if LLM Deployment is allowed for this deployment if self.deployment_is_active_for_environment(deployment=deployment) is not True: diff --git a/tests/test_litellm/test_router_model_cost_isolation.py b/tests/test_litellm/test_router_model_cost_isolation.py index 6db7b04b3b7..2a0713af1ab 100644 --- a/tests/test_litellm/test_router_model_cost_isolation.py +++ b/tests/test_litellm/test_router_model_cost_isolation.py @@ -404,6 +404,60 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override(): _restore_model_cost_entries(model_keys) +def test_deployment_mode_persists_across_model_cost_map_reload(): + """Regression for #32736. + + The proxy periodically replaces ``litellm.model_cost`` wholesale with a + fresh remote cost map. That drops the per-deployment overrides the router + layered on top, so a deployment configured with ``mode: responses`` would + silently revert to the built-in ``mode: chat`` and break the chat -> + responses bridge on the next request. ``re_register_deployments_in_model_cost`` + replays registration so the override survives the reload. + """ + from litellm.main import responses_api_bridge_check + + backend_key = "gpt-5.6" + original = dict(litellm.model_cost) + try: + router = Router( + model_list=[ + { + "model_name": "gpt-5.6", + "litellm_params": { + "model": "gpt-5.6", + "custom_llm_provider": "openai", + "api_key": "fake-key-reload", + }, + "model_info": { + "id": "gpt-56-responses-reload", + "mode": "responses", + }, + } + ], + ) + assert litellm.model_cost[backend_key]["mode"] == "responses" + + # Simulate the proxy reload: swap in a fresh map that only knows the + # built-in mode=chat and carries none of the router's registrations. + litellm.model_cost = {"gpt-5.6": {"litellm_provider": "openai", "mode": "chat"}} + _invalidate_model_cost_lowercase_map() + assert litellm.model_cost[backend_key]["mode"] == "chat" + + router.re_register_deployments_in_model_cost() + + assert litellm.model_cost[backend_key]["mode"] == "responses" + + bridge_model_info, bridge_model = responses_api_bridge_check( + model="gpt-5.6", + custom_llm_provider="openai", + ) + assert bridge_model == "gpt-5.6" + assert bridge_model_info["mode"] == "responses" + finally: + litellm.model_cost = original + _invalidate_model_cost_lowercase_map() + + def test_partial_custom_pricing_inherits_builtin_cache_pricing(): """A deployment that overrides only input/output cost on a cache-supporting model must still bill cache_read and cache_creation tokens. Before the