mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(router): persist deployment mode/overrides across model cost map reload
This commit is contained in:
parent
2a12707372
commit
d822457cfd
3 changed files with 159 additions and 62 deletions
|
|
@ -6062,6 +6062,11 @@ class ProxyConfig:
|
|||
# Repopulate provider model sets (e.g. litellm.anthropic_models) so that
|
||||
# wildcard patterns like "anthropic/*" include any newly added models.
|
||||
litellm.add_known_models(model_cost_map=new_model_cost_map)
|
||||
# Replacing model_cost wholesale drops per-deployment overrides
|
||||
# (custom pricing, mode, ...); replay them so e.g. a configured
|
||||
# mode=responses isn't reverted to the built-in mode=chat.
|
||||
if llm_router is not None:
|
||||
llm_router.re_register_deployments_in_model_cost()
|
||||
|
||||
# Update pod's in-memory last reload time
|
||||
last_model_cost_map_reload = current_time.isoformat()
|
||||
|
|
@ -15131,6 +15136,11 @@ async def reload_model_cost_map(
|
|||
# Repopulate provider model sets (e.g. litellm.anthropic_models) so that
|
||||
# wildcard patterns like "anthropic/*" include any newly added models.
|
||||
litellm.add_known_models(model_cost_map=new_model_cost_map)
|
||||
# Replacing model_cost wholesale drops per-deployment overrides
|
||||
# (custom pricing, mode, ...); replay them so e.g. a configured
|
||||
# mode=responses isn't reverted to the built-in mode=chat.
|
||||
if llm_router is not None:
|
||||
llm_router.re_register_deployments_in_model_cost()
|
||||
|
||||
# Update pod's in-memory last reload time
|
||||
global last_model_cost_map_reload
|
||||
|
|
|
|||
|
|
@ -7340,6 +7340,100 @@ class Router:
|
|||
if backend_value is not None:
|
||||
model_info[field] = backend_value
|
||||
|
||||
def _register_deployment_in_model_cost(self, deployment: Deployment, model_info: dict) -> None:
|
||||
"""Register a deployment's model_info into ``litellm.model_cost``.
|
||||
|
||||
Registers under the deployment's unique ``model_id`` (full pricing) and
|
||||
under the shared provider/model backend key (custom pricing stripped so
|
||||
one alias can't pollute another). The shared registration preserves an
|
||||
existing ``mode == "responses"`` backend so last-write-wins alias
|
||||
registration can't silently downgrade it to ``chat``.
|
||||
|
||||
Written to be idempotent so it can be replayed to restore deployment
|
||||
overrides after ``litellm.model_cost`` is replaced wholesale (e.g. a
|
||||
remote cost-map reload).
|
||||
"""
|
||||
for field in CustomPricingLiteLLMParams.model_fields.keys():
|
||||
if deployment.litellm_params.get(field) is not None:
|
||||
model_info[field] = deployment.litellm_params[field]
|
||||
|
||||
if model_info.get("input_cost_per_token") is not None:
|
||||
Router._inherit_builtin_cache_pricing(
|
||||
model_info=model_info,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
|
||||
## REGISTER MODEL INFO IN LITELLM MODEL COST MAP
|
||||
model_id = deployment.model_info.id
|
||||
if model_id is not None:
|
||||
litellm.register_model(model_cost={model_id: model_info})
|
||||
|
||||
## OLD MODEL REGISTRATION ## Kept to prevent breaking changes
|
||||
backend_model_name = deployment.litellm_params.model
|
||||
if deployment.litellm_params.custom_llm_provider is not None:
|
||||
backend_model_name = deployment.litellm_params.custom_llm_provider + "/" + backend_model_name
|
||||
|
||||
# For the shared backend key, strip custom pricing fields so that
|
||||
# one deployment's pricing overrides don't pollute another
|
||||
# deployment sharing the same backend model name.
|
||||
# Each deployment's full pricing is already stored under its
|
||||
# unique model_id above.
|
||||
_shared_model_info = CustomPricingLiteLLMParams.strip_custom_pricing_fields(model_info)
|
||||
_existing_shared_mode = (cast(Optional[dict], litellm.model_cost.get(backend_model_name, {})) or {}).get("mode")
|
||||
_deployment_mode = _shared_model_info.get("mode")
|
||||
# Keep the built-in bridge mode stable for shared backend keys.
|
||||
# Multiple aliases can point at the same provider/model backend,
|
||||
# but their deployment-level overrides should not downgrade the
|
||||
# backend from responses -> chat via last-write-wins registration.
|
||||
# Only preserve in that specific direction so legitimate upgrades
|
||||
# (e.g. chat -> responses) and unrelated mode changes still apply,
|
||||
# and so a missing deployment mode does not silently clear the
|
||||
# existing shared backend mode.
|
||||
_is_responses_to_chat_downgrade = _existing_shared_mode == "responses" and _deployment_mode == "chat"
|
||||
_would_clear_existing_mode = _existing_shared_mode is not None and _deployment_mode is None
|
||||
if _is_responses_to_chat_downgrade or _would_clear_existing_mode:
|
||||
if _deployment_mode is not None:
|
||||
verbose_router_logger.warning(
|
||||
"Router: preserving existing mode=%s for shared backend "
|
||||
"key %s instead of the deployment-specified mode=%s "
|
||||
"(prevents alias registration from downgrading the "
|
||||
"shared backend mode).",
|
||||
_existing_shared_mode,
|
||||
backend_model_name,
|
||||
_deployment_mode,
|
||||
)
|
||||
_shared_model_info["mode"] = _existing_shared_mode
|
||||
|
||||
# Always register the (possibly mode-preserved) shared backend info.
|
||||
_backend_alias_cost = {backend_model_name: _shared_model_info}
|
||||
if "responses/" in backend_model_name:
|
||||
_backend_alias_cost[backend_model_name.replace("responses/", "")] = _shared_model_info
|
||||
litellm.register_model(model_cost=_backend_alias_cost)
|
||||
|
||||
def re_register_deployments_in_model_cost(self) -> None:
|
||||
"""Replay per-deployment ``model_cost`` registration for all deployments.
|
||||
|
||||
``litellm.model_cost`` is a process-global dict that the proxy replaces
|
||||
wholesale when it reloads the remote cost map. That drops the overrides
|
||||
(custom pricing, ``mode``, ``supported_endpoints``, ...) previously
|
||||
layered on top from the model_list, which e.g. reverts a configured
|
||||
``mode: responses`` back to the built-in ``mode: chat`` and breaks the
|
||||
chat -> responses bridge on the next request. Replaying registration
|
||||
restores the deployment overrides on top of the fresh cost map.
|
||||
"""
|
||||
for deployment_dict in self.model_list:
|
||||
litellm_params = deployment_dict.get("litellm_params") or {}
|
||||
deployment = Deployment(
|
||||
model_name=deployment_dict["model_name"],
|
||||
litellm_params=LiteLLM_Params(**litellm_params),
|
||||
model_info=dict(deployment_dict.get("model_info") or {}),
|
||||
)
|
||||
self._register_deployment_in_model_cost(
|
||||
deployment=deployment,
|
||||
model_info=deployment.model_info.model_dump(exclude_none=True),
|
||||
)
|
||||
|
||||
def _create_deployment(
|
||||
self,
|
||||
deployment_info: dict,
|
||||
|
|
@ -7364,68 +7458,7 @@ class Router:
|
|||
litellm_params=litellm_params,
|
||||
model_info=_model_info,
|
||||
)
|
||||
for field in CustomPricingLiteLLMParams.model_fields.keys():
|
||||
if deployment.litellm_params.get(field) is not None:
|
||||
_model_info[field] = deployment.litellm_params[field]
|
||||
|
||||
if _model_info.get("input_cost_per_token") is not None:
|
||||
Router._inherit_builtin_cache_pricing(
|
||||
model_info=_model_info,
|
||||
backend_model=deployment.litellm_params.model,
|
||||
custom_llm_provider=deployment.litellm_params.custom_llm_provider,
|
||||
)
|
||||
|
||||
## REGISTER MODEL INFO IN LITELLM MODEL COST MAP
|
||||
model_id = deployment.model_info.id
|
||||
if model_id is not None:
|
||||
litellm.register_model(
|
||||
model_cost={
|
||||
model_id: _model_info,
|
||||
}
|
||||
)
|
||||
|
||||
## OLD MODEL REGISTRATION ## Kept to prevent breaking changes
|
||||
_model_name = deployment.litellm_params.model
|
||||
if deployment.litellm_params.custom_llm_provider is not None:
|
||||
_model_name = deployment.litellm_params.custom_llm_provider + "/" + _model_name
|
||||
|
||||
# For the shared backend key, strip custom pricing fields so that
|
||||
# one deployment's pricing overrides don't pollute another
|
||||
# deployment sharing the same backend model name.
|
||||
# Each deployment's full pricing is already stored under its
|
||||
# unique model_id above.
|
||||
_shared_model_info = CustomPricingLiteLLMParams.strip_custom_pricing_fields(_model_info)
|
||||
_existing_shared_mode = (cast(Optional[dict], litellm.model_cost.get(_model_name, {})) or {}).get("mode")
|
||||
_deployment_mode = _shared_model_info.get("mode")
|
||||
# Keep the built-in bridge mode stable for shared backend keys.
|
||||
# Multiple aliases can point at the same provider/model backend,
|
||||
# but their deployment-level overrides should not downgrade the
|
||||
# backend from responses -> chat via last-write-wins registration.
|
||||
# Only preserve in that specific direction so legitimate upgrades
|
||||
# (e.g. chat -> responses) and unrelated mode changes still apply,
|
||||
# and so a missing deployment mode does not silently clear the
|
||||
# existing shared backend mode.
|
||||
_is_responses_to_chat_downgrade = _existing_shared_mode == "responses" and _deployment_mode == "chat"
|
||||
_would_clear_existing_mode = _existing_shared_mode is not None and _deployment_mode is None
|
||||
if _is_responses_to_chat_downgrade or _would_clear_existing_mode:
|
||||
if _deployment_mode is not None:
|
||||
verbose_router_logger.warning(
|
||||
"Router: preserving existing mode=%s for shared backend "
|
||||
"key %s instead of the deployment-specified mode=%s "
|
||||
"(prevents alias registration from downgrading the "
|
||||
"shared backend mode).",
|
||||
_existing_shared_mode,
|
||||
_model_name,
|
||||
_deployment_mode,
|
||||
)
|
||||
_shared_model_info["mode"] = _existing_shared_mode
|
||||
|
||||
# Always register the (possibly mode-preserved) shared backend info.
|
||||
_backend_alias_cost = {_model_name: _shared_model_info}
|
||||
if "responses/" in _model_name:
|
||||
_stripped_model_name = _model_name.replace("responses/", "")
|
||||
_backend_alias_cost[_stripped_model_name] = _shared_model_info
|
||||
litellm.register_model(model_cost=_backend_alias_cost)
|
||||
self._register_deployment_in_model_cost(deployment=deployment, model_info=_model_info)
|
||||
|
||||
## Check if LLM Deployment is allowed for this deployment
|
||||
if self.deployment_is_active_for_environment(deployment=deployment) is not True:
|
||||
|
|
|
|||
|
|
@ -404,6 +404,60 @@ def test_should_not_downgrade_chatgpt_shared_key_mode_with_alias_override():
|
|||
_restore_model_cost_entries(model_keys)
|
||||
|
||||
|
||||
def test_deployment_mode_persists_across_model_cost_map_reload():
|
||||
"""Regression for #32736.
|
||||
|
||||
The proxy periodically replaces ``litellm.model_cost`` wholesale with a
|
||||
fresh remote cost map. That drops the per-deployment overrides the router
|
||||
layered on top, so a deployment configured with ``mode: responses`` would
|
||||
silently revert to the built-in ``mode: chat`` and break the chat ->
|
||||
responses bridge on the next request. ``re_register_deployments_in_model_cost``
|
||||
replays registration so the override survives the reload.
|
||||
"""
|
||||
from litellm.main import responses_api_bridge_check
|
||||
|
||||
backend_key = "gpt-5.6"
|
||||
original = dict(litellm.model_cost)
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-5.6",
|
||||
"litellm_params": {
|
||||
"model": "gpt-5.6",
|
||||
"custom_llm_provider": "openai",
|
||||
"api_key": "fake-key-reload",
|
||||
},
|
||||
"model_info": {
|
||||
"id": "gpt-56-responses-reload",
|
||||
"mode": "responses",
|
||||
},
|
||||
}
|
||||
],
|
||||
)
|
||||
assert litellm.model_cost[backend_key]["mode"] == "responses"
|
||||
|
||||
# Simulate the proxy reload: swap in a fresh map that only knows the
|
||||
# built-in mode=chat and carries none of the router's registrations.
|
||||
litellm.model_cost = {"gpt-5.6": {"litellm_provider": "openai", "mode": "chat"}}
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
assert litellm.model_cost[backend_key]["mode"] == "chat"
|
||||
|
||||
router.re_register_deployments_in_model_cost()
|
||||
|
||||
assert litellm.model_cost[backend_key]["mode"] == "responses"
|
||||
|
||||
bridge_model_info, bridge_model = responses_api_bridge_check(
|
||||
model="gpt-5.6",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
assert bridge_model == "gpt-5.6"
|
||||
assert bridge_model_info["mode"] == "responses"
|
||||
finally:
|
||||
litellm.model_cost = original
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
|
||||
|
||||
def test_partial_custom_pricing_inherits_builtin_cache_pricing():
|
||||
"""A deployment that overrides only input/output cost on a cache-supporting
|
||||
model must still bill cache_read and cache_creation tokens. Before the
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue