feat(proxy): add budget_exceeded_models_policy for model discovery behavior

Adds a general_settings.budget_exceeded_models_policy option with three values:
- 'blocked' (default): current behavior, backward compatible
- 'all': full model list always returned (already handled by MODEL_DISCOVERY_ROUTES exemption)
- 'free_only': returns only zero-cost models when budget would be exceeded

The 'free_only' policy filters the model list using _is_model_cost_zero()
in model_list().

Related to #27923
This commit is contained in:
perseus 2026-06-17 22:06:15 -05:00
parent 3818d6401c
commit f93b8c1e0b
2 changed files with 18 additions and 0 deletions

View file

@ -2396,6 +2396,17 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase):
),
)
budget_exceeded_models_policy: Optional[Literal["blocked", "all", "free_only"]] = Field(
"blocked",
description=(
"Controls model-list behavior when a team/user/key budget is exceeded. "
"'blocked' (default) returns 429 as before. "
"'all' returns the full model list regardless of budget. "
"'free_only' returns only zero-cost models. "
"Inference calls always enforce the budget regardless of this setting."
),
)
class ConfigYAML(LiteLLMPydanticObjectBase):
"""

View file

@ -8832,6 +8832,13 @@ async def model_list(
if hidden_names:
all_models = [m for m in all_models if m not in hidden_names]
# Budget-exceeded models policy: when "free_only", return only zero-cost models
policy = settings.get("budget_exceeded_models_policy", "blocked")
if policy == "free_only":
from litellm.proxy.auth.auth_checks import _is_model_cost_zero
all_models = [m for m in all_models if _is_model_cost_zero(m, llm_router)]
# Surface the public team name by default; legacy internal keys via flag.
# The internal routing key drives the metadata/fallback lookup, while the
# public name is what the client sees as the model id.