From f93b8c1e0b0069314233ceaa14de3850813bcd0c Mon Sep 17 00:00:00 2001 From: perseus <51974392+tcconnally@users.noreply.github.com> Date: Wed, 17 Jun 2026 22:06:15 -0500 Subject: [PATCH] feat(proxy): add budget_exceeded_models_policy for model discovery behavior Adds a general_settings.budget_exceeded_models_policy option with three values: - 'blocked' (default): current behavior, backward compatible - 'all': full model list always returned (already handled by MODEL_DISCOVERY_ROUTES exemption) - 'free_only': returns only zero-cost models when budget would be exceeded The 'free_only' policy filters the model list using _is_model_cost_zero() in model_list(). Related to #27923 --- litellm/proxy/_types.py | 11 +++++++++++ litellm/proxy/proxy_server.py | 7 +++++++ 2 files changed, 18 insertions(+) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 5bba842c7eb..0929fe1c398 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2396,6 +2396,17 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): ), ) + budget_exceeded_models_policy: Optional[Literal["blocked", "all", "free_only"]] = Field( + "blocked", + description=( + "Controls model-list behavior when a team/user/key budget is exceeded. " + "'blocked' (default) returns 429 as before. " + "'all' returns the full model list regardless of budget. " + "'free_only' returns only zero-cost models. " + "Inference calls always enforce the budget regardless of this setting." + ), + ) + class ConfigYAML(LiteLLMPydanticObjectBase): """ diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index e4d933ac699..42718171a82 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -8832,6 +8832,13 @@ async def model_list( if hidden_names: all_models = [m for m in all_models if m not in hidden_names] + # Budget-exceeded models policy: when "free_only", return only zero-cost models + policy = settings.get("budget_exceeded_models_policy", "blocked") + if policy == "free_only": + from litellm.proxy.auth.auth_checks import _is_model_cost_zero + + all_models = [m for m in all_models if _is_model_cost_zero(m, llm_router)] + # Surface the public team name by default; legacy internal keys via flag. # The internal routing key drives the metadata/fallback lookup, while the # public name is what the client sees as the model id.