diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 5bba842c7eb..0929fe1c398 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2396,6 +2396,17 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): ), ) + budget_exceeded_models_policy: Optional[Literal["blocked", "all", "free_only"]] = Field( + "blocked", + description=( + "Controls model-list behavior when a team/user/key budget is exceeded. " + "'blocked' (default) returns 429 as before. " + "'all' returns the full model list regardless of budget. " + "'free_only' returns only zero-cost models. " + "Inference calls always enforce the budget regardless of this setting." + ), + ) + class ConfigYAML(LiteLLMPydanticObjectBase): """ diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index e4d933ac699..42718171a82 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -8832,6 +8832,13 @@ async def model_list( if hidden_names: all_models = [m for m in all_models if m not in hidden_names] + # Budget-exceeded models policy: when "free_only", return only zero-cost models + policy = settings.get("budget_exceeded_models_policy", "blocked") + if policy == "free_only": + from litellm.proxy.auth.auth_checks import _is_model_cost_zero + + all_models = [m for m in all_models if _is_model_cost_zero(m, llm_router)] + # Surface the public team name by default; legacy internal keys via flag. # The internal routing key drives the metadata/fallback lookup, while the # public name is what the client sees as the model id.