fix(router): treat max_budget=0 as a real budget limit, not unlimited

The provider, deployment, and tag budget checks in
RouterBudgetLimiting._filter_out_deployments_above_budget used a truthy test
on max_budget. A configured budget of 0 is falsy in Python, so the spend
comparison never ran and the deployment was treated as having no limit and
left routable. Only None means "no limit". Use an explicit `is not None`
check at all three sites, matching the None check already used just above
the provider comparison.

Adds a regression test covering provider/deployment budgets of 0 and the
None-means-no-limit case.

Fixes #43214
This commit is contained in:
Daisong Gan 2026-09-26 20:27:51 +08:00
parent 115668f43e
commit 3c5b4868fc
2 changed files with 78 additions and 3 deletions

View file

@ -246,7 +246,7 @@ class RouterBudgetLimiting(CustomLogger):
budget_limit=config.max_budget,
)
if config.max_budget and current_spend >= config.max_budget:
if config.max_budget is not None and current_spend >= config.max_budget:
debug_msg = f"Exceeded budget for provider {provider}: {current_spend} >= {config.max_budget}"
deployment_above_budget_info += f"{debug_msg}\n"
is_within_budget = False
@ -261,7 +261,7 @@ class RouterBudgetLimiting(CustomLogger):
if model_id in deployment_configs:
config = deployment_configs[model_id]
current_spend = spend_map.get(f"deployment_spend:{model_id}:{config.budget_duration}", 0.0)
if config.max_budget and current_spend >= config.max_budget:
if config.max_budget is not None and current_spend >= config.max_budget:
debug_msg = f"Exceeded budget for deployment model_name: {_model_name}, litellm_params.model: {_litellm_model_name}, model_id: {model_id}: {current_spend} >= {config.budget_duration}"
verbose_router_logger.debug(debug_msg)
deployment_above_budget_info += f"{debug_msg}\n"
@ -276,7 +276,7 @@ class RouterBudgetLimiting(CustomLogger):
f"tag_spend:{_tag}:{_tag_budget_config.budget_duration}",
0.0,
)
if _tag_budget_config.max_budget and _tag_spend >= _tag_budget_config.max_budget:
if _tag_budget_config.max_budget is not None and _tag_spend >= _tag_budget_config.max_budget:
debug_msg = f"Exceeded budget for tag='{_tag}', tag_spend={_tag_spend}, tag_budget_limit={_tag_budget_config.max_budget}"
verbose_router_logger.debug(debug_msg)
deployment_above_budget_info += f"{debug_msg}\n"

View file

@ -12,6 +12,7 @@ import pytest
from litellm.caching.caching import DualCache
from litellm.router_strategy.budget_limiter import RouterBudgetLimiting
from litellm.types.utils import BudgetConfig
@pytest.fixture
@ -135,3 +136,77 @@ async def test_deployment_budget_tracked_when_provider_is_unresolvable(disable_b
)
assert await limiter.dual_cache.async_get_cache("deployment_spend:deployment-1:1d") == 0.25
@pytest.mark.asyncio
async def test_zero_max_budget_blocks_spend(disable_budget_sync):
"""A configured budget of 0 blocks all spend; only None means "no limit".
The three budget checks used a truthy test on ``max_budget``, so a legitimate
``0`` (an admin fully blocking spend) short-circuited to "no limit" and left
the deployment routable.
"""
healthy_deployments: Final = [
{
"model_name": "gpt-4o",
"litellm_params": {"model": "openai/gpt-4o", "max_budget": 0.0, "budget_duration": "1d"},
"model_info": {"id": "deployment-1"},
}
]
# Provider budget of 0 blocks a deployment that already spent.
limiter = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config={"openai": {"budget_limit": 0.0, "time_period": "1d"}},
)
provider_configs: Final = {"openai": BudgetConfig(max_budget=0.0, budget_duration="1d")}
kept, _ = limiter._filter_out_deployments_above_budget(
potential_deployments=[],
healthy_deployments=healthy_deployments,
provider_configs=provider_configs,
deployment_configs={},
deployment_providers=["openai"],
spend_map={"provider_spend:openai:1d": 5.0},
request_tags=[],
)
assert kept == []
# Deployment budget of 0 blocks even at zero spend.
limiter = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config=None,
model_list=[
{
"model_name": "gpt-4o",
"litellm_params": {"model": "openai/gpt-4o", "max_budget": 0.0, "budget_duration": "1d"},
"model_info": {"id": "deployment-1"},
}
],
)
deployment_configs: Final = {"deployment-1": BudgetConfig(max_budget=0.0, budget_duration="1d")}
kept, _ = limiter._filter_out_deployments_above_budget(
potential_deployments=[],
healthy_deployments=healthy_deployments,
provider_configs={},
deployment_configs=deployment_configs,
deployment_providers=["openai"],
spend_map={"deployment_spend:deployment-1:1d": 0.0},
request_tags=[],
)
assert kept == []
# A None budget still means "no limit" and must stay routable.
limiter = RouterBudgetLimiting(
dual_cache=DualCache(),
provider_budget_config=None,
)
kept, _ = limiter._filter_out_deployments_above_budget(
potential_deployments=[],
healthy_deployments=healthy_deployments,
provider_configs={},
deployment_configs={},
deployment_providers=["openai"],
spend_map={},
request_tags=[],
)
assert len(kept) == 1