From fee77d3323a5ed05a22bf57c5072aa75944ae6bb Mon Sep 17 00:00:00 2001 From: jagjeet-singh-23 Date: Sat, 5 Sep 2026 14:20:59 +0530 Subject: [PATCH] fix(lowest-latency): treat tpm=0/rpm=0 as blocked, not unlimited `_get_available_deployments` read each deployment's tpm/rpm limit through an `or`-chain across three config levels. `or` treats `0` as falsy, so a deployment explicitly configured with `tpm: 0` or `rpm: 0` -- a documented way to take a deployment out of rotation without removing it from the model list -- fell all the way through to `float("inf")` and was treated as having unlimited capacity, the exact opposite of the configured intent. Filter the three levels on `is not None` instead, so a configured `0` wins and only a genuinely unset limit falls back to `float("inf")`. This matches the extraction already used in `lowest_tpm_rpm_v2.py`. Fixes #39744 Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_015ftJdiLtpTbjoZaNuqX6dq --- litellm/router_strategy/lowest_latency.py | 36 ++++++--- .../test_lowest_latency_zero_limits.py | 77 +++++++++++++++++++ 2 files changed, 103 insertions(+), 10 deletions(-) create mode 100644 tests/test_litellm/router_strategy/test_lowest_latency_zero_limits.py diff --git a/litellm/router_strategy/lowest_latency.py b/litellm/router_strategy/lowest_latency.py index a1b67eaeaf9..4127ad9eb2a 100644 --- a/litellm/router_strategy/lowest_latency.py +++ b/litellm/router_strategy/lowest_latency.py @@ -412,18 +412,34 @@ class LowestLatencyLoggingHandler(CustomLogger): if _deployment is None: continue # skip to next one - _deployment_tpm = ( - _deployment.get("tpm", None) - or _deployment.get("litellm_params", {}).get("tpm", None) - or _deployment.get("model_info", {}).get("tpm", None) - or float("inf") + # NOTE: levels are filtered with `is not None`, not truthiness. A configured + # `0` means "block this deployment"; an `or`-chain treats it as unset and + # falls through to `float("inf")`, the opposite of the configured intent. + # Matches the extraction in `lowest_tpm_rpm_v2.py`. See issue #39744. + _deployment_tpm = next( + ( + limit + for limit in ( + _deployment.get("tpm"), + _deployment.get("litellm_params", {}).get("tpm"), + _deployment.get("model_info", {}).get("tpm"), + ) + if limit is not None + ), + float("inf"), ) - _deployment_rpm = ( - _deployment.get("rpm", None) - or _deployment.get("litellm_params", {}).get("rpm", None) - or _deployment.get("model_info", {}).get("rpm", None) - or float("inf") + _deployment_rpm = next( + ( + limit + for limit in ( + _deployment.get("rpm"), + _deployment.get("litellm_params", {}).get("rpm"), + _deployment.get("model_info", {}).get("rpm"), + ) + if limit is not None + ), + float("inf"), ) item_latency = item_map.get("latency", []) item_ttft_latency = item_map.get("time_to_first_token", []) diff --git a/tests/test_litellm/router_strategy/test_lowest_latency_zero_limits.py b/tests/test_litellm/router_strategy/test_lowest_latency_zero_limits.py new file mode 100644 index 00000000000..98d67c2f1ad --- /dev/null +++ b/tests/test_litellm/router_strategy/test_lowest_latency_zero_limits.py @@ -0,0 +1,77 @@ +#### What this tests #### +# lowest-latency routing must treat an explicitly configured tpm=0 / rpm=0 +# as "this deployment is blocked", not as "unlimited". The limits were read +# through an `or`-chain, and `or` treats 0 as falsy, so a configured 0 fell +# through to float("inf") -- the exact opposite of the configured intent. +# Issue #39744. + +import pytest + +from litellm.caching.caching import DualCache +from litellm.router_strategy.lowest_latency import LowestLatencyLoggingHandler + +MODEL_GROUP = "gpt-4o" +DEPLOYMENT_ID = "disabled-deployment" + + +def _deployment(*, level: str, limit_key: str, value: int) -> dict: + """Build a deployment carrying `limit_key` at one of the three lookup levels.""" + deployment = { + "model_name": MODEL_GROUP, + "litellm_params": {"model": "azure/gpt-4o"}, + "model_info": {"id": DEPLOYMENT_ID}, + } + if level == "top_level": + deployment[limit_key] = value + elif level == "litellm_params": + deployment["litellm_params"][limit_key] = value + elif level == "model_info": + deployment["model_info"][limit_key] = value + else: # pragma: no cover - guards against a typo in the parametrize list + raise ValueError(f"unknown level {level!r}") + return deployment + + +def _select(deployment: dict): + handler = LowestLatencyLoggingHandler(router_cache=DualCache(), routing_args={}) + return handler._get_available_deployments( + model_group=MODEL_GROUP, + healthy_deployments=[deployment], + messages=[{"role": "user", "content": "hello there, this is a prompt"}], + request_count_dict={}, + ) + + +@pytest.mark.parametrize("level", ["top_level", "litellm_params", "model_info"]) +@pytest.mark.parametrize("limit_key", ["tpm", "rpm"]) +def test_zero_limit_blocks_deployment(level: str, limit_key: str): + """A deployment configured with tpm=0 or rpm=0 must never be selected.""" + selected = _select(_deployment(level=level, limit_key=limit_key, value=0)) + + assert selected is None, ( + f"deployment with {limit_key}=0 at {level} was selected; 0 was treated as unlimited instead of blocked" + ) + + +@pytest.mark.parametrize("level", ["top_level", "litellm_params", "model_info"]) +@pytest.mark.parametrize("limit_key", ["tpm", "rpm"]) +def test_nonzero_limit_still_allows_deployment(level: str, limit_key: str): + """Control: a generous limit at the same level must still be selectable.""" + selected = _select(_deployment(level=level, limit_key=limit_key, value=1_000_000)) + + assert selected is not None, f"deployment with {limit_key}=1000000 at {level} was wrongly excluded" + assert selected["model_info"]["id"] == DEPLOYMENT_ID + + +def test_unset_limits_are_unlimited(): + """Control: with no tpm/rpm configured at all, the deployment stays selectable.""" + selected = _select( + { + "model_name": MODEL_GROUP, + "litellm_params": {"model": "azure/gpt-4o"}, + "model_info": {"id": DEPLOYMENT_ID}, + } + ) + + assert selected is not None + assert selected["model_info"]["id"] == DEPLOYMENT_ID