feat(proxy): opt-in enforce rpm/tpm when adding a model

Add general_settings toggle 'enforce_rpm_tpm_on_model_add' (default false).
When true, /model/new rejects a model whose rpm or tpm is missing or not a
positive value, so the Admin UI Add Model form surfaces a 400 validation
error instead of silently storing an unbounded model (or one with a
zero/negative limit that would exclude it from routing).
This commit is contained in:
Kunal Nayyar 2026-08-11 13:03:55 +05:30
parent b0fac57fe4
commit 65eae963a7
3 changed files with 71 additions and 0 deletions

View file

@ -239,6 +239,38 @@ def _raise_on_strategy_router_write_violation(
)
ENFORCE_RPM_TPM_ON_MODEL_ADD_SETTING: Final = "enforce_rpm_tpm_on_model_add"
_REQUIRED_RATE_LIMIT_FIELDS: Final = ("rpm", "tpm")
def _raise_if_rate_limits_required_but_missing(*, litellm_params: GenericLiteLLMParams, enforced: bool) -> None:
"""Require both rpm and tpm (each a positive value) when the operator opts in via config.yaml.
Off by default, so deployments keep adding models without limits. When
``enforce_rpm_tpm_on_model_add: true`` is set under general_settings, a model added
without both rpm and tpm set to a positive value is rejected rather than stored
unbounded (or effectively excluded from routing by a zero/negative limit).
"""
if not enforced:
return
missing: Final = tuple(
field
for field in _REQUIRED_RATE_LIMIT_FIELDS
if (value := getattr(litellm_params, field)) is None or value <= 0
)
if not missing:
return
raise ProxyException(
message=(
f"{' and '.join(missing)} must be set to a positive value when "
f"'{ENFORCE_RPM_TPM_ON_MODEL_ADD_SETTING}' is enabled in general_settings"
),
type=ProxyErrorTypes.validation_error.value,
code=status.HTTP_400_BAD_REQUEST,
param=f"litellm_params.{missing[0]}",
)
_PTU_MODEL_INFO_FIELDS: Final = ("ptu_count", "cost_per_ptu_per_hour", "ptu_effective_from", "ptu_effective_to")
@ -1566,6 +1598,11 @@ async def add_new_model(
existing_params=None,
)
_raise_if_rate_limits_required_but_missing(
litellm_params=model_params.litellm_params,
enforced=bool(general_settings.get(ENFORCE_RPM_TPM_ON_MODEL_ADD_SETTING, False)),
)
model_response: LiteLLM_ProxyModelTable | None = None
# update DB
incoming_model_info: Final = model_params.model_info.model_dump(exclude_none=True)

View file

@ -23,6 +23,7 @@ from litellm.proxy._types import (
from litellm.proxy.management_endpoints.model_management_endpoints import (
ModelManagementAuthChecks,
_get_team_deployments,
_raise_if_rate_limits_required_but_missing,
clear_cache,
delete_team_models,
)
@ -3825,3 +3826,35 @@ class TestAutoRouterClassifierDefaultPrompt:
for empty in (None, "", "{}"):
response = await get_auto_router_classifier_default_prompt(context_window_size=5, tier_labels=empty)
assert response.system_prompt == classification_system_prompt(5)
class TestEnforceRpmTpmOnModelAdd:
def test_passes_when_disabled_even_without_limits(self):
_raise_if_rate_limits_required_but_missing(
litellm_params=LiteLLM_Params(model="azure/gpt-5.2"),
enforced=False,
)
def test_passes_when_enabled_and_both_set(self):
_raise_if_rate_limits_required_but_missing(
litellm_params=LiteLLM_Params(model="azure/gpt-5.2", rpm=10, tpm=1000),
enforced=True,
)
@pytest.mark.parametrize(
"params, expected_missing",
[
(LiteLLM_Params(model="azure/gpt-5.2"), "rpm and tpm"),
(LiteLLM_Params(model="azure/gpt-5.2", rpm=10), "tpm"),
(LiteLLM_Params(model="azure/gpt-5.2", tpm=1000), "rpm"),
(LiteLLM_Params(model="azure/gpt-5.2", rpm=0, tpm=1000), "rpm"),
(LiteLLM_Params(model="azure/gpt-5.2", rpm=10, tpm=-1), "tpm"),
],
)
def test_raises_when_enabled_and_missing(self, params, expected_missing):
from litellm.proxy._types import ProxyException
with pytest.raises(ProxyException) as exc_info:
_raise_if_rate_limits_required_but_missing(litellm_params=params, enforced=True)
assert expected_missing in str(exc_info.value.message)
assert exc_info.value.code == "400"

View file

@ -104,6 +104,7 @@ const VALIDATION_MATCH = [
"invalid file type",
"invalid field",
"invalid date format",
"must be set when",
];
const NOT_FOUND_MATCH = [