fix(batch_rate_limiter): warn when no-op per-model skip key is configured
Some checks are pending
Unit Tests: Proxy DB Operations / assert-shard-coverage (push) Waiting to run
Unit Tests: Proxy DB Operations / auth-checks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / budgets (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / custom-logging (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / db-and-spend (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / endpoints-and-responses (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / guardrails-hooks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / jwt-and-keys (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / key-generation (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / logging-misc (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-runtime (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-server-core (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / schema-migration (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-utils (push) Blocked by required conditions
Unit Tests: Security / security (push) Waiting to run

This commit is contained in:
mateo-berri 2026-05-30 05:42:38 +00:00
parent 7a7e24b120
commit f0e7c45d6f
No known key found for this signature in database
2 changed files with 76 additions and 2 deletions

View file

@ -98,6 +98,7 @@ class _PROXY_BatchRateLimiter(CustomLogger):
"""
self.internal_usage_cache = internal_usage_cache
self.parallel_request_limiter = parallel_request_limiter
self._warned_unsupported_model_skip = False
def _get_file_bound_batch_model(self, data: Dict) -> Optional[str]:
"""Resolve the model bound to the batch input file ID.
@ -209,11 +210,13 @@ class _PROXY_BatchRateLimiter(CustomLogger):
rate-limit descriptor list computed for the no-limits check, so the
caller can reuse it for counter enforcement without recomputing.
"""
from litellm.proxy.proxy_server import general_settings
self._warn_if_unsupported_model_skip_configured(general_settings)
if self._key_requires_batch_model_access_check(user_api_key_dict):
return False, None
from litellm.proxy.proxy_server import general_settings
if general_settings.get("disable_batch_input_file_rate_limiting") is True:
return True, None
@ -243,6 +246,26 @@ class _PROXY_BatchRateLimiter(CustomLogger):
return False, descriptors
def _warn_if_unsupported_model_skip_configured(
self, general_settings: Dict
) -> None:
"""Warn once that ``skip_batch_input_file_rate_limiting_for_models`` is a no-op.
A per-model skip is intentionally not honored because the model a batch
runs on is caller-influenced and can be pointed at a skip-listed
deployment while the JSONL routes a different, rate-limited model.
"""
if self._warned_unsupported_model_skip:
return
if general_settings.get("skip_batch_input_file_rate_limiting_for_models"):
self._warned_unsupported_model_skip = True
verbose_proxy_logger.warning(
"general_settings.skip_batch_input_file_rate_limiting_for_models is not "
"supported and has no effect. Use "
"skip_batch_input_file_rate_limiting_for_providers or "
"disable_batch_input_file_rate_limiting instead."
)
@staticmethod
def _key_requires_batch_model_access_check(
user_api_key_dict: UserAPIKeyAuth,

View file

@ -708,6 +708,57 @@ def test_should_not_skip_for_skip_listed_top_level_model():
assert should_skip is False
def test_warns_once_for_unsupported_model_skip_setting():
"""Operators who set the no-op per-model skip key get a single warning so a
misconfigured deployment does not silently leave batch limits unenforced."""
rate_limiter = _make_rate_limiter()
rate_limiter.parallel_request_limiter._create_rate_limit_descriptors.return_value = [
{"rate_limit": {"requests_per_unit": 5}}
]
user = UserAPIKeyAuth(api_key="sk", models=["*"])
with (
patch(
"litellm.proxy.proxy_server.general_settings",
{"skip_batch_input_file_rate_limiting_for_models": ["gpt-4o-mini"]},
),
patch(
"litellm.proxy.hooks.batch_rate_limiter.verbose_proxy_logger"
) as mock_logger,
):
for _ in range(3):
rate_limiter._should_skip_batch_input_file_processing(
data={"model": "gpt-4o-mini", "input_file_id": "file-abc"},
user_api_key_dict=user,
)
assert mock_logger.warning.call_count == 1
assert (
"skip_batch_input_file_rate_limiting_for_models"
in mock_logger.warning.call_args[0][0]
)
def test_no_warning_when_model_skip_setting_absent():
rate_limiter = _make_rate_limiter()
rate_limiter.parallel_request_limiter._create_rate_limit_descriptors.return_value = [
{"rate_limit": {"requests_per_unit": 5}}
]
user = UserAPIKeyAuth(api_key="sk", models=["*"])
with (
patch(
"litellm.proxy.proxy_server.general_settings",
{"skip_batch_input_file_rate_limiting_for_providers": ["openai"]},
),
patch(
"litellm.proxy.hooks.batch_rate_limiter.verbose_proxy_logger"
) as mock_logger,
):
rate_limiter._should_skip_batch_input_file_processing(
data={"model": "gpt-4o-mini", "input_file_id": "file-abc"},
user_api_key_dict=user,
)
mock_logger.warning.assert_not_called()
def test_should_skip_when_no_rate_limits_configured():
rate_limiter = _make_rate_limiter()
rate_limiter.parallel_request_limiter._create_rate_limit_descriptors.return_value = [