mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
fix(batch_rate_limiter): warn when no-op per-model skip key is configured
Some checks are pending
Unit Tests: Proxy DB Operations / assert-shard-coverage (push) Waiting to run
Unit Tests: Proxy DB Operations / auth-checks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / budgets (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / custom-logging (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / db-and-spend (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / endpoints-and-responses (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / guardrails-hooks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / jwt-and-keys (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / key-generation (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / logging-misc (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-runtime (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-server-core (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / schema-migration (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-utils (push) Blocked by required conditions
Unit Tests: Security / security (push) Waiting to run
Some checks are pending
Unit Tests: Proxy DB Operations / assert-shard-coverage (push) Waiting to run
Unit Tests: Proxy DB Operations / auth-checks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / budgets (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / custom-logging (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / db-and-spend (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / endpoints-and-responses (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / guardrails-hooks (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / jwt-and-keys (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / key-generation (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / logging-misc (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-runtime (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-server-core (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / schema-migration (push) Blocked by required conditions
Unit Tests: Proxy DB Operations / proxy-utils (push) Blocked by required conditions
Unit Tests: Security / security (push) Waiting to run
This commit is contained in:
parent
7a7e24b120
commit
f0e7c45d6f
2 changed files with 76 additions and 2 deletions
|
|
@ -98,6 +98,7 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
"""
|
||||
self.internal_usage_cache = internal_usage_cache
|
||||
self.parallel_request_limiter = parallel_request_limiter
|
||||
self._warned_unsupported_model_skip = False
|
||||
|
||||
def _get_file_bound_batch_model(self, data: Dict) -> Optional[str]:
|
||||
"""Resolve the model bound to the batch input file ID.
|
||||
|
|
@ -209,11 +210,13 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
rate-limit descriptor list computed for the no-limits check, so the
|
||||
caller can reuse it for counter enforcement without recomputing.
|
||||
"""
|
||||
from litellm.proxy.proxy_server import general_settings
|
||||
|
||||
self._warn_if_unsupported_model_skip_configured(general_settings)
|
||||
|
||||
if self._key_requires_batch_model_access_check(user_api_key_dict):
|
||||
return False, None
|
||||
|
||||
from litellm.proxy.proxy_server import general_settings
|
||||
|
||||
if general_settings.get("disable_batch_input_file_rate_limiting") is True:
|
||||
return True, None
|
||||
|
||||
|
|
@ -243,6 +246,26 @@ class _PROXY_BatchRateLimiter(CustomLogger):
|
|||
|
||||
return False, descriptors
|
||||
|
||||
def _warn_if_unsupported_model_skip_configured(
|
||||
self, general_settings: Dict
|
||||
) -> None:
|
||||
"""Warn once that ``skip_batch_input_file_rate_limiting_for_models`` is a no-op.
|
||||
|
||||
A per-model skip is intentionally not honored because the model a batch
|
||||
runs on is caller-influenced and can be pointed at a skip-listed
|
||||
deployment while the JSONL routes a different, rate-limited model.
|
||||
"""
|
||||
if self._warned_unsupported_model_skip:
|
||||
return
|
||||
if general_settings.get("skip_batch_input_file_rate_limiting_for_models"):
|
||||
self._warned_unsupported_model_skip = True
|
||||
verbose_proxy_logger.warning(
|
||||
"general_settings.skip_batch_input_file_rate_limiting_for_models is not "
|
||||
"supported and has no effect. Use "
|
||||
"skip_batch_input_file_rate_limiting_for_providers or "
|
||||
"disable_batch_input_file_rate_limiting instead."
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _key_requires_batch_model_access_check(
|
||||
user_api_key_dict: UserAPIKeyAuth,
|
||||
|
|
|
|||
|
|
@ -708,6 +708,57 @@ def test_should_not_skip_for_skip_listed_top_level_model():
|
|||
assert should_skip is False
|
||||
|
||||
|
||||
def test_warns_once_for_unsupported_model_skip_setting():
|
||||
"""Operators who set the no-op per-model skip key get a single warning so a
|
||||
misconfigured deployment does not silently leave batch limits unenforced."""
|
||||
rate_limiter = _make_rate_limiter()
|
||||
rate_limiter.parallel_request_limiter._create_rate_limit_descriptors.return_value = [
|
||||
{"rate_limit": {"requests_per_unit": 5}}
|
||||
]
|
||||
user = UserAPIKeyAuth(api_key="sk", models=["*"])
|
||||
with (
|
||||
patch(
|
||||
"litellm.proxy.proxy_server.general_settings",
|
||||
{"skip_batch_input_file_rate_limiting_for_models": ["gpt-4o-mini"]},
|
||||
),
|
||||
patch(
|
||||
"litellm.proxy.hooks.batch_rate_limiter.verbose_proxy_logger"
|
||||
) as mock_logger,
|
||||
):
|
||||
for _ in range(3):
|
||||
rate_limiter._should_skip_batch_input_file_processing(
|
||||
data={"model": "gpt-4o-mini", "input_file_id": "file-abc"},
|
||||
user_api_key_dict=user,
|
||||
)
|
||||
assert mock_logger.warning.call_count == 1
|
||||
assert (
|
||||
"skip_batch_input_file_rate_limiting_for_models"
|
||||
in mock_logger.warning.call_args[0][0]
|
||||
)
|
||||
|
||||
|
||||
def test_no_warning_when_model_skip_setting_absent():
|
||||
rate_limiter = _make_rate_limiter()
|
||||
rate_limiter.parallel_request_limiter._create_rate_limit_descriptors.return_value = [
|
||||
{"rate_limit": {"requests_per_unit": 5}}
|
||||
]
|
||||
user = UserAPIKeyAuth(api_key="sk", models=["*"])
|
||||
with (
|
||||
patch(
|
||||
"litellm.proxy.proxy_server.general_settings",
|
||||
{"skip_batch_input_file_rate_limiting_for_providers": ["openai"]},
|
||||
),
|
||||
patch(
|
||||
"litellm.proxy.hooks.batch_rate_limiter.verbose_proxy_logger"
|
||||
) as mock_logger,
|
||||
):
|
||||
rate_limiter._should_skip_batch_input_file_processing(
|
||||
data={"model": "gpt-4o-mini", "input_file_id": "file-abc"},
|
||||
user_api_key_dict=user,
|
||||
)
|
||||
mock_logger.warning.assert_not_called()
|
||||
|
||||
|
||||
def test_should_skip_when_no_rate_limits_configured():
|
||||
rate_limiter = _make_rate_limiter()
|
||||
rate_limiter.parallel_request_limiter._create_rate_limit_descriptors.return_value = [
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue