From 64c22e0cdd6781e87be6dfd5430635cc33fab596 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:30:03 -0700 Subject: [PATCH 1/4] feat(proxy): cap batch file records, daily batch uploads, and per-file downloads Adds three opt-in limits for batch jobs, each settable in general_settings as a per-key default and overridable in key or team metadata by a proxy admin: max_batch_file_records rejects a purpose=batch upload with more request lines than allowed with a 413 before it reaches the provider. max_batch_file_uploads_per_day counts accepted batch uploads per key and per team in a UTC day and returns 429 with Retry-After once the count is used. max_file_downloads_per_minute counts GET /v1/files/{id}/content per key and per team for each file in a one-minute window and returns 429 with Retry-After. The existing admin-only guard for batch_enqueued_token_limit now covers all four metadata keys. --- litellm/constants.py | 9 + litellm/proxy/_types.py | 12 + litellm/proxy/auth/auth_utils.py | 30 ++- .../key_management_endpoints.py | 10 +- .../management_endpoints/team_endpoints.py | 6 +- .../batch_file_validation.py | 51 +++- .../openai_files_endpoints/file_usage_caps.py | 239 ++++++++++++++++++ .../openai_files_endpoints/files_endpoints.py | 14 + litellm/proxy/proxy_server.py | 3 + .../test_key_management_endpoints.py | 41 +-- .../test_files_batch_file_validation.py | 36 ++- .../test_files_endpoint.py | 127 ++++++++++ .../proxy/openai_files_endpoints/__init__.py | 0 .../test_file_usage_caps.py | 182 +++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 15 ++ 15 files changed, 724 insertions(+), 51 deletions(-) create mode 100644 litellm/proxy/openai_files_endpoints/file_usage_caps.py create mode 100644 tests/unit/proxy/openai_files_endpoints/__init__.py create mode 100644 tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py diff --git a/litellm/constants.py b/litellm/constants.py index fd9812e2cf5..e69ccad6d12 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -2170,6 +2170,15 @@ BATCH_ENQUEUED_TOKEN_TTL_SECONDS: Final[int] = 8 * 24 * 60 * 60 # admins may write it: when present it replaces the standard RPM/TPM checks for # batch submissions. BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY: Final = "batch_enqueued_token_limit" +MAX_BATCH_FILE_RECORDS_KEY: Final = "max_batch_file_records" +MAX_BATCH_FILE_UPLOADS_PER_DAY_KEY: Final = "max_batch_file_uploads_per_day" +MAX_FILE_DOWNLOADS_PER_MINUTE_KEY: Final = "max_file_downloads_per_minute" +ADMIN_ONLY_BATCH_LIMIT_METADATA_KEYS: Final = ( + BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY, + MAX_BATCH_FILE_RECORDS_KEY, + MAX_BATCH_FILE_UPLOADS_PER_DAY_KEY, + MAX_FILE_DOWNLOADS_PER_MINUTE_KEY, +) # Shared read-only empty mapping, for defaulting optional Mapping parameters without # constructing a fresh mutable dict at each call site. diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index d9fb053035b..29fa7c5c912 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2782,6 +2782,18 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): None, description="max batch input file size in MB for /v1/files uploads with purpose=batch, if a file is larger than this size it will be rejected before being forwarded to the provider", ) + max_batch_file_records: int | None = Field( + None, + description="max records (non-blank lines) per batch input file for /v1/files uploads with purpose=batch, applied per key. A key or team can carry its own value in metadata, set by a proxy admin, and the lowest applicable value wins. Unset means no limit", + ) + max_batch_file_uploads_per_day: int | None = Field( + None, + description="max /v1/files uploads with purpose=batch per key per UTC day. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit", + ) + max_file_downloads_per_minute: int | None = Field( + None, + description="max GET /v1/files/{file_id}/content calls per key per file per minute. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit", + ) max_file_size_mb: int | None = Field( None, description="max file size in MB for /v1/files uploads, for any purpose, if a file is larger than this size it will be rejected before being forwarded to the provider", diff --git a/litellm/proxy/auth/auth_utils.py b/litellm/proxy/auth/auth_utils.py index c0123ae45a3..3d51c2dd044 100644 --- a/litellm/proxy/auth/auth_utils.py +++ b/litellm/proxy/auth/auth_utils.py @@ -14,7 +14,7 @@ import litellm from litellm import Router, constants, provider_list from litellm._logging import verbose_proxy_logger from litellm.constants import ( - BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY, + ADMIN_ONLY_BATCH_LIMIT_METADATA_KEYS, EMPTY_MAPPING, INVALID_VIRTUAL_KEY_ERROR_MARKER, MINIMUM_CUSTOM_KEY_LENGTH, @@ -1226,8 +1226,8 @@ def enforce_output_token_estimates_are_admin_only( ) -class BatchEnqueuedTokenLimitRequest(Protocol): - """The shape of any management request that can carry a batch enqueued-token limit.""" +class BatchLimitRequest(Protocol): + """The shape of any management request that can carry an admin-only batch limit in its metadata.""" @property def metadata(self) -> Mapping[str, object] | None: ... @@ -1236,18 +1236,18 @@ class BatchEnqueuedTokenLimitRequest(Protocol): def model_fields_set(self) -> Collection[str]: ... -def enforce_batch_enqueued_token_limit_is_admin_only( - data: BatchEnqueuedTokenLimitRequest, +def enforce_batch_limits_are_admin_only( + data: BatchLimitRequest, existing_metadata: Mapping[str, object] | None, user_api_key_dict: UserAPIKeyAuth, entity: Literal["key", "team"], ) -> None: - """Only a proxy admin may change a key or team's batch enqueued-token limit. + """Only a proxy admin may change a key or team's batch limits. - When set, ``batch_enqueued_token_limit`` replaces the standard RPM/TPM checks - for batch submissions, so a holder-writable copy would let a caller lift their - own batch quota. Gated on the resulting value rather than on presence, so a - form resending the stored value stays a no-op. + Every key in ``ADMIN_ONLY_BATCH_LIMIT_METADATA_KEYS`` caps what the holder + may do with batches, so a holder-writable copy would let a caller lift their + own quota. Gated on the resulting value rather than on presence, so a form + resending the stored value stays a no-op. """ if user_api_key_dict.user_role == LitellmUserRoles.PROXY_ADMIN.value: return @@ -1255,13 +1255,17 @@ def enforce_batch_enqueued_token_limit_is_admin_only( requested: Final[Mapping[str, object]] = ( (data.metadata or EMPTY_MAPPING) if "metadata" in data.model_fields_set else stored ) - if requested.get(BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY) == stored.get(BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY): + changed: Final = next( + (key for key in ADMIN_ONLY_BATCH_LIMIT_METADATA_KEYS if requested.get(key) != stored.get(key)), + None, + ) + if changed is None: return raise HTTPException( status_code=403, detail={ # mutable-ok: HTTPException.detail has no immutable form - "error": f"Only proxy admins can set {BATCH_ENQUEUED_TOKEN_LIMIT_METADATA_KEY} on a {entity}. " - "It replaces the standard rate limit checks for batch submissions." + "error": f"Only proxy admins can set {changed} on a {entity}. " + "It limits what the holder can do with batches, so the holder cannot raise it." }, ) diff --git a/litellm/proxy/management_endpoints/key_management_endpoints.py b/litellm/proxy/management_endpoints/key_management_endpoints.py index 7e159ec90e7..c4228d615bf 100644 --- a/litellm/proxy/management_endpoints/key_management_endpoints.py +++ b/litellm/proxy/management_endpoints/key_management_endpoints.py @@ -63,7 +63,7 @@ from litellm.proxy.auth.auth_checks import ( ) from litellm.proxy.auth.auth_utils import ( abbreviate_api_key, - enforce_batch_enqueued_token_limit_is_admin_only, + enforce_batch_limits_are_admin_only, enforce_output_token_estimates_are_admin_only, ) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth @@ -1212,7 +1212,7 @@ async def _common_key_generation_helper( user_api_key_dict=user_api_key_dict, entity="key", ) - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=data, existing_metadata=None, user_api_key_dict=user_api_key_dict, @@ -2749,7 +2749,7 @@ async def _process_single_key_update( existing_metadata=existing_key_row.metadata, # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType] # LiteLLM_VerificationToken.metadata is a bare dict ) - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=update_key_request, existing_metadata=existing_key_row.metadata, user_api_key_dict=user_api_key_dict, @@ -3201,7 +3201,7 @@ async def _validate_update_key_data( user_api_key_dict=user_api_key_dict, entity="key", ) - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=data, existing_metadata=_existing_metadata if isinstance(_existing_metadata, dict) else None, user_api_key_dict=user_api_key_dict, @@ -5584,7 +5584,7 @@ async def _execute_virtual_key_regeneration( user_api_key_dict=user_api_key_dict, entity="key", ) - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=data, existing_metadata=_existing_key_metadata if isinstance(_existing_key_metadata, dict) else None, user_api_key_dict=user_api_key_dict, diff --git a/litellm/proxy/management_endpoints/team_endpoints.py b/litellm/proxy/management_endpoints/team_endpoints.py index 6cec3e714ec..c602dd363bc 100644 --- a/litellm/proxy/management_endpoints/team_endpoints.py +++ b/litellm/proxy/management_endpoints/team_endpoints.py @@ -110,7 +110,7 @@ from litellm.proxy.auth.auth_checks import ( invalidate_team_member_spend_state, ) from litellm.proxy.auth.auth_utils import ( - enforce_batch_enqueued_token_limit_is_admin_only, + enforce_batch_limits_are_admin_only, enforce_output_token_estimates_are_admin_only, ) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth @@ -1526,7 +1526,7 @@ async def new_team( user_api_key_dict=user_api_key_dict, entity="team", ) - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=data, existing_metadata=None, user_api_key_dict=user_api_key_dict, @@ -2312,7 +2312,7 @@ async def update_team( user_api_key_dict=user_api_key_dict, entity="team", ) - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=data, existing_metadata=_existing_team_metadata if isinstance(_existing_team_metadata, dict) else None, user_api_key_dict=user_api_key_dict, diff --git a/litellm/proxy/openai_files_endpoints/batch_file_validation.py b/litellm/proxy/openai_files_endpoints/batch_file_validation.py index fcd3be56ae7..d49aaaeba65 100644 --- a/litellm/proxy/openai_files_endpoints/batch_file_validation.py +++ b/litellm/proxy/openai_files_endpoints/batch_file_validation.py @@ -7,6 +7,7 @@ from typing import BinaryIO, Final, NoReturn from typing_extensions import assert_never from litellm.proxy._types import ProxyException +from litellm.proxy.openai_files_endpoints.file_usage_caps import FileUsageLimit, describe_limit_source _MB: Final = 1024 * 1024 @@ -33,6 +34,11 @@ class BatchFileTooLarge: limit_mb: int +@dataclass(frozen=True, slots=True) +class BatchFileTooManyRecords: + limit: FileUsageLimit + + @dataclass(frozen=True, slots=True) class BatchFileWrongExtension: filename: str @@ -62,6 +68,7 @@ class BatchFileMissingLineKey: BatchFileValidationFailure = ( BatchFileTooLarge + | BatchFileTooManyRecords | BatchFileWrongExtension | BatchFileEmpty | BatchFileInvalidJsonLine @@ -99,7 +106,23 @@ def _check_line(line_number: int, raw_line: bytes, line_shape: BatchLineShape) - return BatchFileMissingLineKey(line_number=line_number, key=missing, line_shape=line_shape) -def _scan_lines(file_source: bytes | BinaryIO, line_shape: BatchLineShape) -> BatchFileValidationFailure | None: +def _check_record( + record_number: int, + line_number: int, + raw_line: bytes, + line_shape: BatchLineShape, + max_records: FileUsageLimit | None, +) -> BatchFileValidationFailure | None: + if max_records is not None and record_number > max_records.value: + return BatchFileTooManyRecords(limit=max_records) + return _check_line(line_number, raw_line, line_shape) + + +def _scan_lines( + file_source: bytes | BinaryIO, + line_shape: BatchLineShape, + max_records: FileUsageLimit | None, +) -> BatchFileValidationFailure | None: content_lines: Final = ( (line_number, raw_line) for line_number, raw_line in enumerate(_iter_lines(file_source), start=1) @@ -108,15 +131,11 @@ def _scan_lines(file_source: bytes | BinaryIO, line_shape: BatchLineShape) -> Ba first_line: Final = next(content_lines, None) if first_line is None: return BatchFileEmpty() - return next( - ( - failure - for line_number, raw_line in chain((first_line,), content_lines) - for failure in (_check_line(line_number, raw_line, line_shape),) - if failure is not None - ), - None, + failures: Final = ( + _check_record(record_number, line_number, raw_line, line_shape, max_records) + for record_number, (line_number, raw_line) in enumerate(chain((first_line,), content_lines), start=1) ) + return next((failure for failure in failures if failure is not None), None) def check_batch_file_upload( @@ -124,6 +143,7 @@ def check_batch_file_upload( file_source: bytes | BinaryIO, max_batch_file_size_mb: int | None, line_shape: BatchLineShape = BATCH_LINE_SHAPE, + max_records: FileUsageLimit | None = None, ) -> BatchFileValidationFailure | None: if filename is None or not filename.lower().endswith(".jsonl"): return BatchFileWrongExtension(filename=filename or "") @@ -131,7 +151,7 @@ def check_batch_file_upload( size_bytes: Final = _file_size_bytes(file_source) if size_bytes > max_batch_file_size_mb * _MB: return BatchFileTooLarge(size_bytes=size_bytes, limit_mb=max_batch_file_size_mb) - scan_failure: Final = _scan_lines(file_source, line_shape) + scan_failure: Final = _scan_lines(file_source, line_shape, max_records) if not isinstance(file_source, bytes): file_source.seek(0) return scan_failure @@ -149,6 +169,17 @@ def raise_batch_file_validation_failure(failure: BatchFileValidationFailure) -> param="file", code=413, ) + case BatchFileTooManyRecords(limit=limit): + raise ProxyException( + message=( + f"Batch input file has more than {limit.value} records, which exceeds the " + f"{limit.setting} of {limit.value} set {describe_limit_source(limit.source)}. " + "The file was not forwarded to the provider." + ), + type="invalid_request_error", + param="file", + code=413, + ) case BatchFileWrongExtension(filename=filename): raise ProxyException( message=( diff --git a/litellm/proxy/openai_files_endpoints/file_usage_caps.py b/litellm/proxy/openai_files_endpoints/file_usage_caps.py new file mode 100644 index 00000000000..c8555ac3cbc --- /dev/null +++ b/litellm/proxy/openai_files_endpoints/file_usage_caps.py @@ -0,0 +1,239 @@ +import math +import time +from collections.abc import Callable, Mapping +from dataclasses import dataclass +from typing import TYPE_CHECKING, Annotated, Final, Literal, NoReturn, TypeAlias + +from pydantic import Field, TypeAdapter, ValidationError +from typing_extensions import assert_never + +from litellm._logging import verbose_proxy_logger +from litellm.constants import ( + EMPTY_MAPPING, + MAX_BATCH_FILE_RECORDS_KEY, + MAX_BATCH_FILE_UPLOADS_PER_DAY_KEY, + MAX_FILE_DOWNLOADS_PER_MINUTE_KEY, +) +from litellm.proxy._types import ProxyException, UserAPIKeyAuth + +if TYPE_CHECKING: + from litellm.proxy.utils import InternalUsageCache + +FileUsageSetting: TypeAlias = Literal[ + "max_batch_file_records", + "max_batch_file_uploads_per_day", + "max_file_downloads_per_minute", +] +LimitSource: TypeAlias = Literal["key", "team", "general_settings"] +CounterScope: TypeAlias = Literal["key", "team"] + +_COUNTER_PREFIX: Final = "litellm:file_usage" +_DAY_SECONDS: Final = 24 * 60 * 60 +_MINUTE_SECONDS: Final = 60 +_LIMIT_ADAPTER: Final = TypeAdapter(Annotated[int, Field(gt=0)]) + + +@dataclass(frozen=True, slots=True) +class FileUsageLimit: + setting: FileUsageSetting + value: int + source: LimitSource + + +@dataclass(frozen=True, slots=True) +class ScopedFileUsageLimit: + scope: CounterScope + scope_id: str + limit: FileUsageLimit + + +@dataclass(frozen=True, slots=True) +class FileUsageLimitExceeded: + limit: ScopedFileUsageLimit + retry_after_seconds: int + + +def _read_limit( + settings: Mapping[str, object] | None, + setting: FileUsageSetting, + source: LimitSource, +) -> FileUsageLimit | None: + raw: Final = (settings or EMPTY_MAPPING).get(setting) + if raw is None: + return None + try: + return FileUsageLimit(setting=setting, value=_LIMIT_ADAPTER.validate_python(raw), source=source) + except ValidationError: + verbose_proxy_logger.warning( + "Ignoring invalid %s value %r in %s; expected a positive integer", + setting, + raw, + source, + ) + return None + + +def _key_limit( + user_api_key_dict: UserAPIKeyAuth, + general_settings: Mapping[str, object], + setting: FileUsageSetting, +) -> FileUsageLimit | None: + from_key: Final = _read_limit(user_api_key_dict.metadata, setting, "key") + if from_key is not None: + return from_key + return _read_limit(general_settings, setting, "general_settings") + + +def _team_limit(user_api_key_dict: UserAPIKeyAuth, setting: FileUsageSetting) -> FileUsageLimit | None: + return _read_limit(user_api_key_dict.team_metadata, setting, "team") + + +def batch_file_record_limit( + user_api_key_dict: UserAPIKeyAuth, + general_settings: Mapping[str, object], +) -> FileUsageLimit | None: + applicable: Final = tuple( + limit + for limit in ( + _key_limit(user_api_key_dict, general_settings, MAX_BATCH_FILE_RECORDS_KEY), + _team_limit(user_api_key_dict, MAX_BATCH_FILE_RECORDS_KEY), + ) + if limit is not None + ) + return min(applicable, key=lambda limit: limit.value, default=None) + + +def resolve_scoped_limits( + user_api_key_dict: UserAPIKeyAuth, + general_settings: Mapping[str, object], + setting: FileUsageSetting, +) -> tuple[ScopedFileUsageLimit, ...]: + key_limit: Final = _key_limit(user_api_key_dict, general_settings, setting) + team_limit: Final = _team_limit(user_api_key_dict, setting) + candidates: Final = ( + ScopedFileUsageLimit(scope="key", scope_id=user_api_key_dict.api_key, limit=key_limit) + if key_limit is not None and user_api_key_dict.api_key + else None, + ScopedFileUsageLimit(scope="team", scope_id=user_api_key_dict.team_id, limit=team_limit) + if team_limit is not None and user_api_key_dict.team_id + else None, + ) + return tuple(scoped for scoped in candidates if scoped is not None) + + +async def _increment_all_or_none( + cache: "InternalUsageCache", + counters: tuple[tuple[str, ScopedFileUsageLimit], ...], + ttl_seconds: int, +) -> ScopedFileUsageLimit | None: + if not counters: + return None + (counter_key, scoped), rest = counters[0], counters[1:] + count: Final = await cache.async_increment_cache( + key=counter_key, value=1, litellm_parent_otel_span=None, ttl=ttl_seconds + ) + over_here: Final = count is not None and count > scoped.limit.value + exceeded: Final = scoped if over_here else await _increment_all_or_none(cache, rest, ttl_seconds) + if exceeded is not None: + await cache.async_increment_cache(key=counter_key, value=-1, litellm_parent_otel_span=None, ttl=ttl_seconds) + return exceeded + + +async def consume_file_usage( + cache: "InternalUsageCache", + limits: tuple[ScopedFileUsageLimit, ...], + window_seconds: int, + subject: str, + now: float, +) -> FileUsageLimitExceeded | None: + window_start: Final = int(now // window_seconds) * window_seconds + counters: Final = tuple( + ( + f"{_COUNTER_PREFIX}:{scoped.limit.setting}:{scoped.scope}:{scoped.scope_id}:{subject}:{window_start}", + scoped, + ) + for scoped in limits + ) + exceeded: Final = await _increment_all_or_none(cache, counters, window_seconds) + if exceeded is None: + return None + return FileUsageLimitExceeded( + limit=exceeded, + retry_after_seconds=max(1, math.ceil(window_start + window_seconds - now)), + ) + + +def describe_limit_source(source: LimitSource) -> str: + match source: + case "key": + return "in this key's metadata" + case "team": + return "in this team's metadata" + case "general_settings": + return "in general_settings" + case _: + assert_never(source) + + +def _describe_scope(scoped: ScopedFileUsageLimit) -> str: + match scoped.scope: + case "key": + return "this key" + case "team": + return f"team {scoped.scope_id}" + case _: + assert_never(scoped.scope) + + +def _raise_limit_exceeded(exceeded: FileUsageLimitExceeded, what_ran_out: str, when_it_resets: str) -> NoReturn: + scoped: Final = exceeded.limit + headers: Final = {"retry-after": str(exceeded.retry_after_seconds)} # mutable-ok: ProxyException mutates it + raise ProxyException( + message=( + f"{what_ran_out}: {scoped.limit.setting} is {scoped.limit.value} for {_describe_scope(scoped)} " + f"(set {describe_limit_source(scoped.limit.source)}). {when_it_resets}" + ), + type="rate_limit_error", + param=None, + code=429, + headers=headers, + ) + + +async def enforce_batch_file_upload_limit( + cache: "InternalUsageCache", + user_api_key_dict: UserAPIKeyAuth, + general_settings: Mapping[str, object], + clock: Callable[[], float] = time.time, +) -> None: + limits: Final = resolve_scoped_limits(user_api_key_dict, general_settings, MAX_BATCH_FILE_UPLOADS_PER_DAY_KEY) + if not limits: + return + exceeded: Final = await consume_file_usage(cache, limits, _DAY_SECONDS, "", clock()) + if exceeded is None: + return + _raise_limit_exceeded( + exceeded, + "Batch file upload limit reached, the file was not forwarded to the provider", + f"The count resets at 00:00 UTC, in {exceeded.retry_after_seconds} seconds.", + ) + + +async def enforce_file_download_limit( + cache: "InternalUsageCache", + user_api_key_dict: UserAPIKeyAuth, + general_settings: Mapping[str, object], + file_id: str, + clock: Callable[[], float] = time.time, +) -> None: + limits: Final = resolve_scoped_limits(user_api_key_dict, general_settings, MAX_FILE_DOWNLOADS_PER_MINUTE_KEY) + if not limits: + return + exceeded: Final = await consume_file_usage(cache, limits, _MINUTE_SECONDS, file_id, clock()) + if exceeded is None: + return + _raise_limit_exceeded( + exceeded, + f"Download limit reached for file {file_id}", + f"Retry in {exceeded.retry_after_seconds} seconds.", + ) diff --git a/litellm/proxy/openai_files_endpoints/files_endpoints.py b/litellm/proxy/openai_files_endpoints/files_endpoints.py index f5e62da5962..83b3433130e 100644 --- a/litellm/proxy/openai_files_endpoints/files_endpoints.py +++ b/litellm/proxy/openai_files_endpoints/files_endpoints.py @@ -84,6 +84,11 @@ from litellm.proxy.openai_files_endpoints.common_utils import ( validate_managed_files_requirement, validate_managed_id_requirement, ) +from litellm.proxy.openai_files_endpoints.file_usage_caps import ( + batch_file_record_limit, + enforce_batch_file_upload_limit, + enforce_file_download_limit, +) from litellm.proxy.openai_files_endpoints.general_upload_validation import ( MB, check_allowed_extension, @@ -672,9 +677,13 @@ async def create_file( file_source, _MAX_BATCH_FILE_SIZE_MB_ADAPTER.validate_python(general_settings.get("max_batch_file_size_mb")), PASSTHROUGH_BATCH_LINE_SHAPE if passthrough else BATCH_LINE_SHAPE, + batch_file_record_limit(user_api_key_dict, general_settings), ) if batch_file_failure is not None: raise_batch_file_validation_failure(batch_file_failure) + await enforce_batch_file_upload_limit( + proxy_logging_obj.internal_usage_cache, user_api_key_dict, general_settings + ) data = {"passthrough": True} if passthrough else {} @@ -978,6 +987,9 @@ async def get_file_content( user_api_key_dict=user_api_key_dict, managed_files_obj=proxy_logging_obj.get_proxy_hook("managed_files"), ) + await enforce_file_download_limit( + proxy_logging_obj.internal_usage_cache, user_api_key_dict, general_settings, file_id + ) # Include original request and headers in the data base_llm_response_processor: Final = ProxyBaseLLMRequestProcessing(data=data) @@ -1216,6 +1228,8 @@ async def get_file_content( ) verbose_proxy_logger.exception("litellm.proxy.proxy_server.retrieve_file_content(): Exception occured - %s", e) verbose_proxy_logger.debug(traceback.format_exc()) + if isinstance(e, ProxyException): + raise e if isinstance(e, HTTPException): raise ProxyException( message=getattr(e, "message", str(e.detail)), diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 7cd116c23d7..8b1a3e5dd25 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -17957,6 +17957,9 @@ _GENERAL_SETTINGS_CONFIG_LIST_FIELD_TYPES: Final[Mapping[str, str]] = MappingPro "admission_queue_timeout_seconds": "Float", "max_request_size_mb": "Integer", "max_batch_file_size_mb": "Integer", + "max_batch_file_records": "Integer", + "max_batch_file_uploads_per_day": "Integer", + "max_file_downloads_per_minute": "Integer", "max_file_size_mb": "Integer", "allowed_file_extensions": "List", "blocked_file_extensions": "List", diff --git a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py index aa6be328f4a..df38e7f67ad 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_key_management_endpoints.py @@ -18930,30 +18930,37 @@ async def test_regenerate_key_output_token_estimate_lowered_rejected_for_non_adm _BATCH_LIMIT = "batch_enqueued_token_limit" +_UNTOUCHED = object() + + @pytest.mark.parametrize( - "label, request_body, existing_metadata, allowed", + "limit_key", [ - ("set on a key with none stored", {"metadata": {_BATCH_LIMIT: 50000}}, None, False), - ("raised above the stored limit", {"metadata": {_BATCH_LIMIT: 200000}}, {_BATCH_LIMIT: 100000}, False), - ("cleared by replacing the blob", {"metadata": {}}, {_BATCH_LIMIT: 100000}, False), - ("resent unchanged", {"metadata": {_BATCH_LIMIT: 100000}}, {_BATCH_LIMIT: 100000}, True), - ("left untouched", {}, {_BATCH_LIMIT: 100000}, True), + "batch_enqueued_token_limit", + "max_batch_file_records", + "max_batch_file_uploads_per_day", + "max_file_downloads_per_minute", ], ) -def test_batch_enqueued_token_limit_admin_gate_matrix(label, request_body, existing_metadata, allowed): - """A non-admin may only leave a key's stored batch enqueued-token limit as it is. - - When set, the limit replaces the standard RPM/TPM checks for batch - submissions, so a key holder writing it would pick their own batch quota. - Resending the stored value is what the edit form produces on every save - and has to stay allowed. - """ +@pytest.mark.parametrize( + "label, sent, stored, allowed", + [ + ("set on a key with none stored", 50000, None, False), + ("raised above the stored limit", 200000, 100000, False), + ("cleared by replacing the blob", None, 100000, False), + ("resent unchanged", 100000, 100000, True), + ("left untouched", _UNTOUCHED, 100000, True), + ], +) +def test_batch_limits_admin_gate_matrix(limit_key, label, sent, stored, allowed): + request_body = {} if sent is _UNTOUCHED else {"metadata": {} if sent is None else {limit_key: sent}} + existing_metadata = None if stored is None else {limit_key: stored} from litellm.proxy.auth.auth_utils import ( - enforce_batch_enqueued_token_limit_is_admin_only, + enforce_batch_limits_are_admin_only, ) def _call(caller): - enforce_batch_enqueued_token_limit_is_admin_only( + enforce_batch_limits_are_admin_only( data=UpdateKeyRequest(key="sk-1", **request_body), existing_metadata=existing_metadata, user_api_key_dict=caller, @@ -18971,7 +18978,7 @@ def test_batch_enqueued_token_limit_admin_gate_matrix(label, request_body, exist with pytest.raises(HTTPException) as exc: _call(non_admin) assert exc.value.status_code == 403 - assert "Only proxy admins can set" in str(exc.value.detail) + assert f"Only proxy admins can set {limit_key}" in str(exc.value.detail) _call( UserAPIKeyAuth( diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py index 3a73c39e177..b55ae1500be 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_batch_file_validation.py @@ -11,10 +11,12 @@ from litellm.proxy.openai_files_endpoints.batch_file_validation import ( BatchFileLineNotObject, BatchFileMissingLineKey, BatchFileTooLarge, + BatchFileTooManyRecords, BatchFileWrongExtension, check_batch_file_upload, raise_batch_file_validation_failure, ) +from litellm.proxy.openai_files_endpoints.file_usage_caps import FileUsageLimit VALID_LINE = ( b'{"custom_id": "req-1", "method": "POST", "url": "/v1/chat/completions",' @@ -44,9 +46,7 @@ def test_wrong_extension_rejected(filename): def test_size_over_cap_rejected_for_bytes(): content = b"x" * (2 * 1024 * 1024) - assert check_batch_file_upload("batch.jsonl", content, 1) == BatchFileTooLarge( - size_bytes=len(content), limit_mb=1 - ) + assert check_batch_file_upload("batch.jsonl", content, 1) == BatchFileTooLarge(size_bytes=len(content), limit_mb=1) def test_size_over_cap_rejected_for_binaryio(): @@ -150,6 +150,12 @@ def test_scan_stops_at_first_failure(): "file", ("210.0 MB", "max_batch_file_size_mb", "10 MB", "not forwarded"), ), + ( + BatchFileTooManyRecords(limit=FileUsageLimit("max_batch_file_records", 1000, "team")), + "413", + "file", + ("more than 1000 records", "max_batch_file_records of 1000", "this team's metadata", "not forwarded"), + ), ( BatchFileWrongExtension(filename="batch.csv"), "400", @@ -208,3 +214,27 @@ def test_passthrough_missing_key_message_says_what_a_passthrough_upload_takes(): assert "passthrough upload takes native Vertex batch rows" in exc_info.value.message assert "with a request key." in exc_info.value.message assert "custom_id" not in exc_info.value.message + + +RECORD_LIMIT_3 = FileUsageLimit("max_batch_file_records", 3, "key") + + +@pytest.mark.parametrize( + "content, expected", + [ + ((VALID_LINE + b"\n") * 3, None), + ((VALID_LINE + b"\n") * 4, BatchFileTooManyRecords(limit=RECORD_LIMIT_3)), + (b"\n\n" + (VALID_LINE + b"\n\n \n") * 3, None), + (io.BytesIO((VALID_LINE + b"\n") * 4), BatchFileTooManyRecords(limit=RECORD_LIMIT_3)), + (b"broken\n" + (VALID_LINE + b"\n") * 4, BatchFileInvalidJsonLine(line_number=1)), + ], +) +def test_record_limit_counts_non_blank_request_lines(content, expected): + assert check_batch_file_upload("batch.jsonl", content, None, max_records=RECORD_LIMIT_3) == expected + + +def test_record_limit_applies_to_passthrough_rows(): + content = (NATIVE_VERTEX_LINE + b"\n") * 4 + assert check_batch_file_upload( + "batch.jsonl", content, None, PASSTHROUGH_BATCH_LINE_SHAPE, RECORD_LIMIT_3 + ) == BatchFileTooManyRecords(limit=RECORD_LIMIT_3) diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py index 84cf4ea7c32..5b7b5514a0c 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py @@ -4224,6 +4224,133 @@ def test_create_file_batch_under_max_batch_file_size_mb_forwards(monkeypatch, ll assert len(forwarded_calls) == 1 +FIXED_NOW: Final = 20_000 * 86400 + 3600 + 15 + + +def _pin_file_usage_clock(monkeypatch) -> None: + from functools import partial + + from litellm.proxy.openai_files_endpoints import file_usage_caps + from litellm.proxy.openai_files_endpoints import files_endpoints as fe + + monkeypatch.setattr( + fe, + "enforce_batch_file_upload_limit", + partial(file_usage_caps.enforce_batch_file_upload_limit, clock=lambda: FIXED_NOW), + ) + monkeypatch.setattr( + fe, + "enforce_file_download_limit", + partial(file_usage_caps.enforce_file_download_limit, clock=lambda: FIXED_NOW), + ) + + +def _upload_batch(content: bytes): + return client.post( + "/v1/files", + files={"file": ("batch.jsonl", content, "application/jsonl")}, + data={"purpose": "batch"}, + headers={"Authorization": "Bearer test-key"}, + ) + + +def test_create_file_batch_over_max_batch_file_records_rejected_before_forwarding(monkeypatch, llm_router: Router): + import litellm.proxy.proxy_server as ps + + forwarded_calls = _setup_batch_upload_endpoint(monkeypatch, llm_router) + monkeypatch.setitem(ps.general_settings, "max_batch_file_records", 2) + + try: + over = _upload_batch(VALID_BATCH_LINE * 3) + at_limit = _upload_batch(VALID_BATCH_LINE * 2) + finally: + _teardown_batch_upload_endpoint() + + assert over.status_code == 413, over.text + error = over.json()["error"] + assert error["param"] == "file" + assert "max_batch_file_records of 2 set in general_settings" in error["message"] + assert at_limit.status_code == 200, at_limit.text + assert len(forwarded_calls) == 1 + + +def test_create_file_batch_uploads_over_daily_limit_get_429_and_only_valid_files_count( + monkeypatch, llm_router: Router +): + import litellm.proxy.proxy_server as ps + from litellm.proxy._types import LitellmUserRoles + + forwarded_calls = _setup_batch_upload_endpoint(monkeypatch, llm_router) + _pin_file_usage_clock(monkeypatch) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + api_key="hashed-caller-key", + user_role=LitellmUserRoles.INTERNAL_USER, + user_id="test-user", + metadata={"max_batch_file_uploads_per_day": 2}, + ) + + try: + invalid = _upload_batch(b"not json\n") + allowed = [_upload_batch(VALID_BATCH_LINE) for _ in range(2)] + rejected = _upload_batch(VALID_BATCH_LINE) + finally: + _teardown_batch_upload_endpoint() + + assert invalid.status_code == 400, invalid.text + assert [response.status_code for response in allowed] == [200, 200] + assert rejected.status_code == 429, rejected.text + assert rejected.headers["retry-after"] == str(86400 - 3615) + error = rejected.json()["error"] + assert error["type"] == "rate_limit_error" + assert "max_batch_file_uploads_per_day is 2 for this key (set in this key's metadata)" in error["message"] + assert len(forwarded_calls) == 2 + + +def test_get_file_content_over_per_minute_download_limit_gets_429_per_file(monkeypatch, llm_router: Router): + import litellm.proxy.proxy_server as ps + from litellm.proxy._types import LitellmUserRoles + + setup_proxy_logging_object(monkeypatch, llm_router) + monkeypatch.setattr("litellm.proxy.proxy_server.master_key", None) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None) + monkeypatch.setitem(ps.general_settings, "max_file_downloads_per_minute", 2) + _pin_file_usage_clock(monkeypatch) + provider_calls: list = [] + + async def _mock_afile_content(**kwargs): + provider_calls.append(kwargs["file_id"]) + + async def _stream(): + yield b"output" + + return FileContentStreamingResult(stream_iterator=_stream(), headers={"content-length": "6"}) + + monkeypatch.setattr(litellm, "afile_content", _mock_afile_content) + monkeypatch.setattr( + "litellm.proxy.openai_files_endpoints.files_endpoints.handle_model_based_routing", + AsyncMock(return_value=(False, None, None, None)), + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + api_key="hashed-caller-key", user_role=LitellmUserRoles.INTERNAL_USER, user_id="test-user" + ) + + try: + same_file = [ + client.get("/v1/files/file-out/content", headers={"Authorization": "Bearer test-key"}) for _ in range(3) + ] + other_file = client.get("/v1/files/file-other/content", headers={"Authorization": "Bearer test-key"}) + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + assert [response.status_code for response in same_file] == [200, 200, 429] + assert same_file[2].headers["retry-after"] == "45" + error = same_file[2].json()["error"] + assert error["type"] == "rate_limit_error" + assert "Download limit reached for file file-out" in error["message"] + assert other_file.status_code == 200, other_file.text + assert provider_calls == ["file-out", "file-out", "file-other"] + + def test_create_file_batch_wrong_extension_rejected_before_forwarding(monkeypatch, llm_router: Router): forwarded_calls = _setup_batch_upload_endpoint(monkeypatch, llm_router) diff --git a/tests/unit/proxy/openai_files_endpoints/__init__.py b/tests/unit/proxy/openai_files_endpoints/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py b/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py new file mode 100644 index 00000000000..bbf9b380c98 --- /dev/null +++ b/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py @@ -0,0 +1,182 @@ +from typing import Final + +import pytest + +from litellm.caching.dual_cache import DualCache +from litellm.proxy._types import ProxyException, UserAPIKeyAuth +from litellm.proxy.openai_files_endpoints.file_usage_caps import ( + FileUsageLimit, + ScopedFileUsageLimit, + batch_file_record_limit, + consume_file_usage, + enforce_batch_file_upload_limit, + enforce_file_download_limit, + resolve_scoped_limits, +) +from litellm.proxy.utils import InternalUsageCache + +DAY: Final = 86400 +MIDDAY: Final = 20_000 * DAY + DAY / 2 + + +def _cache() -> InternalUsageCache: + return InternalUsageCache(dual_cache=DualCache()) + + +def _scoped(scope, scope_id, value, source="key", setting="max_batch_file_uploads_per_day"): + return ScopedFileUsageLimit( + scope=scope, scope_id=scope_id, limit=FileUsageLimit(setting=setting, value=value, source=source) + ) + + +@pytest.mark.parametrize( + "key_metadata, team_metadata, general_settings, expected", + [ + ({}, {}, {}, None), + ({}, {}, {"max_batch_file_records": 50}, FileUsageLimit("max_batch_file_records", 50, "general_settings")), + ( + {"max_batch_file_records": 80}, + {}, + {"max_batch_file_records": 50}, + FileUsageLimit("max_batch_file_records", 80, "key"), + ), + ( + {"max_batch_file_records": 80}, + {"max_batch_file_records": 30}, + {}, + FileUsageLimit("max_batch_file_records", 30, "team"), + ), + ( + {}, + {"max_batch_file_records": 90}, + {"max_batch_file_records": 50}, + FileUsageLimit("max_batch_file_records", 50, "general_settings"), + ), + ({"max_batch_file_records": 0}, {"max_batch_file_records": "lots"}, {}, None), + ({"max_batch_file_records": "40"}, {}, {}, FileUsageLimit("max_batch_file_records", 40, "key")), + ], +) +def test_record_limit_is_the_lowest_applicable_limit(key_metadata, team_metadata, general_settings, expected): + caller: Final = UserAPIKeyAuth(api_key="hashed", team_id="t1", metadata=key_metadata, team_metadata=team_metadata) + assert batch_file_record_limit(caller, general_settings) == expected + + +@pytest.mark.parametrize( + "caller, general_settings, expected", + [ + (UserAPIKeyAuth(api_key="hashed", team_id="t1"), {}, ()), + ( + UserAPIKeyAuth(api_key="hashed", team_id="t1"), + {"max_batch_file_uploads_per_day": 5}, + (_scoped("key", "hashed", 5, "general_settings"),), + ), + ( + UserAPIKeyAuth( + api_key="hashed", + team_id="t1", + metadata={"max_batch_file_uploads_per_day": 9}, + team_metadata={"max_batch_file_uploads_per_day": 20}, + ), + {"max_batch_file_uploads_per_day": 5}, + (_scoped("key", "hashed", 9, "key"), _scoped("team", "t1", 20, "team")), + ), + ( + UserAPIKeyAuth(api_key="hashed", team_metadata={"max_batch_file_uploads_per_day": 20}), + {}, + (), + ), + ], +) +def test_counter_scopes_pair_each_limit_with_its_own_counter(caller, general_settings, expected): + assert resolve_scoped_limits(caller, general_settings, "max_batch_file_uploads_per_day") == expected + + +async def test_a_scope_admits_exactly_its_limit_per_window_and_resets_on_the_next(): + cache: Final = _cache() + limits: Final = (_scoped("key", "hashed", 2),) + + outcomes: Final = [await consume_file_usage(cache, limits, DAY, "", MIDDAY + i) for i in range(3)] + next_window: Final = await consume_file_usage(cache, limits, DAY, "", MIDDAY + DAY) + + assert outcomes[:2] == [None, None] + assert outcomes[2] is not None + assert outcomes[2].limit == limits[0] + assert outcomes[2].retry_after_seconds == DAY / 2 - 2 + assert next_window is None + + +async def test_a_rejection_on_one_scope_gives_back_the_slot_it_took_on_the_other(): + cache: Final = _cache() + key_scope: Final = _scoped("key", "hashed", 3) + team_scope: Final = _scoped("team", "t1", 1, "team") + + first: Final = await consume_file_usage(cache, (key_scope, team_scope), DAY, "", MIDDAY) + rejected: Final = [await consume_file_usage(cache, (key_scope, team_scope), DAY, "", MIDDAY) for _ in range(5)] + key_only: Final = [await consume_file_usage(cache, (key_scope,), DAY, "", MIDDAY) for _ in range(3)] + + assert first is None + assert {outcome.limit for outcome in rejected if outcome is not None} == {team_scope} + assert len([outcome for outcome in rejected if outcome is not None]) == 5 + assert key_only[:2] == [None, None] + assert key_only[2] is not None and key_only[2].limit == key_scope + + +async def test_a_rejection_does_not_use_up_the_scope_that_rejected_it(): + cache: Final = _cache() + limits: Final = (_scoped("key", "hashed", 1),) + + await consume_file_usage(cache, limits, DAY, "", MIDDAY) + for _ in range(4): + await consume_file_usage(cache, limits, DAY, "", MIDDAY) + raised_limit: Final = (_scoped("key", "hashed", 2),) + + assert await consume_file_usage(cache, raised_limit, DAY, "", MIDDAY) is None + + +async def test_download_counters_are_per_file(): + cache: Final = _cache() + limits: Final = (_scoped("key", "hashed", 1, setting="max_file_downloads_per_minute"),) + + assert await consume_file_usage(cache, limits, 60, "file-a", MIDDAY) is None + assert await consume_file_usage(cache, limits, 60, "file-a", MIDDAY) is not None + assert await consume_file_usage(cache, limits, 60, "file-b", MIDDAY) is None + + +async def test_upload_limit_error_names_the_setting_value_scope_and_reset(): + cache: Final = _cache() + caller: Final = UserAPIKeyAuth(api_key="hashed", team_id="t1", team_metadata={"max_batch_file_uploads_per_day": 1}) + + await enforce_batch_file_upload_limit(cache, caller, {}, clock=lambda: MIDDAY) + with pytest.raises(ProxyException) as exc: + await enforce_batch_file_upload_limit(cache, caller, {}, clock=lambda: MIDDAY + 100) + + assert exc.value.code == "429" + assert exc.value.type == "rate_limit_error" + assert exc.value.headers == {"retry-after": str(DAY // 2 - 100)} + assert "max_batch_file_uploads_per_day is 1 for team t1" in exc.value.message + assert "this team's metadata" in exc.value.message + + +async def test_download_limit_error_names_the_file_and_general_settings_default(): + cache: Final = _cache() + caller: Final = UserAPIKeyAuth(api_key="hashed") + general_settings: Final = {"max_file_downloads_per_minute": 2} + + for _ in range(2): + await enforce_file_download_limit(cache, caller, general_settings, "file-a", clock=lambda: MIDDAY + 15) + with pytest.raises(ProxyException) as exc: + await enforce_file_download_limit(cache, caller, general_settings, "file-a", clock=lambda: MIDDAY + 15) + + assert exc.value.code == "429" + assert exc.value.headers == {"retry-after": "45"} + assert "file-a" in exc.value.message + assert "max_file_downloads_per_minute is 2 for this key (set in general_settings)" in exc.value.message + + +async def test_no_configured_limit_never_rejects(): + cache: Final = _cache() + caller: Final = UserAPIKeyAuth(api_key="hashed", team_id="t1") + + for _ in range(50): + await enforce_batch_file_upload_limit(cache, caller, {}, clock=lambda: MIDDAY) + await enforce_file_download_limit(cache, caller, {}, "file-a", clock=lambda: MIDDAY) diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 4e64a98dd65..a1709fc808b 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -28203,11 +28203,21 @@ export interface components { * @description require a key for all calls to proxy */ master_key?: string | null; + /** + * Max Batch File Records + * @description max records (non-blank lines) per batch input file for /v1/files uploads with purpose=batch, applied per key. A key or team can carry its own value in metadata, set by a proxy admin, and the lowest applicable value wins. Unset means no limit + */ + max_batch_file_records?: number | null; /** * Max Batch File Size Mb * @description max batch input file size in MB for /v1/files uploads with purpose=batch, if a file is larger than this size it will be rejected before being forwarded to the provider */ max_batch_file_size_mb?: number | null; + /** + * Max Batch File Uploads Per Day + * @description max /v1/files uploads with purpose=batch per key per UTC day. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit + */ + max_batch_file_uploads_per_day?: number | null; /** * Max Failed Login Attempts Per Source * @description Failed Admin UI sign-in attempts allowed from one source address, across every username, within `failed_login_window_seconds`. One more blocks that address for `failed_login_block_seconds`. Half this value, rounded down but at least 1, is the allowance for one username from that address; one more blocks that address for that username only, and its further failures stop counting toward the address limit, so a script stuck on one account does not block everyone behind a shared address. The per-address limit is only enforced when `trusted_proxy_ranges` is set: to the proxies in front of LiteLLM, or to an empty list when clients connect directly. Left unset, the peer address may be a shared ingress and only the per-username half runs. IPv6 addresses are grouped by /64. Set under `general_settings` in config.yaml. Defaults to 10 @@ -28220,6 +28230,11 @@ export interface components { max_failed_login_attempts_per_source_overrides?: { [key: string]: number; } | null; + /** + * Max File Downloads Per Minute + * @description max GET /v1/files/{file_id}/content calls per key per file per minute. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit + */ + max_file_downloads_per_minute?: number | null; /** * Max File Size Mb * @description max file size in MB for /v1/files uploads, for any purpose, if a file is larger than this size it will be rejected before being forwarded to the provider From 3b5f1bb122f8d7f38d3f095a89b02e48fc8d9958 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 17:09:08 -0700 Subject: [PATCH 2/4] fix(proxy): keep file usage counters in their own store and gate batch limits on user creation --- .../internal_user_endpoints.py | 4 ++ .../management_helpers/bulk_user_creation.py | 3 + .../openai_files_endpoints/files_endpoints.py | 4 +- litellm/proxy/utils.py | 2 + .../test_internal_user_endpoints.py | 63 +++++++++++++++++++ .../test_bulk_user_creation.py | 31 +++++++++ .../test_files_endpoint.py | 41 ++++++++++++ 7 files changed, 146 insertions(+), 2 deletions(-) diff --git a/litellm/proxy/management_endpoints/internal_user_endpoints.py b/litellm/proxy/management_endpoints/internal_user_endpoints.py index 59d8dd821d8..6bd92cd78df 100644 --- a/litellm/proxy/management_endpoints/internal_user_endpoints.py +++ b/litellm/proxy/management_endpoints/internal_user_endpoints.py @@ -36,6 +36,7 @@ from litellm.proxy.auth.auth_checks import ( get_team_object, get_user_object, ) +from litellm.proxy.auth.auth_utils import enforce_batch_limits_are_admin_only from litellm.proxy.auth.password_policy import ( validate_password_not_breached, validate_password_policy, @@ -581,6 +582,9 @@ async def new_user( detail=f"Only proxy admins can create administrative users (proxy_admin, proxy_admin_viewer). Attempted to create user with role: {data.user_role}. Your role: {user_api_key_dict.user_role}", ) + if data.auto_create_key and isinstance(user_api_key_dict, UserAPIKeyAuth): + enforce_batch_limits_are_admin_only(data, None, user_api_key_dict, "key") + _check_permissions_caller_permission( data=data, user_api_key_dict=user_api_key_dict, diff --git a/litellm/proxy/management_helpers/bulk_user_creation.py b/litellm/proxy/management_helpers/bulk_user_creation.py index ec8fd312766..ef7ba7eb772 100644 --- a/litellm/proxy/management_helpers/bulk_user_creation.py +++ b/litellm/proxy/management_helpers/bulk_user_creation.py @@ -29,6 +29,7 @@ from litellm.proxy._types import ( UserAPIKeyAuth, ) from litellm.proxy.auth.auth_checks import invalidate_team_member_spend_state +from litellm.proxy.auth.auth_utils import enforce_batch_limits_are_admin_only from litellm.proxy.auth.litellm_license import LicenseCheck from litellm.proxy.common_utils.timezone_utils import get_budget_reset_time from litellm.proxy.db.exception_handler import PrismaDBExceptionHandler @@ -215,6 +216,8 @@ def _row_error(item: BulkNewUserItem, user_api_key_dict: UserAPIKeyAuth) -> str try: validate_budget_duration(item.budget_duration) _check_permissions_caller_permission(data=item, user_api_key_dict=user_api_key_dict) + if item.auto_create_key: + enforce_batch_limits_are_admin_only(item, None, user_api_key_dict, "key") except Exception as exc: # noqa: BLE001 # any validation failure is reported on this row only return _error_message(exc) return None diff --git a/litellm/proxy/openai_files_endpoints/files_endpoints.py b/litellm/proxy/openai_files_endpoints/files_endpoints.py index 83b3433130e..5e52fe1641b 100644 --- a/litellm/proxy/openai_files_endpoints/files_endpoints.py +++ b/litellm/proxy/openai_files_endpoints/files_endpoints.py @@ -682,7 +682,7 @@ async def create_file( if batch_file_failure is not None: raise_batch_file_validation_failure(batch_file_failure) await enforce_batch_file_upload_limit( - proxy_logging_obj.internal_usage_cache, user_api_key_dict, general_settings + proxy_logging_obj.file_usage_cache, user_api_key_dict, general_settings ) data = {"passthrough": True} if passthrough else {} @@ -988,7 +988,7 @@ async def get_file_content( managed_files_obj=proxy_logging_obj.get_proxy_hook("managed_files"), ) await enforce_file_download_limit( - proxy_logging_obj.internal_usage_cache, user_api_key_dict, general_settings, file_id + proxy_logging_obj.file_usage_cache, user_api_key_dict, general_settings, file_id ) # Include original request and headers in the data diff --git a/litellm/proxy/utils.py b/litellm/proxy/utils.py index 1fca50e24c9..203f9098e06 100644 --- a/litellm/proxy/utils.py +++ b/litellm/proxy/utils.py @@ -1206,6 +1206,7 @@ class ProxyLogging: self.internal_usage_cache: InternalUsageCache = InternalUsageCache( dual_cache=DualCache(default_in_memory_ttl=1) # ping redis cache every 1s ) + self.file_usage_cache: Final = InternalUsageCache(dual_cache=DualCache()) self.max_parallel_request_limiter = _PROXY_MaxParallelRequestsHandler(self.internal_usage_cache) self.cache_control_check = _PROXY_CacheControlCheck() self.alerting: list[str] | None = None @@ -1347,6 +1348,7 @@ class ProxyLogging: if redis_cache is not None: self.internal_usage_cache.dual_cache.redis_cache = redis_cache + self.file_usage_cache.dual_cache.redis_cache = redis_cache self.db_spend_update_writer.redis_update_buffer.redis_cache = redis_cache self.db_spend_update_writer.pod_lock_manager.redis_cache = redis_cache diff --git a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py index c663e63414c..75a98ff00a1 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_internal_user_endpoints.py @@ -1241,6 +1241,69 @@ async def test_new_user_admin_can_set_permissions(mocker): assert result is not None +@pytest.mark.asyncio +@pytest.mark.parametrize( + "limit_key", + ["max_batch_file_records", "max_batch_file_uploads_per_day", "max_file_downloads_per_minute"], +) +async def test_new_user_only_proxy_admin_sets_batch_limits_on_the_created_key(mocker, limit_key): + from litellm.proxy.management_endpoints.internal_user_endpoints import new_user + + mock_prisma_client = mocker.MagicMock() + + async def mock_count(*args, **kwargs): + return 5 + + mock_prisma_client.db.litellm_usertable.count = mock_count + + async def mock_check(*_args, **_kwargs): + return None + + mocker.patch( + "litellm.proxy.management_endpoints.internal_user_endpoints._check_duplicate_user_email", + mock_check, + ) + mocker.patch( + "litellm.proxy.management_endpoints.internal_user_endpoints._check_duplicate_user_id", + mock_check, + ) + mock_license_check = mocker.MagicMock() + mock_license_check.is_over_limit.return_value = False + mocker.patch("litellm.proxy.proxy_server.prisma_client", mock_prisma_client) + mocker.patch("litellm.proxy.proxy_server._license_check", mock_license_check) + + created_with: list[dict[str, object]] = [] + + async def stub_helper(**kwargs): + created_with.append(kwargs) + return {"user_id": "alice", "key": "sk-alice", "expires": None} + + mocker.patch( + "litellm.proxy.management_endpoints.internal_user_endpoints.generate_key_helper_fn", + stub_helper, + ) + org_admin = UserAPIKeyAuth(user_id="org-admin", user_role=LitellmUserRoles.ORG_ADMIN) + admin = UserAPIKeyAuth(user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN) + + def request(auto_create_key: bool) -> NewUserRequest: + return NewUserRequest( + user_email="alice@example.com", + user_role=LitellmUserRoles.INTERNAL_USER, + metadata={limit_key: 1000}, + auto_create_key=auto_create_key, + ) + + with pytest.raises(ProxyException) as exc_info: + await new_user(data=request(auto_create_key=True), user_api_key_dict=org_admin) + assert str(exc_info.value.code) == "403" + assert f"Only proxy admins can set {limit_key} on a key" in str(exc_info.value.message) + assert created_with == [] + + await new_user(data=request(auto_create_key=False), user_api_key_dict=org_admin) + await new_user(data=request(auto_create_key=True), user_api_key_dict=admin) + assert [call["metadata"] for call in created_with] == [{limit_key: 1000}, {limit_key: 1000}] + + @pytest.mark.asyncio async def test_update_single_user_non_admin_permissions_rejected(mocker): """`_update_single_user_helper` rejects a non-admin when `permissions` diff --git a/tests/test_litellm/proxy/management_helpers/test_bulk_user_creation.py b/tests/test_litellm/proxy/management_helpers/test_bulk_user_creation.py index b5349fc2387..cf1ac61c93a 100644 --- a/tests/test_litellm/proxy/management_helpers/test_bulk_user_creation.py +++ b/tests/test_litellm/proxy/management_helpers/test_bulk_user_creation.py @@ -390,6 +390,37 @@ async def test_non_admin_cannot_create_admin_users_but_other_rows_proceed(): assert set(prisma.db.litellm_usertable.rows) == {"u2"} +@pytest.mark.asyncio +async def test_only_proxy_admin_sets_batch_limits_on_a_created_key(): + limits = {"max_file_downloads_per_minute": 1000} + calls: list[dict[str, object]] = [] + + async def generate_key(**kwargs: object) -> dict[str, object]: + calls.append(kwargs) + return {"token": f"sk-{kwargs['user_id']}"} + + rows = [ + {"user_id": "u1", "auto_create_key": True, "metadata": limits}, + {"user_id": "u2", "auto_create_key": False, "metadata": limits}, + {"user_id": "u3", "auto_create_key": True}, + ] + + prisma = _FakePrisma() + response = await _run(prisma, rows, caller=INTERNAL, generate_key=generate_key) + + assert [r.success for r in response.data] == [False, True, True] + assert "Only proxy admins can set max_file_downloads_per_minute on a key" in (response.data[0].error or "") + assert set(prisma.db.litellm_usertable.rows) == {"u2", "u3"} + assert [call["user_id"] for call in calls] == ["u3"] + + admin_prisma = _FakePrisma() + admin_response = await _run(admin_prisma, rows[:1], caller=ADMIN, generate_key=generate_key) + + assert [r.key for r in admin_response.data] == ["sk-u1"] + assert calls[-1]["user_id"] == "u1" + assert calls[-1]["metadata"] == limits + + @pytest.mark.asyncio async def test_license_is_checked_once_against_the_whole_batch(): prisma = _FakePrisma() diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py index 5b7b5514a0c..01eed0228a5 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py @@ -4351,6 +4351,47 @@ def test_get_file_content_over_per_minute_download_limit_gets_429_per_file(monke assert provider_calls == ["file-out", "file-out", "file-other"] +def test_download_counters_for_many_files_do_not_evict_rate_limit_state(monkeypatch, llm_router: Router): + import litellm.proxy.proxy_server as ps + from litellm.proxy._types import LitellmUserRoles + + proxy_logging = setup_proxy_logging_object(monkeypatch, llm_router) + monkeypatch.setattr("litellm.proxy.proxy_server.master_key", None) + monkeypatch.setattr("litellm.proxy.proxy_server.prisma_client", None) + monkeypatch.setitem(ps.general_settings, "max_file_downloads_per_minute", 1000) + _pin_file_usage_clock(monkeypatch) + + async def _mock_afile_content(**kwargs): + async def _stream(): + yield b"output" + + return FileContentStreamingResult(stream_iterator=_stream(), headers={"content-length": "6"}) + + monkeypatch.setattr(litellm, "afile_content", _mock_afile_content) + monkeypatch.setattr( + "litellm.proxy.openai_files_endpoints.files_endpoints.handle_model_based_routing", + AsyncMock(return_value=(False, None, None, None)), + ) + app.dependency_overrides[ps.user_api_key_auth] = lambda: UserAPIKeyAuth( + api_key="hashed-caller-key", user_role=LitellmUserRoles.INTERNAL_USER, user_id="test-user" + ) + rate_limit_cache: Final = proxy_logging.internal_usage_cache.dual_cache + rate_limit_window_key: Final = "{api_key:hashed-caller-key}:window" + rate_limit_cache.set_cache(key=rate_limit_window_key, value=12345, local_only=True, ttl=60) + distinct_files: Final = rate_limit_cache.in_memory_cache.max_size_in_memory + 1 + + try: + statuses = [ + client.get(f"/v1/files/file-{index}/content", headers={"Authorization": "Bearer test-key"}).status_code + for index in range(distinct_files) + ] + finally: + app.dependency_overrides.pop(ps.user_api_key_auth, None) + + assert set(statuses) == {200} + assert rate_limit_cache.get_cache(key=rate_limit_window_key, local_only=True) == 12345 + + def test_create_file_batch_wrong_extension_rejected_before_forwarding(monkeypatch, llm_router: Router): forwarded_calls = _setup_batch_upload_endpoint(monkeypatch, llm_router) From cc3316aa32c1ae7a1c1351f3e4ceb05d02ae8acd Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 17:46:33 -0700 Subject: [PATCH 3/4] fix(proxy): count keyless JWT callers per user, require positive file caps, and take the upload slot after request validation --- litellm/proxy/_types.py | 9 ++++--- .../openai_files_endpoints/file_usage_caps.py | 26 ++++++++++++------- .../openai_files_endpoints/files_endpoints.py | 8 +++--- .../test_files_endpoint.py | 7 +++++ tests/test_litellm/proxy/test__types.py | 12 +++++++++ .../test_file_usage_caps.py | 21 +++++++++++++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 6 ++--- 7 files changed, 71 insertions(+), 18 deletions(-) diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 29fa7c5c912..77c45ad544e 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -2784,15 +2784,18 @@ class ConfigGeneralSettings(LiteLLMPydanticObjectBase): ) max_batch_file_records: int | None = Field( None, - description="max records (non-blank lines) per batch input file for /v1/files uploads with purpose=batch, applied per key. A key or team can carry its own value in metadata, set by a proxy admin, and the lowest applicable value wins. Unset means no limit", + gt=0, + description="max records (non-blank lines) per batch input file for /v1/files uploads with purpose=batch, applied per key. A key's metadata can override it and a team's metadata adds a team cap on top, both set by a proxy admin; the lower of the key's value and the team's value wins. Unset means no limit", ) max_batch_file_uploads_per_day: int | None = Field( None, - description="max /v1/files uploads with purpose=batch per key per UTC day. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit", + gt=0, + description="max /v1/files uploads with purpose=batch per key (per user for JWT callers) per UTC day. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit", ) max_file_downloads_per_minute: int | None = Field( None, - description="max GET /v1/files/{file_id}/content calls per key per file per minute. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit", + gt=0, + description="max GET /v1/files/{file_id}/content calls per key (per user for JWT callers) per file per minute. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit", ) max_file_size_mb: int | None = Field( None, diff --git a/litellm/proxy/openai_files_endpoints/file_usage_caps.py b/litellm/proxy/openai_files_endpoints/file_usage_caps.py index c8555ac3cbc..2945d8b54d7 100644 --- a/litellm/proxy/openai_files_endpoints/file_usage_caps.py +++ b/litellm/proxy/openai_files_endpoints/file_usage_caps.py @@ -25,7 +25,7 @@ FileUsageSetting: TypeAlias = Literal[ "max_file_downloads_per_minute", ] LimitSource: TypeAlias = Literal["key", "team", "general_settings"] -CounterScope: TypeAlias = Literal["key", "team"] +CounterScope: TypeAlias = Literal["key", "user", "team"] _COUNTER_PREFIX: Final = "litellm:file_usage" _DAY_SECONDS: Final = 24 * 60 * 60 @@ -65,9 +65,8 @@ def _read_limit( return FileUsageLimit(setting=setting, value=_LIMIT_ADAPTER.validate_python(raw), source=source) except ValidationError: verbose_proxy_logger.warning( - "Ignoring invalid %s value %r in %s; expected a positive integer", + "Ignoring invalid %s in %s; expected a positive integer", setting, - raw, source, ) return None @@ -103,17 +102,24 @@ def batch_file_record_limit( return min(applicable, key=lambda limit: limit.value, default=None) +def _caller_counter(user_api_key_dict: UserAPIKeyAuth, limit: FileUsageLimit | None) -> ScopedFileUsageLimit | None: + if limit is None: + return None + if user_api_key_dict.api_key: + return ScopedFileUsageLimit(scope="key", scope_id=user_api_key_dict.api_key, limit=limit) + if user_api_key_dict.user_id: + return ScopedFileUsageLimit(scope="user", scope_id=user_api_key_dict.user_id, limit=limit) + return None + + def resolve_scoped_limits( user_api_key_dict: UserAPIKeyAuth, general_settings: Mapping[str, object], setting: FileUsageSetting, ) -> tuple[ScopedFileUsageLimit, ...]: - key_limit: Final = _key_limit(user_api_key_dict, general_settings, setting) team_limit: Final = _team_limit(user_api_key_dict, setting) candidates: Final = ( - ScopedFileUsageLimit(scope="key", scope_id=user_api_key_dict.api_key, limit=key_limit) - if key_limit is not None and user_api_key_dict.api_key - else None, + _caller_counter(user_api_key_dict, _key_limit(user_api_key_dict, general_settings, setting)), ScopedFileUsageLimit(scope="team", scope_id=user_api_key_dict.team_id, limit=team_limit) if team_limit is not None and user_api_key_dict.team_id else None, @@ -172,17 +178,19 @@ def describe_limit_source(source: LimitSource) -> str: case "general_settings": return "in general_settings" case _: - assert_never(source) + return assert_never(source) def _describe_scope(scoped: ScopedFileUsageLimit) -> str: match scoped.scope: case "key": return "this key" + case "user": + return f"user {scoped.scope_id}" case "team": return f"team {scoped.scope_id}" case _: - assert_never(scoped.scope) + return assert_never(scoped.scope) def _raise_limit_exceeded(exceeded: FileUsageLimitExceeded, what_ran_out: str, when_it_resets: str) -> NoReturn: diff --git a/litellm/proxy/openai_files_endpoints/files_endpoints.py b/litellm/proxy/openai_files_endpoints/files_endpoints.py index 5e52fe1641b..8875c582f0a 100644 --- a/litellm/proxy/openai_files_endpoints/files_endpoints.py +++ b/litellm/proxy/openai_files_endpoints/files_endpoints.py @@ -681,9 +681,6 @@ async def create_file( ) if batch_file_failure is not None: raise_batch_file_validation_failure(batch_file_failure) - await enforce_batch_file_upload_limit( - proxy_logging_obj.file_usage_cache, user_api_key_dict, general_settings - ) data = {"passthrough": True} if passthrough else {} @@ -757,6 +754,11 @@ async def create_file( seconds=expires_after_seconds, ) + if purpose == "batch": + await enforce_batch_file_upload_limit( + proxy_logging_obj.file_usage_cache, user_api_key_dict, general_settings + ) + # Include original request and headers in the data data = await add_litellm_data_to_request( data=data, diff --git a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py index 01eed0228a5..25c62fc6dcc 100644 --- a/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py +++ b/tests/test_litellm/proxy/openai_files_endpoint/test_files_endpoint.py @@ -4291,12 +4291,19 @@ def test_create_file_batch_uploads_over_daily_limit_get_429_and_only_valid_files try: invalid = _upload_batch(b"not json\n") + malformed_expiry = client.post( + "/v1/files", + files={"file": ("batch.jsonl", VALID_BATCH_LINE, "application/jsonl")}, + data={"purpose": "batch", "expires_after[anchor]": "created_at"}, + headers={"Authorization": "Bearer test-key"}, + ) allowed = [_upload_batch(VALID_BATCH_LINE) for _ in range(2)] rejected = _upload_batch(VALID_BATCH_LINE) finally: _teardown_batch_upload_endpoint() assert invalid.status_code == 400, invalid.text + assert malformed_expiry.status_code == 400, malformed_expiry.text assert [response.status_code for response in allowed] == [200, 200] assert rejected.status_code == 429, rejected.text assert rejected.headers["retry-after"] == str(86400 - 3615) diff --git a/tests/test_litellm/proxy/test__types.py b/tests/test_litellm/proxy/test__types.py index b43a75d3323..d59cb184d9b 100644 --- a/tests/test_litellm/proxy/test__types.py +++ b/tests/test_litellm/proxy/test__types.py @@ -395,3 +395,15 @@ def test_mcp_metadata_rejects_unavailable_upstream_protocol(revision): for model in (NewMCPServerRequest, UpdateMCPServerRequest): with pytest.raises(ValidationError): model.model_validate(payload) + + +@pytest.mark.parametrize( + "setting", ["max_batch_file_records", "max_batch_file_uploads_per_day", "max_file_downloads_per_minute"] +) +def test_batch_file_caps_accept_only_positive_limits(setting): + from litellm.proxy._types import ConfigGeneralSettings + + for invalid in (0, -5): + with pytest.raises(ValidationError): + ConfigGeneralSettings.model_validate({setting: invalid}) + assert getattr(ConfigGeneralSettings.model_validate({setting: 3}), setting) == 3 diff --git a/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py b/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py index bbf9b380c98..0f7b77da89e 100644 --- a/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py +++ b/tests/unit/proxy/openai_files_endpoints/test_file_usage_caps.py @@ -85,6 +85,12 @@ def test_record_limit_is_the_lowest_applicable_limit(key_metadata, team_metadata {}, (), ), + ( + UserAPIKeyAuth(api_key=None, user_id="u1", team_id="t1"), + {"max_batch_file_uploads_per_day": 5}, + (_scoped("user", "u1", 5, "general_settings"),), + ), + (UserAPIKeyAuth(api_key=None), {"max_batch_file_uploads_per_day": 5}, ()), ], ) def test_counter_scopes_pair_each_limit_with_its_own_counter(caller, general_settings, expected): @@ -180,3 +186,18 @@ async def test_no_configured_limit_never_rejects(): for _ in range(50): await enforce_batch_file_upload_limit(cache, caller, {}, clock=lambda: MIDDAY) await enforce_file_download_limit(cache, caller, {}, "file-a", clock=lambda: MIDDAY) + + +async def test_a_caller_without_a_key_is_counted_per_user(): + cache: Final = _cache() + general_settings: Final = {"max_file_downloads_per_minute": 1} + jwt_caller: Final = UserAPIKeyAuth(api_key=None, user_id="u1") + + await enforce_file_download_limit(cache, jwt_caller, general_settings, "file-a", clock=lambda: MIDDAY) + await enforce_file_download_limit( + cache, UserAPIKeyAuth(api_key=None, user_id="u2"), general_settings, "file-a", clock=lambda: MIDDAY + ) + with pytest.raises(ProxyException) as exc: + await enforce_file_download_limit(cache, jwt_caller, general_settings, "file-a", clock=lambda: MIDDAY) + + assert "max_file_downloads_per_minute is 1 for user u1 (set in general_settings)" in exc.value.message diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index a1709fc808b..99ad9ac57e0 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -28205,7 +28205,7 @@ export interface components { master_key?: string | null; /** * Max Batch File Records - * @description max records (non-blank lines) per batch input file for /v1/files uploads with purpose=batch, applied per key. A key or team can carry its own value in metadata, set by a proxy admin, and the lowest applicable value wins. Unset means no limit + * @description max records (non-blank lines) per batch input file for /v1/files uploads with purpose=batch, applied per key. A key's metadata can override it and a team's metadata adds a team cap on top, both set by a proxy admin; the lower of the key's value and the team's value wins. Unset means no limit */ max_batch_file_records?: number | null; /** @@ -28215,7 +28215,7 @@ export interface components { max_batch_file_size_mb?: number | null; /** * Max Batch File Uploads Per Day - * @description max /v1/files uploads with purpose=batch per key per UTC day. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit + * @description max /v1/files uploads with purpose=batch per key (per user for JWT callers) per UTC day. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit */ max_batch_file_uploads_per_day?: number | null; /** @@ -28232,7 +28232,7 @@ export interface components { } | null; /** * Max File Downloads Per Minute - * @description max GET /v1/files/{file_id}/content calls per key per file per minute. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit + * @description max GET /v1/files/{file_id}/content calls per key (per user for JWT callers) per file per minute. A key's metadata can override it and a team's metadata adds a shared team cap, both set by a proxy admin. Unset means no limit */ max_file_downloads_per_minute?: number | null; /** From c13b47730d9da3a8c02eeb50be8d4a6d5ebb1e31 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 18:07:09 -0700 Subject: [PATCH 4/4] refactor(proxy): end file usage cap describers with an explicit return after the match --- litellm/proxy/openai_files_endpoints/file_usage_caps.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/openai_files_endpoints/file_usage_caps.py b/litellm/proxy/openai_files_endpoints/file_usage_caps.py index 2945d8b54d7..698ab71aa80 100644 --- a/litellm/proxy/openai_files_endpoints/file_usage_caps.py +++ b/litellm/proxy/openai_files_endpoints/file_usage_caps.py @@ -177,8 +177,7 @@ def describe_limit_source(source: LimitSource) -> str: return "in this team's metadata" case "general_settings": return "in general_settings" - case _: - return assert_never(source) + return assert_never(source) def _describe_scope(scoped: ScopedFileUsageLimit) -> str: @@ -189,8 +188,7 @@ def _describe_scope(scoped: ScopedFileUsageLimit) -> str: return f"user {scoped.scope_id}" case "team": return f"team {scoped.scope_id}" - case _: - return assert_never(scoped.scope) + return assert_never(scoped.scope) def _raise_limit_exceeded(exceeded: FileUsageLimitExceeded, what_ran_out: str, when_it_resets: str) -> NoReturn: