From 0004fe1b9257b738d08c83ef5c06fa6b423d77ed Mon Sep 17 00:00:00 2001 From: Deepanshu Date: Tue, 11 Aug 2026 10:43:45 -0400 Subject: [PATCH] fix(rate-limiting): satisfy strict lint gate and regenerate stale schema.d.ts - Add the missing __init__ return annotation (ANN204) and drop the now-unused typing.Any import in favor of a concrete object fallback for the optional-otel-Span type alias (TID251), both newly over the ruff-strict budget. - Regenerate ui/litellm-dashboard/src/lib/http/schema.d.ts: the prior comment-trimming pass left it out of sync with the trimmed TagRateLimitEntry/TagRateLimits docstrings. --- litellm/proxy/hooks/tag_rate_limiter.py | 8 +++--- ui/litellm-dashboard/src/lib/http/schema.d.ts | 27 ++----------------- 2 files changed, 6 insertions(+), 29 deletions(-) diff --git a/litellm/proxy/hooks/tag_rate_limiter.py b/litellm/proxy/hooks/tag_rate_limiter.py index 5c7c77f0777..12c291abec6 100644 --- a/litellm/proxy/hooks/tag_rate_limiter.py +++ b/litellm/proxy/hooks/tag_rate_limiter.py @@ -5,7 +5,7 @@ import contextvars from collections.abc import Callable from dataclasses import dataclass from datetime import datetime -from typing import TYPE_CHECKING, Any, Literal +from typing import TYPE_CHECKING, Literal from litellm._logging import verbose_proxy_logger from litellm.caching.dual_cache import DualCache @@ -30,9 +30,9 @@ from litellm.types.utils import StandardLoggingPayload if TYPE_CHECKING: from opentelemetry.trace import Span as _Span - Span = _Span | Any + Span = _Span else: - Span = Any + Span = object _LimitUnit = Literal["tokens", "requests", "dollars", "concurrency"] _LIMIT_UNITS: tuple[_LimitUnit, ...] = ("tokens", "requests", "dollars", "concurrency") @@ -414,7 +414,7 @@ class _PROXY_TagRateLimiter(CustomLogger): self, internal_usage_cache: DualCache, time_provider: Callable[[], datetime] | None = None, - ): + ) -> None: self.internal_usage_cache = InternalUsageCache(dual_cache=internal_usage_cache) self._v3 = _PROXY_MaxParallelRequestsHandler_v3(self.internal_usage_cache, time_provider=time_provider) self._time_provider = time_provider or datetime.now diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index fed743190c6..ea1362b6542 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -35110,25 +35110,7 @@ export interface components { /** Tpm Limit */ tpm_limit?: number | null; }; - /** - * TagRateLimitEntry - * @description One tag-scoped limit: a caller-supplied tag value (identified by `tag_id`, - * e.g. `end_user_id` in a request tag like `end_user_id:user-123`) is capped - * at `limit` units per rolling `period_seconds`-second window. Bucketing is - * `epoch_second // period_seconds`, so `period_seconds=86400` resets at UTC - * midnight and `period_seconds=60` resets on real clock-minute boundaries. - * - * For a `concurrency_limits` entry specifically, `period_seconds` is not a - * window: it is a floor under the safety TTL a reserved in-flight slot - * self-heals after, in case a worker crashes before releasing it (the - * counter, not a window). The effective TTL is at least one hour regardless - * of this value, so a slow but genuinely still-running request never has - * its reservation expire out from under it; set this higher only if an - * even longer self-heal window is wanted. `concurrency_limits` also only - * supports chain-wide entries (declared identically by every deployment - * sharing a `model_name`) -- a divergent per-deployment value is dropped - * with a warning, not silently scoped to a subset of deployments. - */ + /** TagRateLimitEntry */ TagRateLimitEntry: { /** Limit */ limit: number; @@ -35152,12 +35134,7 @@ export interface components { /** Limits */ limits?: components["schemas"]["TagRateLimitEntry"][]; }; - /** - * TagRateLimits - * @description Per-chain/model-group tag rate limits, set under a deployment's - * `model_info.tag_rate_limits`. Each entry carries its own `tag_id`, so two - * entries of the same unit on the same chain can key by different tags. - */ + /** TagRateLimits */ TagRateLimits: { concurrency_limits?: components["schemas"]["TagRateLimitGroup"] | null; dollar_limits?: components["schemas"]["TagRateLimitGroup"] | null;