fix(rate-limiting): satisfy strict lint gate and regenerate stale schema.d.ts

- Add the missing __init__ return annotation (ANN204) and drop the
  now-unused typing.Any import in favor of a concrete object fallback
  for the optional-otel-Span type alias (TID251), both newly over the
  ruff-strict budget.
- Regenerate ui/litellm-dashboard/src/lib/http/schema.d.ts: the prior
  comment-trimming pass left it out of sync with the trimmed
  TagRateLimitEntry/TagRateLimits docstrings.
This commit is contained in:
Deepanshu 2026-08-11 10:43:45 -04:00
parent 11461f6e45
commit 0004fe1b92
2 changed files with 6 additions and 29 deletions

View file

@ -5,7 +5,7 @@ import contextvars
from collections.abc import Callable
from dataclasses import dataclass
from datetime import datetime
from typing import TYPE_CHECKING, Any, Literal
from typing import TYPE_CHECKING, Literal
from litellm._logging import verbose_proxy_logger
from litellm.caching.dual_cache import DualCache
@ -30,9 +30,9 @@ from litellm.types.utils import StandardLoggingPayload
if TYPE_CHECKING:
from opentelemetry.trace import Span as _Span
Span = _Span | Any
Span = _Span
else:
Span = Any
Span = object
_LimitUnit = Literal["tokens", "requests", "dollars", "concurrency"]
_LIMIT_UNITS: tuple[_LimitUnit, ...] = ("tokens", "requests", "dollars", "concurrency")
@ -414,7 +414,7 @@ class _PROXY_TagRateLimiter(CustomLogger):
self,
internal_usage_cache: DualCache,
time_provider: Callable[[], datetime] | None = None,
):
) -> None:
self.internal_usage_cache = InternalUsageCache(dual_cache=internal_usage_cache)
self._v3 = _PROXY_MaxParallelRequestsHandler_v3(self.internal_usage_cache, time_provider=time_provider)
self._time_provider = time_provider or datetime.now

View file

@ -35110,25 +35110,7 @@ export interface components {
/** Tpm Limit */
tpm_limit?: number | null;
};
/**
* TagRateLimitEntry
* @description One tag-scoped limit: a caller-supplied tag value (identified by `tag_id`,
* e.g. `end_user_id` in a request tag like `end_user_id:user-123`) is capped
* at `limit` units per rolling `period_seconds`-second window. Bucketing is
* `epoch_second // period_seconds`, so `period_seconds=86400` resets at UTC
* midnight and `period_seconds=60` resets on real clock-minute boundaries.
*
* For a `concurrency_limits` entry specifically, `period_seconds` is not a
* window: it is a floor under the safety TTL a reserved in-flight slot
* self-heals after, in case a worker crashes before releasing it (the
* counter, not a window). The effective TTL is at least one hour regardless
* of this value, so a slow but genuinely still-running request never has
* its reservation expire out from under it; set this higher only if an
* even longer self-heal window is wanted. `concurrency_limits` also only
* supports chain-wide entries (declared identically by every deployment
* sharing a `model_name`) -- a divergent per-deployment value is dropped
* with a warning, not silently scoped to a subset of deployments.
*/
/** TagRateLimitEntry */
TagRateLimitEntry: {
/** Limit */
limit: number;
@ -35152,12 +35134,7 @@ export interface components {
/** Limits */
limits?: components["schemas"]["TagRateLimitEntry"][];
};
/**
* TagRateLimits
* @description Per-chain/model-group tag rate limits, set under a deployment's
* `model_info.tag_rate_limits`. Each entry carries its own `tag_id`, so two
* entries of the same unit on the same chain can key by different tags.
*/
/** TagRateLimits */
TagRateLimits: {
concurrency_limits?: components["schemas"]["TagRateLimitGroup"] | null;
dollar_limits?: components["schemas"]["TagRateLimitGroup"] | null;