diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 5af21f6f3d4..b582b164609 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -2321,7 +2321,9 @@ class ProxyBaseLLMRequestProcessing: try: if general_settings.get("cancel_on_disconnect", False): - responses = await _await_llm_call_cancelling_on_disconnect(request, llm_responses, self.data) + responses = await _await_llm_call_cancelling_on_disconnect( # rebind-ok: assigned in exactly one of these two mutually exclusive branches + request, llm_responses, self.data + ) else: responses = await llm_responses finally: diff --git a/litellm/proxy/hooks/global_tag_rate_limits_hook.py b/litellm/proxy/hooks/global_tag_rate_limits_hook.py index be90e2ff2a4..c6a6ab556dd 100644 --- a/litellm/proxy/hooks/global_tag_rate_limits_hook.py +++ b/litellm/proxy/hooks/global_tag_rate_limits_hook.py @@ -155,9 +155,9 @@ _request_stash: Final[ContextVar[_GlobalTagRateLimitStash | None]] = ContextVar( def _claim_stash_for_data(data: Mapping[str, object]) -> _GlobalTagRateLimitStash: - stash = _request_stash.get() + stash = _request_stash.get() # rebind-ok: lazily initialized below if this ContextVar has never been set if stash is None: - stash = _GlobalTagRateLimitStash() + stash = _GlobalTagRateLimitStash() # rebind-ok: see above _request_stash.set(stash) owner_call_id: Final = data.get("litellm_call_id") if isinstance(owner_call_id, str): @@ -174,9 +174,7 @@ def _stash_for_call(litellm_call_id: str | None) -> _GlobalTagRateLimitStash | N return stash if litellm_call_id == stash.owner_litellm_call_id else None -def _call_id_from_kwargs(kwargs: object) -> str | None: - if not isinstance(kwargs, dict): - return None +def _call_id_from_kwargs(kwargs: Mapping[str, object]) -> str | None: call_id: Final = kwargs.get("litellm_call_id") return call_id if isinstance(call_id, str) else None @@ -507,7 +505,7 @@ class _PROXY_GlobalTagRateLimitsHook( # pyright: ignore[reportUnusedClass] # o return data async def async_release_disconnect_state_hook(self, request_data: Mapping[str, object]) -> None: - stash: Final = _stash_for_call(_call_id_from_kwargs(dict(request_data))) + stash: Final = _stash_for_call(_call_id_from_kwargs(request_data)) if stash is None or not stash.pending_concurrency_keys: return release_keys: Final = tuple(stash.pending_concurrency_keys) diff --git a/litellm/proxy/hooks/model_based_tag_rate_limits_hook.py b/litellm/proxy/hooks/model_based_tag_rate_limits_hook.py index cb7c14da99e..01e981a8407 100644 --- a/litellm/proxy/hooks/model_based_tag_rate_limits_hook.py +++ b/litellm/proxy/hooks/model_based_tag_rate_limits_hook.py @@ -242,12 +242,12 @@ def _entry_applies(entry: TagRateLimitEntry, tag_value: str, tags: Sequence[str] if entry.included_values is not None and tag_value not in entry.included_values: return False if entry.disabled_for is not None: - gate_value = _extract_identity(tags, entry.disabled_for.tag_id) - if gate_value is not None and gate_value in entry.disabled_for.values: + disabled_gate_value: Final = _extract_identity(tags, entry.disabled_for.tag_id) + if disabled_gate_value is not None and disabled_gate_value in entry.disabled_for.values: return False if entry.enabled_for is not None: - gate_value = _extract_identity(tags, entry.enabled_for.tag_id) - if gate_value is None or gate_value not in entry.enabled_for.values: + enabled_gate_value: Final = _extract_identity(tags, entry.enabled_for.tag_id) + if enabled_gate_value is None or enabled_gate_value not in entry.enabled_for.values: return False if entry.apply_to_key_alias is None: return True @@ -901,9 +901,9 @@ def _queue_pending_concurrency_reservations( model_call_details: Final = getattr(logging_obj, "model_call_details", None) if not isinstance(model_call_details, dict): return - pending = model_call_details.get(_PENDING_CONCURRENCY_KEYS_FIELD) + pending = model_call_details.get(_PENDING_CONCURRENCY_KEYS_FIELD) # rebind-ok: lazily initialized below when absent if pending is None: - pending = [] # mutable-ok: shared, request-scoped accumulator; see field's own docstring + pending = [] # mutable-ok: shared, request-scoped accumulator; see field's own docstring # rebind-ok: lazily initialized only when absent model_call_details[_PENDING_CONCURRENCY_KEYS_FIELD] = pending pending.extend(reservations) # mutable-ok: see comment above diff --git a/litellm/types/router.py b/litellm/types/router.py index 941abf79344..cc079a50c33 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -171,7 +171,7 @@ class TagRateLimitScope(BaseModel): # is silently ignored when constructing via __init__ (only takes # effect via model_validate), so mutating in place is the only way # this normalization reliably applies regardless of construction path. - object.__setattr__(self, "values", tuple(sorted(set(self.values)))) + object.__setattr__(self, "values", tuple(sorted(set(self.values)))) # mutable-ok: frozen before escaping return self @@ -281,11 +281,11 @@ class TagRateLimitEntry(BaseModel): # unsorted tuple would make config-order alone decide whether two # deployments' entries dedup to one shared bucket. if self.included_values is not None: - self.included_values = tuple(sorted(set(self.included_values))) + self.included_values = tuple(sorted(set(self.included_values))) # mutable-ok: frozen before escaping if self.excluded_values is not None: - self.excluded_values = tuple(sorted(set(self.excluded_values))) + self.excluded_values = tuple(sorted(set(self.excluded_values))) # mutable-ok: frozen before escaping if self.apply_to_key_alias is not None: - self.apply_to_key_alias = tuple(sorted(set(self.apply_to_key_alias))) + self.apply_to_key_alias = tuple(sorted(set(self.apply_to_key_alias))) # mutable-ok: frozen before escaping return self diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 018b98ef95b..5e55f6fd018 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -35112,6 +35112,8 @@ export interface components { }; /** TagRateLimitEntry */ TagRateLimitEntry: { + /** Apply To Key Alias */ + apply_to_key_alias?: string[] | null; disabled_for?: components["schemas"]["TagRateLimitScope"] | null; enabled_for?: components["schemas"]["TagRateLimitScope"] | null; /** Excluded Values */