From f1805e231fce4c9925f51098d59a1f20bdeef047 Mon Sep 17 00:00:00 2001 From: Your Name Date: Mon, 18 May 2026 01:25:18 +0530 Subject: [PATCH] style: run black formatter on utils.py and __init__.py --- litellm/__init__.py | 1 + litellm/utils.py | 249 ++++++++++++++++++++++---------------------- 2 files changed, 125 insertions(+), 125 deletions(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 04e7e64cedd..b35b9627d7a 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1172,6 +1172,7 @@ _key_management_settings: KeyManagementSettings = KeyManagementSettings() # client must be imported immediately as it's used as a decorator at function definition time from .utils import client from .utils import retry_with_backoff, async_retry_with_backoff + # Note: Most other utils imports are lazy-loaded via __getattr__ to avoid loading utils.py # (which imports tiktoken) at import time diff --git a/litellm/utils.py b/litellm/utils.py index 4e1bdc687dc..c98371c5d53 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -54,7 +54,6 @@ import litellm.litellm_core_utils from typing import Callable, Any, Tuple, Type - # audio_utils.utils is lazy-loaded - only imported when needed for transcription calls import litellm.litellm_core_utils.json_validation_rule from litellm._internal_context import is_internal_call @@ -9728,165 +9727,165 @@ def __getattr__(name: str) -> Any: raise AttributeError(f"module {__name__!r} has no attribute {name!r}") - # ───────────────────────────────────────────── # Retry with Exponential Backoff Utility # Handles RateLimitError and ServiceUnavailableError # automatically with jitter to avoid thundering herd # ───────────────────────────────────────────── + def retry_with_backoff( - fn: Callable, - max_retries: int = 3, - base_delay: float = 1.0, - max_delay: float = 60.0, - backoff_factor: float = 2.0, - retry_on: Optional[Tuple[Type[Exception], ...]] = None, + fn: Callable, + max_retries: int = 3, + base_delay: float = 1.0, + max_delay: float = 60.0, + backoff_factor: float = 2.0, + retry_on: Optional[Tuple[Type[Exception], ...]] = None, ) -> Any: - """ - Retries a synchronous callable with exponential backoff and jitter. + """ + Retries a synchronous callable with exponential backoff and jitter. - For async callables, use async_retry_with_backoff instead. + For async callables, use async_retry_with_backoff instead. - Args: - fn: The callable to retry (e.g. lambda: litellm.completion(...)) - max_retries: Maximum number of retry attempts (default: 3) - base_delay: Initial delay in seconds (default: 1.0) - max_delay: Maximum delay cap in seconds (default: 60.0) - backoff_factor: Multiplier per retry (default: 2.0) - retry_on: Tuple of exception types to retry on. - Defaults to (RateLimitError, ServiceUnavailableError) + Args: + fn: The callable to retry (e.g. lambda: litellm.completion(...)) + max_retries: Maximum number of retry attempts (default: 3) + base_delay: Initial delay in seconds (default: 1.0) + max_delay: Maximum delay cap in seconds (default: 60.0) + backoff_factor: Multiplier per retry (default: 2.0) + retry_on: Tuple of exception types to retry on. + Defaults to (RateLimitError, ServiceUnavailableError) - Returns: - The return value of fn() on success. + Returns: + The return value of fn() on success. - Raises: - TypeError: If fn is a coroutine function (use async_retry_with_backoff). - The last exception if all retries are exhausted. + Raises: + TypeError: If fn is a coroutine function (use async_retry_with_backoff). + The last exception if all retries are exhausted. - Example: - import litellm - from litellm.utils import retry_with_backoff + Example: + import litellm + from litellm.utils import retry_with_backoff - response = retry_with_backoff( - lambda: litellm.completion( - model="gpt-4o", - messages=[{"role": "user", "content": "Hello"}] - ), - max_retries=5, - base_delay=2.0, - ) - """ - if inspect.iscoroutinefunction(fn): - raise TypeError( - "retry_with_backoff does not support async callables. " - "Use async_retry_with_backoff instead." - ) + response = retry_with_backoff( + lambda: litellm.completion( + model="gpt-4o", + messages=[{"role": "user", "content": "Hello"}] + ), + max_retries=5, + base_delay=2.0, + ) + """ + if inspect.iscoroutinefunction(fn): + raise TypeError( + "retry_with_backoff does not support async callables. " + "Use async_retry_with_backoff instead." + ) - if retry_on is None: - retry_on = (RateLimitError, ServiceUnavailableError) + if retry_on is None: + retry_on = (RateLimitError, ServiceUnavailableError) - last_exception: Optional[Exception] = None + last_exception: Optional[Exception] = None - for attempt in range(max_retries + 1): - try: - result = fn() - if inspect.iscoroutine(result): - result.close() - raise TypeError( - "retry_with_backoff received a coroutine from fn(). " - "Use async_retry_with_backoff for async callables." - ) - return result + for attempt in range(max_retries + 1): + try: + result = fn() + if inspect.iscoroutine(result): + result.close() + raise TypeError( + "retry_with_backoff received a coroutine from fn(). " + "Use async_retry_with_backoff for async callables." + ) + return result - except retry_on as e: - last_exception = e + except retry_on as e: + last_exception = e - if attempt == max_retries: - raise + if attempt == max_retries: + raise - # Exponential backoff with full jitter - delay = min( - base_delay * (backoff_factor ** attempt) + random.uniform(0, 1), - max_delay, - ) + # Exponential backoff with full jitter + delay = min( + base_delay * (backoff_factor**attempt) + random.uniform(0, 1), + max_delay, + ) - verbose_logger.warning( - f"[LiteLLM] Attempt {attempt + 1}/{max_retries + 1} failed " - f"({type(e).__name__}). Retrying in {delay:.2f}s..." - ) - time.sleep(delay) + verbose_logger.warning( + f"[LiteLLM] Attempt {attempt + 1}/{max_retries + 1} failed " + f"({type(e).__name__}). Retrying in {delay:.2f}s..." + ) + time.sleep(delay) - raise last_exception # pragma: no cover + raise last_exception # pragma: no cover async def async_retry_with_backoff( - fn: Callable, - max_retries: int = 3, - base_delay: float = 1.0, - max_delay: float = 60.0, - backoff_factor: float = 2.0, - retry_on: Optional[Tuple[Type[Exception], ...]] = None, + fn: Callable, + max_retries: int = 3, + base_delay: float = 1.0, + max_delay: float = 60.0, + backoff_factor: float = 2.0, + retry_on: Optional[Tuple[Type[Exception], ...]] = None, ) -> Any: - """ - Retries an async callable with exponential backoff and jitter. + """ + Retries an async callable with exponential backoff and jitter. - Args: - fn: The async callable to retry (e.g. lambda: litellm.acompletion(...)) - max_retries: Maximum number of retry attempts (default: 3) - base_delay: Initial delay in seconds (default: 1.0) - max_delay: Maximum delay cap in seconds (default: 60.0) - backoff_factor: Multiplier per retry (default: 2.0) - retry_on: Tuple of exception types to retry on. - Defaults to (RateLimitError, ServiceUnavailableError) + Args: + fn: The async callable to retry (e.g. lambda: litellm.acompletion(...)) + max_retries: Maximum number of retry attempts (default: 3) + base_delay: Initial delay in seconds (default: 1.0) + max_delay: Maximum delay cap in seconds (default: 60.0) + backoff_factor: Multiplier per retry (default: 2.0) + retry_on: Tuple of exception types to retry on. + Defaults to (RateLimitError, ServiceUnavailableError) - Returns: - The return value of await fn() on success. + Returns: + The return value of await fn() on success. - Raises: - The last exception if all retries are exhausted. + Raises: + The last exception if all retries are exhausted. - Example: - import litellm - from litellm.utils import async_retry_with_backoff + Example: + import litellm + from litellm.utils import async_retry_with_backoff - response = await async_retry_with_backoff( - lambda: litellm.acompletion( - model="gpt-4o", - messages=[{"role": "user", "content": "Hello"}] - ), - max_retries=5, - base_delay=2.0, - ) - """ - if retry_on is None: - retry_on = (RateLimitError, ServiceUnavailableError) + response = await async_retry_with_backoff( + lambda: litellm.acompletion( + model="gpt-4o", + messages=[{"role": "user", "content": "Hello"}] + ), + max_retries=5, + base_delay=2.0, + ) + """ + if retry_on is None: + retry_on = (RateLimitError, ServiceUnavailableError) - last_exception: Optional[Exception] = None + last_exception: Optional[Exception] = None - for attempt in range(max_retries + 1): - try: - result = fn() - if inspect.iscoroutine(result): - result = await result - return result + for attempt in range(max_retries + 1): + try: + result = fn() + if inspect.iscoroutine(result): + result = await result + return result - except retry_on as e: - last_exception = e + except retry_on as e: + last_exception = e - if attempt == max_retries: - raise + if attempt == max_retries: + raise - # Exponential backoff with full jitter - delay = min( - base_delay * (backoff_factor ** attempt) + random.uniform(0, 1), - max_delay, - ) + # Exponential backoff with full jitter + delay = min( + base_delay * (backoff_factor**attempt) + random.uniform(0, 1), + max_delay, + ) - verbose_logger.warning( - f"[LiteLLM] Attempt {attempt + 1}/{max_retries + 1} failed " - f"({type(e).__name__}). Retrying in {delay:.2f}s..." - ) - await asyncio.sleep(delay) + verbose_logger.warning( + f"[LiteLLM] Attempt {attempt + 1}/{max_retries + 1} failed " + f"({type(e).__name__}). Retrying in {delay:.2f}s..." + ) + await asyncio.sleep(delay) - raise last_exception # pragma: no cover \ No newline at end of file + raise last_exception # pragma: no cover