diff --git a/litellm/__init__.py b/litellm/__init__.py index 6e2a03b7c7c..7d938a9bd64 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1300,6 +1300,7 @@ from .exceptions import ( ImageFetchError, NotFoundError, PermissionDeniedError, + InsufficientQuotaError, RateLimitError, RateLimitErrorCategory, RateLimitType, diff --git a/litellm/exceptions.py b/litellm/exceptions.py index aca3fb551cc..1703f91fe68 100644 --- a/litellm/exceptions.py +++ b/litellm/exceptions.py @@ -423,6 +423,10 @@ class RateLimitError(openai.RateLimitError): # type: ignore :class:`RateLimitErrorCategory` for the available values. """ + exception_name = "RateLimitError" + error_code = "429" + error_type = "throttling_error" + def __init__( self, message, @@ -438,7 +442,7 @@ class RateLimitError(openai.RateLimitError): # type: ignore detail: Any = None, ): self.status_code = 429 - self.message = "litellm.RateLimitError: {}".format(message) + self.message = f"litellm.{self.exception_name}: {message}" self.llm_provider = llm_provider self.model = model self.litellm_debug_info = litellm_debug_info @@ -480,8 +484,8 @@ class RateLimitError(openai.RateLimitError): # type: ignore super().__init__( self.message, response=self.response, body=None ) # Call the base class constructor with the parameters it needs - self.code = "429" - self.type = "throttling_error" + self.code = self.error_code + self.type = self.error_type def __str__(self): _message = self.message @@ -500,6 +504,12 @@ class RateLimitError(openai.RateLimitError): # type: ignore return _message +class InsufficientQuotaError(RateLimitError): + exception_name = "InsufficientQuotaError" + error_code = "insufficient_quota" + error_type = "insufficient_quota" + + # sub class of rate limit error - meant to give more granularity for error handling context window exceeded errors class ContextWindowExceededError(BadRequestError): # type: ignore def __init__( @@ -944,6 +954,7 @@ LITELLM_EXCEPTION_TYPES = [ Timeout, PermissionDeniedError, RateLimitError, + InsufficientQuotaError, ContextWindowExceededError, RejectedRequestError, ContentPolicyViolationError, diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 9dc202c4717..3e66268e848 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1,6 +1,7 @@ import json import re import traceback +from collections.abc import Mapping from typing import Any, Optional, Protocol, cast import httpx @@ -18,6 +19,7 @@ from ..exceptions import ( BadRequestError, ContentPolicyViolationError, ContextWindowExceededError, + InsufficientQuotaError, InternalServerError, NotFoundError, PermissionDeniedError, @@ -246,6 +248,15 @@ class _ProviderHTTPException(Protocol): llm_provider: str +def _is_insufficient_quota_error(original_exception: _ProviderHTTPException) -> bool: + body = original_exception.body + if not isinstance(body, Mapping): + return False + nested_error = body.get("error") + error_body = nested_error if isinstance(nested_error, Mapping) else body + return error_body.get("code") == "insufficient_quota" or error_body.get("type") == "insufficient_quota" + + def _map_openai_exception( *, model: str, @@ -277,7 +288,14 @@ def _map_openai_exception( else: exception_provider = custom_llm_provider[0].upper() + custom_llm_provider[1:] + "Exception" - if ExceptionCheckers.is_error_str_rate_limit(error_str): + if _is_insufficient_quota_error(original_exception): + raise InsufficientQuotaError( + message=f"InsufficientQuotaError: {exception_provider} - {message}", + model=model, + llm_provider=custom_llm_provider, + response=original_exception.response, + ) + elif ExceptionCheckers.is_error_str_rate_limit(error_str): raise RateLimitError( message=f"RateLimitError: {exception_provider} - {message}", model=model, diff --git a/litellm/main.py b/litellm/main.py index 7d457d9cdd1..fbb258be80b 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -78,7 +78,7 @@ from litellm.constants import ( DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT, DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT, ) -from litellm.exceptions import LiteLLMUnknownProvider +from litellm.exceptions import InsufficientQuotaError, LiteLLMUnknownProvider from litellm.integrations.custom_logger import CustomLogger from litellm.litellm_core_utils.asyncify import run_async_function from litellm.litellm_core_utils.chat_completion_agentic_loop import ( @@ -5679,12 +5679,17 @@ def completion_with_retries(*args, **kwargs): original_function = kwargs.pop("original_function", completion) if retry_strategy == "exponential_backoff_retry": retryer = tenacity.Retrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), wait=tenacity.wait_exponential(multiplier=1, max=10), stop=tenacity.stop_after_attempt(num_retries), reraise=True, ) else: - retryer = tenacity.Retrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True) + retryer = tenacity.Retrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), + stop=tenacity.stop_after_attempt(num_retries), + reraise=True, + ) return retryer(original_function, *args, **kwargs) @@ -5705,12 +5710,17 @@ async def acompletion_with_retries(*args, **kwargs): original_function = kwargs.pop("original_function", completion) if retry_strategy == "exponential_backoff_retry": retryer = tenacity.AsyncRetrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), wait=tenacity.wait_exponential(multiplier=1, max=10), stop=tenacity.stop_after_attempt(num_retries), reraise=True, ) else: - retryer = tenacity.AsyncRetrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True) + retryer = tenacity.AsyncRetrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), + stop=tenacity.stop_after_attempt(num_retries), + reraise=True, + ) return await retryer(original_function, *args, **kwargs) @@ -5735,12 +5745,17 @@ def responses_with_retries(*args, **kwargs): original_function = kwargs.pop("original_function", responses) if retry_strategy == "exponential_backoff_retry": retryer = tenacity.Retrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), wait=tenacity.wait_exponential(multiplier=1, max=10), stop=tenacity.stop_after_attempt(num_retries), reraise=True, ) else: - retryer = tenacity.Retrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True) + retryer = tenacity.Retrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), + stop=tenacity.stop_after_attempt(num_retries), + reraise=True, + ) return retryer(original_function, *args, **kwargs) @@ -5762,12 +5777,17 @@ async def aresponses_with_retries(*args, **kwargs): original_function = kwargs.pop("original_function", aresponses) if retry_strategy == "exponential_backoff_retry": retryer = tenacity.AsyncRetrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), wait=tenacity.wait_exponential(multiplier=1, max=10), stop=tenacity.stop_after_attempt(num_retries), reraise=True, ) else: - retryer = tenacity.AsyncRetrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True) + retryer = tenacity.AsyncRetrying( + retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError), + stop=tenacity.stop_after_attempt(num_retries), + reraise=True, + ) return await retryer(original_function, *args, **kwargs) diff --git a/litellm/router.py b/litellm/router.py index 5ffe60c2da0..b960d580e8c 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -6593,6 +6593,9 @@ class Router: _num_all_deployments = len(all_deployments) ### CHECK IF RATE LIMIT / CONTEXT WINDOW ERROR / CONTENT POLICY VIOLATION ERROR w/ fallbacks available / Bad Request Error + if isinstance(error, litellm.InsufficientQuotaError): + raise error + if isinstance(error, litellm.ContextWindowExceededError) and context_window_fallbacks is not None: raise error diff --git a/litellm/router_utils/get_retry_from_policy.py b/litellm/router_utils/get_retry_from_policy.py index 314917d3f56..b40aa1ee43d 100644 --- a/litellm/router_utils/get_retry_from_policy.py +++ b/litellm/router_utils/get_retry_from_policy.py @@ -10,6 +10,7 @@ from litellm.exceptions import ( AuthenticationError, BadRequestError, ContentPolicyViolationError, + InsufficientQuotaError, RateLimitError, Timeout, ) @@ -39,6 +40,8 @@ def get_num_retries_from_retry_policy( if isinstance(retry_policy, dict): retry_policy = RetryPolicy(**retry_policy) + if isinstance(exception, InsufficientQuotaError): + return 0 if isinstance(exception, AuthenticationError) and retry_policy.AuthenticationErrorRetries is not None: return retry_policy.AuthenticationErrorRetries if isinstance(exception, Timeout) and retry_policy.TimeoutErrorRetries is not None: diff --git a/litellm/utils.py b/litellm/utils.py index 5af7b62b332..18016ce7055 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -411,6 +411,7 @@ from .exceptions import ( BudgetExceededError, ContentPolicyViolationError, ContextWindowExceededError, + InsufficientQuotaError, NotFoundError, OpenAIError, PermissionDeniedError, @@ -1106,6 +1107,9 @@ def _get_wrapper_num_retries(kwargs: Dict[str, Any], exception: Exception) -> Tu Used for the wrapper functions. """ + if isinstance(exception, InsufficientQuotaError): + return 0, kwargs + num_retries = kwargs.get("num_retries", None) if num_retries is None: num_retries = litellm.num_retries @@ -1501,7 +1505,11 @@ def client(original_function): except Exception as e: call_type = original_function.__name__ if call_type == CallTypes.completion.value: - num_retries = kwargs.get("num_retries", None) or litellm.num_retries or None + num_retries = ( + None + if isinstance(e, InsufficientQuotaError) + else kwargs.get("num_retries", None) or litellm.num_retries or None + ) if kwargs.get("retry_policy", None): get_num_retries_from_retry_policy = getattr( sys.modules[__name__], "get_num_retries_from_retry_policy" @@ -1540,7 +1548,11 @@ def client(original_function): kwargs["model"] = context_window_fallback_dict[model] return original_function(*args, **kwargs) elif call_type == CallTypes.responses.value: - num_retries = kwargs.get("num_retries", None) or litellm.num_retries or None + num_retries = ( + None + if isinstance(e, InsufficientQuotaError) + else kwargs.get("num_retries", None) or litellm.num_retries or None + ) if kwargs.get("retry_policy", None): get_num_retries_from_retry_policy = getattr( sys.modules[__name__], "get_num_retries_from_retry_policy" diff --git a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py index f441270be7c..e65086108b1 100644 --- a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py @@ -6,9 +6,7 @@ import pytest import litellm -sys.path.insert( - 0, os.path.abspath("../../..") -) # Adds the parent directory to the system path +sys.path.insert(0, os.path.abspath("../../..")) # Adds the parent directory to the system path from litellm.litellm_core_utils.exception_mapping_utils import ( ExceptionCheckers, @@ -133,9 +131,7 @@ class TestExceptionCheckers: ] for error_str in error_strings: - result = ExceptionCheckers.is_azure_content_policy_violation_error( - error_str - ) + result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str) assert result is True, f"Should detect policy violation in: {error_str}" def test_is_azure_content_policy_violation_error_case_insensitive(self): @@ -149,12 +145,8 @@ class TestExceptionCheckers: ] for error_str in error_strings: - result = ExceptionCheckers.is_azure_content_policy_violation_error( - error_str - ) - assert ( - result is True - ), f"Should detect policy violation in uppercase: {error_str}" + result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str) + assert result is True, f"Should detect policy violation in uppercase: {error_str}" def test_is_azure_content_policy_violation_error_with_non_policy_errors(self): """Test that non-policy violation errors are not detected as policy violations""" @@ -171,12 +163,8 @@ class TestExceptionCheckers: ] for error_str in error_strings: - result = ExceptionCheckers.is_azure_content_policy_violation_error( - error_str - ) - assert ( - result is False - ), f"Should NOT detect policy violation in: {error_str}" + result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str) + assert result is False, f"Should NOT detect policy violation in: {error_str}" def test_is_azure_content_policy_violation_error_with_partial_matches(self): """Test that partial keyword matches work correctly""" @@ -189,9 +177,7 @@ class TestExceptionCheckers: ] for error_str in positive_cases: - result = ExceptionCheckers.is_azure_content_policy_violation_error( - error_str - ) + result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str) assert result is True, f"Should detect policy violation in: {error_str}" # These should not match even though they contain similar words @@ -203,12 +189,73 @@ class TestExceptionCheckers: ] for error_str in negative_cases: - result = ExceptionCheckers.is_azure_content_policy_violation_error( - error_str - ) - assert ( - result is False - ), f"Should NOT detect policy violation in: {error_str}" + result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str) + assert result is False, f"Should NOT detect policy violation in: {error_str}" + + +@pytest.mark.parametrize( + "body", + [ + { + "message": "You exceeded your current quota", + "type": "insufficient_quota", + "param": None, + "code": "insufficient_quota", + }, + { + "error": { + "message": "You exceeded your current quota", + "type": "insufficient_quota", + "param": None, + "code": "insufficient_quota", + } + }, + ], +) +def test_openai_insufficient_quota_maps_to_distinct_rate_limit_subtype(body): + original_exception = OpenAIError( + status_code=429, + message="Error code: 429 - You exceeded your current quota", + headers={}, + body=body, + ) + + with pytest.raises(litellm.InsufficientQuotaError) as exc_info: + exception_type( + model="gpt-5.5", + original_exception=original_exception, + custom_llm_provider="openai", + ) + + assert isinstance(exc_info.value, litellm.RateLimitError) + assert exc_info.value.code == "insufficient_quota" + assert exc_info.value.type == "insufficient_quota" + assert litellm.InsufficientQuotaError in litellm.LITELLM_EXCEPTION_TYPES + + +def test_openai_transient_429_remains_rate_limit_error(): + original_exception = OpenAIError( + status_code=429, + message="Error code: 429 - Rate limit reached for requests", + headers={}, + body={ + "message": "Rate limit reached for requests", + "type": "requests", + "param": None, + "code": "rate_limit_exceeded", + }, + ) + + with pytest.raises(litellm.RateLimitError) as exc_info: + exception_type( + model="gpt-5.5", + original_exception=original_exception, + custom_llm_provider="openai", + ) + + assert type(exc_info.value) is litellm.RateLimitError + assert exc_info.value.code == "429" + assert exc_info.value.type == "throttling_error" gemini_context_window_test_cases = [ @@ -226,12 +273,8 @@ gemini_context_window_test_cases = [ ] -@pytest.mark.parametrize( - "error_message, should_raise_context_window", gemini_context_window_test_cases -) -def test_gemini_context_window_error_mapping( - error_message, should_raise_context_window -): +@pytest.mark.parametrize("error_message, should_raise_context_window", gemini_context_window_test_cases) +def test_gemini_context_window_error_mapping(error_message, should_raise_context_window): """ Tests that the exception_type function correctly maps Gemini's context window exceeded errors to litellm.ContextWindowExceededError. @@ -328,9 +371,7 @@ vertex_rate_limit_test_cases = [ ] -@pytest.mark.parametrize( - "error_message, should_raise_rate_limit", vertex_rate_limit_test_cases -) +@pytest.mark.parametrize("error_message, should_raise_rate_limit", vertex_rate_limit_test_cases) def test_vertex_ai_rate_limit_error_mapping(error_message, should_raise_rate_limit): """ Tests that the exception_type function correctly maps Vertex AI's @@ -365,10 +406,7 @@ class TestGetBodyErrorCode: """Unit tests for _get_body_error_code helper.""" def test_parses_int_code(self): - body = ( - '{"error":{"message":"high demand","type":"upstream_error",' - '"param":"","code":429}}' - ) + body = '{"error":{"message":"high demand","type":"upstream_error","param":"","code":429}}' assert _get_body_error_code(body) == 429 def test_parses_string_code(self): @@ -405,8 +443,7 @@ gemini_body_code_429_test_cases = [ ), ( 503, - '{"error":{"message":"upstream unavailable","type":"upstream_error",' - '"param":"","code":429}}', + '{"error":{"message":"upstream unavailable","type":"upstream_error","param":"","code":429}}', litellm.RateLimitError, "HTTP 503 envelope with body code:429 -> RateLimitError", ), diff --git a/tests/test_litellm/test_router_retry_non_retryable_errors.py b/tests/test_litellm/test_router_retry_non_retryable_errors.py index 0728947eafe..39f9da1f74b 100644 --- a/tests/test_litellm/test_router_retry_non_retryable_errors.py +++ b/tests/test_litellm/test_router_retry_non_retryable_errors.py @@ -16,6 +16,8 @@ import pytest import litellm from litellm import Router +from litellm.router_utils.get_retry_from_policy import get_num_retries_from_retry_policy +from litellm.types.router import RetryPolicy def _make_rate_limit_error(message="Rate limited"): @@ -27,6 +29,14 @@ def _make_rate_limit_error(message="Rate limited"): ) +def _make_insufficient_quota_error(message="You exceeded your current quota"): + return litellm.InsufficientQuotaError( + message=message, + llm_provider="openai", + model="gpt-5.5", + ) + + def _make_context_window_error(message="prompt is too long: 1205821 tokens > 200000"): """Create a ContextWindowExceededError for testing.""" return litellm.ContextWindowExceededError( @@ -236,6 +246,46 @@ async def test_retryable_errors_still_retry_normally(): assert call_count == 4 +def test_router_does_not_retry_insufficient_quota_error(): + router = _create_router(num_retries=3) + error = _make_insufficient_quota_error() + + with pytest.raises(litellm.InsufficientQuotaError): + router.should_retry_this_error( + error=error, + healthy_deployments=["d1", "d2"], + all_deployments=["d1", "d2"], + ) + + +def test_insufficient_quota_error_ignores_rate_limit_retry_policy(): + retries = get_num_retries_from_retry_policy( + exception=_make_insufficient_quota_error(), + retry_policy=RetryPolicy(RateLimitErrorRetries=3), + ) + + assert retries == 0 + + +def test_completion_with_retries_stops_on_insufficient_quota(): + call_count = 0 + + def raise_insufficient_quota(*args, **kwargs): + nonlocal call_count + call_count += 1 + raise _make_insufficient_quota_error() + + with pytest.raises(litellm.InsufficientQuotaError): + litellm.completion_with_retries( + model="gpt-5.5", + messages=[{"role": "user", "content": "hi"}], + num_retries=3, + original_function=raise_insufficient_quota, + ) + + assert call_count == 1 + + @pytest.mark.asyncio async def test_not_found_error_in_retry_loop_raises_immediately(): """