fix: distinguish insufficient quota errors

This commit is contained in:
Devin AI 2026-07-10 16:13:06 +00:00
parent bf02a4a47f
commit 7d0dfe357b
9 changed files with 208 additions and 53 deletions

View file

@ -1300,6 +1300,7 @@ from .exceptions import (
ImageFetchError,
NotFoundError,
PermissionDeniedError,
InsufficientQuotaError,
RateLimitError,
RateLimitErrorCategory,
RateLimitType,

View file

@ -423,6 +423,10 @@ class RateLimitError(openai.RateLimitError): # type: ignore
:class:`RateLimitErrorCategory` for the available values.
"""
exception_name = "RateLimitError"
error_code = "429"
error_type = "throttling_error"
def __init__(
self,
message,
@ -438,7 +442,7 @@ class RateLimitError(openai.RateLimitError): # type: ignore
detail: Any = None,
):
self.status_code = 429
self.message = "litellm.RateLimitError: {}".format(message)
self.message = f"litellm.{self.exception_name}: {message}"
self.llm_provider = llm_provider
self.model = model
self.litellm_debug_info = litellm_debug_info
@ -480,8 +484,8 @@ class RateLimitError(openai.RateLimitError): # type: ignore
super().__init__(
self.message, response=self.response, body=None
) # Call the base class constructor with the parameters it needs
self.code = "429"
self.type = "throttling_error"
self.code = self.error_code
self.type = self.error_type
def __str__(self):
_message = self.message
@ -500,6 +504,12 @@ class RateLimitError(openai.RateLimitError): # type: ignore
return _message
class InsufficientQuotaError(RateLimitError):
exception_name = "InsufficientQuotaError"
error_code = "insufficient_quota"
error_type = "insufficient_quota"
# sub class of rate limit error - meant to give more granularity for error handling context window exceeded errors
class ContextWindowExceededError(BadRequestError): # type: ignore
def __init__(
@ -944,6 +954,7 @@ LITELLM_EXCEPTION_TYPES = [
Timeout,
PermissionDeniedError,
RateLimitError,
InsufficientQuotaError,
ContextWindowExceededError,
RejectedRequestError,
ContentPolicyViolationError,

View file

@ -1,6 +1,7 @@
import json
import re
import traceback
from collections.abc import Mapping
from typing import Any, Optional, Protocol, cast
import httpx
@ -18,6 +19,7 @@ from ..exceptions import (
BadRequestError,
ContentPolicyViolationError,
ContextWindowExceededError,
InsufficientQuotaError,
InternalServerError,
NotFoundError,
PermissionDeniedError,
@ -246,6 +248,15 @@ class _ProviderHTTPException(Protocol):
llm_provider: str
def _is_insufficient_quota_error(original_exception: _ProviderHTTPException) -> bool:
body = original_exception.body
if not isinstance(body, Mapping):
return False
nested_error = body.get("error")
error_body = nested_error if isinstance(nested_error, Mapping) else body
return error_body.get("code") == "insufficient_quota" or error_body.get("type") == "insufficient_quota"
def _map_openai_exception(
*,
model: str,
@ -277,7 +288,14 @@ def _map_openai_exception(
else:
exception_provider = custom_llm_provider[0].upper() + custom_llm_provider[1:] + "Exception"
if ExceptionCheckers.is_error_str_rate_limit(error_str):
if _is_insufficient_quota_error(original_exception):
raise InsufficientQuotaError(
message=f"InsufficientQuotaError: {exception_provider} - {message}",
model=model,
llm_provider=custom_llm_provider,
response=original_exception.response,
)
elif ExceptionCheckers.is_error_str_rate_limit(error_str):
raise RateLimitError(
message=f"RateLimitError: {exception_provider} - {message}",
model=model,

View file

@ -78,7 +78,7 @@ from litellm.constants import (
DEFAULT_MOCK_RESPONSE_COMPLETION_TOKEN_COUNT,
DEFAULT_MOCK_RESPONSE_PROMPT_TOKEN_COUNT,
)
from litellm.exceptions import LiteLLMUnknownProvider
from litellm.exceptions import InsufficientQuotaError, LiteLLMUnknownProvider
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.asyncify import run_async_function
from litellm.litellm_core_utils.chat_completion_agentic_loop import (
@ -5679,12 +5679,17 @@ def completion_with_retries(*args, **kwargs):
original_function = kwargs.pop("original_function", completion)
if retry_strategy == "exponential_backoff_retry":
retryer = tenacity.Retrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
wait=tenacity.wait_exponential(multiplier=1, max=10),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
else:
retryer = tenacity.Retrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True)
retryer = tenacity.Retrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
return retryer(original_function, *args, **kwargs)
@ -5705,12 +5710,17 @@ async def acompletion_with_retries(*args, **kwargs):
original_function = kwargs.pop("original_function", completion)
if retry_strategy == "exponential_backoff_retry":
retryer = tenacity.AsyncRetrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
wait=tenacity.wait_exponential(multiplier=1, max=10),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
else:
retryer = tenacity.AsyncRetrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True)
retryer = tenacity.AsyncRetrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
return await retryer(original_function, *args, **kwargs)
@ -5735,12 +5745,17 @@ def responses_with_retries(*args, **kwargs):
original_function = kwargs.pop("original_function", responses)
if retry_strategy == "exponential_backoff_retry":
retryer = tenacity.Retrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
wait=tenacity.wait_exponential(multiplier=1, max=10),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
else:
retryer = tenacity.Retrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True)
retryer = tenacity.Retrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
return retryer(original_function, *args, **kwargs)
@ -5762,12 +5777,17 @@ async def aresponses_with_retries(*args, **kwargs):
original_function = kwargs.pop("original_function", aresponses)
if retry_strategy == "exponential_backoff_retry":
retryer = tenacity.AsyncRetrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
wait=tenacity.wait_exponential(multiplier=1, max=10),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
else:
retryer = tenacity.AsyncRetrying(stop=tenacity.stop_after_attempt(num_retries), reraise=True)
retryer = tenacity.AsyncRetrying(
retry=tenacity.retry_if_not_exception_type(InsufficientQuotaError),
stop=tenacity.stop_after_attempt(num_retries),
reraise=True,
)
return await retryer(original_function, *args, **kwargs)

View file

@ -6593,6 +6593,9 @@ class Router:
_num_all_deployments = len(all_deployments)
### CHECK IF RATE LIMIT / CONTEXT WINDOW ERROR / CONTENT POLICY VIOLATION ERROR w/ fallbacks available / Bad Request Error
if isinstance(error, litellm.InsufficientQuotaError):
raise error
if isinstance(error, litellm.ContextWindowExceededError) and context_window_fallbacks is not None:
raise error

View file

@ -10,6 +10,7 @@ from litellm.exceptions import (
AuthenticationError,
BadRequestError,
ContentPolicyViolationError,
InsufficientQuotaError,
RateLimitError,
Timeout,
)
@ -39,6 +40,8 @@ def get_num_retries_from_retry_policy(
if isinstance(retry_policy, dict):
retry_policy = RetryPolicy(**retry_policy)
if isinstance(exception, InsufficientQuotaError):
return 0
if isinstance(exception, AuthenticationError) and retry_policy.AuthenticationErrorRetries is not None:
return retry_policy.AuthenticationErrorRetries
if isinstance(exception, Timeout) and retry_policy.TimeoutErrorRetries is not None:

View file

@ -411,6 +411,7 @@ from .exceptions import (
BudgetExceededError,
ContentPolicyViolationError,
ContextWindowExceededError,
InsufficientQuotaError,
NotFoundError,
OpenAIError,
PermissionDeniedError,
@ -1106,6 +1107,9 @@ def _get_wrapper_num_retries(kwargs: Dict[str, Any], exception: Exception) -> Tu
Used for the wrapper functions.
"""
if isinstance(exception, InsufficientQuotaError):
return 0, kwargs
num_retries = kwargs.get("num_retries", None)
if num_retries is None:
num_retries = litellm.num_retries
@ -1501,7 +1505,11 @@ def client(original_function):
except Exception as e:
call_type = original_function.__name__
if call_type == CallTypes.completion.value:
num_retries = kwargs.get("num_retries", None) or litellm.num_retries or None
num_retries = (
None
if isinstance(e, InsufficientQuotaError)
else kwargs.get("num_retries", None) or litellm.num_retries or None
)
if kwargs.get("retry_policy", None):
get_num_retries_from_retry_policy = getattr(
sys.modules[__name__], "get_num_retries_from_retry_policy"
@ -1540,7 +1548,11 @@ def client(original_function):
kwargs["model"] = context_window_fallback_dict[model]
return original_function(*args, **kwargs)
elif call_type == CallTypes.responses.value:
num_retries = kwargs.get("num_retries", None) or litellm.num_retries or None
num_retries = (
None
if isinstance(e, InsufficientQuotaError)
else kwargs.get("num_retries", None) or litellm.num_retries or None
)
if kwargs.get("retry_policy", None):
get_num_retries_from_retry_policy = getattr(
sys.modules[__name__], "get_num_retries_from_retry_policy"

View file

@ -6,9 +6,7 @@ import pytest
import litellm
sys.path.insert(
0, os.path.abspath("../../..")
) # Adds the parent directory to the system path
sys.path.insert(0, os.path.abspath("../../..")) # Adds the parent directory to the system path
from litellm.litellm_core_utils.exception_mapping_utils import (
ExceptionCheckers,
@ -133,9 +131,7 @@ class TestExceptionCheckers:
]
for error_str in error_strings:
result = ExceptionCheckers.is_azure_content_policy_violation_error(
error_str
)
result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str)
assert result is True, f"Should detect policy violation in: {error_str}"
def test_is_azure_content_policy_violation_error_case_insensitive(self):
@ -149,12 +145,8 @@ class TestExceptionCheckers:
]
for error_str in error_strings:
result = ExceptionCheckers.is_azure_content_policy_violation_error(
error_str
)
assert (
result is True
), f"Should detect policy violation in uppercase: {error_str}"
result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str)
assert result is True, f"Should detect policy violation in uppercase: {error_str}"
def test_is_azure_content_policy_violation_error_with_non_policy_errors(self):
"""Test that non-policy violation errors are not detected as policy violations"""
@ -171,12 +163,8 @@ class TestExceptionCheckers:
]
for error_str in error_strings:
result = ExceptionCheckers.is_azure_content_policy_violation_error(
error_str
)
assert (
result is False
), f"Should NOT detect policy violation in: {error_str}"
result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str)
assert result is False, f"Should NOT detect policy violation in: {error_str}"
def test_is_azure_content_policy_violation_error_with_partial_matches(self):
"""Test that partial keyword matches work correctly"""
@ -189,9 +177,7 @@ class TestExceptionCheckers:
]
for error_str in positive_cases:
result = ExceptionCheckers.is_azure_content_policy_violation_error(
error_str
)
result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str)
assert result is True, f"Should detect policy violation in: {error_str}"
# These should not match even though they contain similar words
@ -203,12 +189,73 @@ class TestExceptionCheckers:
]
for error_str in negative_cases:
result = ExceptionCheckers.is_azure_content_policy_violation_error(
error_str
)
assert (
result is False
), f"Should NOT detect policy violation in: {error_str}"
result = ExceptionCheckers.is_azure_content_policy_violation_error(error_str)
assert result is False, f"Should NOT detect policy violation in: {error_str}"
@pytest.mark.parametrize(
"body",
[
{
"message": "You exceeded your current quota",
"type": "insufficient_quota",
"param": None,
"code": "insufficient_quota",
},
{
"error": {
"message": "You exceeded your current quota",
"type": "insufficient_quota",
"param": None,
"code": "insufficient_quota",
}
},
],
)
def test_openai_insufficient_quota_maps_to_distinct_rate_limit_subtype(body):
original_exception = OpenAIError(
status_code=429,
message="Error code: 429 - You exceeded your current quota",
headers={},
body=body,
)
with pytest.raises(litellm.InsufficientQuotaError) as exc_info:
exception_type(
model="gpt-5.5",
original_exception=original_exception,
custom_llm_provider="openai",
)
assert isinstance(exc_info.value, litellm.RateLimitError)
assert exc_info.value.code == "insufficient_quota"
assert exc_info.value.type == "insufficient_quota"
assert litellm.InsufficientQuotaError in litellm.LITELLM_EXCEPTION_TYPES
def test_openai_transient_429_remains_rate_limit_error():
original_exception = OpenAIError(
status_code=429,
message="Error code: 429 - Rate limit reached for requests",
headers={},
body={
"message": "Rate limit reached for requests",
"type": "requests",
"param": None,
"code": "rate_limit_exceeded",
},
)
with pytest.raises(litellm.RateLimitError) as exc_info:
exception_type(
model="gpt-5.5",
original_exception=original_exception,
custom_llm_provider="openai",
)
assert type(exc_info.value) is litellm.RateLimitError
assert exc_info.value.code == "429"
assert exc_info.value.type == "throttling_error"
gemini_context_window_test_cases = [
@ -226,12 +273,8 @@ gemini_context_window_test_cases = [
]
@pytest.mark.parametrize(
"error_message, should_raise_context_window", gemini_context_window_test_cases
)
def test_gemini_context_window_error_mapping(
error_message, should_raise_context_window
):
@pytest.mark.parametrize("error_message, should_raise_context_window", gemini_context_window_test_cases)
def test_gemini_context_window_error_mapping(error_message, should_raise_context_window):
"""
Tests that the exception_type function correctly maps Gemini's
context window exceeded errors to litellm.ContextWindowExceededError.
@ -328,9 +371,7 @@ vertex_rate_limit_test_cases = [
]
@pytest.mark.parametrize(
"error_message, should_raise_rate_limit", vertex_rate_limit_test_cases
)
@pytest.mark.parametrize("error_message, should_raise_rate_limit", vertex_rate_limit_test_cases)
def test_vertex_ai_rate_limit_error_mapping(error_message, should_raise_rate_limit):
"""
Tests that the exception_type function correctly maps Vertex AI's
@ -365,10 +406,7 @@ class TestGetBodyErrorCode:
"""Unit tests for _get_body_error_code helper."""
def test_parses_int_code(self):
body = (
'{"error":{"message":"high demand","type":"upstream_error",'
'"param":"","code":429}}'
)
body = '{"error":{"message":"high demand","type":"upstream_error","param":"","code":429}}'
assert _get_body_error_code(body) == 429
def test_parses_string_code(self):
@ -405,8 +443,7 @@ gemini_body_code_429_test_cases = [
),
(
503,
'{"error":{"message":"upstream unavailable","type":"upstream_error",'
'"param":"","code":429}}',
'{"error":{"message":"upstream unavailable","type":"upstream_error","param":"","code":429}}',
litellm.RateLimitError,
"HTTP 503 envelope with body code:429 -> RateLimitError",
),

View file

@ -16,6 +16,8 @@ import pytest
import litellm
from litellm import Router
from litellm.router_utils.get_retry_from_policy import get_num_retries_from_retry_policy
from litellm.types.router import RetryPolicy
def _make_rate_limit_error(message="Rate limited"):
@ -27,6 +29,14 @@ def _make_rate_limit_error(message="Rate limited"):
)
def _make_insufficient_quota_error(message="You exceeded your current quota"):
return litellm.InsufficientQuotaError(
message=message,
llm_provider="openai",
model="gpt-5.5",
)
def _make_context_window_error(message="prompt is too long: 1205821 tokens > 200000"):
"""Create a ContextWindowExceededError for testing."""
return litellm.ContextWindowExceededError(
@ -236,6 +246,46 @@ async def test_retryable_errors_still_retry_normally():
assert call_count == 4
def test_router_does_not_retry_insufficient_quota_error():
router = _create_router(num_retries=3)
error = _make_insufficient_quota_error()
with pytest.raises(litellm.InsufficientQuotaError):
router.should_retry_this_error(
error=error,
healthy_deployments=["d1", "d2"],
all_deployments=["d1", "d2"],
)
def test_insufficient_quota_error_ignores_rate_limit_retry_policy():
retries = get_num_retries_from_retry_policy(
exception=_make_insufficient_quota_error(),
retry_policy=RetryPolicy(RateLimitErrorRetries=3),
)
assert retries == 0
def test_completion_with_retries_stops_on_insufficient_quota():
call_count = 0
def raise_insufficient_quota(*args, **kwargs):
nonlocal call_count
call_count += 1
raise _make_insufficient_quota_error()
with pytest.raises(litellm.InsufficientQuotaError):
litellm.completion_with_retries(
model="gpt-5.5",
messages=[{"role": "user", "content": "hi"}],
num_retries=3,
original_function=raise_insufficient_quota,
)
assert call_count == 1
@pytest.mark.asyncio
async def test_not_found_error_in_retry_loop_raises_immediately():
"""