From 514fe7926797b2e97e1a7c0efa848f74fce46704 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 10 Jul 2026 16:22:23 +0000 Subject: [PATCH] test: cover insufficient quota retry paths --- .../test_exception_mapping_utils.py | 19 +++++--- tests/test_litellm/test_main.py | 39 ++++++++++++++- .../test_router_retry_non_retryable_errors.py | 19 -------- tests/test_litellm/test_utils.py | 48 ++++++++++++++++++- 4 files changed, 98 insertions(+), 27 deletions(-) diff --git a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py index e65086108b1..25c07926be3 100644 --- a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py @@ -233,17 +233,24 @@ def test_openai_insufficient_quota_maps_to_distinct_rate_limit_subtype(body): assert litellm.InsufficientQuotaError in litellm.LITELLM_EXCEPTION_TYPES -def test_openai_transient_429_remains_rate_limit_error(): - original_exception = OpenAIError( - status_code=429, - message="Error code: 429 - Rate limit reached for requests", - headers={}, - body={ +@pytest.mark.parametrize( + "body", + [ + None, + { "message": "Rate limit reached for requests", "type": "requests", "param": None, "code": "rate_limit_exceeded", }, + ], +) +def test_openai_transient_429_remains_rate_limit_error(body): + original_exception = OpenAIError( + status_code=429, + message="Error code: 429 - Rate limit reached for requests", + headers={}, + body=body, ) with pytest.raises(litellm.RateLimitError) as exc_info: diff --git a/tests/test_litellm/test_main.py b/tests/test_litellm/test_main.py index 28cf4fa0744..bfbbc11f9fb 100644 --- a/tests/test_litellm/test_main.py +++ b/tests/test_litellm/test_main.py @@ -12,7 +12,7 @@ sys.path.insert( ) # Adds the parent directory to the system path import urllib.parse -from unittest.mock import MagicMock, patch +from unittest.mock import AsyncMock, MagicMock, Mock, patch import litellm from litellm import main as litellm_main @@ -2081,3 +2081,40 @@ def test_stream_chunk_builder_text_completion_combines_text_and_usage(): assert response.usage.prompt_tokens > 0 assert response.usage.completion_tokens > 0 assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens + + +@pytest.mark.parametrize( + "retry_function", + [litellm.completion_with_retries, litellm.responses_with_retries], +) +def test_sync_retry_helpers_stop_on_insufficient_quota(retry_function): + error = litellm.InsufficientQuotaError( + message="You exceeded your current quota", + llm_provider="openai", + model="gpt-5.5", + ) + original_function = Mock(side_effect=error) + + with pytest.raises(litellm.InsufficientQuotaError): + retry_function(num_retries=3, original_function=original_function) + + assert original_function.call_count == 1 + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "retry_function", + [litellm.acompletion_with_retries, litellm.aresponses_with_retries], +) +async def test_async_retry_helpers_stop_on_insufficient_quota(retry_function): + error = litellm.InsufficientQuotaError( + message="You exceeded your current quota", + llm_provider="openai", + model="gpt-5.5", + ) + original_function = AsyncMock(side_effect=error) + + with pytest.raises(litellm.InsufficientQuotaError): + await retry_function(num_retries=3, original_function=original_function) + + assert original_function.await_count == 1 diff --git a/tests/test_litellm/test_router_retry_non_retryable_errors.py b/tests/test_litellm/test_router_retry_non_retryable_errors.py index 39f9da1f74b..2dbaa8532d1 100644 --- a/tests/test_litellm/test_router_retry_non_retryable_errors.py +++ b/tests/test_litellm/test_router_retry_non_retryable_errors.py @@ -267,25 +267,6 @@ def test_insufficient_quota_error_ignores_rate_limit_retry_policy(): assert retries == 0 -def test_completion_with_retries_stops_on_insufficient_quota(): - call_count = 0 - - def raise_insufficient_quota(*args, **kwargs): - nonlocal call_count - call_count += 1 - raise _make_insufficient_quota_error() - - with pytest.raises(litellm.InsufficientQuotaError): - litellm.completion_with_retries( - model="gpt-5.5", - messages=[{"role": "user", "content": "hi"}], - num_retries=3, - original_function=raise_insufficient_quota, - ) - - assert call_count == 1 - - @pytest.mark.asyncio async def test_not_found_error_in_retry_loop_raises_immediately(): """ diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index d739f9c116a..ec24ae6513d 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -1,7 +1,7 @@ import json import os import sys -from unittest.mock import AsyncMock, MagicMock, patch +from unittest.mock import AsyncMock, MagicMock, Mock, patch import pytest from jsonschema import validate @@ -23,7 +23,9 @@ from litellm.utils import ( ProviderConfigManager, TextCompletionStreamWrapper, _check_provider_match, + _get_wrapper_num_retries, _is_streaming_request, + client, get_llm_provider, get_optional_params_image_gen, is_cached_message, @@ -4709,3 +4711,47 @@ class TestValidateEnvironmentTencent: assert "TENCENT_API_KEY" in result["missing_keys"] +def test_wrapper_num_retries_is_zero_for_insufficient_quota(): + kwargs = {"num_retries": 3} + retries, returned_kwargs = _get_wrapper_num_retries( + kwargs=kwargs, + exception=litellm.InsufficientQuotaError( + message="You exceeded your current quota", + llm_provider="openai", + model="gpt-5.5", + ), + ) + + assert retries == 0 + assert returned_kwargs is kwargs + + +def test_sync_client_does_not_retry_insufficient_quota(): + completion_mock = Mock( + side_effect=litellm.InsufficientQuotaError( + message="You exceeded your current quota", + llm_provider="openai", + model="gpt-5.5", + ) + ) + responses_mock = Mock( + side_effect=litellm.InsufficientQuotaError( + message="You exceeded your current quota", + llm_provider="openai", + model="gpt-5.5", + ) + ) + + def completion(*args, **kwargs): + return completion_mock(*args, **kwargs) + + def responses(*args, **kwargs): + return responses_mock(*args, **kwargs) + + with pytest.raises(litellm.InsufficientQuotaError): + client(completion)(num_retries=3) + with pytest.raises(litellm.InsufficientQuotaError): + client(responses)(num_retries=3) + + assert completion_mock.call_count == 1 + assert responses_mock.call_count == 1