mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
test: cover insufficient quota retry paths
This commit is contained in:
parent
7d0dfe357b
commit
514fe79267
4 changed files with 98 additions and 27 deletions
|
|
@ -233,17 +233,24 @@ def test_openai_insufficient_quota_maps_to_distinct_rate_limit_subtype(body):
|
|||
assert litellm.InsufficientQuotaError in litellm.LITELLM_EXCEPTION_TYPES
|
||||
|
||||
|
||||
def test_openai_transient_429_remains_rate_limit_error():
|
||||
original_exception = OpenAIError(
|
||||
status_code=429,
|
||||
message="Error code: 429 - Rate limit reached for requests",
|
||||
headers={},
|
||||
body={
|
||||
@pytest.mark.parametrize(
|
||||
"body",
|
||||
[
|
||||
None,
|
||||
{
|
||||
"message": "Rate limit reached for requests",
|
||||
"type": "requests",
|
||||
"param": None,
|
||||
"code": "rate_limit_exceeded",
|
||||
},
|
||||
],
|
||||
)
|
||||
def test_openai_transient_429_remains_rate_limit_error(body):
|
||||
original_exception = OpenAIError(
|
||||
status_code=429,
|
||||
message="Error code: 429 - Rate limit reached for requests",
|
||||
headers={},
|
||||
body=body,
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError) as exc_info:
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ sys.path.insert(
|
|||
) # Adds the parent directory to the system path
|
||||
|
||||
import urllib.parse
|
||||
from unittest.mock import MagicMock, patch
|
||||
from unittest.mock import AsyncMock, MagicMock, Mock, patch
|
||||
|
||||
import litellm
|
||||
from litellm import main as litellm_main
|
||||
|
|
@ -2081,3 +2081,40 @@ def test_stream_chunk_builder_text_completion_combines_text_and_usage():
|
|||
assert response.usage.prompt_tokens > 0
|
||||
assert response.usage.completion_tokens > 0
|
||||
assert response.usage.total_tokens == response.usage.prompt_tokens + response.usage.completion_tokens
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"retry_function",
|
||||
[litellm.completion_with_retries, litellm.responses_with_retries],
|
||||
)
|
||||
def test_sync_retry_helpers_stop_on_insufficient_quota(retry_function):
|
||||
error = litellm.InsufficientQuotaError(
|
||||
message="You exceeded your current quota",
|
||||
llm_provider="openai",
|
||||
model="gpt-5.5",
|
||||
)
|
||||
original_function = Mock(side_effect=error)
|
||||
|
||||
with pytest.raises(litellm.InsufficientQuotaError):
|
||||
retry_function(num_retries=3, original_function=original_function)
|
||||
|
||||
assert original_function.call_count == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
"retry_function",
|
||||
[litellm.acompletion_with_retries, litellm.aresponses_with_retries],
|
||||
)
|
||||
async def test_async_retry_helpers_stop_on_insufficient_quota(retry_function):
|
||||
error = litellm.InsufficientQuotaError(
|
||||
message="You exceeded your current quota",
|
||||
llm_provider="openai",
|
||||
model="gpt-5.5",
|
||||
)
|
||||
original_function = AsyncMock(side_effect=error)
|
||||
|
||||
with pytest.raises(litellm.InsufficientQuotaError):
|
||||
await retry_function(num_retries=3, original_function=original_function)
|
||||
|
||||
assert original_function.await_count == 1
|
||||
|
|
|
|||
|
|
@ -267,25 +267,6 @@ def test_insufficient_quota_error_ignores_rate_limit_retry_policy():
|
|||
assert retries == 0
|
||||
|
||||
|
||||
def test_completion_with_retries_stops_on_insufficient_quota():
|
||||
call_count = 0
|
||||
|
||||
def raise_insufficient_quota(*args, **kwargs):
|
||||
nonlocal call_count
|
||||
call_count += 1
|
||||
raise _make_insufficient_quota_error()
|
||||
|
||||
with pytest.raises(litellm.InsufficientQuotaError):
|
||||
litellm.completion_with_retries(
|
||||
model="gpt-5.5",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
num_retries=3,
|
||||
original_function=raise_insufficient_quota,
|
||||
)
|
||||
|
||||
assert call_count == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_not_found_error_in_retry_loop_raises_immediately():
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
import json
|
||||
import os
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
from unittest.mock import AsyncMock, MagicMock, Mock, patch
|
||||
|
||||
import pytest
|
||||
from jsonschema import validate
|
||||
|
|
@ -23,7 +23,9 @@ from litellm.utils import (
|
|||
ProviderConfigManager,
|
||||
TextCompletionStreamWrapper,
|
||||
_check_provider_match,
|
||||
_get_wrapper_num_retries,
|
||||
_is_streaming_request,
|
||||
client,
|
||||
get_llm_provider,
|
||||
get_optional_params_image_gen,
|
||||
is_cached_message,
|
||||
|
|
@ -4709,3 +4711,47 @@ class TestValidateEnvironmentTencent:
|
|||
assert "TENCENT_API_KEY" in result["missing_keys"]
|
||||
|
||||
|
||||
def test_wrapper_num_retries_is_zero_for_insufficient_quota():
|
||||
kwargs = {"num_retries": 3}
|
||||
retries, returned_kwargs = _get_wrapper_num_retries(
|
||||
kwargs=kwargs,
|
||||
exception=litellm.InsufficientQuotaError(
|
||||
message="You exceeded your current quota",
|
||||
llm_provider="openai",
|
||||
model="gpt-5.5",
|
||||
),
|
||||
)
|
||||
|
||||
assert retries == 0
|
||||
assert returned_kwargs is kwargs
|
||||
|
||||
|
||||
def test_sync_client_does_not_retry_insufficient_quota():
|
||||
completion_mock = Mock(
|
||||
side_effect=litellm.InsufficientQuotaError(
|
||||
message="You exceeded your current quota",
|
||||
llm_provider="openai",
|
||||
model="gpt-5.5",
|
||||
)
|
||||
)
|
||||
responses_mock = Mock(
|
||||
side_effect=litellm.InsufficientQuotaError(
|
||||
message="You exceeded your current quota",
|
||||
llm_provider="openai",
|
||||
model="gpt-5.5",
|
||||
)
|
||||
)
|
||||
|
||||
def completion(*args, **kwargs):
|
||||
return completion_mock(*args, **kwargs)
|
||||
|
||||
def responses(*args, **kwargs):
|
||||
return responses_mock(*args, **kwargs)
|
||||
|
||||
with pytest.raises(litellm.InsufficientQuotaError):
|
||||
client(completion)(num_retries=3)
|
||||
with pytest.raises(litellm.InsufficientQuotaError):
|
||||
client(responses)(num_retries=3)
|
||||
|
||||
assert completion_mock.call_count == 1
|
||||
assert responses_mock.call_count == 1
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue