mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Enhance exception mapping for Fireworks AI: add better handling for 429 status codes and text-based rate limit detection. Update tests to verify correct mapping to RateLimitError for both 429 and related error messages.
This commit is contained in:
parent
7c513856dc
commit
fda99ecb41
2 changed files with 77 additions and 2 deletions
|
|
@ -274,7 +274,7 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
+ "Exception"
|
||||
)
|
||||
|
||||
if "429" in error_str:
|
||||
if "429" in error_str or "rate limit" in error_str.lower():
|
||||
exception_mapping_worked = True
|
||||
raise RateLimitError(
|
||||
message=f"RateLimitError: {exception_provider} - {message}",
|
||||
|
|
@ -451,6 +451,15 @@ def exception_type( # type: ignore # noqa: PLR0915
|
|||
response=getattr(original_exception, "response", None),
|
||||
litellm_debug_info=extra_information,
|
||||
)
|
||||
elif original_exception.status_code == 500:
|
||||
exception_mapping_worked = True
|
||||
raise InternalServerError(
|
||||
message=f"InternalServerError: {exception_provider} - {message}",
|
||||
model=model,
|
||||
llm_provider=custom_llm_provider,
|
||||
response=getattr(original_exception, "response", None),
|
||||
litellm_debug_info=extra_information,
|
||||
)
|
||||
elif original_exception.status_code == 503:
|
||||
exception_mapping_worked = True
|
||||
raise ServiceUnavailableError(
|
||||
|
|
|
|||
|
|
@ -773,7 +773,7 @@ def test_litellm_predibase_exception():
|
|||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"provider", ["predibase", "vertex_ai_beta", "anthropic", "databricks", "watsonx"]
|
||||
"provider", ["predibase", "vertex_ai_beta", "anthropic", "databricks", "watsonx", "fireworks_ai"]
|
||||
)
|
||||
def test_exception_mapping(provider):
|
||||
"""
|
||||
|
|
@ -819,6 +819,72 @@ def test_exception_mapping(provider):
|
|||
pass
|
||||
|
||||
|
||||
def test_fireworks_ai_429_exception_mapping():
|
||||
"""
|
||||
Test that Fireworks AI 429 status codes are properly mapped to RateLimitError
|
||||
instead of being converted to 400 BadRequestError.
|
||||
|
||||
Related issue: https://github.com/BerriAI/litellm/issues/xxxxx
|
||||
"""
|
||||
import litellm
|
||||
from litellm.llms.fireworks_ai.common_utils import FireworksAIException
|
||||
|
||||
# Create a mock FireworksAIException with 429 status code
|
||||
mock_exception = FireworksAIException(
|
||||
status_code=429,
|
||||
message="Rate limit exceeded",
|
||||
headers={"retry-after": "60"}
|
||||
)
|
||||
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="fireworks_ai/llama-v3p1-8b-instruct",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
mock_response=mock_exception,
|
||||
)
|
||||
pytest.fail("Expected RateLimitError to be raised")
|
||||
except litellm.RateLimitError as e:
|
||||
# This is the expected behavior - 429 should map to RateLimitError
|
||||
assert e.status_code == 429
|
||||
assert "Rate limit exceeded" in str(e)
|
||||
except Exception as e:
|
||||
pytest.fail(f"Expected RateLimitError but got {type(e).__name__}: {e}")
|
||||
|
||||
|
||||
def test_fireworks_ai_rate_limit_text_detection():
|
||||
"""
|
||||
Test that Fireworks AI rate limit errors are detected by text matching when no 429 status code is present.
|
||||
|
||||
This tests the exact scenario from the real failure where Fireworks AI returns:
|
||||
{"error":{"object":"error","type":"invalid_request_error","message":"rate limit exceeded, please try again later"}}
|
||||
|
||||
This should be detected as a RateLimitError, not a BadRequestError.
|
||||
"""
|
||||
import litellm
|
||||
from litellm.llms.fireworks_ai.common_utils import FireworksAIException
|
||||
|
||||
# Create a mock exception that simulates the exact Fireworks AI response
|
||||
# with "invalid_request_error" type and "rate limit exceeded" message
|
||||
mock_exception = FireworksAIException(
|
||||
status_code=400, # Fireworks AI is returning 400 instead of 429
|
||||
message='{"error":{"object":"error","type":"invalid_request_error","message":"rate limit exceeded, please try again later"}}',
|
||||
headers={}
|
||||
)
|
||||
|
||||
try:
|
||||
response = litellm.completion(
|
||||
model="fireworks_ai/qwen3-235b-a22b",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
mock_response=mock_exception,
|
||||
)
|
||||
pytest.fail("Expected RateLimitError to be raised")
|
||||
except litellm.RateLimitError as e:
|
||||
# This is the expected behavior - should detect "rate limit" text and map to RateLimitError
|
||||
assert "rate limit exceeded" in str(e).lower()
|
||||
except Exception as e:
|
||||
pytest.fail(f"Expected RateLimitError but got {type(e).__name__}: {e}")
|
||||
|
||||
|
||||
def test_anthropic_tool_calling_exception():
|
||||
"""
|
||||
Related - https://github.com/BerriAI/litellm/issues/4348
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue