diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index c514ffd12f9..3a78519074f 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -274,7 +274,7 @@ def exception_type( # type: ignore # noqa: PLR0915 + "Exception" ) - if "429" in error_str: + if "429" in error_str or "rate limit" in error_str.lower(): exception_mapping_worked = True raise RateLimitError( message=f"RateLimitError: {exception_provider} - {message}", @@ -451,6 +451,15 @@ def exception_type( # type: ignore # noqa: PLR0915 response=getattr(original_exception, "response", None), litellm_debug_info=extra_information, ) + elif original_exception.status_code == 500: + exception_mapping_worked = True + raise InternalServerError( + message=f"InternalServerError: {exception_provider} - {message}", + model=model, + llm_provider=custom_llm_provider, + response=getattr(original_exception, "response", None), + litellm_debug_info=extra_information, + ) elif original_exception.status_code == 503: exception_mapping_worked = True raise ServiceUnavailableError( diff --git a/tests/local_testing/test_exceptions.py b/tests/local_testing/test_exceptions.py index 32f0a4f84af..275172291bb 100644 --- a/tests/local_testing/test_exceptions.py +++ b/tests/local_testing/test_exceptions.py @@ -773,7 +773,7 @@ def test_litellm_predibase_exception(): @pytest.mark.parametrize( - "provider", ["predibase", "vertex_ai_beta", "anthropic", "databricks", "watsonx"] + "provider", ["predibase", "vertex_ai_beta", "anthropic", "databricks", "watsonx", "fireworks_ai"] ) def test_exception_mapping(provider): """ @@ -819,6 +819,72 @@ def test_exception_mapping(provider): pass +def test_fireworks_ai_429_exception_mapping(): + """ + Test that Fireworks AI 429 status codes are properly mapped to RateLimitError + instead of being converted to 400 BadRequestError. + + Related issue: https://github.com/BerriAI/litellm/issues/xxxxx + """ + import litellm + from litellm.llms.fireworks_ai.common_utils import FireworksAIException + + # Create a mock FireworksAIException with 429 status code + mock_exception = FireworksAIException( + status_code=429, + message="Rate limit exceeded", + headers={"retry-after": "60"} + ) + + try: + response = litellm.completion( + model="fireworks_ai/llama-v3p1-8b-instruct", + messages=[{"role": "user", "content": "Hello"}], + mock_response=mock_exception, + ) + pytest.fail("Expected RateLimitError to be raised") + except litellm.RateLimitError as e: + # This is the expected behavior - 429 should map to RateLimitError + assert e.status_code == 429 + assert "Rate limit exceeded" in str(e) + except Exception as e: + pytest.fail(f"Expected RateLimitError but got {type(e).__name__}: {e}") + + +def test_fireworks_ai_rate_limit_text_detection(): + """ + Test that Fireworks AI rate limit errors are detected by text matching when no 429 status code is present. + + This tests the exact scenario from the real failure where Fireworks AI returns: + {"error":{"object":"error","type":"invalid_request_error","message":"rate limit exceeded, please try again later"}} + + This should be detected as a RateLimitError, not a BadRequestError. + """ + import litellm + from litellm.llms.fireworks_ai.common_utils import FireworksAIException + + # Create a mock exception that simulates the exact Fireworks AI response + # with "invalid_request_error" type and "rate limit exceeded" message + mock_exception = FireworksAIException( + status_code=400, # Fireworks AI is returning 400 instead of 429 + message='{"error":{"object":"error","type":"invalid_request_error","message":"rate limit exceeded, please try again later"}}', + headers={} + ) + + try: + response = litellm.completion( + model="fireworks_ai/qwen3-235b-a22b", + messages=[{"role": "user", "content": "Hello"}], + mock_response=mock_exception, + ) + pytest.fail("Expected RateLimitError to be raised") + except litellm.RateLimitError as e: + # This is the expected behavior - should detect "rate limit" text and map to RateLimitError + assert "rate limit exceeded" in str(e).lower() + except Exception as e: + pytest.fail(f"Expected RateLimitError but got {type(e).__name__}: {e}") + + def test_anthropic_tool_calling_exception(): """ Related - https://github.com/BerriAI/litellm/issues/4348