diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index d6340d182ae..94706e19ed9 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -358,6 +358,27 @@ def _embedding_request_without_sdk_defaults( return body, options +def _status_code_for_openai_sdk_error(e: Exception) -> int: + """ + Status code to attribute to an exception raised by the OpenAI SDK. + + A bare ``openai.OpenAIError`` is raised at client construction, before any HTTP + exchange, so it carries no status code; missing credentials is the common case. + Defaulting it to 500 asserts that a server responded with a server error, which + makes a permanently unfixable configuration error retryable. Subclasses such as + ``APIConnectionError`` and ``APIStatusError`` keep their existing treatment, so + genuinely transient failures are still retried. + + See https://github.com/BerriAI/litellm/issues/35860 + """ + status_code: Final = getattr(e, "status_code", None) + if status_code is not None: + return status_code + if type(e) is openai.OpenAIError: + return 400 + return 500 + + class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): def __init__(self) -> None: super().__init__() @@ -998,7 +1019,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): # e.message except Exception as e: exception_response = getattr(e, "response", None) - status_code = getattr(e, "status_code", 500) + status_code = _status_code_for_openai_sdk_error(e) exception_body = getattr(e, "body", None) error_headers = getattr(e, "headers", None) if error_headers is None and exception_response: @@ -1153,7 +1174,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): raise e error_headers = getattr(e, "headers", None) - status_code = getattr(e, "status_code", 500) + status_code = _status_code_for_openai_sdk_error(e) error_response = getattr(e, "response", None) exception_body = getattr(e, "body", None) if error_headers is None and error_response: @@ -1181,8 +1202,14 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): body=exception_body, ) else: + # status_code was computed above by + # _status_code_for_openai_sdk_error; hard-coding 500 + # here made the no-response branch discard it, so an + # async streaming call with no configured API key + # surfaced as a retryable 500 instead of the 400 the + # non-streaming path reports. raise OpenAIError( - status_code=500, + status_code=status_code, message=f"{e}", headers=error_headers, body=exception_body, diff --git a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py new file mode 100644 index 00000000000..adda6a9f447 --- /dev/null +++ b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py @@ -0,0 +1,74 @@ +import httpx +import openai +import pytest + +from litellm.llms.openai.openai import _status_code_for_openai_sdk_error + + +def test_missing_credentials_is_not_a_server_error(): + """ + Regression test for https://github.com/BerriAI/litellm/issues/35860 + + The SDK raises a bare OpenAIError at client construction when no key is + configured. That happens before any HTTP exchange, so calling it a 500 marks a + permanently unfixable configuration error as a retryable server error. + """ + with pytest.raises(openai.OpenAIError) as exc_info: + openai.OpenAI(api_key=None) + + error = exc_info.value + assert not hasattr(error, "status_code") + assert _status_code_for_openai_sdk_error(error) == 400 + + +def test_connection_error_stays_retryable(): + """Transient failures must keep the 500 default so retries still happen.""" + error = openai.APIConnectionError(request=httpx.Request("POST", "http://example.com")) + + assert not hasattr(error, "status_code") + assert _status_code_for_openai_sdk_error(error) == 500 + + +def test_existing_status_code_is_preserved(): + class _WithStatus(Exception): + status_code = 429 + + assert _status_code_for_openai_sdk_error(_WithStatus()) == 429 + + +def test_unknown_exception_keeps_the_server_error_default(): + """Anything that is not a bare OpenAIError keeps the previous behaviour.""" + assert _status_code_for_openai_sdk_error(ValueError("boom")) == 500 + + +def test_status_code_zero_is_preserved(): + """A falsy-but-present status code must not fall through to the default.""" + + class _ZeroStatus(Exception): + status_code = 0 + + assert _status_code_for_openai_sdk_error(_ZeroStatus()) == 0 + + +@pytest.mark.asyncio +async def test_async_streaming_missing_credentials_is_not_a_server_error(monkeypatch): + """ + The helper alone is not enough: async_streaming's no-response branch used + to hard-code 500, discarding the status computed above it, so the same + missing-key error that maps to 400 on the non-streaming path surfaced as a + retryable 500 when stream=True. This exercises the full call path. + """ + import litellm + + monkeypatch.delenv("OPENAI_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", None, raising=False) + monkeypatch.setattr(litellm, "openai_key", None, raising=False) + + with pytest.raises(openai.OpenAIError) as exc_info: + await litellm.acompletion( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "hi"}], + stream=True, + ) + + assert getattr(exc_info.value, "status_code", None) == 400