From 3dac50ef9269f701f5cf9c6fb811081e80377e1b Mon Sep 17 00:00:00 2001 From: Ambuj Upadhyay <34904987+lets-order-some-fries@users.noreply.github.com> Date: Thu, 6 Aug 2026 17:35:44 +0530 Subject: [PATCH 1/3] fix(openai): stop reporting a missing API key as a retryable server error The OpenAI SDK raises a bare openai.OpenAIError at client construction when no credentials are configured. That happens before any HTTP exchange, so the exception carries no status code, and the broad handler defaulted it to 500. Downstream that becomes an InternalServerError, which every standard retry policy treats as transient, so a permanently unfixable configuration error is retried until the attempts run out and the useful message arrives last Attribute 400 to a bare OpenAIError instead. The check is on the exact type, so APIConnectionError and APIStatusError keep the existing default and genuinely transient failures are still retried Worth noting for triage: the existing string match in exception_mapping_utils targets 'The api_key client option must be set...', which current SDK versions no longer emit; the message is now 'Missing credentials. Please pass an api_key...', so that branch no longer catches this case Fixes #35860 --- litellm/llms/openai/openai.py | 25 +++++++++- .../test_openai_sdk_error_status_code.py | 50 +++++++++++++++++++ 2 files changed, 73 insertions(+), 2 deletions(-) create mode 100644 tests/unit/llms/openai/test_openai_sdk_error_status_code.py diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index d6340d182ae..49465ade3e1 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -358,6 +358,27 @@ def _embedding_request_without_sdk_defaults( return body, options +def _status_code_for_openai_sdk_error(e: Exception) -> int: + """ + Status code to attribute to an exception raised by the OpenAI SDK. + + A bare ``openai.OpenAIError`` is raised at client construction, before any HTTP + exchange, so it carries no status code; missing credentials is the common case. + Defaulting it to 500 asserts that a server responded with a server error, which + makes a permanently unfixable configuration error retryable. Subclasses such as + ``APIConnectionError`` and ``APIStatusError`` keep their existing treatment, so + genuinely transient failures are still retried. + + See https://github.com/BerriAI/litellm/issues/35860 + """ + status_code: Final = getattr(e, "status_code", None) + if status_code is not None: + return status_code + if type(e) is openai.OpenAIError: + return 400 + return 500 + + class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): def __init__(self) -> None: super().__init__() @@ -998,7 +1019,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): # e.message except Exception as e: exception_response = getattr(e, "response", None) - status_code = getattr(e, "status_code", 500) + status_code = _status_code_for_openai_sdk_error(e) exception_body = getattr(e, "body", None) error_headers = getattr(e, "headers", None) if error_headers is None and exception_response: @@ -1153,7 +1174,7 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): raise e error_headers = getattr(e, "headers", None) - status_code = getattr(e, "status_code", 500) + status_code = _status_code_for_openai_sdk_error(e) error_response = getattr(e, "response", None) exception_body = getattr(e, "body", None) if error_headers is None and error_response: diff --git a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py new file mode 100644 index 00000000000..795dc90c93f --- /dev/null +++ b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py @@ -0,0 +1,50 @@ +import httpx +import openai +import pytest + +from litellm.llms.openai.openai import _status_code_for_openai_sdk_error + + +def test_missing_credentials_is_not_a_server_error(): + """ + Regression test for https://github.com/BerriAI/litellm/issues/35860 + + The SDK raises a bare OpenAIError at client construction when no key is + configured. That happens before any HTTP exchange, so calling it a 500 marks a + permanently unfixable configuration error as a retryable server error. + """ + with pytest.raises(openai.OpenAIError) as exc_info: + openai.OpenAI(api_key=None) + + error = exc_info.value + assert not hasattr(error, "status_code") + assert _status_code_for_openai_sdk_error(error) == 400 + + +def test_connection_error_stays_retryable(): + """Transient failures must keep the 500 default so retries still happen.""" + error = openai.APIConnectionError(request=httpx.Request("POST", "http://example.com")) + + assert not hasattr(error, "status_code") + assert _status_code_for_openai_sdk_error(error) == 500 + + +def test_existing_status_code_is_preserved(): + class _WithStatus(Exception): + status_code = 429 + + assert _status_code_for_openai_sdk_error(_WithStatus()) == 429 + + +def test_unknown_exception_keeps_the_server_error_default(): + """Anything that is not a bare OpenAIError keeps the previous behaviour.""" + assert _status_code_for_openai_sdk_error(ValueError("boom")) == 500 + + +def test_status_code_zero_is_preserved(): + """A falsy-but-present status code must not fall through to the default.""" + + class _ZeroStatus(Exception): + status_code = 0 + + assert _status_code_for_openai_sdk_error(_ZeroStatus()) == 0 From 91a3d4ea5eb2c5bf73f940687d4ece701104ce64 Mon Sep 17 00:00:00 2001 From: Ambuj Upadhyay <34904987+lets-order-some-fries@users.noreply.github.com> Date: Tue, 18 Aug 2026 22:44:52 +0530 Subject: [PATCH 2/3] fix(openai): async streaming no-response branch now uses the computed status Review follow-up: the helper computed 400 for a bare construction-time OpenAIError, but async_streaming's no-response branch hard-coded 500 and discarded it -- so with stream=True a missing API key still surfaced as a retryable server error, the exact class of misreport this PR exists to fix. Verified at runtime: acompletion(stream=True) with no key raised InternalServerError(500) before, BadRequestError(400) after. Adds an end-to-end async-streaming regression test (red without the fix, green with it); the existing tests covered the helper only, which is why this slipped through. Transient classes are unaffected: _status_code_for_openai_sdk_error still returns 500 for anything without a status_code that is not a bare OpenAIError, and the ReadTimeout->408 branch is untouched. --- litellm/llms/openai/openai.py | 8 ++++++- .../test_openai_sdk_error_status_code.py | 24 +++++++++++++++++++ 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 49465ade3e1..94706e19ed9 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -1202,8 +1202,14 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): body=exception_body, ) else: + # status_code was computed above by + # _status_code_for_openai_sdk_error; hard-coding 500 + # here made the no-response branch discard it, so an + # async streaming call with no configured API key + # surfaced as a retryable 500 instead of the 400 the + # non-streaming path reports. raise OpenAIError( - status_code=500, + status_code=status_code, message=f"{e}", headers=error_headers, body=exception_body, diff --git a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py index 795dc90c93f..c0857fe6e0f 100644 --- a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py +++ b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py @@ -48,3 +48,27 @@ def test_status_code_zero_is_preserved(): status_code = 0 assert _status_code_for_openai_sdk_error(_ZeroStatus()) == 0 + + +@pytest.mark.asyncio +async def test_async_streaming_missing_credentials_is_not_a_server_error(monkeypatch): + """ + The helper alone is not enough: async_streaming's no-response branch used + to hard-code 500, discarding the status computed above it, so the same + missing-key error that maps to 400 on the non-streaming path surfaced as a + retryable 500 when stream=True. This exercises the full call path. + """ + import litellm + + monkeypatch.delenv("OPENAI_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", None, raising=False) + monkeypatch.setattr(litellm, "openai_key", None, raising=False) + + with pytest.raises(Exception) as exc_info: + await litellm.acompletion( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "hi"}], + stream=True, + ) + + assert getattr(exc_info.value, "status_code", None) == 400 From 2bc189ff8fa306962818dd5f2ac0d685fcf16798 Mon Sep 17 00:00:00 2001 From: Ambuj Upadhyay Date: Wed, 2 Sep 2026 07:27:01 +0530 Subject: [PATCH 3/3] test: narrow the async-streaming regression to openai.OpenAIError Ruff PT011 (test tree) rejects a bare pytest.raises(Exception). Every litellm exception subclasses an openai one, so OpenAIError is the accurate common base here and matches the assertion the first test in this file already uses. The status_code == 400 assertion is unchanged. --- tests/unit/llms/openai/test_openai_sdk_error_status_code.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py index c0857fe6e0f..adda6a9f447 100644 --- a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py +++ b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py @@ -64,7 +64,7 @@ async def test_async_streaming_missing_credentials_is_not_a_server_error(monkeyp monkeypatch.setattr(litellm, "api_key", None, raising=False) monkeypatch.setattr(litellm, "openai_key", None, raising=False) - with pytest.raises(Exception) as exc_info: + with pytest.raises(openai.OpenAIError) as exc_info: await litellm.acompletion( model="gpt-4o-mini", messages=[{"role": "user", "content": "hi"}],