From 91a3d4ea5eb2c5bf73f940687d4ece701104ce64 Mon Sep 17 00:00:00 2001 From: Ambuj Upadhyay <34904987+lets-order-some-fries@users.noreply.github.com> Date: Tue, 18 Aug 2026 22:44:52 +0530 Subject: [PATCH] fix(openai): async streaming no-response branch now uses the computed status Review follow-up: the helper computed 400 for a bare construction-time OpenAIError, but async_streaming's no-response branch hard-coded 500 and discarded it -- so with stream=True a missing API key still surfaced as a retryable server error, the exact class of misreport this PR exists to fix. Verified at runtime: acompletion(stream=True) with no key raised InternalServerError(500) before, BadRequestError(400) after. Adds an end-to-end async-streaming regression test (red without the fix, green with it); the existing tests covered the helper only, which is why this slipped through. Transient classes are unaffected: _status_code_for_openai_sdk_error still returns 500 for anything without a status_code that is not a bare OpenAIError, and the ReadTimeout->408 branch is untouched. --- litellm/llms/openai/openai.py | 8 ++++++- .../test_openai_sdk_error_status_code.py | 24 +++++++++++++++++++ 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai/openai.py b/litellm/llms/openai/openai.py index 49465ade3e1..94706e19ed9 100644 --- a/litellm/llms/openai/openai.py +++ b/litellm/llms/openai/openai.py @@ -1202,8 +1202,14 @@ class OpenAIChatCompletion(BaseLLM, BaseOpenAILLM): body=exception_body, ) else: + # status_code was computed above by + # _status_code_for_openai_sdk_error; hard-coding 500 + # here made the no-response branch discard it, so an + # async streaming call with no configured API key + # surfaced as a retryable 500 instead of the 400 the + # non-streaming path reports. raise OpenAIError( - status_code=500, + status_code=status_code, message=f"{e}", headers=error_headers, body=exception_body, diff --git a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py index 795dc90c93f..c0857fe6e0f 100644 --- a/tests/unit/llms/openai/test_openai_sdk_error_status_code.py +++ b/tests/unit/llms/openai/test_openai_sdk_error_status_code.py @@ -48,3 +48,27 @@ def test_status_code_zero_is_preserved(): status_code = 0 assert _status_code_for_openai_sdk_error(_ZeroStatus()) == 0 + + +@pytest.mark.asyncio +async def test_async_streaming_missing_credentials_is_not_a_server_error(monkeypatch): + """ + The helper alone is not enough: async_streaming's no-response branch used + to hard-code 500, discarding the status computed above it, so the same + missing-key error that maps to 400 on the non-streaming path surfaced as a + retryable 500 when stream=True. This exercises the full call path. + """ + import litellm + + monkeypatch.delenv("OPENAI_API_KEY", raising=False) + monkeypatch.setattr(litellm, "api_key", None, raising=False) + monkeypatch.setattr(litellm, "openai_key", None, raising=False) + + with pytest.raises(Exception) as exc_info: + await litellm.acompletion( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "hi"}], + stream=True, + ) + + assert getattr(exc_info.value, "status_code", None) == 400