From 86345dd2a1d5f856e210e93c74abd0ae0963ba27 Mon Sep 17 00:00:00 2001 From: Aakash Pydi Date: Sun, 27 Sep 2026 16:12:37 -0700 Subject: [PATCH 1/3] fix(streaming): log text completion streams that fail before the first byte A streaming text completion that fails before the first byte, e.g. the deployment refuses the connection or answers 5xx, never reached failure callbacks. The request is sent before the stream wrapper that logs failures exists, so nothing logged it, unlike the same request non-streamed or on chat completions Log that failure the same way the stream wrapper does, then re-raise --- litellm/llms/openai/completion/handler.py | 10 ++++- .../completion/test_completion_handler.py | 37 +++++++++++++++++++ 2 files changed, 46 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai/completion/handler.py b/litellm/llms/openai/completion/handler.py index c7b59509eb0..6dc738afe83 100644 --- a/litellm/llms/openai/completion/handler.py +++ b/litellm/llms/openai/completion/handler.py @@ -1,4 +1,6 @@ +import asyncio import json +import traceback from collections.abc import Callable from typing import Final @@ -298,7 +300,13 @@ class OpenAITextCompletion(BaseLLM): else: openai_client = client - raw_response: Final = await openai_client.completions.with_raw_response.create(**data) + try: + raw_response: Final = await openai_client.completions.with_raw_response.create(**data) + except Exception as e: + asyncio.create_task( + logging_obj.dispatch_failure_handlers(e, traceback.format_exc(), prefer_async_handlers=True) + ) + raise response: Final = raw_response.parse() streamwrapper: Final = CustomStreamWrapper( completion_stream=response, diff --git a/tests/unit/llms/openai/completion/test_completion_handler.py b/tests/unit/llms/openai/completion/test_completion_handler.py index 329956605ab..6850c6a9cca 100644 --- a/tests/unit/llms/openai/completion/test_completion_handler.py +++ b/tests/unit/llms/openai/completion/test_completion_handler.py @@ -6,6 +6,9 @@ Regression tests for https://github.com/BerriAI/litellm/issues/27410 """ +import asyncio + +import httpx import pytest import respx from httpx import Response @@ -13,6 +16,7 @@ from httpx import Response import litellm from litellm import atext_completion, text_completion +from litellm.integrations.custom_logger import CustomLogger @pytest.fixture(autouse=True) @@ -88,3 +92,36 @@ async def test_acompletion_forwards_client_headers_to_provider( request_headers = mock_completions_endpoint.calls.last.request.headers assert request_headers["x-mycorp-llmcall-id"] == "abc-123" + +class _FailureCounter(CustomLogger): + def __init__(self): + self.failures = 0 + + async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): + self.failures += 1 + + +@pytest.mark.parametrize( + ("provider_response", "expected_error"), + [ + pytest.param(httpx.ConnectError("connection refused"), litellm.APIConnectionError, id="connection-refused"), + pytest.param(Response(500, json={"error": {"message": "boom"}}), litellm.InternalServerError, id="5xx-before-first-byte"), + ], +) +@respx.mock +async def test_astream_failing_before_first_byte_logs_one_failure(provider_response, expected_error, monkeypatch): + respx.post("https://api.openai.com/v1/completions").mock(side_effect=provider_response) + counter = _FailureCounter() + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + monkeypatch.setattr(litellm, "callbacks", [counter]) + monkeypatch.setattr(litellm, "_async_failure_callback", [counter]) + + response = await atext_completion( + model="gpt-3.5-turbo-instruct", prompt="hello", max_tokens=5, stream=True, max_retries=0 + ) + with pytest.raises(expected_error): + async for _ in response: + pass + await asyncio.sleep(0.5) + + assert counter.failures == 1 From 5fc869c91583af627a60d31657695afa3dbac76b Mon Sep 17 00:00:00 2001 From: Aakash Pydi Date: Sun, 27 Sep 2026 16:46:06 -0700 Subject: [PATCH 2/3] fix(streaming): await failure logging before re-raising a failed text stream Await the failure handlers instead of scheduling them, matching the non-streaming path, so the failure is logged even if the event loop shuts down right after. The test now checks the failure payload and no longer waits on a timer --- litellm/llms/openai/completion/handler.py | 5 +---- .../completion/test_completion_handler.py | 21 ++++++++++--------- 2 files changed, 12 insertions(+), 14 deletions(-) diff --git a/litellm/llms/openai/completion/handler.py b/litellm/llms/openai/completion/handler.py index 6dc738afe83..52bc64e92ec 100644 --- a/litellm/llms/openai/completion/handler.py +++ b/litellm/llms/openai/completion/handler.py @@ -1,4 +1,3 @@ -import asyncio import json import traceback from collections.abc import Callable @@ -303,9 +302,7 @@ class OpenAITextCompletion(BaseLLM): try: raw_response: Final = await openai_client.completions.with_raw_response.create(**data) except Exception as e: - asyncio.create_task( - logging_obj.dispatch_failure_handlers(e, traceback.format_exc(), prefer_async_handlers=True) - ) + await logging_obj.dispatch_failure_handlers(e, traceback.format_exc(), prefer_async_handlers=True) raise response: Final = raw_response.parse() streamwrapper: Final = CustomStreamWrapper( diff --git a/tests/unit/llms/openai/completion/test_completion_handler.py b/tests/unit/llms/openai/completion/test_completion_handler.py index 6850c6a9cca..647495121f1 100644 --- a/tests/unit/llms/openai/completion/test_completion_handler.py +++ b/tests/unit/llms/openai/completion/test_completion_handler.py @@ -6,8 +6,6 @@ Regression tests for https://github.com/BerriAI/litellm/issues/27410 """ -import asyncio - import httpx import pytest import respx @@ -93,12 +91,12 @@ async def test_acompletion_forwards_client_headers_to_provider( request_headers = mock_completions_endpoint.calls.last.request.headers assert request_headers["x-mycorp-llmcall-id"] == "abc-123" -class _FailureCounter(CustomLogger): +class _FailureRecorder(CustomLogger): def __init__(self): - self.failures = 0 + self.payloads = [] async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): - self.failures += 1 + self.payloads.append(kwargs["standard_logging_object"]) @pytest.mark.parametrize( @@ -111,10 +109,10 @@ class _FailureCounter(CustomLogger): @respx.mock async def test_astream_failing_before_first_byte_logs_one_failure(provider_response, expected_error, monkeypatch): respx.post("https://api.openai.com/v1/completions").mock(side_effect=provider_response) - counter = _FailureCounter() + recorder = _FailureRecorder() monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) - monkeypatch.setattr(litellm, "callbacks", [counter]) - monkeypatch.setattr(litellm, "_async_failure_callback", [counter]) + monkeypatch.setattr(litellm, "callbacks", [recorder]) + monkeypatch.setattr(litellm, "_async_failure_callback", [recorder]) response = await atext_completion( model="gpt-3.5-turbo-instruct", prompt="hello", max_tokens=5, stream=True, max_retries=0 @@ -122,6 +120,9 @@ async def test_astream_failing_before_first_byte_logs_one_failure(provider_respo with pytest.raises(expected_error): async for _ in response: pass - await asyncio.sleep(0.5) - assert counter.failures == 1 + assert len(recorder.payloads) == 1 + payload = recorder.payloads[0] + assert payload["status"] == "failure" + assert payload["custom_llm_provider"] == "text-completion-openai" + assert payload["model"] == "gpt-3.5-turbo-instruct" From d980bbf4c93ea48d98465b83c9bf0219c2ad7d36 Mon Sep 17 00:00:00 2001 From: Aakash Pydi Date: Sun, 27 Sep 2026 16:58:43 -0700 Subject: [PATCH 3/3] fix(streaming): schedule text stream failure logging like the stream wrapper Awaiting the handlers let a slow or raising callback delay or replace the provider error. Schedule them the way CustomStreamWrapper already logs stream failures, and have the test wait on the callback instead of a timer --- litellm/llms/openai/completion/handler.py | 5 ++++- tests/unit/llms/openai/completion/test_completion_handler.py | 5 +++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/litellm/llms/openai/completion/handler.py b/litellm/llms/openai/completion/handler.py index 52bc64e92ec..6dc738afe83 100644 --- a/litellm/llms/openai/completion/handler.py +++ b/litellm/llms/openai/completion/handler.py @@ -1,3 +1,4 @@ +import asyncio import json import traceback from collections.abc import Callable @@ -302,7 +303,9 @@ class OpenAITextCompletion(BaseLLM): try: raw_response: Final = await openai_client.completions.with_raw_response.create(**data) except Exception as e: - await logging_obj.dispatch_failure_handlers(e, traceback.format_exc(), prefer_async_handlers=True) + asyncio.create_task( + logging_obj.dispatch_failure_handlers(e, traceback.format_exc(), prefer_async_handlers=True) + ) raise response: Final = raw_response.parse() streamwrapper: Final = CustomStreamWrapper( diff --git a/tests/unit/llms/openai/completion/test_completion_handler.py b/tests/unit/llms/openai/completion/test_completion_handler.py index 647495121f1..4af8c3ad3c4 100644 --- a/tests/unit/llms/openai/completion/test_completion_handler.py +++ b/tests/unit/llms/openai/completion/test_completion_handler.py @@ -6,6 +6,8 @@ Regression tests for https://github.com/BerriAI/litellm/issues/27410 """ +import asyncio + import httpx import pytest import respx @@ -94,9 +96,11 @@ async def test_acompletion_forwards_client_headers_to_provider( class _FailureRecorder(CustomLogger): def __init__(self): self.payloads = [] + self.logged = asyncio.Event() async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time): self.payloads.append(kwargs["standard_logging_object"]) + self.logged.set() @pytest.mark.parametrize( @@ -120,6 +124,7 @@ async def test_astream_failing_before_first_byte_logs_one_failure(provider_respo with pytest.raises(expected_error): async for _ in response: pass + await asyncio.wait_for(recorder.logged.wait(), timeout=5) assert len(recorder.payloads) == 1 payload = recorder.payloads[0]