From ff9563eb17c307980f08560423632fee9ab65704 Mon Sep 17 00:00:00 2001 From: Stephen Chin <1290231+steveonjava@users.noreply.github.com> Date: Thu, 13 Aug 2026 03:39:54 -0700 Subject: [PATCH] fix(responses): sync logging stream state Providers such as ChatGPT can force a Responses API request to stream after the client requested a non-streaming chat completion. Record the resolved wire behavior on the shared logging object so chunk logging does not latch non-stream dedup state before the assembled response is logged. Tests: uv run pytest tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py -q --- litellm/llms/custom_httpx/llm_http_handler.py | 4 +++ .../custom_httpx/test_llm_http_handler.py | 28 ++++++++++++++++++- 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 9eccfe12e71..56ed57daede 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -2423,6 +2423,8 @@ class BaseLLMHTTPHandler: if extra_body: data.update(extra_body) stream = bool(stream or data.get("stream")) + logging_obj.stream = stream + logging_obj.model_call_details.update(stream=stream) # Preserve the OpenAI-style request context (not sent to the provider) for streaming # hooks/metadata; the streaming iterator now consumes this to run deployment hooks @@ -2611,6 +2613,8 @@ class BaseLLMHTTPHandler: if extra_body: data.update(extra_body) stream = bool(stream or data.get("stream")) + logging_obj.stream = stream + logging_obj.model_call_details.update(stream=stream) # Preserve the OpenAI-style request context (not sent to the provider) for streaming # hooks/metadata; the streaming iterator now consumes this to run deployment hooks diff --git a/tests/unit/llms/custom_httpx/test_llm_http_handler.py b/tests/unit/llms/custom_httpx/test_llm_http_handler.py index 399e4dbf206..4b2f054cf19 100644 --- a/tests/unit/llms/custom_httpx/test_llm_http_handler.py +++ b/tests/unit/llms/custom_httpx/test_llm_http_handler.py @@ -197,6 +197,10 @@ def test_prepare_fake_stream_request(): def test_response_api_handler_streams_when_provider_transform_adds_stream(): + from datetime import datetime + + from litellm.litellm_core_utils.litellm_logging import Logging + handler = BaseLLMHTTPHandler() config = Mock() config.validate_environment.return_value = {} @@ -214,7 +218,22 @@ def test_response_api_handler_streams_when_provider_transform_adds_stream(): request=httpx.Request("POST", "https://chatgpt.example.com/responses"), ) ) - logging_obj = Mock() + logging_obj = Logging( + model="gpt-5.3-codex", + messages=[{"role": "user", "content": "hi"}], + stream=False, + call_type="completion", + start_time=datetime.now(), + litellm_call_id="test-call", + function_id="test-function", + ) + logging_obj.update_environment_variables( + model="gpt-5.3-codex", + user=None, + optional_params={"stream": False}, + litellm_params={}, + custom_llm_provider="chatgpt", + ) handler.response_api_handler( model="gpt-5.3-codex", @@ -229,6 +248,10 @@ def test_response_api_handler_streams_when_provider_transform_adds_stream(): assert client.post.call_args.kwargs["stream"] is True assert client.post.call_args.kwargs["json"]["stream"] is True + assert logging_obj.stream is True + assert logging_obj.model_call_details["stream"] is True + logging_obj.has_run_logging(event_type="async_success") + assert logging_obj.should_run_logging(event_type="async_success") is True def test_response_api_handler_runs_agentic_hooks_in_sync_path(monkeypatch): @@ -356,6 +379,7 @@ async def test_async_response_api_handler_streams_when_provider_transform_adds_s ) ) logging_obj = Mock() + logging_obj.model_call_details = {} await handler.async_response_api_handler( model="gpt-5.3-codex", @@ -370,6 +394,8 @@ async def test_async_response_api_handler_streams_when_provider_transform_adds_s assert client.post.call_args.kwargs["stream"] is True assert client.post.call_args.kwargs["json"]["stream"] is True + assert logging_obj.stream is True + assert logging_obj.model_call_details["stream"] is True @pytest.mark.asyncio