fix(responses): sync logging stream state

Providers such as ChatGPT can force a Responses API request to stream after the client requested a non-streaming chat completion. Record the resolved wire behavior on the shared logging object so chunk logging does not latch non-stream dedup state before the assembled response is logged.

Tests: uv run pytest tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py -q
This commit is contained in:
Stephen Chin 2026-08-13 03:39:54 -07:00
parent 09889e1986
commit 91534fa481
2 changed files with 31 additions and 1 deletions

View file

@ -2544,6 +2544,8 @@ class BaseLLMHTTPHandler:
if extra_body:
data.update(extra_body)
stream = bool(stream or data.get("stream"))
logging_obj.stream = stream
logging_obj.model_call_details.update(stream=stream)
# Preserve the OpenAI-style request context (not sent to the provider) for streaming
# hooks/metadata; the streaming iterator now consumes this to run deployment hooks
@ -2722,6 +2724,8 @@ class BaseLLMHTTPHandler:
if extra_body:
data.update(extra_body)
stream = bool(stream or data.get("stream"))
logging_obj.stream = stream
logging_obj.model_call_details.update(stream=stream)
# Preserve the OpenAI-style request context (not sent to the provider) for streaming
# hooks/metadata; the streaming iterator now consumes this to run deployment hooks

View file

@ -89,6 +89,10 @@ def test_prepare_fake_stream_request():
def test_response_api_handler_streams_when_provider_transform_adds_stream():
from datetime import datetime
from litellm.litellm_core_utils.litellm_logging import Logging
handler = BaseLLMHTTPHandler()
config = Mock()
config.validate_environment.return_value = {}
@ -106,7 +110,22 @@ def test_response_api_handler_streams_when_provider_transform_adds_stream():
request=httpx.Request("POST", "https://chatgpt.example.com/responses"),
)
)
logging_obj = Mock()
logging_obj = Logging(
model="gpt-5.3-codex",
messages=[{"role": "user", "content": "hi"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-call",
function_id="test-function",
)
logging_obj.update_environment_variables(
model="gpt-5.3-codex",
user=None,
optional_params={"stream": False},
litellm_params={},
custom_llm_provider="chatgpt",
)
handler.response_api_handler(
model="gpt-5.3-codex",
@ -121,6 +140,10 @@ def test_response_api_handler_streams_when_provider_transform_adds_stream():
assert client.post.call_args.kwargs["stream"] is True
assert client.post.call_args.kwargs["json"]["stream"] is True
assert logging_obj.stream is True
assert logging_obj.model_call_details["stream"] is True
logging_obj.has_run_logging(event_type="async_success")
assert logging_obj.should_run_logging(event_type="async_success") is True
def test_response_api_handler_runs_agentic_hooks_in_sync_path(monkeypatch):
@ -250,6 +273,7 @@ async def test_async_response_api_handler_streams_when_provider_transform_adds_s
)
)
logging_obj = Mock()
logging_obj.model_call_details = {}
await handler.async_response_api_handler(
model="gpt-5.3-codex",
@ -264,6 +288,8 @@ async def test_async_response_api_handler_streams_when_provider_transform_adds_s
assert client.post.call_args.kwargs["stream"] is True
assert client.post.call_args.kwargs["json"]["stream"] is True
assert logging_obj.stream is True
assert logging_obj.model_call_details["stream"] is True
def test_get_agentic_loop_settings_defaults_and_overrides():