fix(responses): sync logging stream state

Providers such as ChatGPT can force a Responses API request to stream after the client requested a non-streaming chat completion. Record the resolved wire behavior on the shared logging object so chunk logging does not latch non-stream dedup state before the assembled response is logged.

Tests: uv run pytest tests/test_litellm/llms/custom_httpx/test_llm_http_handler.py tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_handler.py tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py -q
This commit is contained in:
Stephen Chin 2026-08-13 03:39:54 -07:00
parent 601c75a475
commit ff9563eb17
2 changed files with 31 additions and 1 deletions

View file

@ -2423,6 +2423,8 @@ class BaseLLMHTTPHandler:
if extra_body:
data.update(extra_body)
stream = bool(stream or data.get("stream"))
logging_obj.stream = stream
logging_obj.model_call_details.update(stream=stream)
# Preserve the OpenAI-style request context (not sent to the provider) for streaming
# hooks/metadata; the streaming iterator now consumes this to run deployment hooks
@ -2611,6 +2613,8 @@ class BaseLLMHTTPHandler:
if extra_body:
data.update(extra_body)
stream = bool(stream or data.get("stream"))
logging_obj.stream = stream
logging_obj.model_call_details.update(stream=stream)
# Preserve the OpenAI-style request context (not sent to the provider) for streaming
# hooks/metadata; the streaming iterator now consumes this to run deployment hooks

View file

@ -197,6 +197,10 @@ def test_prepare_fake_stream_request():
def test_response_api_handler_streams_when_provider_transform_adds_stream():
from datetime import datetime
from litellm.litellm_core_utils.litellm_logging import Logging
handler = BaseLLMHTTPHandler()
config = Mock()
config.validate_environment.return_value = {}
@ -214,7 +218,22 @@ def test_response_api_handler_streams_when_provider_transform_adds_stream():
request=httpx.Request("POST", "https://chatgpt.example.com/responses"),
)
)
logging_obj = Mock()
logging_obj = Logging(
model="gpt-5.3-codex",
messages=[{"role": "user", "content": "hi"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="test-call",
function_id="test-function",
)
logging_obj.update_environment_variables(
model="gpt-5.3-codex",
user=None,
optional_params={"stream": False},
litellm_params={},
custom_llm_provider="chatgpt",
)
handler.response_api_handler(
model="gpt-5.3-codex",
@ -229,6 +248,10 @@ def test_response_api_handler_streams_when_provider_transform_adds_stream():
assert client.post.call_args.kwargs["stream"] is True
assert client.post.call_args.kwargs["json"]["stream"] is True
assert logging_obj.stream is True
assert logging_obj.model_call_details["stream"] is True
logging_obj.has_run_logging(event_type="async_success")
assert logging_obj.should_run_logging(event_type="async_success") is True
def test_response_api_handler_runs_agentic_hooks_in_sync_path(monkeypatch):
@ -356,6 +379,7 @@ async def test_async_response_api_handler_streams_when_provider_transform_adds_s
)
)
logging_obj = Mock()
logging_obj.model_call_details = {}
await handler.async_response_api_handler(
model="gpt-5.3-codex",
@ -370,6 +394,8 @@ async def test_async_response_api_handler_streams_when_provider_transform_adds_s
assert client.post.call_args.kwargs["stream"] is True
assert client.post.call_args.kwargs["json"]["stream"] is True
assert logging_obj.stream is True
assert logging_obj.model_call_details["stream"] is True
@pytest.mark.asyncio