diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 7d2954fd2dd..9296e9635cd 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -920,9 +920,17 @@ class ProxyBaseLLMRequestProcessing: ### AUTO STREAM USAGE TRACKING ### # If always_include_stream_usage is enabled and this is a streaming request # automatically add stream_options={'include_usage': True} if not already set + # NOTE: Only apply to chat completions, NOT Responses API routes. + # Azure/OpenAI Responses API does not support stream_options (usage is + # included automatically in response.completed events). + _is_responses_api_route = route_type in { + "aresponses", + "_aresponses_websocket", + } if ( general_settings.get("always_include_stream_usage", False) is True and self.data.get("stream", False) is True + and not _is_responses_api_route ): # Only set if stream_options is not already provided by the client if "stream_options" not in self.data: diff --git a/tests/test_litellm/test_responses_api_stream_options.py b/tests/test_litellm/test_responses_api_stream_options.py new file mode 100644 index 00000000000..5351fdfdf38 --- /dev/null +++ b/tests/test_litellm/test_responses_api_stream_options.py @@ -0,0 +1,75 @@ +""" +Unit tests for fix #28553: stream_options should NOT be injected for Responses API routes. + +The proxy's `common_processing_pre_call_logic` injects `stream_options={'include_usage': True}` +when `always_include_stream_usage` is enabled. This must NOT happen for Responses API routes +(`aresponses`, `_aresponses_websocket`) because the Responses API does not support +`stream_options` — usage is included automatically in response.completed events. +""" + +import pytest + + +def _apply_stream_options_logic(data: dict, general_settings: dict, route_type: str): + """Reproduces the stream_options injection logic from common_processing_pre_call_logic.""" + _is_responses_api_route = route_type in { + "aresponses", + "_aresponses_websocket", + } + if ( + general_settings.get("always_include_stream_usage", False) is True + and data.get("stream", False) is True + and not _is_responses_api_route + ): + if "stream_options" not in data: + data["stream_options"] = {"include_usage": True} + elif ( + isinstance(data["stream_options"], dict) + and "include_usage" not in data["stream_options"] + ): + data["stream_options"]["include_usage"] = True + + +class TestStreamOptionsNotInjectedForResponsesAPI: + """Verify stream_options is skipped for Responses API routes.""" + + @pytest.mark.parametrize("route_type", ["aresponses", "_aresponses_websocket"]) + def test_stream_options_not_injected_for_responses_routes(self, route_type): + """stream_options must NOT be added when route is a Responses API route.""" + data = {"stream": True, "model": "gpt-4"} + _apply_stream_options_logic( + data, {"always_include_stream_usage": True}, route_type + ) + assert "stream_options" not in data + + def test_stream_options_injected_for_chat_completions(self): + """stream_options SHOULD be added for acompletion route.""" + data = {"stream": True, "model": "gpt-4"} + _apply_stream_options_logic( + data, {"always_include_stream_usage": True}, "acompletion" + ) + assert data["stream_options"] == {"include_usage": True} + + def test_stream_options_not_injected_when_disabled(self): + """stream_options should NOT be added when always_include_stream_usage is False.""" + data = {"stream": True, "model": "gpt-4"} + _apply_stream_options_logic( + data, {"always_include_stream_usage": False}, "acompletion" + ) + assert "stream_options" not in data + + def test_existing_stream_options_not_overwritten(self): + """If client already set stream_options with include_usage, don't overwrite.""" + data = {"stream": True, "model": "gpt-4", "stream_options": {"include_usage": False}} + _apply_stream_options_logic( + data, {"always_include_stream_usage": True}, "acompletion" + ) + assert data["stream_options"] == {"include_usage": False} + + def test_non_streaming_request_skipped(self): + """stream_options should NOT be added for non-streaming requests.""" + data = {"stream": False, "model": "gpt-4"} + _apply_stream_options_logic( + data, {"always_include_stream_usage": True}, "acompletion" + ) + assert "stream_options" not in data