diff --git a/litellm/llms/chatgpt/responses/transformation.py b/litellm/llms/chatgpt/responses/transformation.py index 8e4bbf1d3c9..f74992ae122 100644 --- a/litellm/llms/chatgpt/responses/transformation.py +++ b/litellm/llms/chatgpt/responses/transformation.py @@ -1,3 +1,4 @@ +from collections.abc import Mapping from typing import Any, Final from litellm.exceptions import AuthenticationError @@ -28,6 +29,25 @@ from ..common_utils import ( get_chatgpt_default_instructions, ) +# Codex >= 0.144 sends this header for models served over the "Responses +# Lite" transport (gpt-5.6-* at the time of writing). When the header is +# present the backend rejects the request unless `parallel_tool_calls` is +# exactly false and `reasoning.context` is "all_turns": +# "X-OpenAI-Internal-Codex-Responses-Lite requires `parallel_tool_calls` +# to be false." (param=parallel_tool_calls, code=unsupported_value) +CODEX_RESPONSES_LITE_HEADER: Final[str] = "x-openai-internal-codex-responses-lite" +_FALSY_HEADER_VALUES: Final[frozenset[str]] = frozenset({"", "0", "false", "no"}) + + +def is_codex_responses_lite_request(headers: Mapping[str, object] | None) -> bool: + """True when the outbound headers carry the Codex Responses-Lite marker.""" + if not headers: + return False + for key, value in headers.items(): + if key.lower() == CODEX_RESPONSES_LITE_HEADER: + return str(value).strip().lower() not in _FALSY_HEADER_VALUES + return False + class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): def __init__(self) -> None: @@ -96,12 +116,32 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig): "include", "tools", "tool_choice", + "parallel_tool_calls", "reasoning", "previous_response_id", "truncation", } - return {k: v for k, v in request.items() if k in allowed_keys} + filtered: Final[dict[str, object]] = { # mutable-ok: outgoing JSON request body + k: v for k, v in request.items() if k in allowed_keys + } + + if is_codex_responses_lite_request(headers): + # The Responses-Lite backend hard-rejects any other combination, + # so normalize even when the caller omitted these fields (codex + # 0.144.x can omit the reasoning object on a metadata race). + filtered["parallel_tool_calls"] = False + raw_reasoning: Final = filtered.get("reasoning") + reasoning_items: Final = raw_reasoning.items() if isinstance(raw_reasoning, dict) else () + reasoning: dict[str, object] = { # mutable-ok: rebuilt to force the required context key + reasoning_key: reasoning_value + for reasoning_key, reasoning_value in reasoning_items + if isinstance(reasoning_key, str) + } + reasoning["context"] = "all_turns" + filtered["reasoning"] = reasoning + + return filtered def transform_response_api_response( self, diff --git a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py index 8e0415d50de..ae7cf1cfae4 100644 --- a/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py +++ b/tests/test_litellm/llms/chatgpt/responses/test_chatgpt_responses_transformation.py @@ -155,6 +155,106 @@ class TestChatGPTResponsesAPITransformation: "function": {"name": "hello"}, } + @pytest.mark.parametrize("parallel_tool_calls", [True, False]) + def test_chatgpt_preserves_parallel_tool_calls(self, parallel_tool_calls): + """parallel_tool_calls is validated by the Codex backend and must not + be dropped by the allowed-keys filter.""" + config = ChatGPTResponsesAPIConfig() + request = config.transform_responses_api_request( + model="chatgpt/gpt-5.3-codex", + input="hi", + response_api_optional_request_params={ + "parallel_tool_calls": parallel_tool_calls, + }, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert request["parallel_tool_calls"] is parallel_tool_calls + + def test_chatgpt_responses_lite_restores_stripped_false(self): + """Regression: a Responses-Lite request (codex >= 0.144, gpt-5.6-*) + carries parallel_tool_calls=false; dropping it made the backend + reject every request with + "X-OpenAI-Internal-Codex-Responses-Lite requires `parallel_tool_calls` + to be false.".""" + config = ChatGPTResponsesAPIConfig() + request = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input="hi", + response_api_optional_request_params={ + "parallel_tool_calls": False, + "reasoning": {"context": "all_turns"}, + }, + litellm_params=GenericLiteLLMParams(), + headers={"X-OpenAI-Internal-Codex-Responses-Lite": "true"}, + ) + + assert request["parallel_tool_calls"] is False + assert request["reasoning"]["context"] == "all_turns" + + def test_chatgpt_responses_lite_forces_lite_contract(self): + """Under the Lite header the backend hard-rejects any other value, + so true is normalized to false and reasoning.context is pinned.""" + config = ChatGPTResponsesAPIConfig() + request = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input="hi", + response_api_optional_request_params={ + "parallel_tool_calls": True, + "reasoning": {"effort": "high"}, + }, + litellm_params=GenericLiteLLMParams(), + headers={"x-openai-internal-codex-responses-lite": "true"}, + ) + + assert request["parallel_tool_calls"] is False + assert request["reasoning"] == {"effort": "high", "context": "all_turns"} + + def test_chatgpt_responses_lite_normalizes_omitted_fields(self): + """codex 0.144.x can omit the reasoning object entirely; the Lite + validator rejects the omission the same way.""" + config = ChatGPTResponsesAPIConfig() + request = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input="hi", + response_api_optional_request_params={}, + litellm_params=GenericLiteLLMParams(), + headers={"x-openai-internal-codex-responses-lite": "true"}, + ) + + assert request["parallel_tool_calls"] is False + assert request["reasoning"] == {"context": "all_turns"} + + def test_chatgpt_unrelated_headers_do_not_trigger_lite_normalization(self): + """Ordinary outbound headers (auth etc.) without the Lite marker must + leave the request untouched.""" + config = ChatGPTResponsesAPIConfig() + request = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input="hi", + response_api_optional_request_params={}, + litellm_params=GenericLiteLLMParams(), + headers={"Authorization": "Bearer token", "originator": "codex_cli_rs"}, + ) + + assert "parallel_tool_calls" not in request + assert "reasoning" not in request + + @pytest.mark.parametrize("header_value", ["false", "0", "no", ""]) + def test_chatgpt_responses_lite_falsy_header_is_ignored(self, header_value): + config = ChatGPTResponsesAPIConfig() + request = config.transform_responses_api_request( + model="chatgpt/gpt-5.6-sol", + input="hi", + response_api_optional_request_params={}, + litellm_params=GenericLiteLLMParams(), + headers={"x-openai-internal-codex-responses-lite": header_value}, + ) + + assert "parallel_tool_calls" not in request + assert "reasoning" not in request + @pytest.mark.parametrize( ("model_name", "response_model"), [