This commit is contained in:
Bryan Li 2026-08-29 07:16:27 -04:00 • committed by GitHub
commit e3d2a7db29
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 141 additions and 1 deletions

View file

@ -1,3 +1,4 @@
from collections.abc import Mapping
from typing import Any, Final
from litellm.exceptions import AuthenticationError
@ -28,6 +29,25 @@ from ..common_utils import (
get_chatgpt_default_instructions,
)
# Codex >= 0.144 sends this header for models served over the "Responses
# Lite" transport (gpt-5.6-* at the time of writing). When the header is
# present the backend rejects the request unless `parallel_tool_calls` is
# exactly false and `reasoning.context` is "all_turns":
# "X-OpenAI-Internal-Codex-Responses-Lite requires `parallel_tool_calls`
# to be false." (param=parallel_tool_calls, code=unsupported_value)
CODEX_RESPONSES_LITE_HEADER: Final[str] = "x-openai-internal-codex-responses-lite"
_FALSY_HEADER_VALUES: Final[frozenset[str]] = frozenset({"", "0", "false", "no"})
def is_codex_responses_lite_request(headers: Mapping[str, object] | None) -> bool:
"""True when the outbound headers carry the Codex Responses-Lite marker."""
if not headers:
return False
for key, value in headers.items():
if key.lower() == CODEX_RESPONSES_LITE_HEADER:
return str(value).strip().lower() not in _FALSY_HEADER_VALUES
return False
class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
def __init__(self) -> None:
@ -96,12 +116,32 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
"include",
"tools",
"tool_choice",
"parallel_tool_calls",
"reasoning",
"previous_response_id",
"truncation",
}
return {k: v for k, v in request.items() if k in allowed_keys}
filtered: Final[dict[str, object]] = { # mutable-ok: outgoing JSON request body
k: v for k, v in request.items() if k in allowed_keys
}
if is_codex_responses_lite_request(headers):
# The Responses-Lite backend hard-rejects any other combination,
# so normalize even when the caller omitted these fields (codex
# 0.144.x can omit the reasoning object on a metadata race).
filtered["parallel_tool_calls"] = False
raw_reasoning: Final = filtered.get("reasoning")
reasoning_items: Final = raw_reasoning.items() if isinstance(raw_reasoning, dict) else ()
reasoning: dict[str, object] = { # mutable-ok: rebuilt to force the required context key
reasoning_key: reasoning_value
for reasoning_key, reasoning_value in reasoning_items
if isinstance(reasoning_key, str)
}
reasoning["context"] = "all_turns"
filtered["reasoning"] = reasoning
return filtered
def transform_response_api_response(
self,

View file

@ -155,6 +155,106 @@ class TestChatGPTResponsesAPITransformation:
"function": {"name": "hello"},
}
@pytest.mark.parametrize("parallel_tool_calls", [True, False])
def test_chatgpt_preserves_parallel_tool_calls(self, parallel_tool_calls):
"""parallel_tool_calls is validated by the Codex backend and must not
be dropped by the allowed-keys filter."""
config = ChatGPTResponsesAPIConfig()
request = config.transform_responses_api_request(
model="chatgpt/gpt-5.3-codex",
input="hi",
response_api_optional_request_params={
"parallel_tool_calls": parallel_tool_calls,
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert request["parallel_tool_calls"] is parallel_tool_calls
def test_chatgpt_responses_lite_restores_stripped_false(self):
"""Regression: a Responses-Lite request (codex >= 0.144, gpt-5.6-*)
carries parallel_tool_calls=false; dropping it made the backend
reject every request with
"X-OpenAI-Internal-Codex-Responses-Lite requires `parallel_tool_calls`
to be false."."""
config = ChatGPTResponsesAPIConfig()
request = config.transform_responses_api_request(
model="chatgpt/gpt-5.6-sol",
input="hi",
response_api_optional_request_params={
"parallel_tool_calls": False,
"reasoning": {"context": "all_turns"},
},
litellm_params=GenericLiteLLMParams(),
headers={"X-OpenAI-Internal-Codex-Responses-Lite": "true"},
)
assert request["parallel_tool_calls"] is False
assert request["reasoning"]["context"] == "all_turns"
def test_chatgpt_responses_lite_forces_lite_contract(self):
"""Under the Lite header the backend hard-rejects any other value,
so true is normalized to false and reasoning.context is pinned."""
config = ChatGPTResponsesAPIConfig()
request = config.transform_responses_api_request(
model="chatgpt/gpt-5.6-sol",
input="hi",
response_api_optional_request_params={
"parallel_tool_calls": True,
"reasoning": {"effort": "high"},
},
litellm_params=GenericLiteLLMParams(),
headers={"x-openai-internal-codex-responses-lite": "true"},
)
assert request["parallel_tool_calls"] is False
assert request["reasoning"] == {"effort": "high", "context": "all_turns"}
def test_chatgpt_responses_lite_normalizes_omitted_fields(self):
"""codex 0.144.x can omit the reasoning object entirely; the Lite
validator rejects the omission the same way."""
config = ChatGPTResponsesAPIConfig()
request = config.transform_responses_api_request(
model="chatgpt/gpt-5.6-sol",
input="hi",
response_api_optional_request_params={},
litellm_params=GenericLiteLLMParams(),
headers={"x-openai-internal-codex-responses-lite": "true"},
)
assert request["parallel_tool_calls"] is False
assert request["reasoning"] == {"context": "all_turns"}
def test_chatgpt_unrelated_headers_do_not_trigger_lite_normalization(self):
"""Ordinary outbound headers (auth etc.) without the Lite marker must
leave the request untouched."""
config = ChatGPTResponsesAPIConfig()
request = config.transform_responses_api_request(
model="chatgpt/gpt-5.6-sol",
input="hi",
response_api_optional_request_params={},
litellm_params=GenericLiteLLMParams(),
headers={"Authorization": "Bearer token", "originator": "codex_cli_rs"},
)
assert "parallel_tool_calls" not in request
assert "reasoning" not in request
@pytest.mark.parametrize("header_value", ["false", "0", "no", ""])
def test_chatgpt_responses_lite_falsy_header_is_ignored(self, header_value):
config = ChatGPTResponsesAPIConfig()
request = config.transform_responses_api_request(
model="chatgpt/gpt-5.6-sol",
input="hi",
response_api_optional_request_params={},
litellm_params=GenericLiteLLMParams(),
headers={"x-openai-internal-codex-responses-lite": header_value},
)
assert "parallel_tool_calls" not in request
assert "reasoning" not in request
@pytest.mark.parametrize(
("model_name", "response_model"),
[