mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge eb1424c350 into 6a3333d3c8
This commit is contained in:
commit
e3d2a7db29
2 changed files with 141 additions and 1 deletions
|
|
@ -1,3 +1,4 @@
|
|||
from collections.abc import Mapping
|
||||
from typing import Any, Final
|
||||
|
||||
from litellm.exceptions import AuthenticationError
|
||||
|
|
@ -28,6 +29,25 @@ from ..common_utils import (
|
|||
get_chatgpt_default_instructions,
|
||||
)
|
||||
|
||||
# Codex >= 0.144 sends this header for models served over the "Responses
|
||||
# Lite" transport (gpt-5.6-* at the time of writing). When the header is
|
||||
# present the backend rejects the request unless `parallel_tool_calls` is
|
||||
# exactly false and `reasoning.context` is "all_turns":
|
||||
# "X-OpenAI-Internal-Codex-Responses-Lite requires `parallel_tool_calls`
|
||||
# to be false." (param=parallel_tool_calls, code=unsupported_value)
|
||||
CODEX_RESPONSES_LITE_HEADER: Final[str] = "x-openai-internal-codex-responses-lite"
|
||||
_FALSY_HEADER_VALUES: Final[frozenset[str]] = frozenset({"", "0", "false", "no"})
|
||||
|
||||
|
||||
def is_codex_responses_lite_request(headers: Mapping[str, object] | None) -> bool:
|
||||
"""True when the outbound headers carry the Codex Responses-Lite marker."""
|
||||
if not headers:
|
||||
return False
|
||||
for key, value in headers.items():
|
||||
if key.lower() == CODEX_RESPONSES_LITE_HEADER:
|
||||
return str(value).strip().lower() not in _FALSY_HEADER_VALUES
|
||||
return False
|
||||
|
||||
|
||||
class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
||||
def __init__(self) -> None:
|
||||
|
|
@ -96,12 +116,32 @@ class ChatGPTResponsesAPIConfig(OpenAIResponsesAPIConfig):
|
|||
"include",
|
||||
"tools",
|
||||
"tool_choice",
|
||||
"parallel_tool_calls",
|
||||
"reasoning",
|
||||
"previous_response_id",
|
||||
"truncation",
|
||||
}
|
||||
|
||||
return {k: v for k, v in request.items() if k in allowed_keys}
|
||||
filtered: Final[dict[str, object]] = { # mutable-ok: outgoing JSON request body
|
||||
k: v for k, v in request.items() if k in allowed_keys
|
||||
}
|
||||
|
||||
if is_codex_responses_lite_request(headers):
|
||||
# The Responses-Lite backend hard-rejects any other combination,
|
||||
# so normalize even when the caller omitted these fields (codex
|
||||
# 0.144.x can omit the reasoning object on a metadata race).
|
||||
filtered["parallel_tool_calls"] = False
|
||||
raw_reasoning: Final = filtered.get("reasoning")
|
||||
reasoning_items: Final = raw_reasoning.items() if isinstance(raw_reasoning, dict) else ()
|
||||
reasoning: dict[str, object] = { # mutable-ok: rebuilt to force the required context key
|
||||
reasoning_key: reasoning_value
|
||||
for reasoning_key, reasoning_value in reasoning_items
|
||||
if isinstance(reasoning_key, str)
|
||||
}
|
||||
reasoning["context"] = "all_turns"
|
||||
filtered["reasoning"] = reasoning
|
||||
|
||||
return filtered
|
||||
|
||||
def transform_response_api_response(
|
||||
self,
|
||||
|
|
|
|||
|
|
@ -155,6 +155,106 @@ class TestChatGPTResponsesAPITransformation:
|
|||
"function": {"name": "hello"},
|
||||
}
|
||||
|
||||
@pytest.mark.parametrize("parallel_tool_calls", [True, False])
|
||||
def test_chatgpt_preserves_parallel_tool_calls(self, parallel_tool_calls):
|
||||
"""parallel_tool_calls is validated by the Codex backend and must not
|
||||
be dropped by the allowed-keys filter."""
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
request = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-5.3-codex",
|
||||
input="hi",
|
||||
response_api_optional_request_params={
|
||||
"parallel_tool_calls": parallel_tool_calls,
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert request["parallel_tool_calls"] is parallel_tool_calls
|
||||
|
||||
def test_chatgpt_responses_lite_restores_stripped_false(self):
|
||||
"""Regression: a Responses-Lite request (codex >= 0.144, gpt-5.6-*)
|
||||
carries parallel_tool_calls=false; dropping it made the backend
|
||||
reject every request with
|
||||
"X-OpenAI-Internal-Codex-Responses-Lite requires `parallel_tool_calls`
|
||||
to be false."."""
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
request = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-5.6-sol",
|
||||
input="hi",
|
||||
response_api_optional_request_params={
|
||||
"parallel_tool_calls": False,
|
||||
"reasoning": {"context": "all_turns"},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"X-OpenAI-Internal-Codex-Responses-Lite": "true"},
|
||||
)
|
||||
|
||||
assert request["parallel_tool_calls"] is False
|
||||
assert request["reasoning"]["context"] == "all_turns"
|
||||
|
||||
def test_chatgpt_responses_lite_forces_lite_contract(self):
|
||||
"""Under the Lite header the backend hard-rejects any other value,
|
||||
so true is normalized to false and reasoning.context is pinned."""
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
request = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-5.6-sol",
|
||||
input="hi",
|
||||
response_api_optional_request_params={
|
||||
"parallel_tool_calls": True,
|
||||
"reasoning": {"effort": "high"},
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"x-openai-internal-codex-responses-lite": "true"},
|
||||
)
|
||||
|
||||
assert request["parallel_tool_calls"] is False
|
||||
assert request["reasoning"] == {"effort": "high", "context": "all_turns"}
|
||||
|
||||
def test_chatgpt_responses_lite_normalizes_omitted_fields(self):
|
||||
"""codex 0.144.x can omit the reasoning object entirely; the Lite
|
||||
validator rejects the omission the same way."""
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
request = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-5.6-sol",
|
||||
input="hi",
|
||||
response_api_optional_request_params={},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"x-openai-internal-codex-responses-lite": "true"},
|
||||
)
|
||||
|
||||
assert request["parallel_tool_calls"] is False
|
||||
assert request["reasoning"] == {"context": "all_turns"}
|
||||
|
||||
def test_chatgpt_unrelated_headers_do_not_trigger_lite_normalization(self):
|
||||
"""Ordinary outbound headers (auth etc.) without the Lite marker must
|
||||
leave the request untouched."""
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
request = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-5.6-sol",
|
||||
input="hi",
|
||||
response_api_optional_request_params={},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"Authorization": "Bearer token", "originator": "codex_cli_rs"},
|
||||
)
|
||||
|
||||
assert "parallel_tool_calls" not in request
|
||||
assert "reasoning" not in request
|
||||
|
||||
@pytest.mark.parametrize("header_value", ["false", "0", "no", ""])
|
||||
def test_chatgpt_responses_lite_falsy_header_is_ignored(self, header_value):
|
||||
config = ChatGPTResponsesAPIConfig()
|
||||
request = config.transform_responses_api_request(
|
||||
model="chatgpt/gpt-5.6-sol",
|
||||
input="hi",
|
||||
response_api_optional_request_params={},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={"x-openai-internal-codex-responses-lite": header_value},
|
||||
)
|
||||
|
||||
assert "parallel_tool_calls" not in request
|
||||
assert "reasoning" not in request
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model_name", "response_model"),
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue