Merge pull request #41469 from BerriAI/litellm_bedrock_mantle_responses_drop_top_p

fix(responses): drop top_p for gpt-5 reasoning models when drop_params is set
This commit is contained in:
Mateo Wang 2026-09-17 18:04:19 -07:00 • committed by GitHub
commit c25c098bc1
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 119 additions and 7 deletions

View file

@ -344,6 +344,10 @@ class BedrockMantleResponsesAPIConfig(BedrockMantleAuthMixin, OpenAIResponsesAPI
kept: Final = [item for item, _ in normalized if item is not None] # mutable-ok: ResponseInputParam is a list
return kept # pyright: ignore[reportReturnType] # Codex passthrough items sit outside the OpenAI input union
@staticmethod
def _model_map_lookup_name(model: str) -> str:
return model.split("/")[-1].removeprefix("openai.")
def map_openai_params(
self,
response_api_optional_params: ResponsesAPIOptionalRequestParams,

View file

@ -125,6 +125,10 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
return False
return is_gpt_reasoning_series_name(model)
@staticmethod
def _model_map_lookup_name(model: str) -> str:
return model
@staticmethod
def _supports_reasoning_effort_none(model: str) -> bool:
"""Return True if the model supports reasoning.effort='none'."""
@ -208,8 +212,9 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
) -> dict:
"""No mapping applied since inputs are in OpenAI spec already.
GPT-5 models have restrictions on temperature (only temperature=1
is accepted unless reasoning_effort='none' on models that support it).
GPT-5 models have restrictions on temperature and top_p (only temperature=1
is accepted, and top_p is rejected, unless reasoning.effort resolves to
'none' on models that support it).
Apply the same validation used by the chat completions path.
"""
params: Final = dict(response_api_optional_params)
@ -234,13 +239,16 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
status_code=400,
)
if self._is_gpt_5_model(model=model):
lookup_name: Final = self._model_map_lookup_name(model)
if self._is_gpt_5_model(model=lookup_name):
reasoning: Final = params.get("reasoning") or {}
effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None
supports_none: Final = self._supports_reasoning_effort_none(model=lookup_name)
effort_is_none: Final = supports_none and self._effort_resolves_to_none(lookup_name, effort)
temperature: Final = params.get("temperature")
if temperature is not None and temperature != 1:
reasoning: Final = params.get("reasoning") or {}
effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None
supports_none: Final = self._supports_reasoning_effort_none(model=model)
if supports_none and self._effort_resolves_to_none(model, effort):
if effort_is_none:
pass # flexible temperature allowed
elif drop_params or litellm.drop_params:
params.pop("temperature", None)
@ -256,6 +264,20 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
status_code=400,
)
if "top_p" in params and not effort_is_none:
if drop_params or litellm.drop_params:
params.pop("top_p", None)
else:
raise litellm.UnsupportedParamsError(
message=(
f"{model} only supports top_p when reasoning.effort resolves to 'none', "
"either set explicitly on the request or declared as the model's "
"default_reasoning_effort. "
"To drop unsupported params set `litellm.drop_params = True`"
),
status_code=400,
)
return params
def transform_responses_api_request(

View file

@ -369,6 +369,52 @@ class TestBedrockMantleResponsesTools:
assert "file_search" in str(mock_warning.call_args)
class TestBedrockMantleSamplingParams:
"""Mantle serves OpenAI's gpt-5 models under their OpenAI sampling rule: top_p and a
non-default temperature are accepted only when reasoning.effort resolves to none, so
the `openai.` catalogue name (region-prefixed on GovCloud) must answer from the OpenAI
model's map entry instead of dropping both params on every request."""
@pytest.mark.parametrize(
"model, effort, survives",
[
("openai.gpt-5.4", None, True),
("openai.gpt-5.5", None, False),
("openai.gpt-5.6-luna", None, False),
("openai.gpt-5.6-luna", "none", True),
("openai.gpt-5.6-luna", "low", False),
("us-gov-west-1/openai.gpt-5.4", None, True),
("us-gov-west-1/openai.gpt-5.6-luna", None, False),
],
)
def test_top_p_and_temperature_follow_the_resolved_effort(self, local_cost_map, model, effort, survives):
params = {"top_p": 0.9, "temperature": 0.2}
if effort is not None:
params["reasoning"] = {"effort": effort}
mapped = BedrockMantleResponsesAPIConfig().map_openai_params(
response_api_optional_params=params,
model=model,
drop_params=True,
)
assert ("top_p" in mapped) is survives
assert ("temperature" in mapped) is survives
def test_top_p_without_drop_params_raises_only_while_reasoning_is_active(self, local_cost_map):
with pytest.raises(litellm.UnsupportedParamsError):
BedrockMantleResponsesAPIConfig().map_openai_params(
response_api_optional_params={"top_p": 0.9},
model="openai.gpt-5.6-luna",
drop_params=False,
)
mapped = BedrockMantleResponsesAPIConfig().map_openai_params(
response_api_optional_params={"top_p": 0.9},
model="openai.gpt-5.4",
drop_params=False,
)
assert mapped["top_p"] == 0.9
class TestBedrockMantleResponsesWebSearch:
"""Web Search on Amazon Bedrock is a server-side built-in tool that Mantle runs
itself when the caller passes {"type": "web_search"} on the Responses path, so

View file

@ -1835,6 +1835,46 @@ class TestResponsesSurfaceSharesTheEffortRule:
)
assert ("temperature" in mapped) is temperature_survives
@pytest.mark.parametrize(
"model, effort, top_p_survives",
[
("gpt-5.1", None, True),
("gpt-5.4", None, True),
("gpt-5.5", None, False),
("gpt-5.6-terra", None, False),
("gpt-5.6-sol", None, False),
("gpt-5.6-terra", "none", True),
("gpt-5.6-terra", "medium", False),
("gpt-6-astra", None, False),
("gpt-6-astra", "low", False),
],
)
def test_top_p_follows_the_resolved_effort(self, local_model_cost_map, model, effort, top_p_survives):
params = {"top_p": 0.9}
if effort is not None:
params["reasoning"] = {"effort": effort}
mapped = OpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params=params,
model=model,
drop_params=True,
)
assert ("top_p" in mapped) is top_p_survives
def test_top_p_raises_without_drop_params(self, local_model_cost_map):
with pytest.raises(litellm.UnsupportedParamsError):
OpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params={"top_p": 0.9},
model="gpt-5.5",
drop_params=False,
)
mapped = OpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params={"top_p": 0.9, "reasoning": {"effort": "none"}},
model="gpt-5.6-terra",
drop_params=False,
)
assert mapped["top_p"] == 0.9
class TestFlattenToolSchemaCombinatorsWiring:
"""Regression tests for MCP tools with a top-level anyOf schema (Codex Desktop).