mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Merge pull request #41469 from BerriAI/litellm_bedrock_mantle_responses_drop_top_p
fix(responses): drop top_p for gpt-5 reasoning models when drop_params is set
This commit is contained in:
commit
c25c098bc1
4 changed files with 119 additions and 7 deletions
|
|
@ -344,6 +344,10 @@ class BedrockMantleResponsesAPIConfig(BedrockMantleAuthMixin, OpenAIResponsesAPI
|
|||
kept: Final = [item for item, _ in normalized if item is not None] # mutable-ok: ResponseInputParam is a list
|
||||
return kept # pyright: ignore[reportReturnType] # Codex passthrough items sit outside the OpenAI input union
|
||||
|
||||
@staticmethod
|
||||
def _model_map_lookup_name(model: str) -> str:
|
||||
return model.split("/")[-1].removeprefix("openai.")
|
||||
|
||||
def map_openai_params(
|
||||
self,
|
||||
response_api_optional_params: ResponsesAPIOptionalRequestParams,
|
||||
|
|
|
|||
|
|
@ -125,6 +125,10 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
return False
|
||||
return is_gpt_reasoning_series_name(model)
|
||||
|
||||
@staticmethod
|
||||
def _model_map_lookup_name(model: str) -> str:
|
||||
return model
|
||||
|
||||
@staticmethod
|
||||
def _supports_reasoning_effort_none(model: str) -> bool:
|
||||
"""Return True if the model supports reasoning.effort='none'."""
|
||||
|
|
@ -208,8 +212,9 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
) -> dict:
|
||||
"""No mapping applied since inputs are in OpenAI spec already.
|
||||
|
||||
GPT-5 models have restrictions on temperature (only temperature=1
|
||||
is accepted unless reasoning_effort='none' on models that support it).
|
||||
GPT-5 models have restrictions on temperature and top_p (only temperature=1
|
||||
is accepted, and top_p is rejected, unless reasoning.effort resolves to
|
||||
'none' on models that support it).
|
||||
Apply the same validation used by the chat completions path.
|
||||
"""
|
||||
params: Final = dict(response_api_optional_params)
|
||||
|
|
@ -234,13 +239,16 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
status_code=400,
|
||||
)
|
||||
|
||||
if self._is_gpt_5_model(model=model):
|
||||
lookup_name: Final = self._model_map_lookup_name(model)
|
||||
if self._is_gpt_5_model(model=lookup_name):
|
||||
reasoning: Final = params.get("reasoning") or {}
|
||||
effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None
|
||||
supports_none: Final = self._supports_reasoning_effort_none(model=lookup_name)
|
||||
effort_is_none: Final = supports_none and self._effort_resolves_to_none(lookup_name, effort)
|
||||
|
||||
temperature: Final = params.get("temperature")
|
||||
if temperature is not None and temperature != 1:
|
||||
reasoning: Final = params.get("reasoning") or {}
|
||||
effort: Final = reasoning.get("effort") if isinstance(reasoning, dict) else None
|
||||
supports_none: Final = self._supports_reasoning_effort_none(model=model)
|
||||
if supports_none and self._effort_resolves_to_none(model, effort):
|
||||
if effort_is_none:
|
||||
pass # flexible temperature allowed
|
||||
elif drop_params or litellm.drop_params:
|
||||
params.pop("temperature", None)
|
||||
|
|
@ -256,6 +264,20 @@ class OpenAIResponsesAPIConfig(BaseResponsesAPIConfig):
|
|||
status_code=400,
|
||||
)
|
||||
|
||||
if "top_p" in params and not effort_is_none:
|
||||
if drop_params or litellm.drop_params:
|
||||
params.pop("top_p", None)
|
||||
else:
|
||||
raise litellm.UnsupportedParamsError(
|
||||
message=(
|
||||
f"{model} only supports top_p when reasoning.effort resolves to 'none', "
|
||||
"either set explicitly on the request or declared as the model's "
|
||||
"default_reasoning_effort. "
|
||||
"To drop unsupported params set `litellm.drop_params = True`"
|
||||
),
|
||||
status_code=400,
|
||||
)
|
||||
|
||||
return params
|
||||
|
||||
def transform_responses_api_request(
|
||||
|
|
|
|||
|
|
@ -369,6 +369,52 @@ class TestBedrockMantleResponsesTools:
|
|||
assert "file_search" in str(mock_warning.call_args)
|
||||
|
||||
|
||||
class TestBedrockMantleSamplingParams:
|
||||
"""Mantle serves OpenAI's gpt-5 models under their OpenAI sampling rule: top_p and a
|
||||
non-default temperature are accepted only when reasoning.effort resolves to none, so
|
||||
the `openai.` catalogue name (region-prefixed on GovCloud) must answer from the OpenAI
|
||||
model's map entry instead of dropping both params on every request."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, effort, survives",
|
||||
[
|
||||
("openai.gpt-5.4", None, True),
|
||||
("openai.gpt-5.5", None, False),
|
||||
("openai.gpt-5.6-luna", None, False),
|
||||
("openai.gpt-5.6-luna", "none", True),
|
||||
("openai.gpt-5.6-luna", "low", False),
|
||||
("us-gov-west-1/openai.gpt-5.4", None, True),
|
||||
("us-gov-west-1/openai.gpt-5.6-luna", None, False),
|
||||
],
|
||||
)
|
||||
def test_top_p_and_temperature_follow_the_resolved_effort(self, local_cost_map, model, effort, survives):
|
||||
params = {"top_p": 0.9, "temperature": 0.2}
|
||||
if effort is not None:
|
||||
params["reasoning"] = {"effort": effort}
|
||||
mapped = BedrockMantleResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params=params,
|
||||
model=model,
|
||||
drop_params=True,
|
||||
)
|
||||
assert ("top_p" in mapped) is survives
|
||||
assert ("temperature" in mapped) is survives
|
||||
|
||||
def test_top_p_without_drop_params_raises_only_while_reasoning_is_active(self, local_cost_map):
|
||||
with pytest.raises(litellm.UnsupportedParamsError):
|
||||
BedrockMantleResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params={"top_p": 0.9},
|
||||
model="openai.gpt-5.6-luna",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
mapped = BedrockMantleResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params={"top_p": 0.9},
|
||||
model="openai.gpt-5.4",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["top_p"] == 0.9
|
||||
|
||||
|
||||
class TestBedrockMantleResponsesWebSearch:
|
||||
"""Web Search on Amazon Bedrock is a server-side built-in tool that Mantle runs
|
||||
itself when the caller passes {"type": "web_search"} on the Responses path, so
|
||||
|
|
|
|||
|
|
@ -1835,6 +1835,46 @@ class TestResponsesSurfaceSharesTheEffortRule:
|
|||
)
|
||||
assert ("temperature" in mapped) is temperature_survives
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, effort, top_p_survives",
|
||||
[
|
||||
("gpt-5.1", None, True),
|
||||
("gpt-5.4", None, True),
|
||||
("gpt-5.5", None, False),
|
||||
("gpt-5.6-terra", None, False),
|
||||
("gpt-5.6-sol", None, False),
|
||||
("gpt-5.6-terra", "none", True),
|
||||
("gpt-5.6-terra", "medium", False),
|
||||
("gpt-6-astra", None, False),
|
||||
("gpt-6-astra", "low", False),
|
||||
],
|
||||
)
|
||||
def test_top_p_follows_the_resolved_effort(self, local_model_cost_map, model, effort, top_p_survives):
|
||||
params = {"top_p": 0.9}
|
||||
if effort is not None:
|
||||
params["reasoning"] = {"effort": effort}
|
||||
mapped = OpenAIResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params=params,
|
||||
model=model,
|
||||
drop_params=True,
|
||||
)
|
||||
assert ("top_p" in mapped) is top_p_survives
|
||||
|
||||
def test_top_p_raises_without_drop_params(self, local_model_cost_map):
|
||||
with pytest.raises(litellm.UnsupportedParamsError):
|
||||
OpenAIResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params={"top_p": 0.9},
|
||||
model="gpt-5.5",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
mapped = OpenAIResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params={"top_p": 0.9, "reasoning": {"effort": "none"}},
|
||||
model="gpt-5.6-terra",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["top_p"] == 0.9
|
||||
|
||||
|
||||
class TestFlattenToolSchemaCombinatorsWiring:
|
||||
"""Regression tests for MCP tools with a top-level anyOf schema (Codex Desktop).
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue