fix(bedrock): drop reasoning_effort none for grok on the native chat completions route

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
mateo-berri 2026-09-26 17:28:06 +00:00 • committed by mateo
parent c60249fa88
commit f4d84f8f5d
2 changed files with 70 additions and 1 deletions

View file

@ -60,6 +60,31 @@ def chat_completions_params_refused_for(model: str) -> frozenset[str]:
)
CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))})
def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]:
"""The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model.
Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every
``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before.
"""
model_id: Final = split_bedrock_region_path(model)[1]
return frozenset().union(
*(
refused
for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items()
if family in model_id
)
)
def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]:
if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model):
return params
return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"})
def _held_close_tag_prefix(text: str) -> int:
return next(
(
@ -277,7 +302,9 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
drop_params=drop_params,
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
)
return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict
return dict( # mutable-ok: get_optional_params keeps filling this dict
without_refused_reasoning_effort(model, with_max_completion_tokens(mapped))
)
def _inference_params(
self, optional_params: Mapping[str, object]

View file

@ -11,6 +11,7 @@ from litellm.llms.bedrock.chat.chat_completions.transformation import (
AmazonBedrockRuntimeChatCompletionsConfig,
BedrockRuntimeChatCompletionsStreamingHandler,
ReasoningTagSplitter,
chat_completions_reasoning_efforts_refused_for,
split_reasoning_tag,
with_max_completion_tokens,
)
@ -333,6 +334,47 @@ def test_with_max_completion_tokens_leaves_other_params_alone():
assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5}
@pytest.mark.parametrize(
"model",
["us.xai.grok-4.6", "bedrock/us-gov-west-1/us.xai.grok-4.6"],
)
def test_map_openai_params_drops_reasoning_effort_none_for_grok(model):
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
mapped = cfg.map_openai_params(
non_default_params={"reasoning_effort": "none", "max_tokens": 64},
optional_params={},
model=model,
drop_params=False,
)
assert "reasoning_effort" not in mapped
def test_map_openai_params_keeps_reasoning_effort_low_for_grok():
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
mapped = cfg.map_openai_params(
non_default_params={"reasoning_effort": "low", "max_tokens": 64},
optional_params={},
model="us.xai.grok-4.6",
drop_params=False,
)
assert mapped["reasoning_effort"] == "low"
def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56():
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
mapped = cfg.map_openai_params(
non_default_params={"reasoning_effort": "none", "max_tokens": 64},
optional_params={},
model="global.openai.gpt-5.6-sol",
drop_params=False,
)
assert mapped["reasoning_effort"] == "none"
def test_reasoning_efforts_refused_for_is_empty_outside_xai():
assert chat_completions_reasoning_efforts_refused_for("openai.gpt-oss-20b-1:0") == frozenset()
def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol")