From f4d84f8f5db0c26e0aa3c50bc2ddb77d07122f5b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:28:06 +0000 Subject: [PATCH] fix(bedrock): drop reasoning_effort none for grok on the native chat completions route Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../chat/chat_completions/transformation.py | 29 ++++++++++++- ...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++++ 2 files changed, 70 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 4c5e5768119..be2eb8c7713 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -60,6 +60,31 @@ def chat_completions_params_refused_for(model: str) -> frozenset[str]: ) +CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))}) + + +def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]: + """The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model. + + Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every + ``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before. + """ + model_id: Final = split_bedrock_region_path(model)[1] + return frozenset().union( + *( + refused + for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items() + if family in model_id + ) + ) + + +def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]: + if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model): + return params + return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"}) + + def _held_close_tag_prefix(text: str) -> int: return next( ( @@ -277,7 +302,9 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): drop_params=drop_params, replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, ) - return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict + return dict( # mutable-ok: get_optional_params keeps filling this dict + without_refused_reasoning_effort(model, with_max_completion_tokens(mapped)) + ) def _inference_params( self, optional_params: Mapping[str, object] diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 3c347e4bed2..061df201c15 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -11,6 +11,7 @@ from litellm.llms.bedrock.chat.chat_completions.transformation import ( AmazonBedrockRuntimeChatCompletionsConfig, BedrockRuntimeChatCompletionsStreamingHandler, ReasoningTagSplitter, + chat_completions_reasoning_efforts_refused_for, split_reasoning_tag, with_max_completion_tokens, ) @@ -333,6 +334,47 @@ def test_with_max_completion_tokens_leaves_other_params_alone(): assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5} +@pytest.mark.parametrize( + "model", + ["us.xai.grok-4.6", "bedrock/us-gov-west-1/us.xai.grok-4.6"], +) +def test_map_openai_params_drops_reasoning_effort_none_for_grok(model): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "none", "max_tokens": 64}, + optional_params={}, + model=model, + drop_params=False, + ) + assert "reasoning_effort" not in mapped + + +def test_map_openai_params_keeps_reasoning_effort_low_for_grok(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "low", "max_tokens": 64}, + optional_params={}, + model="us.xai.grok-4.6", + drop_params=False, + ) + assert mapped["reasoning_effort"] == "low" + + +def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "none", "max_tokens": 64}, + optional_params={}, + model="global.openai.gpt-5.6-sol", + drop_params=False, + ) + assert mapped["reasoning_effort"] == "none" + + +def test_reasoning_efforts_refused_for_is_empty_outside_xai(): + assert chat_completions_reasoning_efforts_refused_for("openai.gpt-oss-20b-1:0") == frozenset() + + def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): cfg = AmazonBedrockRuntimeChatCompletionsConfig() assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol")