From 22e35c6b216b2954c0e7961d030e2a4b60e139b5 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:30:52 -0700 Subject: [PATCH] fix(bedrock): refuse temperature and top_p natively on GPT 5.6 and newer like Converse does AWS answers temperature and top_p with a 400 on the native Chat Completions endpoint for the GPT 5.6+ models, the same models whose Converse route already dropped both under drop_params via supports_sampling_params: false. The native config now honors that price-map flag, the gpt-6 and gpt-6.1 rows carry it, and the gpt-6 family joins gpt-5 in refusing frequency_penalty, presence_penalty, logprobs, and top_logprobs before the request reaches AWS. --- .../chat/chat_completions/transformation.py | 27 ++++++++++++++----- litellm/llms/bedrock/common_utils.py | 13 +++++++++ ...odel_prices_and_context_window_backup.json | 11 ++++++++ model_prices_and_context_window.json | 11 ++++++++ ...bedrock_chat_completions_transformation.py | 16 ++++++++--- 5 files changed, 68 insertions(+), 10 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 91a643fcd7b..34a756a54e2 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -32,7 +32,11 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( ) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM -from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path +from litellm.llms.bedrock.common_utils import ( + BedrockError, + bedrock_model_supports_sampling_params, + split_bedrock_region_path, +) from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig from litellm.types.llms.openai import AllMessageValues @@ -46,26 +50,37 @@ if TYPE_CHECKING: REASONING_OPEN_TAG: Final = "" REASONING_CLOSE_TAG: Final = "" +GPT_CHAT_COMPLETIONS_REFUSED_PARAMS: Final = frozenset( + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs") +) CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( { - "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")), + "openai.gpt-5": GPT_CHAT_COMPLETIONS_REFUSED_PARAMS, + "openai.gpt-6": GPT_CHAT_COMPLETIONS_REFUSED_PARAMS, "openai.gpt-oss": frozenset(("logit_bias",)), "xai.": frozenset(("frequency_penalty", "presence_penalty")), } ) +CHAT_COMPLETIONS_SAMPLING_PARAMS: Final = frozenset(("temperature", "top_p")) + + def chat_completions_params_refused_for(model: str) -> frozenset[str]: """The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says. - Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same - params under ``drop_params``, so the native config leaves them out of its supported list and the usual - drop-or-raise handling applies before the request reaches AWS. + Each family answers them with a 400 (GPT 5.6 and newer, gpt-oss) or a 503 (Grok), and a model whose price-map row + says ``supports_sampling_params: false`` answers ``temperature`` and ``top_p`` with a 400 too, where + Converse dropped the same params under ``drop_params``, so the native config leaves them out of its + supported list and the usual drop-or-raise handling applies before the request reaches AWS. """ model_id: Final = split_bedrock_region_path(model)[1] - return frozenset().union( + family_refused: Final = frozenset().union( *(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id) ) + if bedrock_model_supports_sampling_params(model): + return family_refused + return family_refused | CHAT_COMPLETIONS_SAMPLING_PARAMS CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))}) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 14ab71bd63a..bf40e9e93be 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -882,6 +882,19 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format") +def bedrock_model_supports_sampling_params(model: str) -> bool: + """Whether the model takes ``temperature`` and ``top_p``: false only when a price-map row says so. + + The GPT 5.6 and newer rows carry ``supports_sampling_params: false`` because AWS answers either param with + a 400 on Converse and on native Chat Completions alike, so both routes drop them under ``drop_params`` and + refuse them otherwise. + """ + return not any( + entry is not None and entry.get("supports_sampling_params") is False + for entry in _bedrock_price_map_entries(model) + ) + + BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ( "guardrailConfig", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7c70f07329e..28690ed5afe 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -58479,6 +58479,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58516,6 +58517,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58553,6 +58555,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58590,6 +58593,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58626,6 +58630,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { @@ -58659,6 +58664,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58695,6 +58701,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { @@ -58728,6 +58735,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -79359,6 +79367,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "openai.gpt-6.1-sol": { @@ -79391,6 +79400,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "bedrock_mantle/openai.gpt-6.1-sol": { @@ -79466,6 +79476,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "vertex_ai/gemini-3.8-flash-tts": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7c70f07329e..28690ed5afe 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -58479,6 +58479,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58516,6 +58517,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58553,6 +58555,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58590,6 +58593,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58626,6 +58630,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { @@ -58659,6 +58664,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58695,6 +58701,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { @@ -58728,6 +58735,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -79359,6 +79367,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "openai.gpt-6.1-sol": { @@ -79391,6 +79400,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "bedrock_mantle/openai.gpt-6.1-sol": { @@ -79466,6 +79476,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "vertex_ai/gemini-3.8-flash-tts": { diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 7539e8c8b43..37d96c45d6d 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -463,7 +463,7 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): mapped = cfg.map_openai_params( non_default_params={"max_tokens": 64, "temperature": 0.1}, optional_params={}, - model="global.openai.gpt-5.6-sol", + model="us.xai.grok-4.6", drop_params=False, ) assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} @@ -599,13 +599,18 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): [ ( "bedrock/global.openai.gpt-5.6-sol", - ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"), - ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"), + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "temperature", "top_p", "n"), + ("logit_bias", "reasoning_effort", "tools", "functions", "stop"), + ), + ( + "bedrock/us.openai.gpt-6.1-sol", + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "temperature", "top_p", "n"), + ("logit_bias", "reasoning_effort", "tools", "functions", "stop"), ), ( "us.xai.grok-4.6", ("frequency_penalty", "presence_penalty", "n"), - ("stop", "logprobs", "top_p", "logit_bias", "reasoning_effort"), + ("stop", "logprobs", "temperature", "top_p", "logit_bias", "reasoning_effort"), ), ( "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", @@ -625,6 +630,9 @@ def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, mod [ ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), + ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"temperature": 0.2}), + ("bedrock/chat_completions/global.openai.gpt-6-sol", {"top_p": 0.9}), + ("bedrock/chat_completions/global.openai.gpt-6-sol", {"presence_penalty": 0.5}), ("bedrock/chat_completions/us.xai.grok-4.6", {"presence_penalty": 0.5}), ("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), ],