From da1c77be2c0f80dbc99e29d117927cb724ae29e4 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Sat, 26 Sep 2026 17:43:22 +0000 Subject: [PATCH] fix(bedrock): keep native chat completions when only top_k is set Converse only forwards top_k for Anthropic and Nova, so treating it as a converse-only key sent gpt-oss, GPT-5.6, and Grok off the native route and still dropped the value --- litellm/llms/bedrock/common_utils.py | 5 ++-- ...bedrock_chat_completions_transformation.py | 28 +++++++++++++++---- 2 files changed, 24 insertions(+), 9 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 098be7082d3..cb7cfdf9640 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -898,7 +898,6 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( "outputConfig", "thinking", "additionalModelRequestFields", - "top_k", ) ) @@ -919,8 +918,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje """Whether a request on a runtime-Chat-Completions model must still be served by Converse. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` - block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse - forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on + block and ``additionalModelRequestFields`` included, which only Converse + forwards as ``additionalModelRequestFields``) have no field on AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 17c0f8d3c1d..3f1499cda8b 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -259,17 +259,33 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): @pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) -@pytest.mark.parametrize( - "request_params", - [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}], - ids=["additionalModelRequestFields", "top_k"], -) -def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): +def test_additional_model_request_fields_fall_back_to_converse(local_cost_map, model): + request_params = {"additionalModelRequestFields": {"reasoning_effort": "high"}} assert bedrock_request_needs_converse(model, request_params) is True assert BedrockModelInfo.get_bedrock_route(model, request_params) == "converse" assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions" +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]) +def test_top_k_stays_on_chat_completions(local_cost_map, fake_aws_env, model): + request_params = {"top_k": 40} + assert bedrock_request_needs_converse(model, request_params) is False + assert BedrockModelInfo.get_bedrock_route(model, request_params) == "chat_completions" + + requests, client = _recording_client(json=_chat_completion_json("ok", model)) + litellm.completion( + model=f"bedrock/{model}", + messages=[{"role": "user", "content": "hello"}], + top_k=40, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["top_k"] == 40 + assert "inferenceConfig" not in body + + @pytest.mark.parametrize( "request_params, expected_route", [