From 9f5de05b9db330df1f7899f57ccc99c9de56449d Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 9 Oct 2026 15:00:24 -0700 Subject: [PATCH] fix(bedrock): drop stop sequences for GPT 5.6 and newer instead of forwarding them to a 400 (#45664) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/bedrock/common_utils.py | 8 +++--- ...t_bedrock_runtime_chat_completions_wire.py | 13 +++------ ...bedrock_chat_completions_transformation.py | 28 ++++++++++++------- 3 files changed, 26 insertions(+), 23 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 6b28ecd02df..76c85b438d3 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -896,11 +896,11 @@ def bedrock_runtime_chat_completions_is_default(model: str) -> bool: def bedrock_rejects_stop_sequences(model: str) -> bool: """Whether AWS refuses stop sequences for this model on every Bedrock route. - Grok answers ``stopSequences`` on Converse and ``stop`` on native Chat Completions alike with - ``This model doesn't support the stopSequences field`` (Grok 4.6 and 4.7 checked live on 2026-10-09), - so litellm drops ``stop`` for it instead of forwarding it to a 400. + Grok and GPT 5.6 and newer answer ``stopSequences`` on Converse and ``stop`` on native Chat Completions + alike with ``This model doesn't support the stopSequences field`` (Grok 4.6 and 4.7, GPT 5.6 and GPT 6.1 + checked live on 2026-10-09), so litellm drops ``stop`` for them instead of forwarding it to a 400. """ - return _XAI_GROK_MODEL_RE.search(model) is not None + return _bedrock_runtime_chat_completions_default_family(model) def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py index d44d9f154ec..c8f9e5c1f5e 100644 --- a/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py @@ -354,19 +354,14 @@ def test_model_id_application_inference_profile_keeps_converse_at_the_profile_ur ) -def test_stop_sequences_keep_converse(gateway: Gateway) -> None: +def test_stop_sequences_are_dropped_and_the_request_stays_native(gateway: Gateway) -> None: marker: Final = uuid.uuid4().hex with wire_server(respond) as wire, gateway.scenario() as scenario: model: Final = _deployment(scenario, wire) response: Final = _chat(gateway, model, marker, stop=["END"]) - payload: Final = _payload(response) - assert payload["choices"] == [ - {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} - ], response.text - body: Final = _body(_converse_request(wire)) - assert body["messages"] == _converse_messages(marker), body - assert body["inferenceConfig"] == {"stopSequences": ["END"]}, body - assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}") + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") def test_json_object_response_format_keeps_converse(gateway: Gateway) -> None: diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index a2d3ba2a1bb..1a952966514 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -451,17 +451,24 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions" -@pytest.mark.parametrize( - "model", - ["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/global.openai.gpt-5.6-sol", "bedrock/us.openai.gpt-6.1-sol"], -) -def test_stop_keeps_other_models_on_converse(local_cost_map, model): +def test_stop_keeps_gpt_oss_on_converse(local_cost_map): + model = "chat_completions/openai.gpt-oss-20b-1:0" assert bedrock_request_needs_converse(model, {"stop": ["END"]}) is True assert BedrockModelInfo.get_bedrock_route(model, {"stop": ["END"]}) == "converse" -@pytest.mark.parametrize("model", ["bedrock/global.xai.grok-4.7", "bedrock/us.xai.grok-4.6"]) -def test_stop_is_dropped_on_chat_completions_for_grok(local_cost_map, fake_aws_env, model): +@pytest.mark.parametrize( + "model", + [ + "bedrock/global.xai.grok-4.7", + "bedrock/us.xai.grok-4.6", + "bedrock/global.openai.gpt-5.6-sol", + "bedrock/us.openai.gpt-5.6-terra", + "bedrock/global.openai.gpt-6-sol", + "bedrock/us.openai.gpt-6.1-sol", + ], +) +def test_stop_is_dropped_on_chat_completions_for_models_rejecting_stop_sequences(local_cost_map, fake_aws_env, model): requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/"))) response = litellm.completion( model=model, messages=[{"role": "user", "content": "hello"}], stop=[""], max_tokens=64, client=client @@ -474,17 +481,18 @@ def test_stop_is_dropped_on_chat_completions_for_grok(local_cost_map, fake_aws_e assert response.choices[0].message.content == "ok" -def test_stop_is_dropped_on_converse_for_grok(local_cost_map, fake_aws_env): +@pytest.mark.parametrize("model_id", ["global.xai.grok-4.7", "global.openai.gpt-5.6-sol", "us.openai.gpt-6.1-sol"]) +def test_stop_is_dropped_on_converse_for_models_rejecting_stop_sequences(local_cost_map, fake_aws_env, model_id): requests, client = _recording_client(json=CONVERSE_JSON) litellm.completion( - model="bedrock/converse/global.xai.grok-4.7", + model=f"bedrock/converse/{model_id}", messages=[{"role": "user", "content": "hello"}], stop=[""], max_tokens=64, client=client, ) - assert requests[0].url.raw_path == b"/model/global.xai.grok-4.7/converse" + assert requests[0].url.raw_path == f"/model/{model_id}/converse".encode() assert json.loads(requests[0].content)["inferenceConfig"] == {"maxTokens": 64}