fix(bedrock): drop stop sequences for GPT 5.6 and newer instead of forwarding them to a 400 (#45664)

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-09 15:00:24 -07:00 • committed by GitHub
parent dcd0630e98
commit 9f5de05b9d
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 26 additions and 23 deletions

View file

@ -896,11 +896,11 @@ def bedrock_runtime_chat_completions_is_default(model: str) -> bool:
def bedrock_rejects_stop_sequences(model: str) -> bool:
"""Whether AWS refuses stop sequences for this model on every Bedrock route.
Grok answers ``stopSequences`` on Converse and ``stop`` on native Chat Completions alike with
``This model doesn't support the stopSequences field`` (Grok 4.6 and 4.7 checked live on 2026-10-09),
so litellm drops ``stop`` for it instead of forwarding it to a 400.
Grok and GPT 5.6 and newer answer ``stopSequences`` on Converse and ``stop`` on native Chat Completions
alike with ``This model doesn't support the stopSequences field`` (Grok 4.6 and 4.7, GPT 5.6 and GPT 6.1
checked live on 2026-10-09), so litellm drops ``stop`` for them instead of forwarding it to a 400.
"""
return _XAI_GROK_MODEL_RE.search(model) is not None
return _bedrock_runtime_chat_completions_default_family(model)
def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:

View file

@ -354,19 +354,14 @@ def test_model_id_application_inference_profile_keeps_converse_at_the_profile_ur
)
def test_stop_sequences_keep_converse(gateway: Gateway) -> None:
def test_stop_sequences_are_dropped_and_the_request_stays_native(gateway: Gateway) -> None:
marker: Final = uuid.uuid4().hex
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = _deployment(scenario, wire)
response: Final = _chat(gateway, model, marker, stop=["END"])
payload: Final = _payload(response)
assert payload["choices"] == [
{"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}}
], response.text
body: Final = _body(_converse_request(wire))
assert body["messages"] == _converse_messages(marker), body
assert body["inferenceConfig"] == {"stopSequences": ["END"]}, body
assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}")
assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text
assert _body(_native_request(wire)) == _native_body(GPT, marker)
assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}")
def test_json_object_response_format_keeps_converse(gateway: Gateway) -> None:

View file

@ -451,17 +451,24 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model,
assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions"
@pytest.mark.parametrize(
"model",
["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/global.openai.gpt-5.6-sol", "bedrock/us.openai.gpt-6.1-sol"],
)
def test_stop_keeps_other_models_on_converse(local_cost_map, model):
def test_stop_keeps_gpt_oss_on_converse(local_cost_map):
model = "chat_completions/openai.gpt-oss-20b-1:0"
assert bedrock_request_needs_converse(model, {"stop": ["END"]}) is True
assert BedrockModelInfo.get_bedrock_route(model, {"stop": ["END"]}) == "converse"
@pytest.mark.parametrize("model", ["bedrock/global.xai.grok-4.7", "bedrock/us.xai.grok-4.6"])
def test_stop_is_dropped_on_chat_completions_for_grok(local_cost_map, fake_aws_env, model):
@pytest.mark.parametrize(
"model",
[
"bedrock/global.xai.grok-4.7",
"bedrock/us.xai.grok-4.6",
"bedrock/global.openai.gpt-5.6-sol",
"bedrock/us.openai.gpt-5.6-terra",
"bedrock/global.openai.gpt-6-sol",
"bedrock/us.openai.gpt-6.1-sol",
],
)
def test_stop_is_dropped_on_chat_completions_for_models_rejecting_stop_sequences(local_cost_map, fake_aws_env, model):
requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/")))
response = litellm.completion(
model=model, messages=[{"role": "user", "content": "hello"}], stop=["</block>"], max_tokens=64, client=client
@ -474,17 +481,18 @@ def test_stop_is_dropped_on_chat_completions_for_grok(local_cost_map, fake_aws_e
assert response.choices[0].message.content == "ok"
def test_stop_is_dropped_on_converse_for_grok(local_cost_map, fake_aws_env):
@pytest.mark.parametrize("model_id", ["global.xai.grok-4.7", "global.openai.gpt-5.6-sol", "us.openai.gpt-6.1-sol"])
def test_stop_is_dropped_on_converse_for_models_rejecting_stop_sequences(local_cost_map, fake_aws_env, model_id):
requests, client = _recording_client(json=CONVERSE_JSON)
litellm.completion(
model="bedrock/converse/global.xai.grok-4.7",
model=f"bedrock/converse/{model_id}",
messages=[{"role": "user", "content": "hello"}],
stop=["</block>"],
max_tokens=64,
client=client,
)
assert requests[0].url.raw_path == b"/model/global.xai.grok-4.7/converse"
assert requests[0].url.raw_path == f"/model/{model_id}/converse".encode()
assert json.loads(requests[0].content)["inferenceConfig"] == {"maxTokens": 64}