mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(bedrock): drop stop sequences for GPT 5.6 and newer instead of forwarding them to a 400 (#45664)
Co-authored-by: kerry <kerry@berri.ai> Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
dcd0630e98
commit
9f5de05b9d
3 changed files with 26 additions and 23 deletions
|
|
@ -896,11 +896,11 @@ def bedrock_runtime_chat_completions_is_default(model: str) -> bool:
|
|||
def bedrock_rejects_stop_sequences(model: str) -> bool:
|
||||
"""Whether AWS refuses stop sequences for this model on every Bedrock route.
|
||||
|
||||
Grok answers ``stopSequences`` on Converse and ``stop`` on native Chat Completions alike with
|
||||
``This model doesn't support the stopSequences field`` (Grok 4.6 and 4.7 checked live on 2026-10-09),
|
||||
so litellm drops ``stop`` for it instead of forwarding it to a 400.
|
||||
Grok and GPT 5.6 and newer answer ``stopSequences`` on Converse and ``stop`` on native Chat Completions
|
||||
alike with ``This model doesn't support the stopSequences field`` (Grok 4.6 and 4.7, GPT 5.6 and GPT 6.1
|
||||
checked live on 2026-10-09), so litellm drops ``stop`` for them instead of forwarding it to a 400.
|
||||
"""
|
||||
return _XAI_GROK_MODEL_RE.search(model) is not None
|
||||
return _bedrock_runtime_chat_completions_default_family(model)
|
||||
|
||||
|
||||
def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:
|
||||
|
|
|
|||
|
|
@ -354,19 +354,14 @@ def test_model_id_application_inference_profile_keeps_converse_at_the_profile_ur
|
|||
)
|
||||
|
||||
|
||||
def test_stop_sequences_keep_converse(gateway: Gateway) -> None:
|
||||
def test_stop_sequences_are_dropped_and_the_request_stays_native(gateway: Gateway) -> None:
|
||||
marker: Final = uuid.uuid4().hex
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = _deployment(scenario, wire)
|
||||
response: Final = _chat(gateway, model, marker, stop=["END"])
|
||||
payload: Final = _payload(response)
|
||||
assert payload["choices"] == [
|
||||
{"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}}
|
||||
], response.text
|
||||
body: Final = _body(_converse_request(wire))
|
||||
assert body["messages"] == _converse_messages(marker), body
|
||||
assert body["inferenceConfig"] == {"stopSequences": ["END"]}, body
|
||||
assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}")
|
||||
assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text
|
||||
assert _body(_native_request(wire)) == _native_body(GPT, marker)
|
||||
assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}")
|
||||
|
||||
|
||||
def test_json_object_response_format_keeps_converse(gateway: Gateway) -> None:
|
||||
|
|
|
|||
|
|
@ -451,17 +451,24 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model,
|
|||
assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/global.openai.gpt-5.6-sol", "bedrock/us.openai.gpt-6.1-sol"],
|
||||
)
|
||||
def test_stop_keeps_other_models_on_converse(local_cost_map, model):
|
||||
def test_stop_keeps_gpt_oss_on_converse(local_cost_map):
|
||||
model = "chat_completions/openai.gpt-oss-20b-1:0"
|
||||
assert bedrock_request_needs_converse(model, {"stop": ["END"]}) is True
|
||||
assert BedrockModelInfo.get_bedrock_route(model, {"stop": ["END"]}) == "converse"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["bedrock/global.xai.grok-4.7", "bedrock/us.xai.grok-4.6"])
|
||||
def test_stop_is_dropped_on_chat_completions_for_grok(local_cost_map, fake_aws_env, model):
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"bedrock/global.xai.grok-4.7",
|
||||
"bedrock/us.xai.grok-4.6",
|
||||
"bedrock/global.openai.gpt-5.6-sol",
|
||||
"bedrock/us.openai.gpt-5.6-terra",
|
||||
"bedrock/global.openai.gpt-6-sol",
|
||||
"bedrock/us.openai.gpt-6.1-sol",
|
||||
],
|
||||
)
|
||||
def test_stop_is_dropped_on_chat_completions_for_models_rejecting_stop_sequences(local_cost_map, fake_aws_env, model):
|
||||
requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/")))
|
||||
response = litellm.completion(
|
||||
model=model, messages=[{"role": "user", "content": "hello"}], stop=["</block>"], max_tokens=64, client=client
|
||||
|
|
@ -474,17 +481,18 @@ def test_stop_is_dropped_on_chat_completions_for_grok(local_cost_map, fake_aws_e
|
|||
assert response.choices[0].message.content == "ok"
|
||||
|
||||
|
||||
def test_stop_is_dropped_on_converse_for_grok(local_cost_map, fake_aws_env):
|
||||
@pytest.mark.parametrize("model_id", ["global.xai.grok-4.7", "global.openai.gpt-5.6-sol", "us.openai.gpt-6.1-sol"])
|
||||
def test_stop_is_dropped_on_converse_for_models_rejecting_stop_sequences(local_cost_map, fake_aws_env, model_id):
|
||||
requests, client = _recording_client(json=CONVERSE_JSON)
|
||||
litellm.completion(
|
||||
model="bedrock/converse/global.xai.grok-4.7",
|
||||
model=f"bedrock/converse/{model_id}",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
stop=["</block>"],
|
||||
max_tokens=64,
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert requests[0].url.raw_path == b"/model/global.xai.grok-4.7/converse"
|
||||
assert requests[0].url.raw_path == f"/model/{model_id}/converse".encode()
|
||||
assert json.loads(requests[0].content)["inferenceConfig"] == {"maxTokens": 64}
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue