From 0a85e2799814b4582d110f93c0b9c87ed3f66d8b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:57:27 -0700 Subject: [PATCH] fix(bedrock): keep schema-less json_object on Converse for the chat completions models --- litellm/llms/bedrock/common_utils.py | 20 +++++---- ...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++-- 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 8798b05cfa4..2fe1af9edf4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -853,10 +853,15 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ) -def _response_format_constrains_output(response_format: object) -> bool: +def _response_format_needs_converse(model: str, response_format: object) -> bool: if response_format is None: return False - return not (isinstance(response_format, Mapping) and response_format.get("type") == "text") + if not isinstance(response_format, Mapping): + return not bedrock_runtime_chat_completions_enforces_response_format(model) + if response_format.get("type") == "text": + return False + carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format + return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: @@ -867,16 +872,17 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless - ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without - ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse. + ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema + (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with + ``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only + honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's + native surface rejects it with a 400 unless the prompt mentions json. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True if bedrock_request_metadata_is_owned(): return True - if _response_format_constrains_output( - request_params.get("response_format") - ) and not bedrock_runtime_chat_completions_enforces_response_format(model): + if _response_format_needs_converse(model, request_params.get("response_format")): return True if not (request_params.get("tools") or request_params.get("functions")): return False diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 2852ed3ae84..e49a2f7f252 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -799,13 +799,32 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route -@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]) -def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model): - params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA} +RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"] + + +@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) +@pytest.mark.parametrize( + "response_format", + [ + RESPONSE_FORMAT_JSON_SCHEMA, + {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]}, + Answer, + ], + ids=["json_schema", "response_schema", "pydantic"], +) +def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format): + params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is False assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" +@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) +def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model): + params = {"response_format": {"type": "json_object"}} + assert bedrock_request_needs_converse(model, params) is True + assert BedrockModelInfo.get_bedrock_route(model, params) == "converse" + + SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" @@ -886,3 +905,20 @@ def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA assert response.choices[0].message.content == '{"word": "pong"}' + + +def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format={"type": "json_object"}, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert "toolConfig" not in body + assert "response_format" not in body + assert body["inferenceConfig"]["maxTokens"] == 64