mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(bedrock): keep every json_object response_format on Converse for the chat completions models
This commit is contained in:
parent
0a85e27998
commit
0139dd08a6
2 changed files with 43 additions and 19 deletions
|
|
@ -858,10 +858,11 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool
|
|||
return False
|
||||
if not isinstance(response_format, Mapping):
|
||||
return not bedrock_runtime_chat_completions_enforces_response_format(model)
|
||||
if response_format.get("type") == "text":
|
||||
response_format_type: Final = response_format.get("type")
|
||||
if response_format_type == "text":
|
||||
return False
|
||||
carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format
|
||||
return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
|
||||
is_json_schema: Final = response_format_type == "json_schema" and "json_schema" in response_format
|
||||
return not (is_json_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
|
||||
|
||||
|
||||
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
|
||||
|
|
@ -872,11 +873,12 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
|
|||
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
|
||||
function tools (``tools`` or legacy ``functions``) on a model without
|
||||
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
|
||||
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema
|
||||
(a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with
|
||||
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as
|
||||
``{"type": "json_schema", "json_schema": ...}`` (a pydantic model is converted to that) on a model with
|
||||
``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only
|
||||
honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's
|
||||
native surface rejects it with a 400 unless the prompt mentions json.
|
||||
honored by Converse, and every ``json_object`` form (``response_schema`` included) keeps Converse's
|
||||
handling everywhere, since AWS's native surface rejects that type with a 400 unless the prompt
|
||||
mentions json.
|
||||
"""
|
||||
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
|
||||
return True
|
||||
|
|
|
|||
|
|
@ -802,25 +802,30 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r
|
|||
RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]
|
||||
|
||||
|
||||
JSON_OBJECT_WITH_RESPONSE_SCHEMA = {
|
||||
"type": "json_object",
|
||||
"response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS)
|
||||
@pytest.mark.parametrize(
|
||||
"response_format",
|
||||
[
|
||||
RESPONSE_FORMAT_JSON_SCHEMA,
|
||||
{"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]},
|
||||
Answer,
|
||||
],
|
||||
ids=["json_schema", "response_schema", "pydantic"],
|
||||
)
|
||||
def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format):
|
||||
@pytest.mark.parametrize("response_format", [RESPONSE_FORMAT_JSON_SCHEMA, Answer], ids=["json_schema", "pydantic"])
|
||||
def test_json_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(
|
||||
local_cost_map, model, response_format
|
||||
):
|
||||
params = {"response_format": response_format}
|
||||
assert bedrock_request_needs_converse(model, params) is False
|
||||
assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS)
|
||||
def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model):
|
||||
params = {"response_format": {"type": "json_object"}}
|
||||
@pytest.mark.parametrize(
|
||||
"response_format",
|
||||
[{"type": "json_object"}, JSON_OBJECT_WITH_RESPONSE_SCHEMA],
|
||||
ids=["json_object", "json_object_with_response_schema"],
|
||||
)
|
||||
def test_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model, response_format):
|
||||
params = {"response_format": response_format}
|
||||
assert bedrock_request_needs_converse(model, params) is True
|
||||
assert BedrockModelInfo.get_bedrock_route(model, params) == "converse"
|
||||
|
||||
|
|
@ -922,3 +927,20 @@ def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(lo
|
|||
assert "toolConfig" not in body
|
||||
assert "response_format" not in body
|
||||
assert body["inferenceConfig"]["maxTokens"] == 64
|
||||
|
||||
|
||||
def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool(local_cost_map, fake_aws_env):
|
||||
requests, client = _recording_client(json=CONVERSE_JSON)
|
||||
litellm.completion(
|
||||
model="bedrock/global.openai.gpt-5.6-sol",
|
||||
messages=[{"role": "user", "content": "Reply with the single word pong."}],
|
||||
response_format=JSON_OBJECT_WITH_RESPONSE_SCHEMA,
|
||||
max_tokens=64,
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse")
|
||||
body = json.loads(requests[0].content)
|
||||
assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call"
|
||||
assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}}
|
||||
assert "response_format" not in body
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue