fix(bedrock): skip synthetic tool injection for json_object with no schema (#25740)

When response_format={"type": "json_object"} is sent without a JSON
schema, _create_json_tool_call_for_response_format builds a tool with an
empty schema (properties: {}). The model follows the empty schema and
returns {} instead of the actual JSON the caller asked for.

This patch:
- Skips synthetic json_tool_call injection when no schema is provided.
  The model already returns JSON when the prompt asks for it.
- Fixes finish_reason: after _filter_json_mode_tools strips all
  synthetic tool calls, finish_reason stays "tool_calls" instead of
  "stop". Callers (like the OpenAI SDK) misinterpret this as a pending
  tool invocation.

json_schema requests with an explicit schema are unchanged.

Co-authored-by: Claude <noreply@anthropic.com>
This commit is contained in:
Amiram Mizne 2026-04-14 20:29:45 -07:00 committed by GitHub
parent 69dabab373
commit 20758e69d8
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 161 additions and 6 deletions

View file

@ -1003,7 +1003,7 @@ class AmazonConverseConfig(BaseConfig):
description=description,
)
optional_params["outputConfig"] = output_config
else:
elif json_schema is not None:
# Fallback: translate to a synthetic tool call
# https://docs.anthropic.com/en/docs/build-with-claude/tool-use#json-mode
_tool = self._create_json_tool_call_for_response_format(
@ -1025,6 +1025,12 @@ class AmazonConverseConfig(BaseConfig):
)
if non_default_params.get("stream", False) is True:
optional_params["fake_stream"] = True
# else: response_format=json_object with no schema.
# Don't inject the synthetic json_tool_call tool here. When no
# schema is given, _create_json_tool_call_for_response_format
# produces an empty schema (properties: {}), and the model
# returns {} instead of the requested JSON. The model already
# returns JSON when the prompt asks for it.
optional_params["json_mode"] = True
return optional_params
@ -2027,6 +2033,12 @@ class AmazonConverseConfig(BaseConfig):
_message = Message(**chat_completion_message)
initial_finish_reason = map_finish_reason(completion_response["stopReason"])
# When json_mode filtered out all synthetic tool calls the response
# is plain content, not a pending tool invocation. Fix finish_reason
# so callers (e.g. OpenAI SDK) don't misinterpret it.
if json_mode and not filtered_tools and tools:
initial_finish_reason = "stop"
(
returned_message,
returned_finish_reason,

View file

@ -3140,9 +3140,14 @@ def test_add_additional_properties_definitions():
assert result["definitions"]["Item"]["properties"]["details"]["additionalProperties"] is False
def test_json_object_no_schema_falls_back_to_tool_call():
"""response_format: {type: json_object} with no schema should use tool-call fallback,
even for models that support native structured outputs."""
def test_json_object_no_schema_skips_tool_injection():
"""response_format: {type: json_object} with no schema should NOT inject
the synthetic json_tool_call tool.
When no schema is given, _create_json_tool_call_for_response_format builds
a tool with an empty schema (properties: {}). The model follows the schema
and returns {} instead of the requested JSON. Skipping tool injection lets
the model respond naturally with the JSON the caller asked for."""
old_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP")
old_cost = litellm.model_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
@ -3162,8 +3167,9 @@ def test_json_object_no_schema_falls_back_to_tool_call():
# Should NOT use native outputConfig (no schema provided)
assert "outputConfig" not in result
# Should use tool-call fallback
assert "tools" in result
# Should NOT inject tools - empty schema causes model to return {}
assert "tools" not in result
assert "tool_choice" not in result
assert result["json_mode"] is True
finally:
litellm.model_cost = old_cost
@ -3655,3 +3661,140 @@ def test_cache_control_injection_tool_config_not_added_without_injection_point()
tools = result["toolConfig"]["tools"]
# No cachePoint should be appended
assert all("cachePoint" not in tool for tool in tools)
def test_translate_response_format_json_schema_still_injects_tool():
"""
response_format with an explicit json_schema should still use the
synthetic tool call approach (for models that don't support native
structured outputs).
"""
config = AmazonConverseConfig()
response_format = {
"type": "json_schema",
"json_schema": {
"name": "FactResult",
"schema": {
"type": "object",
"properties": {
"facts": {
"type": "array",
"items": {"type": "string"},
},
},
"required": ["facts"],
},
},
}
optional_params: dict = {}
result = config._translate_response_format_param(
value=response_format,
model="anthropic.claude-3-haiku-20240307-v1:0",
optional_params=optional_params,
non_default_params={"response_format": response_format},
is_thinking_enabled=False,
)
assert result["json_mode"] is True
assert "tools" in result
assert "tool_choice" in result
def test_transform_response_finish_reason_stop_when_json_mode_filters_all_tools():
"""
When json_mode is True and _filter_json_mode_tools strips all synthetic
tool calls, finish_reason should be "stop", not "tool_calls".
Bedrock returns stopReason="tool_use" for json_tool_call responses.
After filtering, the response is plain JSON content and should not look
like a pending tool invocation to callers.
"""
from litellm.llms.bedrock.chat.converse_transformation import AmazonConverseConfig
from litellm.types.utils import ModelResponse
response_json = {
"metrics": {"latencyMs": 100},
"output": {
"message": {
"role": "assistant",
"content": [
{
"toolUse": {
"toolUseId": "tooluse_001",
"name": "json_tool_call",
"input": {
"facts": ["Bob is a software engineer"],
},
}
}
],
}
},
"stopReason": "tool_use",
"usage": {
"inputTokens": 50,
"outputTokens": 20,
"totalTokens": 70,
"cacheReadInputTokenCount": 0,
"cacheReadInputTokens": 0,
"cacheWriteInputTokenCount": 0,
"cacheWriteInputTokens": 0,
},
}
class MockResponse:
def json(self):
return response_json
@property
def text(self):
return json.dumps(response_json)
config = AmazonConverseConfig()
model_response = ModelResponse()
# Simulate what happens when json_tool_call was injected for a
# json_schema request: optional_params has the synthetic tool
optional_params = {
"json_mode": True,
"tools": [
{
"type": "function",
"function": {
"name": "json_tool_call",
"parameters": {
"type": "object",
"additionalProperties": True,
"properties": {},
},
},
}
],
}
result = config._transform_response(
model="bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
response=MockResponse(),
model_response=model_response,
stream=False,
logging_obj=None,
optional_params=optional_params,
api_key=None,
data=None,
messages=[],
encoding=None,
)
# Content should have the JSON from the tool call arguments
content = result.choices[0].message.content
assert content is not None
parsed = json.loads(content)
assert parsed["facts"] == ["Bob is a software engineer"]
# No tool_calls on the message
assert result.choices[0].message.tool_calls is None
# finish_reason must be "stop", not "tool_calls"
assert result.choices[0].finish_reason == "stop"