fix(responses): merge bridged tool calls into the same choice as the text (#44346)

* revert(responses): revert "fix(responses): keep gpt-5.4/5.5 tool calls on chat and merge bridged tool calls into one choice" (#44295)

This reverts commit ca1994e403.

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* fix(responses): merge bridged tool calls into the same choice as the text

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-03 07:00:52 +00:00 • committed by GitHub
parent 8e32d4568c
commit 5724117116
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 744 additions and 2 deletions

View file

@ -187,7 +187,7 @@ def _reasoning_items_from_output_items(output_items: Sequence[object]) -> tuple[
def _as_chat_reasoning_items(
reasoning_items: Sequence[_BuiltReasoningItem],
reasoning_items: Sequence[_BuiltReasoningItem | ChatCompletionReasoningItem],
) -> list[ChatCompletionReasoningItem] | None:
if not reasoning_items:
return None
@ -788,7 +788,32 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
else:
pass # don't fail request if item in list is not supported
# If we accumulated tool calls, create a single choice with all of them
if accumulated_tool_calls and choices:
last_choice: Final = choices[-1]
last_reasoning_content: Final = getattr(last_choice.message, "reasoning_content", None)
last_reasoning_items: Final = getattr(last_choice.message, "reasoning_items", None)
merged_reasoning_content: Final = (
" ".join(value for value in (last_reasoning_content, reasoning_content) if value) or None
)
merged_reasoning_items: Final = _as_chat_reasoning_items(
(
*(last_reasoning_items or ()),
*(() if pending_reasoning_item is None else (pending_reasoning_item,)),
)
)
merged_message: Final = Message(
role=last_choice.message.role,
content=last_choice.message.content,
annotations=getattr(last_choice.message, "annotations", None),
tool_calls=accumulated_tool_calls,
reasoning_content=merged_reasoning_content,
reasoning_items=merged_reasoning_items,
)
return [
*choices[:-1],
Choices(message=merged_message, finish_reason="tool_calls", index=last_choice.index),
]
if accumulated_tool_calls:
msg = Message(
content=None,

View file

@ -1,5 +1,6 @@
import json
import uuid
from itertools import chain
from typing import Final
import pytest
@ -64,3 +65,190 @@ def test_openai_chat_tool_choice_without_tools_is_not_forwarded(gateway: Gateway
}
]
assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/chat/completions")]
def test_azure_gpt_6_bridged_stream_returns_text_and_tool_call_on_one_choice(gateway: Gateway) -> None:
identity: Final = f"azure-gpt-6-sol-stream-{uuid.uuid4().hex}"
expected_text: Final = "Let me check the weather."
events: Final = (
{
"type": "response.created",
"response": {
"id": "resp_weather",
"object": "response",
"created_at": 1,
"status": "in_progress",
"model": "gpt-6-sol",
},
},
{
"type": "response.output_item.added",
"output_index": 0,
"item": {
"id": "msg_weather",
"type": "message",
"status": "in_progress",
"role": "assistant",
"content": [],
},
},
{
"type": "response.output_text.delta",
"item_id": "msg_weather",
"output_index": 0,
"content_index": 0,
"delta": "Let me check ",
},
{
"type": "response.output_text.delta",
"item_id": "msg_weather",
"output_index": 0,
"content_index": 0,
"delta": "the weather.",
},
{
"type": "response.output_item.done",
"output_index": 0,
"item": {
"id": "msg_weather",
"type": "message",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": expected_text, "annotations": []}],
},
},
{
"type": "response.output_item.added",
"output_index": 1,
"item": {
"id": "fc_1",
"type": "function_call",
"status": "in_progress",
"call_id": "call_1",
"name": "get_weather",
"arguments": "",
},
},
{
"type": "response.function_call_arguments.delta",
"item_id": "fc_1",
"output_index": 1,
"delta": '{"city":',
},
{
"type": "response.function_call_arguments.delta",
"item_id": "fc_1",
"output_index": 1,
"delta": '"Paris"}',
},
{
"type": "response.output_item.done",
"output_index": 1,
"item": {
"id": "fc_1",
"type": "function_call",
"status": "completed",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
},
},
{
"type": "response.completed",
"response": {
"id": "resp_weather",
"object": "response",
"created_at": 1,
"status": "completed",
"model": "gpt-6-sol",
"output": [
{
"id": "msg_weather",
"type": "message",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": expected_text, "annotations": []}],
},
{
"id": "fc_1",
"type": "function_call",
"status": "completed",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
},
],
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15},
},
},
)
stream_chunks: Final = tuple(f"data: {json.dumps(event)}\n\n".encode() for event in events)
def respond(request: Request) -> Reply:
assert request.method == "POST"
assert request.target == "/openai/responses?api-version=2025-04-01-preview"
body: Final = _JSON_OBJECT.validate_json(request.body)
assert body["model"] == "gpt-6-sol"
return Reply(content_type="text/event-stream", chunks=stream_chunks)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model="azure/gpt-6-sol",
api_base=wire.url,
api_key=_API_KEY,
api_version="2025-04-01-preview",
)
with gateway.client.stream(
"POST",
"/v1/chat/completions",
headers={"Authorization": f"Bearer {gateway.key}"},
json={
"model": model,
"messages": [{"role": "user", "content": f"What is the weather in Paris? {identity}"}],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the weather for a city.",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
"stream": True,
"cache": {"no-cache": True},
},
) as response:
response_body: Final = response.read()
assert response.status_code == 200, response.text
chunks: Final = tuple(
_JSON_OBJECT.validate_json(line.removeprefix("data: "))
for line in response_body.decode().splitlines()
if line.startswith("data: ") and line != "data: [DONE]"
)
choices: Final = tuple(chain.from_iterable(chunk["choices"] for chunk in chunks))
assert choices, response.text
assert all(choice["index"] == 0 for choice in choices), response.text
assert "".join(str(choice["delta"].get("content") or "") for choice in choices) == expected_text, (
response.text
)
tool_call_chunks: Final = tuple(
chain.from_iterable(choice["delta"].get("tool_calls", []) for choice in choices)
)
assert (
"".join(str(tool_call["function"].get("name") or "") for tool_call in tool_call_chunks) == "get_weather"
), response.text
assert (
"".join(str(tool_call["function"].get("arguments") or "") for tool_call in tool_call_chunks)
== '{"city":"Paris"}'
), response.text
assert tuple(
choice.get("finish_reason") for choice in choices if choice.get("finish_reason") is not None
) == ("tool_calls",), response.text
assert [(request.method, request.target) for request in wire.drain()] == [
("POST", "/openai/responses?api-version=2025-04-01-preview")
]

View file

@ -194,3 +194,381 @@ def test_messages_over_responses_deployment_with_max_tokens_one_reaches_openai_a
assert len(tuple(request for request in wire.drain() if request.method == "POST")) == 1
assert body["content"] == [{"type": "text", "text": "ok"}], response.text
assert body["usage"]["input_tokens"] == 9 and body["usage"]["output_tokens"] == 1, response.text
def test_chat_over_responses_deployment_merges_message_and_function_call(gateway: Gateway) -> None:
identity: Final = "responses-bridge-" + uuid.uuid4().hex
def respond(request: Request) -> Reply:
if request.method == "GET" and request.target == "/v1/models":
return Reply(body=b'{"object":"list","data":[]}')
assert request.method == "POST" and request.target == "/responses", request.target
return Reply(
body=json.dumps(
{
"id": "resp_weather",
"object": "response",
"created_at": 1789788253,
"status": "completed",
"model": "gpt-6-sol",
"output": [
{
"type": "message",
"id": "msg_weather",
"status": "completed",
"role": "assistant",
"content": [
{
"type": "output_text",
"text": "Let me check the weather.",
"annotations": [],
}
],
},
{
"type": "function_call",
"id": "fc_1",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
"status": "completed",
},
],
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15},
}
).encode()
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model="openai/responses/gpt-6-sol", api_base=wire.url, api_key="synthetic-openai-key"
)
response: Final = gateway.request(
"POST",
"/v1/chat/completions",
{
"model": model,
"messages": [{"role": "user", "content": f"What is the weather in Paris? {identity}"}],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the weather for a city.",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
"cache": {"no-cache": True},
},
)
assert response.status_code == 200, response.text
body: Final = response.json()
assert body["choices"] == [
{
"finish_reason": "tool_calls",
"index": 0,
"message": {
"role": "assistant",
"content": "Let me check the weather.",
"tool_calls": [
{
"id": "fc_1",
"type": "function",
"function": {
"name": "get_weather",
"arguments": '{"city":"Paris"}',
},
"index": 0,
}
],
},
}
], response.text
assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/responses")]
def test_chat_over_responses_deployment_keeps_reasoning_with_merged_tool_call(gateway: Gateway) -> None:
identity: Final = "responses-bridge-reasoning-" + uuid.uuid4().hex
def respond(request: Request) -> Reply:
if request.method == "GET" and request.target == "/v1/models":
return Reply(body=b'{"object":"list","data":[]}')
assert request.method == "POST" and request.target == "/responses", request.target
return Reply(
body=json.dumps(
{
"id": "resp_weather_reasoning",
"object": "response",
"created_at": 1789788253,
"status": "completed",
"model": "gpt-6-sol",
"output": [
{
"type": "message",
"id": "msg_weather_reasoning",
"status": "completed",
"role": "assistant",
"content": [
{
"type": "output_text",
"text": "Let me check the weather.",
"annotations": [],
}
],
},
{
"type": "reasoning",
"id": "rs_weather",
"summary": [{"type": "summary_text", "text": "Checking the forecast."}],
},
{
"type": "function_call",
"id": "fc_1",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
"status": "completed",
},
],
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15},
}
).encode()
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model="openai/responses/gpt-6-sol", api_base=wire.url, api_key="synthetic-openai-key"
)
response: Final = gateway.request(
"POST",
"/v1/chat/completions",
{
"model": model,
"messages": [{"role": "user", "content": f"What is the weather in Paris? {identity}"}],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the weather for a city.",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
"cache": {"no-cache": True},
},
)
assert response.status_code == 200, response.text
body: Final = response.json()
assert body["choices"] == [
{
"finish_reason": "tool_calls",
"index": 0,
"message": {
"role": "assistant",
"content": "Let me check the weather.",
"reasoning_content": "Checking the forecast.",
"reasoning_items": [
{
"type": "reasoning",
"id": "rs_weather",
"summary": [{"type": "summary_text", "text": "Checking the forecast."}],
}
],
"tool_calls": [
{
"id": "fc_1",
"type": "function",
"function": {
"name": "get_weather",
"arguments": '{"city":"Paris"}',
},
"index": 0,
}
],
},
}
], response.text
assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/responses")]
def test_chat_over_responses_deployment_returns_tool_call_only_reply_as_one_choice(gateway: Gateway) -> None:
identity: Final = "responses-bridge-tool-only-" + uuid.uuid4().hex
def respond(request: Request) -> Reply:
if request.method == "GET" and request.target == "/v1/models":
return Reply(body=b'{"object":"list","data":[]}')
assert request.method == "POST" and request.target == "/responses", request.target
return Reply(
body=json.dumps(
{
"id": "resp_weather_tool_only",
"object": "response",
"created_at": 1789788253,
"status": "completed",
"model": "gpt-6-sol",
"output": [
{
"type": "function_call",
"id": "fc_1",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
"status": "completed",
}
],
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15},
}
).encode()
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model="openai/responses/gpt-6-sol", api_base=wire.url, api_key="synthetic-openai-key"
)
response: Final = gateway.request(
"POST",
"/v1/chat/completions",
{
"model": model,
"messages": [{"role": "user", "content": f"What is the weather in Paris? {identity}"}],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the weather for a city.",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
"cache": {"no-cache": True},
},
)
assert response.status_code == 200, response.text
body: Final = response.json()
assert body["choices"] == [
{
"finish_reason": "tool_calls",
"index": 0,
"message": {
"role": "assistant",
"content": None,
"tool_calls": [
{
"id": "fc_1",
"type": "function",
"function": {
"name": "get_weather",
"arguments": '{"city":"Paris"}',
},
"index": 0,
}
],
},
}
], response.text
assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/responses")]
def test_chat_over_responses_deployment_merges_function_call_followed_by_message(gateway: Gateway) -> None:
identity: Final = "responses-bridge-tool-then-message-" + uuid.uuid4().hex
def respond(request: Request) -> Reply:
if request.method == "GET" and request.target == "/v1/models":
return Reply(body=b'{"object":"list","data":[]}')
assert request.method == "POST" and request.target == "/responses", request.target
return Reply(
body=json.dumps(
{
"id": "resp_weather_tool_then_message",
"object": "response",
"created_at": 1789788253,
"status": "completed",
"model": "gpt-6-sol",
"output": [
{
"type": "function_call",
"id": "fc_1",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
"status": "completed",
},
{
"type": "message",
"id": "msg_after_tool",
"status": "completed",
"role": "assistant",
"content": [{"type": "output_text", "text": "After the tool.", "annotations": []}],
},
],
"usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15},
}
).encode()
)
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = scenario.model(
model="openai/responses/gpt-6-sol", api_base=wire.url, api_key="synthetic-openai-key"
)
response: Final = gateway.request(
"POST",
"/v1/chat/completions",
{
"model": model,
"messages": [{"role": "user", "content": f"What is the weather in Paris? {identity}"}],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the weather for a city.",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
"cache": {"no-cache": True},
},
)
assert response.status_code == 200, response.text
body: Final = response.json()
assert body["choices"] == [
{
"finish_reason": "tool_calls",
"index": 0,
"message": {
"role": "assistant",
"content": "After the tool.",
"tool_calls": [
{
"id": "fc_1",
"type": "function",
"function": {
"name": "get_weather",
"arguments": '{"city":"Paris"}',
},
"index": 0,
}
],
},
}
], response.text
assert [(request.method, request.target) for request in wire.drain()] == [("POST", "/responses")]

View file

@ -7,6 +7,15 @@ from unittest.mock import ANY, MagicMock, Mock, patch
import httpx
import pytest
from openai.types.responses import (
ResponseFunctionToolCall,
ResponseOutputMessage,
ResponseOutputText,
)
from openai.types.responses.response_reasoning_item import (
ResponseReasoningItem,
Summary,
)
import litellm
from litellm.completion_extras.litellm_responses_transformation.transformation import (
@ -3307,6 +3316,148 @@ def test_convert_response_output_generic_pydantic_message_item():
assert choices[0].finish_reason == "stop"
def test_convert_response_output_merges_message_reasoning_and_function_call() -> None:
message: Final = ResponseOutputMessage(
id="msg_weather",
content=[
ResponseOutputText(
annotations=[
{
"type": "url_citation",
"start_index": 0,
"end_index": 5,
"title": "Forecast",
"url": "https://example.com/forecast",
}
],
text="Sunny.",
type="output_text",
logprobs=[],
)
],
role="assistant",
status="completed",
type="message",
)
reasoning: Final = ResponseReasoningItem(
id="rs_before",
summary=[Summary(type="summary_text", text="Checking the forecast.")],
type="reasoning",
content=None,
encrypted_content=None,
status=None,
)
pending_reasoning: Final = ResponseReasoningItem(
id="rs_after",
summary=[Summary(type="summary_text", text="The location is Paris.")],
type="reasoning",
content=None,
encrypted_content=None,
status=None,
)
function_call: Final = ResponseFunctionToolCall(
id="fc_1",
type="function_call",
status="completed",
arguments='{"city":"Paris"}',
call_id="call_1",
name="get_weather",
)
message_and_call: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
(message, function_call)
)
assert len(message_and_call) == 1
assert message_and_call[0].index == 0
assert message_and_call[0].finish_reason == "tool_calls"
assert message_and_call[0].message.role == "assistant"
assert message_and_call[0].message.content == "Sunny."
assert message_and_call[0].message.annotations == [
{
"type": "url_citation",
"start_index": 0,
"end_index": 5,
"title": "Forecast",
"url": "https://example.com/forecast",
}
]
function_calls: Final = message_and_call[0].message.tool_calls
assert function_calls is not None
assert len(function_calls) == 1
assert function_calls[0].function.name == "get_weather"
assert function_calls[0].function.arguments == '{"city":"Paris"}'
reasoning_before_message: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
(reasoning, message, function_call)
)
assert len(reasoning_before_message) == 1
assert reasoning_before_message[0].message.reasoning_content == "Checking the forecast."
reasoning_before_items: Final = reasoning_before_message[0].message.reasoning_items
assert reasoning_before_items is not None
assert reasoning_before_items[0]["id"] == "rs_before"
reasoning_after_message: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
(message, pending_reasoning, function_call)
)
assert len(reasoning_after_message) == 1
assert reasoning_after_message[0].message.reasoning_content == "The location is Paris."
reasoning_after_items: Final = reasoning_after_message[0].message.reasoning_items
assert reasoning_after_items is not None
assert reasoning_after_items[0]["id"] == "rs_after"
merged_reasoning: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
(reasoning, message, pending_reasoning, function_call)
)
assert len(merged_reasoning) == 1
assert merged_reasoning[0].message.reasoning_content == "Checking the forecast. The location is Paris."
merged_reasoning_items: Final = merged_reasoning[0].message.reasoning_items
assert merged_reasoning_items is not None
assert [item["id"] for item in merged_reasoning_items] == ["rs_before", "rs_after"]
tool_only: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices((function_call,))
assert len(tool_only) == 1
assert tool_only[0].index == 0
assert tool_only[0].finish_reason == "tool_calls"
assert tool_only[0].message.content is None
assert tool_only[0].message.tool_calls is not None
assert len(tool_only[0].message.tool_calls) == 1
message_only: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices((message,))
assert len(message_only) == 1
assert message_only[0].index == 0
assert message_only[0].finish_reason == "stop"
assert message_only[0].message.content == "Sunny."
assert message_only[0].message.tool_calls is None
def test_convert_response_output_merges_raw_dict_message_and_function_call() -> None:
handler: Final = LiteLLMResponsesTransformationHandler()
raw_message: Final = {
"type": "message",
"role": "assistant",
"content": [{"type": "output_text", "text": "Let me check.", "annotations": []}],
}
raw_function_call: Final = {
"type": "function_call",
"id": "fc_1",
"call_id": "call_1",
"name": "get_weather",
"arguments": '{"city":"Paris"}',
}
choices: Final = LiteLLMResponsesTransformationHandler._convert_response_output_to_choices(
(raw_message, raw_function_call),
handle_raw_dict_callback=handler._handle_raw_dict_response_item,
)
assert len(choices) == 1
assert choices[0].index == 0
assert choices[0].finish_reason == "tool_calls"
assert choices[0].message.role == "assistant"
assert choices[0].message.content == "Let me check."
assert choices[0].message.tool_calls is not None
assert len(choices[0].message.tool_calls) == 1
def test_convert_tools_to_responses_format_flattens_nested_custom_tool():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,