fix(litellm): make the responses bridge and cursor routing total over the surfaces they now serve

Three gaps from the bridge becoming a mainstream path for chat traffic.
The chat to responses message converter only mapped function tool_calls,
so history carrying the native custom tool calls this PR introduced
raised "tool call not supported" on follow-up turns; custom entries now
map to custom_tool_call items and their results to
custom_tool_call_output. The stream translator returned an empty delta
for output_item.done on tool items, which left the responses guardrail
handler's tool extraction permanently empty (dead on staging too, where
the built chunk was discarded); stateless callers now receive the
complete tool call while per-stream callers keep the suppressed delta
that prevents client-side duplication. Cursor routing keyed on the
presence of a messages key, so a null or empty stub next to a real
agent-mode input array picked the chat arm; routing now keys on
messages content
This commit is contained in:
Tin Chi Lo 2026-07-21 20:51:21 -07:00
parent bbba450301
commit e9d16bc35c
4 changed files with 222 additions and 9 deletions

View file

@ -221,6 +221,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
) -> Tuple[List[Any], Optional[str]]:
input_items: List[Any] = []
instructions: Optional[str] = None
custom_tool_call_ids: set = set()
for msg in messages:
role = msg.get("role")
@ -266,18 +267,28 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
else:
# Fallback: convert unexpected types to input_text
tool_output = [{"type": "input_text", "text": str(content)}]
input_items.append(
{
"type": "function_call_output",
"call_id": tool_call_id,
"output": tool_output,
}
)
if tool_call_id in custom_tool_call_ids:
input_items.append(
{
"type": "custom_tool_call_output",
"call_id": tool_call_id,
"output": content if isinstance(content, str) else tool_output,
}
)
else:
input_items.append(
{
"type": "function_call_output",
"call_id": tool_call_id,
"output": tool_output,
}
)
elif role == "assistant" and tool_calls and isinstance(tool_calls, list):
for r_item in _get_reasoning_items(msg):
input_items.append(_reasoning_item_to_response_input(r_item))
for tool_call in tool_calls:
function = tool_call.get("function")
custom = tool_call.get("custom")
if function:
input_tool_call: Dict[str, Any] = {
"type": "function_call",
@ -288,6 +299,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
if "arguments" in function:
input_tool_call["arguments"] = function["arguments"]
input_items.append(input_tool_call)
elif isinstance(custom, dict):
custom_tool_call_ids.add(tool_call["id"])
input_items.append(
{
"type": "custom_tool_call",
"call_id": tool_call["id"],
"name": custom.get("name", ""),
"input": custom.get("input", ""),
}
)
else:
raise ValueError(f"tool call not supported: {tool_call}")
elif content is not None:
@ -1272,6 +1293,27 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
# New output item added
output_item = parsed_chunk.get("item", {})
if output_item.get("type") in ("function_call", "custom_tool_call"):
if tool_call_index_map is None:
# Stateless callers (the responses guardrail handler extracting
# tool calls from a buffered output_item.done) get the complete
# tool call; per-stream callers already received it via
# output_item.added and the argument delta events
return ModelResponseStream(
choices=[
StreamingChoices(
index=0,
delta=Delta(
tool_calls=[
{
**_tool_call_dict_from_output_item(dict(output_item)),
"index": parsed_chunk.get("output_index", 0),
}
]
),
finish_reason=None,
)
]
)
# Do NOT emit finish_reason here — response.completed handles the terminal
# finish_reason. Emitting "tool_calls" here would prematurely terminate
# the stream before subsequent tool calls arrive (same fix as #17246 for

View file

@ -95,6 +95,13 @@ def _flatten_chat_tools_for_responses(tools: list) -> list:
return [_flatten_chat_tool_for_responses(tool) for tool in tools]
def _is_chat_completions_body(data: dict) -> bool:
messages = data.get("messages")
if isinstance(messages, list) and len(messages) > 0:
return True
return "messages" in data and "input" not in data
def _flatten_chat_tool_choice_for_responses(tool_choice: object) -> object:
if not isinstance(tool_choice, dict):
return tool_choice
@ -456,9 +463,11 @@ async def cursor_chat_completions(
data = await _read_request_body(request=request)
if "messages" in data:
if _is_chat_completions_body(data):
# Genuine chat completions body (Cursor sends these for models whose BYOK it
# already fixed); delegate so behavior matches /chat/completions exactly
# already fixed); delegate so behavior matches /chat/completions exactly.
# Keyed on messages CONTENT, not key presence: Cursor can send a null or
# empty messages stub alongside a real agent-mode input array
tools = data.get("tools")
tool_choice = data.get("tool_choice")
normalized: dict = {}

View file

@ -3297,3 +3297,106 @@ def test_convert_tools_to_responses_format_text_format_passes_through():
[{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}]
)
assert converted[0] == {"type": "custom", "name": "A", "format": {"type": "text"}}
def test_convert_chat_completion_messages_maps_custom_tool_call_history():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
handler = LiteLLMResponsesTransformationHandler()
input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
[
{"role": "user", "content": "use ApplyPatch"},
{
"role": "assistant",
"tool_calls": [
{
"id": "call_c",
"type": "custom",
"custom": {"name": "ApplyPatch", "input": "*** Begin Patch"},
},
{
"id": "call_f",
"type": "function",
"function": {"name": "shell", "arguments": '{"cmd": "ls"}'},
},
],
},
{"role": "tool", "tool_call_id": "call_c", "content": "patch applied"},
{"role": "tool", "tool_call_id": "call_f", "content": "a.py"},
]
)
assert {
"type": "custom_tool_call",
"call_id": "call_c",
"name": "ApplyPatch",
"input": "*** Begin Patch",
} in input_items
assert {"type": "custom_tool_call_output", "call_id": "call_c", "output": "patch applied"} in input_items
assert {"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'} in input_items
assert {
"type": "function_call_output",
"call_id": "call_f",
"output": [{"type": "input_text", "text": "a.py"}],
} in input_items
def test_convert_chat_completion_messages_still_rejects_unknown_tool_call_shape():
import pytest
from litellm.completion_extras.litellm_responses_transformation.transformation import (
LiteLLMResponsesTransformationHandler,
)
handler = LiteLLMResponsesTransformationHandler()
with pytest.raises(ValueError, match="tool call not supported"):
handler.convert_chat_completion_messages_to_responses_api(
[{"role": "assistant", "tool_calls": [{"id": "call_x", "type": "mystery"}]}]
)
def test_output_item_done_stateless_emits_complete_tool_call():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
for item, expected_name, expected_args in (
(
{"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'},
"shell",
'{"cmd": "ls"}',
),
(
{"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "*** Begin Patch"},
"ApplyPatch",
"*** Begin Patch",
),
):
chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
{"type": "response.output_item.done", "output_index": 2, "item": item}
)
tool_calls = chunk.choices[0].delta.tool_calls
assert tool_calls is not None and len(tool_calls) == 1
assert tool_calls[0].id == item["call_id"]
assert tool_calls[0].function.name == expected_name
assert tool_calls[0].function.arguments == expected_args
assert tool_calls[0].index == 2
assert chunk.choices[0].finish_reason is None
def test_output_item_done_with_stream_map_keeps_empty_delta():
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream(
{
"type": "response.output_item.done",
"output_index": 0,
"item": {"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "x"},
},
tool_call_index_map={0: 0},
)
assert chunk.choices[0].delta.tool_calls is None
assert chunk.choices[0].finish_reason is None

View file

@ -1287,3 +1287,62 @@ class TestCursorInputArmFlattening:
{"type": "function", "name": "read_file", "parameters": {"type": "object"}},
]
assert call_kwargs["tool_choice"] == {"type": "custom", "name": "ApplyPatch"}
class TestChatCompletionsBodyDetection:
def test_routing_matrix(self):
from litellm.proxy.response_api_endpoints.endpoints import _is_chat_completions_body
assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}]}) is True
assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}], "input": []}) is True
assert _is_chat_completions_body({"messages": None, "input": [{"role": "user", "content": "hi"}]}) is False
assert _is_chat_completions_body({"messages": [], "input": [{"role": "user", "content": "hi"}]}) is False
assert _is_chat_completions_body({"messages": None}) is True
assert _is_chat_completions_body({"messages": []}) is True
assert _is_chat_completions_body({"input": [{"role": "user", "content": "hi"}]}) is False
assert _is_chat_completions_body({}) is False
@pytest.mark.asyncio
async def test_null_messages_stub_with_input_reaches_responses_arm(self):
from openai.types.responses import ResponseOutputMessage, ResponseOutputText
from litellm.proxy._types import UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.types.llms.openai import ResponsesAPIResponse
mock_response = ResponsesAPIResponse(
id="resp_stub1",
created_at=1234567890,
model="gpt-5.6",
object="response",
output=[
ResponseOutputMessage(
id="msg_stub1",
type="message",
role="assistant",
status="completed",
content=[ResponseOutputText(type="output_text", text="ok", annotations=[])],
)
],
)
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234")
try:
with patch("litellm.proxy.proxy_server.llm_router") as mock_router:
mock_router.aresponses = AsyncMock(return_value=mock_response)
client = TestClient(app)
response = client.post(
"/cursor/chat/completions",
json={
"model": "gpt-5.6",
"messages": None,
"input": [{"role": "user", "content": "hello"}],
},
headers={"Authorization": "Bearer sk-1234"},
)
finally:
app.dependency_overrides.pop(user_api_key_auth, None)
assert response.status_code == 200
assert mock_router.aresponses.call_args is not None
assert mock_router.aresponses.call_args.kwargs["input"] == [{"role": "user", "content": "hello"}]