From e9d16bc35cb7e1a754114a1977cdcab441c9f39b Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Tue, 21 Jul 2026 20:51:21 -0700 Subject: [PATCH] fix(litellm): make the responses bridge and cursor routing total over the surfaces they now serve Three gaps from the bridge becoming a mainstream path for chat traffic. The chat to responses message converter only mapped function tool_calls, so history carrying the native custom tool calls this PR introduced raised "tool call not supported" on follow-up turns; custom entries now map to custom_tool_call items and their results to custom_tool_call_output. The stream translator returned an empty delta for output_item.done on tool items, which left the responses guardrail handler's tool extraction permanently empty (dead on staging too, where the built chunk was discarded); stateless callers now receive the complete tool call while per-stream callers keep the suppressed delta that prevents client-side duplication. Cursor routing keyed on the presence of a messages key, so a null or empty stub next to a real agent-mode input array picked the chat arm; routing now keys on messages content --- .../transformation.py | 56 ++++++++-- .../proxy/response_api_endpoints/endpoints.py | 13 ++- ...responses_transformation_transformation.py | 103 ++++++++++++++++++ .../response_api_endpoints/test_endpoints.py | 59 ++++++++++ 4 files changed, 222 insertions(+), 9 deletions(-) diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index e1842a62d56..768bf6c3e66 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -221,6 +221,7 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): ) -> Tuple[List[Any], Optional[str]]: input_items: List[Any] = [] instructions: Optional[str] = None + custom_tool_call_ids: set = set() for msg in messages: role = msg.get("role") @@ -266,18 +267,28 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): else: # Fallback: convert unexpected types to input_text tool_output = [{"type": "input_text", "text": str(content)}] - input_items.append( - { - "type": "function_call_output", - "call_id": tool_call_id, - "output": tool_output, - } - ) + if tool_call_id in custom_tool_call_ids: + input_items.append( + { + "type": "custom_tool_call_output", + "call_id": tool_call_id, + "output": content if isinstance(content, str) else tool_output, + } + ) + else: + input_items.append( + { + "type": "function_call_output", + "call_id": tool_call_id, + "output": tool_output, + } + ) elif role == "assistant" and tool_calls and isinstance(tool_calls, list): for r_item in _get_reasoning_items(msg): input_items.append(_reasoning_item_to_response_input(r_item)) for tool_call in tool_calls: function = tool_call.get("function") + custom = tool_call.get("custom") if function: input_tool_call: Dict[str, Any] = { "type": "function_call", @@ -288,6 +299,16 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): if "arguments" in function: input_tool_call["arguments"] = function["arguments"] input_items.append(input_tool_call) + elif isinstance(custom, dict): + custom_tool_call_ids.add(tool_call["id"]) + input_items.append( + { + "type": "custom_tool_call", + "call_id": tool_call["id"], + "name": custom.get("name", ""), + "input": custom.get("input", ""), + } + ) else: raise ValueError(f"tool call not supported: {tool_call}") elif content is not None: @@ -1272,6 +1293,27 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator): # New output item added output_item = parsed_chunk.get("item", {}) if output_item.get("type") in ("function_call", "custom_tool_call"): + if tool_call_index_map is None: + # Stateless callers (the responses guardrail handler extracting + # tool calls from a buffered output_item.done) get the complete + # tool call; per-stream callers already received it via + # output_item.added and the argument delta events + return ModelResponseStream( + choices=[ + StreamingChoices( + index=0, + delta=Delta( + tool_calls=[ + { + **_tool_call_dict_from_output_item(dict(output_item)), + "index": parsed_chunk.get("output_index", 0), + } + ] + ), + finish_reason=None, + ) + ] + ) # Do NOT emit finish_reason here — response.completed handles the terminal # finish_reason. Emitting "tool_calls" here would prematurely terminate # the stream before subsequent tool calls arrive (same fix as #17246 for diff --git a/litellm/proxy/response_api_endpoints/endpoints.py b/litellm/proxy/response_api_endpoints/endpoints.py index 8a678601284..8ee742c0d69 100644 --- a/litellm/proxy/response_api_endpoints/endpoints.py +++ b/litellm/proxy/response_api_endpoints/endpoints.py @@ -95,6 +95,13 @@ def _flatten_chat_tools_for_responses(tools: list) -> list: return [_flatten_chat_tool_for_responses(tool) for tool in tools] +def _is_chat_completions_body(data: dict) -> bool: + messages = data.get("messages") + if isinstance(messages, list) and len(messages) > 0: + return True + return "messages" in data and "input" not in data + + def _flatten_chat_tool_choice_for_responses(tool_choice: object) -> object: if not isinstance(tool_choice, dict): return tool_choice @@ -456,9 +463,11 @@ async def cursor_chat_completions( data = await _read_request_body(request=request) - if "messages" in data: + if _is_chat_completions_body(data): # Genuine chat completions body (Cursor sends these for models whose BYOK it - # already fixed); delegate so behavior matches /chat/completions exactly + # already fixed); delegate so behavior matches /chat/completions exactly. + # Keyed on messages CONTENT, not key presence: Cursor can send a null or + # empty messages stub alongside a real agent-mode input array tools = data.get("tools") tool_choice = data.get("tool_choice") normalized: dict = {} diff --git a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 64112ed43e8..b8bd5c951ee 100644 --- a/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/test_litellm/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -3297,3 +3297,106 @@ def test_convert_tools_to_responses_format_text_format_passes_through(): [{"type": "custom", "custom": {"name": "A", "format": {"type": "text"}}}] ) assert converted[0] == {"type": "custom", "name": "A", "format": {"type": "text"}} + + +def test_convert_chat_completion_messages_maps_custom_tool_call_history(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + input_items, instructions = handler.convert_chat_completion_messages_to_responses_api( + [ + {"role": "user", "content": "use ApplyPatch"}, + { + "role": "assistant", + "tool_calls": [ + { + "id": "call_c", + "type": "custom", + "custom": {"name": "ApplyPatch", "input": "*** Begin Patch"}, + }, + { + "id": "call_f", + "type": "function", + "function": {"name": "shell", "arguments": '{"cmd": "ls"}'}, + }, + ], + }, + {"role": "tool", "tool_call_id": "call_c", "content": "patch applied"}, + {"role": "tool", "tool_call_id": "call_f", "content": "a.py"}, + ] + ) + assert { + "type": "custom_tool_call", + "call_id": "call_c", + "name": "ApplyPatch", + "input": "*** Begin Patch", + } in input_items + assert {"type": "custom_tool_call_output", "call_id": "call_c", "output": "patch applied"} in input_items + assert {"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'} in input_items + assert { + "type": "function_call_output", + "call_id": "call_f", + "output": [{"type": "input_text", "text": "a.py"}], + } in input_items + + +def test_convert_chat_completion_messages_still_rejects_unknown_tool_call_shape(): + import pytest + + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + LiteLLMResponsesTransformationHandler, + ) + + handler = LiteLLMResponsesTransformationHandler() + with pytest.raises(ValueError, match="tool call not supported"): + handler.convert_chat_completion_messages_to_responses_api( + [{"role": "assistant", "tool_calls": [{"id": "call_x", "type": "mystery"}]}] + ) + + +def test_output_item_done_stateless_emits_complete_tool_call(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + for item, expected_name, expected_args in ( + ( + {"type": "function_call", "call_id": "call_f", "name": "shell", "arguments": '{"cmd": "ls"}'}, + "shell", + '{"cmd": "ls"}', + ), + ( + {"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "*** Begin Patch"}, + "ApplyPatch", + "*** Begin Patch", + ), + ): + chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( + {"type": "response.output_item.done", "output_index": 2, "item": item} + ) + tool_calls = chunk.choices[0].delta.tool_calls + assert tool_calls is not None and len(tool_calls) == 1 + assert tool_calls[0].id == item["call_id"] + assert tool_calls[0].function.name == expected_name + assert tool_calls[0].function.arguments == expected_args + assert tool_calls[0].index == 2 + assert chunk.choices[0].finish_reason is None + + +def test_output_item_done_with_stream_map_keeps_empty_delta(): + from litellm.completion_extras.litellm_responses_transformation.transformation import ( + OpenAiResponsesToChatCompletionStreamIterator, + ) + + chunk = OpenAiResponsesToChatCompletionStreamIterator.translate_responses_chunk_to_openai_stream( + { + "type": "response.output_item.done", + "output_index": 0, + "item": {"type": "custom_tool_call", "call_id": "call_c", "name": "ApplyPatch", "input": "x"}, + }, + tool_call_index_map={0: 0}, + ) + assert chunk.choices[0].delta.tool_calls is None + assert chunk.choices[0].finish_reason is None diff --git a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py index 04f027ac4bb..6a1e0d0a494 100644 --- a/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py +++ b/tests/test_litellm/proxy/response_api_endpoints/test_endpoints.py @@ -1287,3 +1287,62 @@ class TestCursorInputArmFlattening: {"type": "function", "name": "read_file", "parameters": {"type": "object"}}, ] assert call_kwargs["tool_choice"] == {"type": "custom", "name": "ApplyPatch"} + + +class TestChatCompletionsBodyDetection: + def test_routing_matrix(self): + from litellm.proxy.response_api_endpoints.endpoints import _is_chat_completions_body + + assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}]}) is True + assert _is_chat_completions_body({"messages": [{"role": "user", "content": "hi"}], "input": []}) is True + assert _is_chat_completions_body({"messages": None, "input": [{"role": "user", "content": "hi"}]}) is False + assert _is_chat_completions_body({"messages": [], "input": [{"role": "user", "content": "hi"}]}) is False + assert _is_chat_completions_body({"messages": None}) is True + assert _is_chat_completions_body({"messages": []}) is True + assert _is_chat_completions_body({"input": [{"role": "user", "content": "hi"}]}) is False + assert _is_chat_completions_body({}) is False + + @pytest.mark.asyncio + async def test_null_messages_stub_with_input_reaches_responses_arm(self): + from openai.types.responses import ResponseOutputMessage, ResponseOutputText + + from litellm.proxy._types import UserAPIKeyAuth + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.types.llms.openai import ResponsesAPIResponse + + mock_response = ResponsesAPIResponse( + id="resp_stub1", + created_at=1234567890, + model="gpt-5.6", + object="response", + output=[ + ResponseOutputMessage( + id="msg_stub1", + type="message", + role="assistant", + status="completed", + content=[ResponseOutputText(type="output_text", text="ok", annotations=[])], + ) + ], + ) + + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(api_key="sk-1234") + try: + with patch("litellm.proxy.proxy_server.llm_router") as mock_router: + mock_router.aresponses = AsyncMock(return_value=mock_response) + client = TestClient(app) + response = client.post( + "/cursor/chat/completions", + json={ + "model": "gpt-5.6", + "messages": None, + "input": [{"role": "user", "content": "hello"}], + }, + headers={"Authorization": "Bearer sk-1234"}, + ) + finally: + app.dependency_overrides.pop(user_api_key_auth, None) + + assert response.status_code == 200 + assert mock_router.aresponses.call_args is not None + assert mock_router.aresponses.call_args.kwargs["input"] == [{"role": "user", "content": "hello"}]