From d3e695f0552a6637d9a0f573a1c86f83eecfa47c Mon Sep 17 00:00:00 2001 From: Guilherme Pires Date: Thu, 12 Feb 2026 22:26:49 -0800 Subject: [PATCH] Fix Gemini parallel tool call ID mismatch in Responses API translation When litellm converts Responses API function_call input items to Chat Completion messages, each function_call item becomes a separate assistant message with a single tool_calls entry. For providers like Gemini that issue parallel tool calls in one model turn, this produces multiple assistant messages instead of one with multiple tool_calls. The Gemini message converter later walks the history looking for the 'last assistant message with tool calls' to match a tool result. With split messages, it only finds the last call's assistant message and fails to match earlier calls' tool results, producing: Missing corresponding tool call for tool response message. Received - message={'role': 'tool', 'content': '...', 'tool_call_id': 'call_xxx'} Fix: merge consecutive function_call-derived assistant messages into a single assistant message with multiple tool_calls entries, re-indexing as needed. This preserves the one-turn-one-message invariant that downstream converters expect. Affects: Gemini 3 models with parallel tool calls via the Responses API (litellm.aresponses / /v1/responses endpoint). --- .../transformation.py | 46 ++++++++++++++++++- 1 file changed, 45 insertions(+), 1 deletion(-) diff --git a/litellm/responses/litellm_completion_transformation/transformation.py b/litellm/responses/litellm_completion_transformation/transformation.py index 08e31c59662..fdd70657cf6 100644 --- a/litellm/responses/litellm_completion_transformation/transformation.py +++ b/litellm/responses/litellm_completion_transformation/transformation.py @@ -434,7 +434,51 @@ class LiteLLMCompletionResponsesConfig: messages.extend(deduped_in_place) continue - messages.extend(chat_completion_messages) + ########################################################### + # Merge consecutive function_call items into a single + # assistant message. Providers like Gemini require that + # parallel tool calls that were issued in one model turn + # appear as *one* assistant message with multiple + # tool_calls, not as separate assistant messages. + ########################################################### + for m in chat_completion_messages: + role = m.get("role") if isinstance(m, dict) else getattr(m, "role", "") + if role != "assistant": + messages.append(m) + continue + + new_tcs: Any = ( + m.get("tool_calls") + if isinstance(m, dict) + else getattr(m, "tool_calls", None) + ) or [] + + # Try to merge into the last message if it is an + # assistant message that already carries tool_calls. + merged = False + if messages: + last = messages[-1] + last_role = last.get("role") if isinstance(last, dict) else getattr(last, "role", "") + if last_role == "assistant": + existing_tcs: Any = ( + last.get("tool_calls") + if isinstance(last, dict) + else getattr(last, "tool_calls", None) + ) + if existing_tcs is not None: + # Re-index and append + for tc in new_tcs: + idx = len(existing_tcs) + if isinstance(tc, dict): + tc["index"] = idx + elif hasattr(tc, "index"): + tc.index = idx + existing_tcs.append(tc) + merged = True + + if not merged: + messages.append(m) + return messages @staticmethod