Merge pull request #37541 from ljogeiger/litellm_vertex_parallel_fc_thought_signatures

fix(vertex_ai): only fall back to a placeholder thought signature on the first parallel function call
This commit is contained in:
Mateo Wang 2026-08-20 17:53:05 -07:00 • committed by GitHub
commit 6eacdbfbf0
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 389 additions and 19 deletions

View file

@ -1200,13 +1200,14 @@ def _encode_tool_call_id_with_signature(tool_call_id: str, thought_signature: st
return tool_call_id
def _get_thought_signature_from_tool(tool: dict, model: str | None = None) -> str | None:
def _get_thought_signature_from_tool(tool: dict) -> str | None:
"""Extract thought signature from tool call's provider_specific_fields.
If not provided try to extract thought signature from tool call id
Checks both tool.provider_specific_fields and tool.function.provider_specific_fields.
If no signature is found and model is gemini-3, returns a dummy signature.
Returns None when the tool call carries no signature; callers decide whether a
placeholder signature is needed.
"""
# First check tool's provider_specific_fields
provider_fields: Final = tool.get("provider_specific_fields") or {}
@ -1236,13 +1237,6 @@ def _get_thought_signature_from_tool(tool: dict, model: str | None = None) -> st
if len(parts) == 2:
_, signature = parts
return signature
# If no signature found and model is gemini-3, return dummy signature
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
VertexGeminiConfig,
)
if model and VertexGeminiConfig._is_gemini_3_or_newer(model):
return _get_dummy_thought_signature()
return None
@ -1251,10 +1245,14 @@ def _get_dummy_thought_signature() -> str:
This is used when transferring conversation history from older models
(like gemini-2.5-flash) to gemini-3, which requires thought_signature
for strict validation.
for strict validation. Google documents it as a last resort that "will
negatively impact model performance", so callers must only fall back to it
when no real signature is available.
See:
https://ai.google.dev/gemini-api/docs/thought-signatures#faqs
https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures
"""
# Return a base64-encoded dummy signature string
# Below dummy signature is recommended by google - https://ai.google.dev/gemini-api/docs/thought-signatures#faqs
dummy_data: Final = b"skip_thought_signature_validator"
return base64.b64encode(dummy_data).decode("utf-8")
@ -1312,8 +1310,10 @@ def convert_to_gemini_tool_call_invoke(
VertexGeminiConfig,
)
needs_dummy_signature: Final = model is not None and VertexGeminiConfig._is_gemini_3_or_newer(model)
if tool_calls is not None:
for idx, tool in enumerate(tool_calls):
for tool in tool_calls:
if "function" in tool:
gemini_function_call: VertexFunctionCall | None = _gemini_tool_call_invoke_helper(
function_call_params=tool["function"],
@ -1321,7 +1321,13 @@ def convert_to_gemini_tool_call_invoke(
)
if gemini_function_call is not None:
part_dict: VertexPartType = {"function_call": gemini_function_call}
thought_signature = _get_thought_signature_from_tool(dict(tool), model=model)
thought_signature = _get_thought_signature_from_tool(dict(tool))
# Gemini signs only the first functionCall part of a parallel batch, so scope the
# placeholder fallback to that part instead of fabricating one per sibling call:
# https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures#parallel_function_calling_example
is_first_function_call = len(_parts_list) == 0
if not thought_signature and is_first_function_call and needs_dummy_signature:
thought_signature = _get_dummy_thought_signature()
if thought_signature:
part_dict["thoughtSignature"] = thought_signature
@ -1344,7 +1350,7 @@ def convert_to_gemini_tool_call_invoke(
thought_signature = provider_fields.get("thought_signature")
# If no signature found and model is gemini-3, use dummy signature
if not thought_signature and model and VertexGeminiConfig._is_gemini_3_or_newer(model):
if not thought_signature and needs_dummy_signature:
thought_signature = _get_dummy_thought_signature()
if thought_signature:

View file

@ -645,10 +645,9 @@ def _collect_tool_call_thought_signatures(
the text part as well would send two copies and double-bill the previous
turn's reasoning tokens on gemini-3 and newer models.
Detection deliberately calls _get_thought_signature_from_tool without the
model argument: with a gemini-3 model that helper synthesizes a dummy
signature for unsigned tool calls, which must not suppress a real
text-part signature (e.g. replaying gemini-2.5 history to a newer model).
Only real signatures count here; a synthesized placeholder must not
suppress a genuine text-part signature (e.g. replaying gemini-2.5 history
to a newer model).
"""
signatures: tuple[str, ...] = ()

View file

@ -1,3 +1,7 @@
import base64
import pytest
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_result,
)
@ -784,6 +788,367 @@ def test_dummy_signature_with_function_call_mode():
assert gemini_parts[0]["thoughtSignature"] == expected_dummy
def _parallel_tool_calls(*signatures):
return [
{
"id": f"call_{idx}",
"type": "function",
"function": {
"name": f"tool_{idx}",
"arguments": '{"location": "Paris"}',
**(
{"provider_specific_fields": {"thought_signature": signature}}
if signature is not None
else {}
),
},
"index": idx,
}
for idx, signature in enumerate(signatures)
]
def _parallel_tool_calls_signed_via_id(*signatures):
"""Parallel tool calls in the shape LiteLLM actually hands back to clients.
The signature rides in the tool call id behind __thought__, which is what an
OpenAI-format client echoes back on the next turn.
"""
from litellm.litellm_core_utils.prompt_templates.factory import (
_encode_tool_call_id_with_signature,
)
return [
{
"id": _encode_tool_call_id_with_signature(f"call_{idx}", signature),
"type": "function",
"function": {"name": f"tool_{idx}", "arguments": '{"location": "Paris"}'},
"index": idx,
}
for idx, signature in enumerate(signatures)
]
REAL_THOUGHT_SIGNATURE = "Co4CAdHtim/rWgXbz2Ghp4tShzLeMASrPw6JJyYIC3cbVyZnKzU3uv8/wVzyS2sKRPL2m8QQHHXbNQhEEz500G7n"
PLACEHOLDER_SIGNATURE = base64.b64encode(b"skip_thought_signature_validator").decode(
"utf-8"
)
def test_dummy_signature_only_on_first_parallel_tool_call():
"""Google documents the placeholder as a last resort that degrades quality, so an unsigned
parallel turn replayed to gemini-3 gets a budget of exactly one."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(None, None, None),
},
model="gemini-3-pro-preview",
)
assert len(gemini_parts) == 3
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
assert "thoughtSignature" not in gemini_parts[2]
def test_real_signature_on_first_parallel_tool_call_leaves_siblings_empty():
"""Gemini signs only the first of N parallel function calls, so a faithful replay has
nothing to attach to the siblings."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(REAL_THOUGHT_SIGNATURE, None, None),
},
model="gemini-3-pro-preview",
)
assert len(gemini_parts) == 3
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
assert "thoughtSignature" not in gemini_parts[2]
def test_real_signature_on_later_parallel_tool_call_is_preserved():
"""Clients may reorder or drop calls, so a signature that lands on a non-first call is
still the model's own and must survive the round trip."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(None, REAL_THOUGHT_SIGNATURE),
},
model="gemini-3-pro-preview",
)
assert len(gemini_parts) == 2
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
assert gemini_parts[1]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
def test_no_signatures_on_parallel_tool_calls_for_gemini_2_5():
"""Non-gemini-3 models never get a placeholder signature, on any call."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(None, None),
},
model="gemini-2.5-flash",
)
assert len(gemini_parts) == 2
assert all("thoughtSignature" not in part for part in gemini_parts)
def test_signature_embedded_in_tool_call_id_only_on_first_parallel_call():
"""The production shape: the signature arrives inside the first call's id, siblings have bare ids."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls_signed_via_id(
REAL_THOUGHT_SIGNATURE, None, None
),
},
model="gemini-3-pro-preview",
)
assert len(gemini_parts) == 3
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
assert "thoughtSignature" not in gemini_parts[2]
def test_tool_level_provider_specific_fields_signature_leaves_siblings_empty():
"""A signature on the tool call itself, rather than on its function, behaves the same way."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
tool_calls = _parallel_tool_calls(None, None)
tool_calls[0]["provider_specific_fields"] = {
"thought_signature": REAL_THOUGHT_SIGNATURE
}
gemini_parts = convert_to_gemini_tool_call_invoke(
{"role": "assistant", "content": None, "tool_calls": tool_calls},
model="gemini-3-pro-preview",
)
assert len(gemini_parts) == 2
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
def test_placeholder_lands_on_first_emitted_part_not_first_tool_call_entry():
"""A non-function entry (e.g. an OpenAI custom tool call) emits no part, so it must not
consume the one placeholder slot and leave the real first function call bare."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
tool_calls = [
{"id": "call_custom", "type": "custom", "custom": {"name": "noop", "input": ""}}
] + _parallel_tool_calls(None, None)
gemini_parts = convert_to_gemini_tool_call_invoke(
{"role": "assistant", "content": None, "tool_calls": tool_calls},
model="gemini-3-pro-preview",
)
assert len(gemini_parts) == 2
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
def test_no_placeholder_when_model_is_unknown():
"""Without a model there is nothing to prove the target needs a placeholder, so none is added."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(None, None),
},
)
assert len(gemini_parts) == 2
assert all("thoughtSignature" not in part for part in gemini_parts)
def test_real_signature_forwarded_to_gemini_2_5_without_placeholder_siblings():
"""Older models still receive a real signature that a client replays, and still get no placeholder."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(REAL_THOUGHT_SIGNATURE, None),
},
model="gemini-2.5-flash",
)
assert len(gemini_parts) == 2
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
def test_parallel_tool_call_history_replayed_through_full_message_conversion():
"""End to end through the message-history converter, the path a real /chat/completions replay takes."""
from litellm.llms.vertex_ai.gemini.transformation import (
_gemini_convert_messages_with_history,
)
messages = [
{"role": "user", "content": "Weather in Paris, London and Tokyo?"},
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls_signed_via_id(
REAL_THOUGHT_SIGNATURE, None, None
),
},
]
contents = _gemini_convert_messages_with_history(
messages=messages, model="gemini-3-pro-preview"
)
model_parts = contents[1]["parts"]
assert len(model_parts) == 3
assert model_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
assert "thoughtSignature" not in model_parts[1]
assert "thoughtSignature" not in model_parts[2]
@pytest.mark.parametrize(
"model",
["gemini-3.5-flash", "vertex_ai/gemini-3.5-flash", "gemini/gemini-3.5-flash"],
)
def test_natively_signed_parallel_turn_never_carries_a_placeholder(model):
"""A native gemini-3.5 parallel turn replays with zero skip_thought_signature_validator parts.
Fabricating the placeholder alongside a real signature is what produced empty text responses
on gemini-3.5 parallel function calling, so the whole payload has to stay placeholder-free.
"""
import json
from litellm.llms.vertex_ai.gemini.transformation import (
_gemini_convert_messages_with_history,
)
messages = [
{"role": "user", "content": "Weather in Paris, London and Tokyo?"},
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls_signed_via_id(
REAL_THOUGHT_SIGNATURE, None, None
),
},
]
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
model_parts = contents[1]["parts"]
assert len(model_parts) == 3
assert model_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
assert "thoughtSignature" not in model_parts[1]
assert "thoughtSignature" not in model_parts[2]
assert PLACEHOLDER_SIGNATURE not in json.dumps(contents)
@pytest.mark.parametrize(
"model",
[
"gemini-3-pro-preview",
"gemini-3-flash-preview",
"gemini-3.1-pro-preview",
"gemini-3.5-flash",
"gemini-3.6-flash",
"gemini-3.7-flash",
"vertex_ai/gemini-3.5-flash",
"vertex_ai/gemini-3.7-flash",
"gemini/gemini-3.5-flash",
"gemini/gemini-3.7-flash",
],
)
def test_placeholder_scoped_to_first_call_across_gemini_3_variants(model):
"""The gemini-3 gate is a substring match, so every family member and prefix form has to
land on the same one-placeholder budget rather than only the versions we happened to try."""
from litellm.litellm_core_utils.prompt_templates.factory import (
convert_to_gemini_tool_call_invoke,
)
gemini_parts = convert_to_gemini_tool_call_invoke(
{
"role": "assistant",
"content": None,
"tool_calls": _parallel_tool_calls(None, None, None),
},
model=model,
)
assert len(gemini_parts) == 3
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
assert "thoughtSignature" not in gemini_parts[1]
assert "thoughtSignature" not in gemini_parts[2]
def test_signed_text_part_survives_alongside_unsigned_parallel_tool_calls():
"""Text-part and function-call signatures are collected by separate code paths, so scoping the
placeholder must not disturb a real signature that arrived on the text part."""
from litellm.llms.vertex_ai.gemini.transformation import (
_gemini_convert_messages_with_history,
)
msg = {
"role": "assistant",
"content": "Checking all three cities.",
"provider_specific_fields": {"thought_signatures": ["real_25_signature"]},
"tool_calls": _parallel_tool_calls(None, None, None),
}
parts = _gemini_convert_messages_with_history(
messages=[msg], model="gemini-3-pro-preview"
)[0]["parts"]
assert parts[0]["text"] == "Checking all three cities."
assert parts[0]["thoughtSignature"] == "real_25_signature"
assert parts[1]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
assert "thoughtSignature" not in parts[2]
assert "thoughtSignature" not in parts[3]
# Tests for media_resolution (detail parameter) handling - Issue #17084
class TestMediaResolution:
"""Tests for media_resolution handling in Gemini 2.x models"""