mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Merge pull request #37541 from ljogeiger/litellm_vertex_parallel_fc_thought_signatures
fix(vertex_ai): only fall back to a placeholder thought signature on the first parallel function call
This commit is contained in:
commit
6eacdbfbf0
3 changed files with 389 additions and 19 deletions
|
|
@ -1200,13 +1200,14 @@ def _encode_tool_call_id_with_signature(tool_call_id: str, thought_signature: st
|
|||
return tool_call_id
|
||||
|
||||
|
||||
def _get_thought_signature_from_tool(tool: dict, model: str | None = None) -> str | None:
|
||||
def _get_thought_signature_from_tool(tool: dict) -> str | None:
|
||||
"""Extract thought signature from tool call's provider_specific_fields.
|
||||
|
||||
If not provided try to extract thought signature from tool call id
|
||||
|
||||
Checks both tool.provider_specific_fields and tool.function.provider_specific_fields.
|
||||
If no signature is found and model is gemini-3, returns a dummy signature.
|
||||
Returns None when the tool call carries no signature; callers decide whether a
|
||||
placeholder signature is needed.
|
||||
"""
|
||||
# First check tool's provider_specific_fields
|
||||
provider_fields: Final = tool.get("provider_specific_fields") or {}
|
||||
|
|
@ -1236,13 +1237,6 @@ def _get_thought_signature_from_tool(tool: dict, model: str | None = None) -> st
|
|||
if len(parts) == 2:
|
||||
_, signature = parts
|
||||
return signature
|
||||
# If no signature found and model is gemini-3, return dummy signature
|
||||
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
|
||||
VertexGeminiConfig,
|
||||
)
|
||||
|
||||
if model and VertexGeminiConfig._is_gemini_3_or_newer(model):
|
||||
return _get_dummy_thought_signature()
|
||||
return None
|
||||
|
||||
|
||||
|
|
@ -1251,10 +1245,14 @@ def _get_dummy_thought_signature() -> str:
|
|||
|
||||
This is used when transferring conversation history from older models
|
||||
(like gemini-2.5-flash) to gemini-3, which requires thought_signature
|
||||
for strict validation.
|
||||
for strict validation. Google documents it as a last resort that "will
|
||||
negatively impact model performance", so callers must only fall back to it
|
||||
when no real signature is available.
|
||||
|
||||
See:
|
||||
https://ai.google.dev/gemini-api/docs/thought-signatures#faqs
|
||||
https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures
|
||||
"""
|
||||
# Return a base64-encoded dummy signature string
|
||||
# Below dummy signature is recommended by google - https://ai.google.dev/gemini-api/docs/thought-signatures#faqs
|
||||
dummy_data: Final = b"skip_thought_signature_validator"
|
||||
return base64.b64encode(dummy_data).decode("utf-8")
|
||||
|
||||
|
|
@ -1312,8 +1310,10 @@ def convert_to_gemini_tool_call_invoke(
|
|||
VertexGeminiConfig,
|
||||
)
|
||||
|
||||
needs_dummy_signature: Final = model is not None and VertexGeminiConfig._is_gemini_3_or_newer(model)
|
||||
|
||||
if tool_calls is not None:
|
||||
for idx, tool in enumerate(tool_calls):
|
||||
for tool in tool_calls:
|
||||
if "function" in tool:
|
||||
gemini_function_call: VertexFunctionCall | None = _gemini_tool_call_invoke_helper(
|
||||
function_call_params=tool["function"],
|
||||
|
|
@ -1321,7 +1321,13 @@ def convert_to_gemini_tool_call_invoke(
|
|||
)
|
||||
if gemini_function_call is not None:
|
||||
part_dict: VertexPartType = {"function_call": gemini_function_call}
|
||||
thought_signature = _get_thought_signature_from_tool(dict(tool), model=model)
|
||||
thought_signature = _get_thought_signature_from_tool(dict(tool))
|
||||
# Gemini signs only the first functionCall part of a parallel batch, so scope the
|
||||
# placeholder fallback to that part instead of fabricating one per sibling call:
|
||||
# https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/thinking/thought-signatures#parallel_function_calling_example
|
||||
is_first_function_call = len(_parts_list) == 0
|
||||
if not thought_signature and is_first_function_call and needs_dummy_signature:
|
||||
thought_signature = _get_dummy_thought_signature()
|
||||
if thought_signature:
|
||||
part_dict["thoughtSignature"] = thought_signature
|
||||
|
||||
|
|
@ -1344,7 +1350,7 @@ def convert_to_gemini_tool_call_invoke(
|
|||
thought_signature = provider_fields.get("thought_signature")
|
||||
|
||||
# If no signature found and model is gemini-3, use dummy signature
|
||||
if not thought_signature and model and VertexGeminiConfig._is_gemini_3_or_newer(model):
|
||||
if not thought_signature and needs_dummy_signature:
|
||||
thought_signature = _get_dummy_thought_signature()
|
||||
|
||||
if thought_signature:
|
||||
|
|
|
|||
|
|
@ -645,10 +645,9 @@ def _collect_tool_call_thought_signatures(
|
|||
the text part as well would send two copies and double-bill the previous
|
||||
turn's reasoning tokens on gemini-3 and newer models.
|
||||
|
||||
Detection deliberately calls _get_thought_signature_from_tool without the
|
||||
model argument: with a gemini-3 model that helper synthesizes a dummy
|
||||
signature for unsigned tool calls, which must not suppress a real
|
||||
text-part signature (e.g. replaying gemini-2.5 history to a newer model).
|
||||
Only real signatures count here; a synthesized placeholder must not
|
||||
suppress a genuine text-part signature (e.g. replaying gemini-2.5 history
|
||||
to a newer model).
|
||||
"""
|
||||
signatures: tuple[str, ...] = ()
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,7 @@
|
|||
import base64
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_result,
|
||||
)
|
||||
|
|
@ -784,6 +788,367 @@ def test_dummy_signature_with_function_call_mode():
|
|||
assert gemini_parts[0]["thoughtSignature"] == expected_dummy
|
||||
|
||||
|
||||
def _parallel_tool_calls(*signatures):
|
||||
return [
|
||||
{
|
||||
"id": f"call_{idx}",
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": f"tool_{idx}",
|
||||
"arguments": '{"location": "Paris"}',
|
||||
**(
|
||||
{"provider_specific_fields": {"thought_signature": signature}}
|
||||
if signature is not None
|
||||
else {}
|
||||
),
|
||||
},
|
||||
"index": idx,
|
||||
}
|
||||
for idx, signature in enumerate(signatures)
|
||||
]
|
||||
|
||||
|
||||
def _parallel_tool_calls_signed_via_id(*signatures):
|
||||
"""Parallel tool calls in the shape LiteLLM actually hands back to clients.
|
||||
|
||||
The signature rides in the tool call id behind __thought__, which is what an
|
||||
OpenAI-format client echoes back on the next turn.
|
||||
"""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
_encode_tool_call_id_with_signature,
|
||||
)
|
||||
|
||||
return [
|
||||
{
|
||||
"id": _encode_tool_call_id_with_signature(f"call_{idx}", signature),
|
||||
"type": "function",
|
||||
"function": {"name": f"tool_{idx}", "arguments": '{"location": "Paris"}'},
|
||||
"index": idx,
|
||||
}
|
||||
for idx, signature in enumerate(signatures)
|
||||
]
|
||||
|
||||
|
||||
REAL_THOUGHT_SIGNATURE = "Co4CAdHtim/rWgXbz2Ghp4tShzLeMASrPw6JJyYIC3cbVyZnKzU3uv8/wVzyS2sKRPL2m8QQHHXbNQhEEz500G7n"
|
||||
PLACEHOLDER_SIGNATURE = base64.b64encode(b"skip_thought_signature_validator").decode(
|
||||
"utf-8"
|
||||
)
|
||||
|
||||
|
||||
def test_dummy_signature_only_on_first_parallel_tool_call():
|
||||
"""Google documents the placeholder as a last resort that degrades quality, so an unsigned
|
||||
parallel turn replayed to gemini-3 gets a budget of exactly one."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(None, None, None),
|
||||
},
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 3
|
||||
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
assert "thoughtSignature" not in gemini_parts[2]
|
||||
|
||||
|
||||
def test_real_signature_on_first_parallel_tool_call_leaves_siblings_empty():
|
||||
"""Gemini signs only the first of N parallel function calls, so a faithful replay has
|
||||
nothing to attach to the siblings."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(REAL_THOUGHT_SIGNATURE, None, None),
|
||||
},
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 3
|
||||
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
assert "thoughtSignature" not in gemini_parts[2]
|
||||
|
||||
|
||||
def test_real_signature_on_later_parallel_tool_call_is_preserved():
|
||||
"""Clients may reorder or drop calls, so a signature that lands on a non-first call is
|
||||
still the model's own and must survive the round trip."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(None, REAL_THOUGHT_SIGNATURE),
|
||||
},
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 2
|
||||
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
|
||||
assert gemini_parts[1]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
|
||||
|
||||
def test_no_signatures_on_parallel_tool_calls_for_gemini_2_5():
|
||||
"""Non-gemini-3 models never get a placeholder signature, on any call."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(None, None),
|
||||
},
|
||||
model="gemini-2.5-flash",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 2
|
||||
assert all("thoughtSignature" not in part for part in gemini_parts)
|
||||
|
||||
|
||||
def test_signature_embedded_in_tool_call_id_only_on_first_parallel_call():
|
||||
"""The production shape: the signature arrives inside the first call's id, siblings have bare ids."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls_signed_via_id(
|
||||
REAL_THOUGHT_SIGNATURE, None, None
|
||||
),
|
||||
},
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 3
|
||||
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
assert "thoughtSignature" not in gemini_parts[2]
|
||||
|
||||
|
||||
def test_tool_level_provider_specific_fields_signature_leaves_siblings_empty():
|
||||
"""A signature on the tool call itself, rather than on its function, behaves the same way."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
tool_calls = _parallel_tool_calls(None, None)
|
||||
tool_calls[0]["provider_specific_fields"] = {
|
||||
"thought_signature": REAL_THOUGHT_SIGNATURE
|
||||
}
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{"role": "assistant", "content": None, "tool_calls": tool_calls},
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 2
|
||||
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
|
||||
|
||||
def test_placeholder_lands_on_first_emitted_part_not_first_tool_call_entry():
|
||||
"""A non-function entry (e.g. an OpenAI custom tool call) emits no part, so it must not
|
||||
consume the one placeholder slot and leave the real first function call bare."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
tool_calls = [
|
||||
{"id": "call_custom", "type": "custom", "custom": {"name": "noop", "input": ""}}
|
||||
] + _parallel_tool_calls(None, None)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{"role": "assistant", "content": None, "tool_calls": tool_calls},
|
||||
model="gemini-3-pro-preview",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 2
|
||||
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
|
||||
|
||||
def test_no_placeholder_when_model_is_unknown():
|
||||
"""Without a model there is nothing to prove the target needs a placeholder, so none is added."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(None, None),
|
||||
},
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 2
|
||||
assert all("thoughtSignature" not in part for part in gemini_parts)
|
||||
|
||||
|
||||
def test_real_signature_forwarded_to_gemini_2_5_without_placeholder_siblings():
|
||||
"""Older models still receive a real signature that a client replays, and still get no placeholder."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(REAL_THOUGHT_SIGNATURE, None),
|
||||
},
|
||||
model="gemini-2.5-flash",
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 2
|
||||
assert gemini_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
|
||||
|
||||
def test_parallel_tool_call_history_replayed_through_full_message_conversion():
|
||||
"""End to end through the message-history converter, the path a real /chat/completions replay takes."""
|
||||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_gemini_convert_messages_with_history,
|
||||
)
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Weather in Paris, London and Tokyo?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls_signed_via_id(
|
||||
REAL_THOUGHT_SIGNATURE, None, None
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
contents = _gemini_convert_messages_with_history(
|
||||
messages=messages, model="gemini-3-pro-preview"
|
||||
)
|
||||
|
||||
model_parts = contents[1]["parts"]
|
||||
assert len(model_parts) == 3
|
||||
assert model_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
assert "thoughtSignature" not in model_parts[1]
|
||||
assert "thoughtSignature" not in model_parts[2]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
["gemini-3.5-flash", "vertex_ai/gemini-3.5-flash", "gemini/gemini-3.5-flash"],
|
||||
)
|
||||
def test_natively_signed_parallel_turn_never_carries_a_placeholder(model):
|
||||
"""A native gemini-3.5 parallel turn replays with zero skip_thought_signature_validator parts.
|
||||
|
||||
Fabricating the placeholder alongside a real signature is what produced empty text responses
|
||||
on gemini-3.5 parallel function calling, so the whole payload has to stay placeholder-free.
|
||||
"""
|
||||
import json
|
||||
|
||||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_gemini_convert_messages_with_history,
|
||||
)
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Weather in Paris, London and Tokyo?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls_signed_via_id(
|
||||
REAL_THOUGHT_SIGNATURE, None, None
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
contents = _gemini_convert_messages_with_history(messages=messages, model=model)
|
||||
|
||||
model_parts = contents[1]["parts"]
|
||||
assert len(model_parts) == 3
|
||||
assert model_parts[0]["thoughtSignature"] == REAL_THOUGHT_SIGNATURE
|
||||
assert "thoughtSignature" not in model_parts[1]
|
||||
assert "thoughtSignature" not in model_parts[2]
|
||||
assert PLACEHOLDER_SIGNATURE not in json.dumps(contents)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"gemini-3-pro-preview",
|
||||
"gemini-3-flash-preview",
|
||||
"gemini-3.1-pro-preview",
|
||||
"gemini-3.5-flash",
|
||||
"gemini-3.6-flash",
|
||||
"gemini-3.7-flash",
|
||||
"vertex_ai/gemini-3.5-flash",
|
||||
"vertex_ai/gemini-3.7-flash",
|
||||
"gemini/gemini-3.5-flash",
|
||||
"gemini/gemini-3.7-flash",
|
||||
],
|
||||
)
|
||||
def test_placeholder_scoped_to_first_call_across_gemini_3_variants(model):
|
||||
"""The gemini-3 gate is a substring match, so every family member and prefix form has to
|
||||
land on the same one-placeholder budget rather than only the versions we happened to try."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
convert_to_gemini_tool_call_invoke,
|
||||
)
|
||||
|
||||
gemini_parts = convert_to_gemini_tool_call_invoke(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": _parallel_tool_calls(None, None, None),
|
||||
},
|
||||
model=model,
|
||||
)
|
||||
|
||||
assert len(gemini_parts) == 3
|
||||
assert gemini_parts[0]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
|
||||
assert "thoughtSignature" not in gemini_parts[1]
|
||||
assert "thoughtSignature" not in gemini_parts[2]
|
||||
|
||||
|
||||
def test_signed_text_part_survives_alongside_unsigned_parallel_tool_calls():
|
||||
"""Text-part and function-call signatures are collected by separate code paths, so scoping the
|
||||
placeholder must not disturb a real signature that arrived on the text part."""
|
||||
from litellm.llms.vertex_ai.gemini.transformation import (
|
||||
_gemini_convert_messages_with_history,
|
||||
)
|
||||
|
||||
msg = {
|
||||
"role": "assistant",
|
||||
"content": "Checking all three cities.",
|
||||
"provider_specific_fields": {"thought_signatures": ["real_25_signature"]},
|
||||
"tool_calls": _parallel_tool_calls(None, None, None),
|
||||
}
|
||||
|
||||
parts = _gemini_convert_messages_with_history(
|
||||
messages=[msg], model="gemini-3-pro-preview"
|
||||
)[0]["parts"]
|
||||
|
||||
assert parts[0]["text"] == "Checking all three cities."
|
||||
assert parts[0]["thoughtSignature"] == "real_25_signature"
|
||||
assert parts[1]["thoughtSignature"] == PLACEHOLDER_SIGNATURE
|
||||
assert "thoughtSignature" not in parts[2]
|
||||
assert "thoughtSignature" not in parts[3]
|
||||
|
||||
|
||||
# Tests for media_resolution (detail parameter) handling - Issue #17084
|
||||
class TestMediaResolution:
|
||||
"""Tests for media_resolution handling in Gemini 2.x models"""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue