Handled edge case where gemini-2.5-flash-lite returns content {'role': 'model'} with no parts field when finishReason: STOP. After the existing fallback, added a second pass that corrects finish_reason from "stop" to "tool_calls" on choices that have no delta tool_calls, when has_seen_tool_calls is True.

This commit is contained in:
RoyVivat 2026-03-23 16:25:20 -07:00
parent 50f88c8642
commit 1bfe1dd956
3 changed files with 214 additions and 24 deletions

View file

@ -3001,6 +3001,23 @@ class ModelResponseIterator:
)
model_response.choices.append(choice)
# Handle the case where Gemini returns content: {"role": "model"} with
# no "parts" (e.g., gemini-2.5-flash-lite with finishReason: STOP and
# empty content). _process_candidates creates a choice for this candidate
# because "content" IS present, but it sets finish_reason="stop" without
# knowing about tool_calls seen in earlier streaming chunks.
# Per the OpenAI spec, finish_reason must be "tool_calls" when the model
# previously called a tool.
if self.has_seen_tool_calls and model_response.choices:
for choice in model_response.choices:
if (
hasattr(choice, "finish_reason")
and choice.finish_reason == "stop"
and hasattr(choice, "delta")
and (choice.delta is None or not choice.delta.tool_calls)
):
choice.finish_reason = "tool_calls"
setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata) # type: ignore
setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata) # type: ignore
setattr(model_response, "vertex_ai_safety_ratings", safety_ratings) # type: ignore

View file

@ -230,3 +230,110 @@ def test_streaming_content_filter_finish_reason_preserved():
assert response is not None
assert len(response.choices) == 1
assert response.choices[0].finish_reason == "content_filter"
def test_streaming_empty_content_no_parts_finish_reason_is_stop():
"""
When Gemini returns content: {"role": "model"} with no "parts" and
finishReason: STOP (a known Gemini quirk, e.g. gemini-2.5-flash-lite),
the finish_reason should be "stop" (no prior tool calls).
Ref: https://github.com/BerriAI/litellm/issues/24442
"""
logging_obj = _make_logging_obj()
iterator = ModelResponseIterator(
streaming_response=iter([]),
sync_stream=True,
logging_obj=logging_obj,
)
# Final chunk: content present but no "parts", finishReason="STOP"
chunk_empty_content = {
"candidates": [
{
"content": {"role": "model"},
"finishReason": "STOP",
"index": 0,
}
],
"usageMetadata": {
"promptTokenCount": 3130,
"totalTokenCount": 3130,
},
}
response = iterator.chunk_parser(chunk_empty_content)
assert response is not None
assert len(response.choices) == 1
assert response.choices[0].finish_reason == "stop"
assert response.choices[0].delta.content is None
def test_streaming_tool_calls_then_empty_content_finish_reason_is_tool_calls():
"""
When Gemini streams tool calls in one chunk and then returns a final chunk
with content: {"role": "model"} (no "parts") and finishReason: STOP,
the finish_reason must be "tool_calls" per the OpenAI spec.
This tests the case where Gemini returns empty content (no parts) instead of
the usual empty candidate (no "content" field) as the final chunk.
Ref: https://github.com/BerriAI/litellm/issues/24442
"""
logging_obj = _make_logging_obj()
iterator = ModelResponseIterator(
streaming_response=iter([]),
sync_stream=True,
logging_obj=logging_obj,
)
# Chunk 1: tool call with no finishReason
chunk_with_tool_calls = {
"candidates": [
{
"content": {
"parts": [
{
"functionCall": {
"name": "get_current_weather",
"args": {"location": "Boston, MA"},
}
}
],
"role": "model",
},
"index": 0,
}
],
}
# Chunk 2: finishReason="STOP" with content present but no "parts"
# (Gemini quirk: content: {"role": "model"} instead of no content field)
chunk_empty_content_with_finish = {
"candidates": [
{
"content": {"role": "model"},
"finishReason": "STOP",
"index": 0,
}
],
"usageMetadata": {
"promptTokenCount": 50,
"candidatesTokenCount": 20,
"totalTokenCount": 70,
},
}
# Process chunk 1: tool calls
response1 = iterator.chunk_parser(chunk_with_tool_calls)
assert response1 is not None
assert len(response1.choices) == 1
assert response1.choices[0].delta.tool_calls is not None
assert iterator.has_seen_tool_calls is True
# Process chunk 2: empty content with finishReason
response2 = iterator.chunk_parser(chunk_empty_content_with_finish)
assert response2 is not None
assert len(response2.choices) == 1
assert response2.choices[0].finish_reason == "tool_calls"
assert response2.choices[0].delta.content is None

View file

@ -1,38 +1,104 @@
"""
Tests for Gemini responses where content is present but "parts" is missing.
Gemini can return content: {"role": "model"} with no "parts" field when
finishReason is STOP. This is a known Gemini quirk (e.g. gemini-2.5-flash-lite
in long-running agentic tasks).
LiteLLM must handle this gracefully: return a valid response with
content=None and finish_reason="stop" rather than raising an exception.
Ref: https://github.com/BerriAI/litellm/issues/24442
"""
import pytest
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import VertexGeminiConfig
import litellm
from litellm import ModelResponse
from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import (
VertexGeminiConfig,
)
def test_process_candidates_empty_content_no_parts_returns_valid_response():
"""
When a candidate has content: {"role": "model"} with no "parts",
_process_candidates must not raise and must produce a choice with
content=None and finish_reason="stop".
"""
candidates = [
{
"content": {
"role": "model"
# "parts" intentionally absent — the Gemini quirk under test
},
"finishReason": "STOP",
"index": 0,
}
]
model_response = ModelResponse()
model_response.choices = []
VertexGeminiConfig._process_candidates(
_candidates=candidates,
model_response=model_response,
standard_optional_params={},
cumulative_tool_call_index=0,
)
assert len(model_response.choices) == 1
choice = model_response.choices[0]
assert choice.finish_reason == "stop"
assert choice.message.content is None
def test_process_candidates_empty_content_no_parts_no_finish_reason():
"""
When a candidate has content: {"role": "model"} with no "parts" and
no finishReason, _process_candidates must not raise and must produce
a choice with content=None.
"""
candidates = [
{
"content": {"role": "model"},
"index": 0,
}
]
model_response = ModelResponse()
model_response.choices = []
VertexGeminiConfig._process_candidates(
_candidates=candidates,
model_response=model_response,
standard_optional_params={},
cumulative_tool_call_index=0,
)
assert len(model_response.choices) == 1
choice = model_response.choices[0]
assert choice.message.content is None
def test_process_candidates_unbound_local_error_fix():
# Setup
"""
Regression test: _process_candidates must not raise UnboundLocalError
when content is present but "parts" is missing.
"""
candidates = [
{
"content": {
"role": "model"
# "parts" is missing intentionally to trigger the issue
},
"finishReason": "STOP"
"finishReason": "STOP",
}
]
model_response = ModelResponse()
# Execution
try:
VertexGeminiConfig._process_candidates(
_candidates=candidates,
model_response=model_response,
standard_optional_params={},
cumulative_tool_call_index=0
)
except UnboundLocalError as e:
pytest.fail(f"UnboundLocalError raised: {e}")
except Exception as e:
# Other exceptions might be okay if they are not UnboundLocalError,
# but ideally it should pass without error or raise a specific error if parts are required.
# However, the goal is to verify thought_signatures doesn't crash.
pass
# Verify that we didn't crash with UnboundLocalError
if __name__ == "__main__":
test_process_candidates_unbound_local_error_fix()
print("Test passed!")
# Must not raise UnboundLocalError
VertexGeminiConfig._process_candidates(
_candidates=candidates,
model_response=model_response,
standard_optional_params={},
cumulative_tool_call_index=0,
)