From 1bfe1dd9568c20abd2284188a16c1bb79427536a Mon Sep 17 00:00:00 2001 From: RoyVivat Date: Mon, 23 Mar 2026 16:25:20 -0700 Subject: [PATCH] Handled edge case where gemini-2.5-flash-lite returns content {'role': 'model'} with no parts field when finishReason: STOP. After the existing fallback, added a second pass that corrects finish_reason from "stop" to "tool_calls" on choices that have no delta tool_calls, when has_seen_tool_calls is True. --- .../vertex_and_google_ai_studio_gemini.py | 17 +++ ...emini_streaming_tool_call_finish_reason.py | 107 ++++++++++++++++ .../test_vertex_gemini_unbound_local_error.py | 114 ++++++++++++++---- 3 files changed, 214 insertions(+), 24 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 3f1bccaccfc..63955cec1bf 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -3001,6 +3001,23 @@ class ModelResponseIterator: ) model_response.choices.append(choice) + # Handle the case where Gemini returns content: {"role": "model"} with + # no "parts" (e.g., gemini-2.5-flash-lite with finishReason: STOP and + # empty content). _process_candidates creates a choice for this candidate + # because "content" IS present, but it sets finish_reason="stop" without + # knowing about tool_calls seen in earlier streaming chunks. + # Per the OpenAI spec, finish_reason must be "tool_calls" when the model + # previously called a tool. + if self.has_seen_tool_calls and model_response.choices: + for choice in model_response.choices: + if ( + hasattr(choice, "finish_reason") + and choice.finish_reason == "stop" + and hasattr(choice, "delta") + and (choice.delta is None or not choice.delta.tool_calls) + ): + choice.finish_reason = "tool_calls" + setattr(model_response, "vertex_ai_grounding_metadata", grounding_metadata) # type: ignore setattr(model_response, "vertex_ai_url_context_metadata", url_context_metadata) # type: ignore setattr(model_response, "vertex_ai_safety_ratings", safety_ratings) # type: ignore diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_gemini_streaming_tool_call_finish_reason.py b/tests/test_litellm/llms/vertex_ai/gemini/test_gemini_streaming_tool_call_finish_reason.py index 3f8efd47fa3..df3281b3197 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_gemini_streaming_tool_call_finish_reason.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_gemini_streaming_tool_call_finish_reason.py @@ -230,3 +230,110 @@ def test_streaming_content_filter_finish_reason_preserved(): assert response is not None assert len(response.choices) == 1 assert response.choices[0].finish_reason == "content_filter" + + +def test_streaming_empty_content_no_parts_finish_reason_is_stop(): + """ + When Gemini returns content: {"role": "model"} with no "parts" and + finishReason: STOP (a known Gemini quirk, e.g. gemini-2.5-flash-lite), + the finish_reason should be "stop" (no prior tool calls). + + Ref: https://github.com/BerriAI/litellm/issues/24442 + """ + logging_obj = _make_logging_obj() + iterator = ModelResponseIterator( + streaming_response=iter([]), + sync_stream=True, + logging_obj=logging_obj, + ) + + # Final chunk: content present but no "parts", finishReason="STOP" + chunk_empty_content = { + "candidates": [ + { + "content": {"role": "model"}, + "finishReason": "STOP", + "index": 0, + } + ], + "usageMetadata": { + "promptTokenCount": 3130, + "totalTokenCount": 3130, + }, + } + + response = iterator.chunk_parser(chunk_empty_content) + assert response is not None + assert len(response.choices) == 1 + assert response.choices[0].finish_reason == "stop" + assert response.choices[0].delta.content is None + + +def test_streaming_tool_calls_then_empty_content_finish_reason_is_tool_calls(): + """ + When Gemini streams tool calls in one chunk and then returns a final chunk + with content: {"role": "model"} (no "parts") and finishReason: STOP, + the finish_reason must be "tool_calls" per the OpenAI spec. + + This tests the case where Gemini returns empty content (no parts) instead of + the usual empty candidate (no "content" field) as the final chunk. + + Ref: https://github.com/BerriAI/litellm/issues/24442 + """ + logging_obj = _make_logging_obj() + iterator = ModelResponseIterator( + streaming_response=iter([]), + sync_stream=True, + logging_obj=logging_obj, + ) + + # Chunk 1: tool call with no finishReason + chunk_with_tool_calls = { + "candidates": [ + { + "content": { + "parts": [ + { + "functionCall": { + "name": "get_current_weather", + "args": {"location": "Boston, MA"}, + } + } + ], + "role": "model", + }, + "index": 0, + } + ], + } + + # Chunk 2: finishReason="STOP" with content present but no "parts" + # (Gemini quirk: content: {"role": "model"} instead of no content field) + chunk_empty_content_with_finish = { + "candidates": [ + { + "content": {"role": "model"}, + "finishReason": "STOP", + "index": 0, + } + ], + "usageMetadata": { + "promptTokenCount": 50, + "candidatesTokenCount": 20, + "totalTokenCount": 70, + }, + } + + # Process chunk 1: tool calls + response1 = iterator.chunk_parser(chunk_with_tool_calls) + assert response1 is not None + assert len(response1.choices) == 1 + assert response1.choices[0].delta.tool_calls is not None + assert iterator.has_seen_tool_calls is True + + # Process chunk 2: empty content with finishReason + response2 = iterator.chunk_parser(chunk_empty_content_with_finish) + assert response2 is not None + assert len(response2.choices) == 1 + assert response2.choices[0].finish_reason == "tool_calls" + assert response2.choices[0].delta.content is None diff --git a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_gemini_unbound_local_error.py b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_gemini_unbound_local_error.py index 0a1ac7e2a54..23e16871c96 100644 --- a/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_gemini_unbound_local_error.py +++ b/tests/test_litellm/llms/vertex_ai/gemini/test_vertex_gemini_unbound_local_error.py @@ -1,38 +1,104 @@ +""" +Tests for Gemini responses where content is present but "parts" is missing. + +Gemini can return content: {"role": "model"} with no "parts" field when +finishReason is STOP. This is a known Gemini quirk (e.g. gemini-2.5-flash-lite +in long-running agentic tasks). + +LiteLLM must handle this gracefully: return a valid response with +content=None and finish_reason="stop" rather than raising an exception. + +Ref: https://github.com/BerriAI/litellm/issues/24442 +""" + import pytest -from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import VertexGeminiConfig + +import litellm from litellm import ModelResponse +from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + VertexGeminiConfig, +) + + +def test_process_candidates_empty_content_no_parts_returns_valid_response(): + """ + When a candidate has content: {"role": "model"} with no "parts", + _process_candidates must not raise and must produce a choice with + content=None and finish_reason="stop". + """ + candidates = [ + { + "content": { + "role": "model" + # "parts" intentionally absent — the Gemini quirk under test + }, + "finishReason": "STOP", + "index": 0, + } + ] + model_response = ModelResponse() + model_response.choices = [] + + VertexGeminiConfig._process_candidates( + _candidates=candidates, + model_response=model_response, + standard_optional_params={}, + cumulative_tool_call_index=0, + ) + + assert len(model_response.choices) == 1 + choice = model_response.choices[0] + assert choice.finish_reason == "stop" + assert choice.message.content is None + + +def test_process_candidates_empty_content_no_parts_no_finish_reason(): + """ + When a candidate has content: {"role": "model"} with no "parts" and + no finishReason, _process_candidates must not raise and must produce + a choice with content=None. + """ + candidates = [ + { + "content": {"role": "model"}, + "index": 0, + } + ] + model_response = ModelResponse() + model_response.choices = [] + + VertexGeminiConfig._process_candidates( + _candidates=candidates, + model_response=model_response, + standard_optional_params={}, + cumulative_tool_call_index=0, + ) + + assert len(model_response.choices) == 1 + choice = model_response.choices[0] + assert choice.message.content is None + def test_process_candidates_unbound_local_error_fix(): - # Setup + """ + Regression test: _process_candidates must not raise UnboundLocalError + when content is present but "parts" is missing. + """ candidates = [ { "content": { "role": "model" # "parts" is missing intentionally to trigger the issue }, - "finishReason": "STOP" + "finishReason": "STOP", } ] model_response = ModelResponse() - - # Execution - try: - VertexGeminiConfig._process_candidates( - _candidates=candidates, - model_response=model_response, - standard_optional_params={}, - cumulative_tool_call_index=0 - ) - except UnboundLocalError as e: - pytest.fail(f"UnboundLocalError raised: {e}") - except Exception as e: - # Other exceptions might be okay if they are not UnboundLocalError, - # but ideally it should pass without error or raise a specific error if parts are required. - # However, the goal is to verify thought_signatures doesn't crash. - pass - # Verify that we didn't crash with UnboundLocalError - -if __name__ == "__main__": - test_process_candidates_unbound_local_error_fix() - print("Test passed!") + # Must not raise UnboundLocalError + VertexGeminiConfig._process_candidates( + _candidates=candidates, + model_response=model_response, + standard_optional_params={}, + cumulative_tool_call_index=0, + )