From 874e9bb67f92c6eec6411da5ea94808b0c3fdf6c Mon Sep 17 00:00:00 2001 From: basil-k-aji-dev <70605804+basil-k-aji-dev@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:15:23 +0530 Subject: [PATCH 1/4] fix(gemini): keep grounding citations on streamed responses The conversion from groundingMetadata to OpenAI annotations sat inside the branch that handles a candidate's content parts, so it only ran when the same chunk also carried text. Gemini does not promise that. In a streamed response the grounding commonly arrives on the final candidate, which has a finishReason and no parts at all, and those citations were dropped That is why the same prompt returned a Sources card on some runs and not others, while non-streaming returned one every time: non-streaming has a single candidate and it always has parts The conversion now runs for every candidate that reaches it, content or not. Nothing else moves, and the grounding was already being collected here for web-search request counting, so cost accounting is unchanged Fixes #41492 --- .../vertex_and_google_ai_studio_gemini.py | 24 ++- ...test_vertex_and_google_ai_studio_gemini.py | 145 ++++++++++++++++++ 2 files changed, 162 insertions(+), 7 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index b5f32d57061..b70f4295d3b 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -2217,6 +2217,7 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): continue image_response: list[ImageURLListItem] | None = None + content: str | None = None chat_completion_message: ChatCompletionResponseMessage = {"role": "assistant"} chat_completion_logprobs: ChoiceLogprobs | None = None tools: list[ChatCompletionToolCallChunk] | None = None @@ -2278,13 +2279,6 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): if reasoning_content is not None: chat_completion_message["reasoning_content"] = reasoning_content - if candidate_grounding_metadata: - annotations = VertexGeminiConfig._convert_grounding_metadata_to_annotations( - grounding_metadata=candidate_grounding_metadata, - content_text=content, - ) - if annotations: - chat_completion_message["annotations"] = annotations ( functions, tools, @@ -2295,6 +2289,22 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): is_function_call=is_function_call(standard_optional_params), ) + # Outside the "content"/"parts" branch on purpose. Gemini does not + # promise to put groundingMetadata on a chunk that also carries text: + # streaming commonly delivers it on the final candidate, which has a + # finishReason and no parts at all. Converting only when parts were + # present dropped the citations for exactly those responses, which is + # why the same prompt produced annotations on some runs and not + # others while non-streaming -- one candidate, always with parts -- + # produced them every time. + if candidate_grounding_metadata: + annotations = VertexGeminiConfig._convert_grounding_metadata_to_annotations( + grounding_metadata=candidate_grounding_metadata, + content_text=content, + ) + if annotations: + chat_completion_message["annotations"] = annotations + if "logprobsResult" in candidate: chat_completion_logprobs = VertexGeminiConfig._transform_logprobs( logprobs_result=candidate["logprobsResult"] diff --git a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index fd735afb16e..393732bb12a 100644 --- a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -6356,3 +6356,148 @@ def test_gemini_multi_candidate_messages_do_not_share_state(): assert resp.choices[1].message.tool_calls is None assert getattr(resp.choices[1].message, "reasoning_content", None) is None assert resp.choices[1].provider_specific_fields["native_finish_reason"] == "STOP" + + +_GROUNDING_METADATA: Final = { + "webSearchQueries": ["current price of gold"], + "groundingChunks": [ + {"web": {"uri": "https://vertexaisearch.cloud.google.com/grounding-api-redirect/AbC123", "title": "kitco"}}, + {"web": {"uri": "https://vertexaisearch.cloud.google.com/grounding-api-redirect/DeF456", "title": "reuters"}}, + ], + "groundingSupports": [ + {"segment": {"startIndex": 0, "endIndex": 42}, "groundingChunkIndices": [0]}, + {"segment": {"startIndex": 43, "endIndex": 80}, "groundingChunkIndices": [1]}, + ], +} +_GROUNDED_TEXT: Final = "As of today spot gold trades near $4,270" +_GROUNDING_URIS: Final = [ + "https://vertexaisearch.cloud.google.com/grounding-api-redirect/AbC123", + "https://vertexaisearch.cloud.google.com/grounding-api-redirect/DeF456", +] + + +def _gemini_stream_iterator(): + from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ModelResponseIterator + + logging_obj = MagicMock() + logging_obj.optional_params = {} + return ModelResponseIterator(streaming_response=iter([]), sync_stream=True, logging_obj=logging_obj) + + +def _delta_annotations(model_response): + found: list = [] + for choice in model_response.choices: + annotations = getattr(getattr(choice, "delta", None), "annotations", None) + if annotations: + found.extend(annotations) + return found + + +def _citation_urls(annotations): + return [annotation["url_citation"]["url"] for annotation in annotations] + + +def test_streaming_grounding_on_the_final_chunk_produces_annotations(): + """Gemini often sends groundingMetadata on a candidate with finishReason and no parts""" + iterator = _gemini_stream_iterator() + iterator.chunk_parser({"candidates": [{"index": 0, "content": {"role": "model", "parts": [{"text": _GROUNDED_TEXT}]}}]}) + + final = iterator.chunk_parser( + {"candidates": [{"index": 0, "finishReason": "STOP", "groundingMetadata": _GROUNDING_METADATA}]} + ) + + assert _citation_urls(_delta_annotations(final)) == _GROUNDING_URIS + + +def test_streaming_grounding_alongside_text_produces_annotations(): + """The shape that already worked keeps working""" + iterator = _gemini_stream_iterator() + + chunk = iterator.chunk_parser( + { + "candidates": [ + { + "index": 0, + "content": {"role": "model", "parts": [{"text": _GROUNDED_TEXT}]}, + "groundingMetadata": _GROUNDING_METADATA, + "finishReason": "STOP", + } + ] + } + ) + + annotations = _delta_annotations(chunk) + assert _citation_urls(annotations) == _GROUNDING_URIS + assert annotations[0]["type"] == "url_citation" + assert annotations[0]["url_citation"]["start_index"] == 0 + + +def test_streaming_without_grounding_carries_no_annotations(): + iterator = _gemini_stream_iterator() + + chunk = iterator.chunk_parser( + {"candidates": [{"index": 0, "content": {"role": "model", "parts": [{"text": "Paris."}]}, "finishReason": "STOP"}]} + ) + + assert _delta_annotations(chunk) == [] + + +def test_streaming_grounding_without_web_chunks_carries_no_annotations(): + """Maps grounding has no web URI to cite, so it must not invent one""" + iterator = _gemini_stream_iterator() + + chunk = iterator.chunk_parser( + { + "candidates": [ + { + "index": 0, + "finishReason": "STOP", + "groundingMetadata": { + "groundingChunks": [{"maps": {"placeId": "abc"}}], + "groundingSupports": [ + {"segment": {"startIndex": 0, "endIndex": 5}, "groundingChunkIndices": [0]} + ], + }, + } + ] + } + ) + + assert _delta_annotations(chunk) == [] + + +def test_non_streaming_grounding_annotations_are_unchanged(): + model_response = ModelResponse() + VertexGeminiConfig._process_candidates( + [ + { + "index": 0, + "content": {"role": "model", "parts": [{"text": _GROUNDED_TEXT}]}, + "groundingMetadata": _GROUNDING_METADATA, + "finishReason": "STOP", + } + ], + model_response, + {}, + ) + + annotations = getattr(model_response.choices[-1].message, "annotations", None) + assert _citation_urls(annotations or []) == _GROUNDING_URIS + + +def test_streamed_grounding_survives_reassembly(): + """A client rebuilding the stream ends up with the citations a non-streaming call returns""" + iterator = _gemini_stream_iterator() + chunks = [ + iterator.chunk_parser( + {"candidates": [{"index": 0, "content": {"role": "model", "parts": [{"text": _GROUNDED_TEXT}]}}]} + ), + iterator.chunk_parser( + {"candidates": [{"index": 0, "finishReason": "STOP", "groundingMetadata": _GROUNDING_METADATA}]} + ), + ] + + rebuilt = litellm.stream_chunk_builder(chunks=[chunk for chunk in chunks if chunk is not None]) + + assert rebuilt is not None + assert _citation_urls(getattr(rebuilt.choices[0].message, "annotations", None) or []) == _GROUNDING_URIS From 75fe8989087897cf3b65b61f05a45f65e63cf6a2 Mon Sep 17 00:00:00 2001 From: basil-k-aji-dev <70605804+basil-k-aji-dev@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:27:49 +0530 Subject: [PATCH 2/4] style(gemini): drop the narrative comment and type the test helpers Both from the Greptile review: AGENTS.md limits comments to complex logic, tool inputs and TODOs, and requires fully typed code. The reasoning that was in the comment lives in the PR description and the commit message instead --- .../vertex_and_google_ai_studio_gemini.py | 8 ------- ...test_vertex_and_google_ai_studio_gemini.py | 24 ++++++++----------- 2 files changed, 10 insertions(+), 22 deletions(-) diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index b70f4295d3b..7ee32260fe0 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -2289,14 +2289,6 @@ class VertexGeminiConfig(VertexAIBaseConfig, BaseConfig): is_function_call=is_function_call(standard_optional_params), ) - # Outside the "content"/"parts" branch on purpose. Gemini does not - # promise to put groundingMetadata on a chunk that also carries text: - # streaming commonly delivers it on the final candidate, which has a - # finishReason and no parts at all. Converting only when parts were - # present dropped the citations for exactly those responses, which is - # why the same prompt produced annotations on some runs and not - # others while non-streaming -- one candidate, always with parts -- - # produced them every time. if candidate_grounding_metadata: annotations = VertexGeminiConfig._convert_grounding_metadata_to_annotations( grounding_metadata=candidate_grounding_metadata, diff --git a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 393732bb12a..86c0ab49414 100644 --- a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -6376,7 +6376,7 @@ _GROUNDING_URIS: Final = [ ] -def _gemini_stream_iterator(): +def _gemini_stream_iterator() -> "ModelResponseIterator": from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ModelResponseIterator logging_obj = MagicMock() @@ -6384,8 +6384,8 @@ def _gemini_stream_iterator(): return ModelResponseIterator(streaming_response=iter([]), sync_stream=True, logging_obj=logging_obj) -def _delta_annotations(model_response): - found: list = [] +def _delta_annotations(model_response: object) -> List[dict]: + found: List[dict] = [] for choice in model_response.choices: annotations = getattr(getattr(choice, "delta", None), "annotations", None) if annotations: @@ -6393,12 +6393,11 @@ def _delta_annotations(model_response): return found -def _citation_urls(annotations): +def _citation_urls(annotations: List[dict]) -> List[str]: return [annotation["url_citation"]["url"] for annotation in annotations] -def test_streaming_grounding_on_the_final_chunk_produces_annotations(): - """Gemini often sends groundingMetadata on a candidate with finishReason and no parts""" +def test_streaming_grounding_on_the_final_chunk_produces_annotations() -> None: iterator = _gemini_stream_iterator() iterator.chunk_parser({"candidates": [{"index": 0, "content": {"role": "model", "parts": [{"text": _GROUNDED_TEXT}]}}]}) @@ -6409,8 +6408,7 @@ def test_streaming_grounding_on_the_final_chunk_produces_annotations(): assert _citation_urls(_delta_annotations(final)) == _GROUNDING_URIS -def test_streaming_grounding_alongside_text_produces_annotations(): - """The shape that already worked keeps working""" +def test_streaming_grounding_alongside_text_produces_annotations() -> None: iterator = _gemini_stream_iterator() chunk = iterator.chunk_parser( @@ -6432,7 +6430,7 @@ def test_streaming_grounding_alongside_text_produces_annotations(): assert annotations[0]["url_citation"]["start_index"] == 0 -def test_streaming_without_grounding_carries_no_annotations(): +def test_streaming_without_grounding_carries_no_annotations() -> None: iterator = _gemini_stream_iterator() chunk = iterator.chunk_parser( @@ -6442,8 +6440,7 @@ def test_streaming_without_grounding_carries_no_annotations(): assert _delta_annotations(chunk) == [] -def test_streaming_grounding_without_web_chunks_carries_no_annotations(): - """Maps grounding has no web URI to cite, so it must not invent one""" +def test_streaming_grounding_without_web_chunks_carries_no_annotations() -> None: iterator = _gemini_stream_iterator() chunk = iterator.chunk_parser( @@ -6466,7 +6463,7 @@ def test_streaming_grounding_without_web_chunks_carries_no_annotations(): assert _delta_annotations(chunk) == [] -def test_non_streaming_grounding_annotations_are_unchanged(): +def test_non_streaming_grounding_annotations_are_unchanged() -> None: model_response = ModelResponse() VertexGeminiConfig._process_candidates( [ @@ -6485,8 +6482,7 @@ def test_non_streaming_grounding_annotations_are_unchanged(): assert _citation_urls(annotations or []) == _GROUNDING_URIS -def test_streamed_grounding_survives_reassembly(): - """A client rebuilding the stream ends up with the citations a non-streaming call returns""" +def test_streamed_grounding_survives_reassembly() -> None: iterator = _gemini_stream_iterator() chunks = [ iterator.chunk_parser( From a27c4a67a5c698521aa546e2238d0da0bee4e340 Mon Sep 17 00:00:00 2001 From: basil-k-aji-dev <70605804+basil-k-aji-dev@users.noreply.github.com> Date: Sun, 20 Sep 2026 17:58:08 +0530 Subject: [PATCH 3/4] style(gemini): build the annotation helpers without mutation AGENTS.md asks new code to build values in one shot rather than seeding an empty collection and mutating it. From the Greptile review --- ...test_vertex_and_google_ai_studio_gemini.py | 29 +++++++++---------- 1 file changed, 14 insertions(+), 15 deletions(-) diff --git a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 86c0ab49414..70ce0dc24b0 100644 --- a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -6370,10 +6370,10 @@ _GROUNDING_METADATA: Final = { ], } _GROUNDED_TEXT: Final = "As of today spot gold trades near $4,270" -_GROUNDING_URIS: Final = [ +_GROUNDING_URIS: Final = ( "https://vertexaisearch.cloud.google.com/grounding-api-redirect/AbC123", "https://vertexaisearch.cloud.google.com/grounding-api-redirect/DeF456", -] +) def _gemini_stream_iterator() -> "ModelResponseIterator": @@ -6384,17 +6384,16 @@ def _gemini_stream_iterator() -> "ModelResponseIterator": return ModelResponseIterator(streaming_response=iter([]), sync_stream=True, logging_obj=logging_obj) -def _delta_annotations(model_response: object) -> List[dict]: - found: List[dict] = [] - for choice in model_response.choices: - annotations = getattr(getattr(choice, "delta", None), "annotations", None) - if annotations: - found.extend(annotations) - return found +def _delta_annotations(model_response: object) -> tuple[dict, ...]: + return tuple( + annotation + for choice in getattr(model_response, "choices", ()) + for annotation in (getattr(getattr(choice, "delta", None), "annotations", None) or ()) + ) -def _citation_urls(annotations: List[dict]) -> List[str]: - return [annotation["url_citation"]["url"] for annotation in annotations] +def _citation_urls(annotations: tuple[dict, ...]) -> tuple[str, ...]: + return tuple(annotation["url_citation"]["url"] for annotation in annotations) def test_streaming_grounding_on_the_final_chunk_produces_annotations() -> None: @@ -6437,7 +6436,7 @@ def test_streaming_without_grounding_carries_no_annotations() -> None: {"candidates": [{"index": 0, "content": {"role": "model", "parts": [{"text": "Paris."}]}, "finishReason": "STOP"}]} ) - assert _delta_annotations(chunk) == [] + assert _delta_annotations(chunk) == () def test_streaming_grounding_without_web_chunks_carries_no_annotations() -> None: @@ -6460,7 +6459,7 @@ def test_streaming_grounding_without_web_chunks_carries_no_annotations() -> None } ) - assert _delta_annotations(chunk) == [] + assert _delta_annotations(chunk) == () def test_non_streaming_grounding_annotations_are_unchanged() -> None: @@ -6479,7 +6478,7 @@ def test_non_streaming_grounding_annotations_are_unchanged() -> None: ) annotations = getattr(model_response.choices[-1].message, "annotations", None) - assert _citation_urls(annotations or []) == _GROUNDING_URIS + assert _citation_urls(tuple(annotations or ())) == _GROUNDING_URIS def test_streamed_grounding_survives_reassembly() -> None: @@ -6496,4 +6495,4 @@ def test_streamed_grounding_survives_reassembly() -> None: rebuilt = litellm.stream_chunk_builder(chunks=[chunk for chunk in chunks if chunk is not None]) assert rebuilt is not None - assert _citation_urls(getattr(rebuilt.choices[0].message, "annotations", None) or []) == _GROUNDING_URIS + assert _citation_urls(tuple(getattr(rebuilt.choices[0].message, "annotations", None) or ())) == _GROUNDING_URIS From fb9f3ee24b9e8c4d8fc96dcd8b12f75bee727549 Mon Sep 17 00:00:00 2001 From: basil-k-aji-dev <70605804+basil-k-aji-dev@users.noreply.github.com> Date: Tue, 22 Sep 2026 20:51:10 +0530 Subject: [PATCH 4/4] fix(test): hoist the ModelResponseIterator import so ruff can resolve it F821: the annotation referenced a name imported inside the function, so it was undefined at module scope where ruff evaluates it. --- .../gemini/test_vertex_and_google_ai_studio_gemini.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py index 70ce0dc24b0..c01353a3568 100644 --- a/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py +++ b/tests/unit/llms/vertex_ai/gemini/test_vertex_and_google_ai_studio_gemini.py @@ -16,6 +16,7 @@ from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from litellm.llms.gemini.chat.transformation import GoogleAIStudioGeminiConfig from litellm.llms.vertex_ai.common_utils import VertexAIError from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ( + ModelResponseIterator, VertexGeminiConfig, ) from litellm.types.llms.vertex_ai import GeminiFinishReason, UsageMetadata @@ -6376,9 +6377,7 @@ _GROUNDING_URIS: Final = ( ) -def _gemini_stream_iterator() -> "ModelResponseIterator": - from litellm.llms.vertex_ai.gemini.vertex_and_google_ai_studio_gemini import ModelResponseIterator - +def _gemini_stream_iterator() -> ModelResponseIterator: logging_obj = MagicMock() logging_obj.optional_params = {} return ModelResponseIterator(streaming_response=iter([]), sync_stream=True, logging_obj=logging_obj)