From 03d6546b44470c6d1f79ac6d1a2b03ad00c6214f Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Mon, 9 Jun 2025 18:15:09 -0700 Subject: [PATCH] fix(vertex_and_google_ai_studio_gemini.py): bubble up stream is not finished, even if stop reason is given prevents early completion of stream Closes https://github.com/BerriAI/litellm/issues/11549 --- litellm/litellm_core_utils/streaming_handler.py | 8 +++++++- .../vertex_and_google_ai_studio_gemini.py | 2 ++ litellm/proxy/_new_secret_config.yaml | 9 ++------- .../test_amazing_vertex_completion.py | 17 +++++++++++++++++ 4 files changed, 28 insertions(+), 8 deletions(-) diff --git a/litellm/litellm_core_utils/streaming_handler.py b/litellm/litellm_core_utils/streaming_handler.py index 25799dc2dc5..0b019cf5300 100644 --- a/litellm/litellm_core_utils/streaming_handler.py +++ b/litellm/litellm_core_utils/streaming_handler.py @@ -439,7 +439,13 @@ class CustomStreamWrapper: else: # function/tool calling chunk - when content is None. in this case we just return the original chunk from openai pass if str_line.choices[0].finish_reason: - is_finished = True + is_finished = ( + True # check if str_line._hidden_params["is_finished"] is True + ) + if hasattr( + str_line, "_hidden_params" + ) and str_line._hidden_params.get("is_finished"): + is_finished = str_line._hidden_params.get("is_finished") finish_reason = str_line.choices[0].finish_reason # checking for logprobs diff --git a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py index 495597a3812..5dc89ba11da 100644 --- a/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py +++ b/litellm/llms/vertex_ai/gemini/vertex_and_google_ai_studio_gemini.py @@ -1886,6 +1886,8 @@ class ModelResponseIterator: ).web_search_requests = web_search_requests setattr(model_response, "usage", usage) # type: ignore + + model_response._hidden_params["is_finished"] = False return model_response except json.JSONDecodeError: diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index 2f605bc76c3..65b07b5f5f3 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -1,9 +1,4 @@ model_list: - - model_name: "gemini-2.0-flash" - litellm_params: - model: vertex_ai/gemini-2.0-flash - vertex_project: my-project-id - vertex_location: us-central1 - model_name: "gpt-4o-mini-openai" litellm_params: model: gpt-4o-mini @@ -42,9 +37,9 @@ model_list: litellm_params: model: text-embedding-ada-002 api_key: os.environ/OPENAI_API_KEY - - model_name: gemini/gemini-2.0-flash + - model_name: gemini/* litellm_params: - model: gemini/gemini-2.0-flash + model: gemini/* - model_name: llama-qwen litellm_params: model: ollama/qwen2:0.5b diff --git a/tests/local_testing/test_amazing_vertex_completion.py b/tests/local_testing/test_amazing_vertex_completion.py index 729f4dc5c51..95bf7e01a95 100644 --- a/tests/local_testing/test_amazing_vertex_completion.py +++ b/tests/local_testing/test_amazing_vertex_completion.py @@ -3942,3 +3942,20 @@ def test_vertex_ai_streaming_response_id(): iterator = iter(iterator) first_chunk = next(iterator) assert first_chunk.id == "vertex_ai_response_stream_123" + + +def test_vertex_ai_gemini_2_5_pro_streaming(): + load_vertex_ai_credentials() + litellm._turn_on_debug() + response = completion( + model="vertex_ai/gemini-2.5-pro-preview-06-05", + messages=[{"role": "user", "content": "Hi!"}], + vertex_location="global", + stream=True, + ) + has_real_content = False + for chunk in response: + print(chunk) + if chunk.choices[0].delta.content is not None and len(chunk.choices[0].delta.content) > 0: + has_real_content = True + assert has_real_content \ No newline at end of file