fix(vertex_and_google_ai_studio_gemini.py): bubble up stream is not finished, even if stop reason is given

prevents early completion of stream

Closes https://github.com/BerriAI/litellm/issues/11549
This commit is contained in:
Krrish Dholakia 2025-06-09 18:15:09 -07:00
parent 49a7833861
commit 03d6546b44
4 changed files with 28 additions and 8 deletions

View file

@ -439,7 +439,13 @@ class CustomStreamWrapper:
else: # function/tool calling chunk - when content is None. in this case we just return the original chunk from openai
pass
if str_line.choices[0].finish_reason:
is_finished = True
is_finished = (
True # check if str_line._hidden_params["is_finished"] is True
)
if hasattr(
str_line, "_hidden_params"
) and str_line._hidden_params.get("is_finished"):
is_finished = str_line._hidden_params.get("is_finished")
finish_reason = str_line.choices[0].finish_reason
# checking for logprobs

View file

@ -1886,6 +1886,8 @@ class ModelResponseIterator:
).web_search_requests = web_search_requests
setattr(model_response, "usage", usage) # type: ignore
model_response._hidden_params["is_finished"] = False
return model_response
except json.JSONDecodeError:

View file

@ -1,9 +1,4 @@
model_list:
- model_name: "gemini-2.0-flash"
litellm_params:
model: vertex_ai/gemini-2.0-flash
vertex_project: my-project-id
vertex_location: us-central1
- model_name: "gpt-4o-mini-openai"
litellm_params:
model: gpt-4o-mini
@ -42,9 +37,9 @@ model_list:
litellm_params:
model: text-embedding-ada-002
api_key: os.environ/OPENAI_API_KEY
- model_name: gemini/gemini-2.0-flash
- model_name: gemini/*
litellm_params:
model: gemini/gemini-2.0-flash
model: gemini/*
- model_name: llama-qwen
litellm_params:
model: ollama/qwen2:0.5b

View file

@ -3942,3 +3942,20 @@ def test_vertex_ai_streaming_response_id():
iterator = iter(iterator)
first_chunk = next(iterator)
assert first_chunk.id == "vertex_ai_response_stream_123"
def test_vertex_ai_gemini_2_5_pro_streaming():
load_vertex_ai_credentials()
litellm._turn_on_debug()
response = completion(
model="vertex_ai/gemini-2.5-pro-preview-06-05",
messages=[{"role": "user", "content": "Hi!"}],
vertex_location="global",
stream=True,
)
has_real_content = False
for chunk in response:
print(chunk)
if chunk.choices[0].delta.content is not None and len(chunk.choices[0].delta.content) > 0:
has_real_content = True
assert has_real_content