mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(vertex_and_google_ai_studio_gemini.py): bubble up stream is not finished, even if stop reason is given
prevents early completion of stream Closes https://github.com/BerriAI/litellm/issues/11549
This commit is contained in:
parent
49a7833861
commit
03d6546b44
4 changed files with 28 additions and 8 deletions
|
|
@ -439,7 +439,13 @@ class CustomStreamWrapper:
|
|||
else: # function/tool calling chunk - when content is None. in this case we just return the original chunk from openai
|
||||
pass
|
||||
if str_line.choices[0].finish_reason:
|
||||
is_finished = True
|
||||
is_finished = (
|
||||
True # check if str_line._hidden_params["is_finished"] is True
|
||||
)
|
||||
if hasattr(
|
||||
str_line, "_hidden_params"
|
||||
) and str_line._hidden_params.get("is_finished"):
|
||||
is_finished = str_line._hidden_params.get("is_finished")
|
||||
finish_reason = str_line.choices[0].finish_reason
|
||||
|
||||
# checking for logprobs
|
||||
|
|
|
|||
|
|
@ -1886,6 +1886,8 @@ class ModelResponseIterator:
|
|||
).web_search_requests = web_search_requests
|
||||
|
||||
setattr(model_response, "usage", usage) # type: ignore
|
||||
|
||||
model_response._hidden_params["is_finished"] = False
|
||||
return model_response
|
||||
|
||||
except json.JSONDecodeError:
|
||||
|
|
|
|||
|
|
@ -1,9 +1,4 @@
|
|||
model_list:
|
||||
- model_name: "gemini-2.0-flash"
|
||||
litellm_params:
|
||||
model: vertex_ai/gemini-2.0-flash
|
||||
vertex_project: my-project-id
|
||||
vertex_location: us-central1
|
||||
- model_name: "gpt-4o-mini-openai"
|
||||
litellm_params:
|
||||
model: gpt-4o-mini
|
||||
|
|
@ -42,9 +37,9 @@ model_list:
|
|||
litellm_params:
|
||||
model: text-embedding-ada-002
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
- model_name: gemini/gemini-2.0-flash
|
||||
- model_name: gemini/*
|
||||
litellm_params:
|
||||
model: gemini/gemini-2.0-flash
|
||||
model: gemini/*
|
||||
- model_name: llama-qwen
|
||||
litellm_params:
|
||||
model: ollama/qwen2:0.5b
|
||||
|
|
|
|||
|
|
@ -3942,3 +3942,20 @@ def test_vertex_ai_streaming_response_id():
|
|||
iterator = iter(iterator)
|
||||
first_chunk = next(iterator)
|
||||
assert first_chunk.id == "vertex_ai_response_stream_123"
|
||||
|
||||
|
||||
def test_vertex_ai_gemini_2_5_pro_streaming():
|
||||
load_vertex_ai_credentials()
|
||||
litellm._turn_on_debug()
|
||||
response = completion(
|
||||
model="vertex_ai/gemini-2.5-pro-preview-06-05",
|
||||
messages=[{"role": "user", "content": "Hi!"}],
|
||||
vertex_location="global",
|
||||
stream=True,
|
||||
)
|
||||
has_real_content = False
|
||||
for chunk in response:
|
||||
print(chunk)
|
||||
if chunk.choices[0].delta.content is not None and len(chunk.choices[0].delta.content) > 0:
|
||||
has_real_content = True
|
||||
assert has_real_content
|
||||
Loading…
Add table
Reference in a new issue