From 7a4c6344ea19e0df44d630d65ddf8d9c2f232521 Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Fri, 10 Oct 2025 11:22:15 -0700 Subject: [PATCH] fix gemma: stream param --- .../vertex_gemma_models/transformation.py | 25 +++++++++++-------- .../test_vertex_gemma_transformation.py | 3 +++ 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/litellm/llms/vertex_ai/vertex_gemma_models/transformation.py b/litellm/llms/vertex_ai/vertex_gemma_models/transformation.py index 779eb69c760..ece1cef37fc 100644 --- a/litellm/llms/vertex_ai/vertex_gemma_models/transformation.py +++ b/litellm/llms/vertex_ai/vertex_gemma_models/transformation.py @@ -30,6 +30,18 @@ class VertexGemmaConfig(OpenAIGPTConfig): def __init__(self) -> None: super().__init__() + def should_fake_stream( + self, + model: Optional[str], + stream: Optional[bool], + custom_llm_provider: Optional[str] = None, + ) -> bool: + """ + Vertex AI Gemma models do not support streaming. + Return True to enable fake streaming on the client side. + """ + return True + def transform_request( self, model: str, @@ -53,8 +65,9 @@ class VertexGemmaConfig(OpenAIGPTConfig): headers=headers, ) - # Remove 'model' from the request as it's not needed in the instance + # Remove params not needed/supported by Vertex Gemma openai_request.pop("model", None) + openai_request.pop("stream", None) # Streaming not supported, will be faked client-side # Wrap in Vertex Gemma format return { @@ -105,16 +118,8 @@ class VertexGemmaConfig(OpenAIGPTConfig): ): """ Make completion request to Vertex Gemma endpoint. - Supports both sync and async requests. + Supports both sync and async requests with fake streaming. """ - # Handle streaming - stream = optional_params.get("stream", False) - if stream: - raise BaseLLMException( - status_code=400, - message="Streaming is not yet supported for Vertex AI Gemma models", - ) - if acompletion: return self._async_completion( model=model, diff --git a/tests/test_litellm/llms/vertex_ai/vertex_gemma_models/test_vertex_gemma_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_gemma_models/test_vertex_gemma_transformation.py index fb988485791..f40024665de 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_gemma_models/test_vertex_gemma_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_gemma_models/test_vertex_gemma_transformation.py @@ -155,6 +155,9 @@ class TestVertexGemmaCompletion: assert instance["messages"][0]["content"] == "What is machine learning?" assert instance["max_tokens"] == 100 + # Verify stream parameter is NOT sent to Vertex (will be faked client-side) + assert "stream" not in instance + # Validate LiteLLM Response (OpenAI format) assert response.id == "chatcmpl-aaa4288f-2b8e-4bc0-8b14-4e444decd2c4" assert response.object == "chat.completion"