From f8334dfeda51d16d03193ab834406d6cdb491ba6 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 14 Feb 2026 09:18:29 -0800 Subject: [PATCH] ci/cd fixes - streaming role & bedrock model cost (#21200) * fix(model_cost): add missing supports_system_messages and supports_tool_choice to bedrock/moonshotai.kimi-k2.5 * fix(streaming): ensure role=assistant is set on first streaming chunk via strip_role_from_delta * fix(vertex_ai): ensure role=assistant on first streaming chunk for Llama models Add VertexAILlama3StreamingHandler that injects role='assistant' into the first streaming chunk delta when the Vertex AI Llama API omits it. --- .../llama3/transformation.py | 41 ++++++++- ...odel_prices_and_context_window_backup.json | 2 + model_prices_and_context_window.json | 2 + ...ai_partner_models_llama3_transformation.py | 92 ++++++++++++++++++- 4 files changed, 133 insertions(+), 4 deletions(-) diff --git a/litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/transformation.py b/litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/transformation.py index 748a5f5fb40..a70e16000b6 100644 --- a/litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/transformation.py +++ b/litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/transformation.py @@ -1,12 +1,15 @@ import types -from typing import Any, List, Optional +from typing import Any, AsyncIterator, Iterator, List, Optional, Union import httpx from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj -from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig +from litellm.llms.openai.chat.gpt_transformation import ( + OpenAIChatCompletionStreamingHandler, + OpenAIGPTConfig, +) from litellm.types.llms.openai import AllMessageValues, OpenAIChatCompletionResponse -from litellm.types.utils import ModelResponse, Usage +from litellm.types.utils import ModelResponse, ModelResponseStream, Usage from ...common_utils import VertexAIError @@ -79,6 +82,18 @@ class VertexAILlama3Config(OpenAIGPTConfig): drop_params=drop_params, ) + def get_model_response_iterator( + self, + streaming_response: Union[Iterator[str], AsyncIterator[str], ModelResponse], + sync_stream: bool, + json_mode: Optional[bool] = False, + ) -> Any: + return VertexAILlama3StreamingHandler( + streaming_response=streaming_response, + sync_stream=sync_stream, + json_mode=json_mode, + ) + def transform_response( self, model: str, @@ -124,3 +139,23 @@ class VertexAILlama3Config(OpenAIGPTConfig): ) return model_response + + +class VertexAILlama3StreamingHandler(OpenAIChatCompletionStreamingHandler): + """ + Vertex AI Llama models may not include role in streaming chunk deltas. + This handler ensures the first chunk always has role="assistant". + """ + + def __init__(self, **kwargs): + super().__init__(**kwargs) + self.sent_role = False + + def chunk_parser(self, chunk: dict) -> ModelResponseStream: + result = super().chunk_parser(chunk) + if not self.sent_role and result.choices: + delta = result.choices[0].delta + if delta.role is None: + delta.role = "assistant" + self.sent_role = True + return result diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 18d0f0079ba..64eceae3297 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -6191,6 +6191,8 @@ "source": "https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart", "supports_function_calling": true, "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": true, "supports_video_input": true, "supports_vision": true }, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 18d0f0079ba..64eceae3297 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -6191,6 +6191,8 @@ "source": "https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart", "supports_function_calling": true, "supports_reasoning": true, + "supports_system_messages": true, + "supports_tool_choice": true, "supports_video_input": true, "supports_vision": true }, diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/test_vertex_ai_partner_models_llama3_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/test_vertex_ai_partner_models_llama3_transformation.py index b6228bc2e10..242a89d729a 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/test_vertex_ai_partner_models_llama3_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/llama3/test_vertex_ai_partner_models_llama3_transformation.py @@ -12,6 +12,7 @@ sys.path.insert( from litellm.llms.vertex_ai.vertex_ai_partner_models.llama3.transformation import ( VertexAILlama3Config, + VertexAILlama3StreamingHandler, ) @@ -59,4 +60,93 @@ class TestVertexAILlama3Config: ) assert response[0].message.tool_calls is not None assert response[0].finish_reason == "tool_calls" - # response = config.transform_response( + + +class TestVertexAILlama3StreamingHandler: + def test_first_chunk_has_role_assistant_when_missing(self): + """ + Vertex AI Llama streaming may return chunks without role in delta. + The handler should inject role='assistant' on the first chunk. + """ + handler = VertexAILlama3StreamingHandler( + streaming_response=iter([]), + sync_stream=True, + ) + chunk = { + "id": "test-id", + "object": "chat.completion.chunk", + "created": 123, + "model": "meta/llama-4-scout-17b-16e-instruct-maas", + "choices": [ + { + "index": 0, + "delta": {"content": None, "role": None}, + "finish_reason": None, + } + ], + } + result = handler.chunk_parser(chunk) + assert result.choices[0].delta.role == "assistant" + + def test_subsequent_chunks_no_role_override(self): + """ + Only the first chunk should have role injected. + """ + handler = VertexAILlama3StreamingHandler( + streaming_response=iter([]), + sync_stream=True, + ) + first_chunk = { + "id": "test-id", + "object": "chat.completion.chunk", + "created": 123, + "model": "meta/llama-4-scout-17b-16e-instruct-maas", + "choices": [ + { + "index": 0, + "delta": {"content": "Hello", "role": None}, + "finish_reason": None, + } + ], + } + second_chunk = { + "id": "test-id", + "object": "chat.completion.chunk", + "created": 123, + "model": "meta/llama-4-scout-17b-16e-instruct-maas", + "choices": [ + { + "index": 0, + "delta": {"content": " world", "role": None}, + "finish_reason": None, + } + ], + } + first_result = handler.chunk_parser(first_chunk) + second_result = handler.chunk_parser(second_chunk) + assert first_result.choices[0].delta.role == "assistant" + assert second_result.choices[0].delta.role is None + + def test_first_chunk_preserves_existing_role(self): + """ + If the API already provides role, don't overwrite it. + """ + handler = VertexAILlama3StreamingHandler( + streaming_response=iter([]), + sync_stream=True, + ) + chunk = { + "id": "test-id", + "object": "chat.completion.chunk", + "created": 123, + "model": "meta/llama-4-scout-17b-16e-instruct-maas", + "choices": [ + { + "index": 0, + "delta": {"content": None, "role": "assistant"}, + "finish_reason": None, + } + ], + } + result = handler.chunk_parser(chunk) + assert result.choices[0].delta.role == "assistant"