ci/cd fixes - streaming role & bedrock model cost (#21200)

* fix(model_cost): add missing supports_system_messages and supports_tool_choice to bedrock/moonshotai.kimi-k2.5

* fix(streaming): ensure role=assistant is set on first streaming chunk via strip_role_from_delta

* fix(vertex_ai): ensure role=assistant on first streaming chunk for Llama models

Add VertexAILlama3StreamingHandler that injects role='assistant' into the
first streaming chunk delta when the Vertex AI Llama API omits it.
This commit is contained in:
Ishaan Jaff 2026-02-14 09:18:29 -08:00 • committed by GitHub
parent c32544a284
commit f8334dfeda
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 133 additions and 4 deletions

View file

@ -1,12 +1,15 @@
import types
from typing import Any, List, Optional
from typing import Any, AsyncIterator, Iterator, List, Optional, Union
import httpx
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from litellm.llms.openai.chat.gpt_transformation import (
OpenAIChatCompletionStreamingHandler,
OpenAIGPTConfig,
)
from litellm.types.llms.openai import AllMessageValues, OpenAIChatCompletionResponse
from litellm.types.utils import ModelResponse, Usage
from litellm.types.utils import ModelResponse, ModelResponseStream, Usage
from ...common_utils import VertexAIError
@ -79,6 +82,18 @@ class VertexAILlama3Config(OpenAIGPTConfig):
drop_params=drop_params,
)
def get_model_response_iterator(
self,
streaming_response: Union[Iterator[str], AsyncIterator[str], ModelResponse],
sync_stream: bool,
json_mode: Optional[bool] = False,
) -> Any:
return VertexAILlama3StreamingHandler(
streaming_response=streaming_response,
sync_stream=sync_stream,
json_mode=json_mode,
)
def transform_response(
self,
model: str,
@ -124,3 +139,23 @@ class VertexAILlama3Config(OpenAIGPTConfig):
)
return model_response
class VertexAILlama3StreamingHandler(OpenAIChatCompletionStreamingHandler):
"""
Vertex AI Llama models may not include role in streaming chunk deltas.
This handler ensures the first chunk always has role="assistant".
"""
def __init__(self, **kwargs):
super().__init__(**kwargs)
self.sent_role = False
def chunk_parser(self, chunk: dict) -> ModelResponseStream:
result = super().chunk_parser(chunk)
if not self.sent_role and result.choices:
delta = result.choices[0].delta
if delta.role is None:
delta.role = "assistant"
self.sent_role = True
return result

View file

@ -6191,6 +6191,8 @@
"source": "https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_video_input": true,
"supports_vision": true
},

View file

@ -6191,6 +6191,8 @@
"source": "https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart",
"supports_function_calling": true,
"supports_reasoning": true,
"supports_system_messages": true,
"supports_tool_choice": true,
"supports_video_input": true,
"supports_vision": true
},

View file

@ -12,6 +12,7 @@ sys.path.insert(
from litellm.llms.vertex_ai.vertex_ai_partner_models.llama3.transformation import (
VertexAILlama3Config,
VertexAILlama3StreamingHandler,
)
@ -59,4 +60,93 @@ class TestVertexAILlama3Config:
)
assert response[0].message.tool_calls is not None
assert response[0].finish_reason == "tool_calls"
# response = config.transform_response(
class TestVertexAILlama3StreamingHandler:
def test_first_chunk_has_role_assistant_when_missing(self):
"""
Vertex AI Llama streaming may return chunks without role in delta.
The handler should inject role='assistant' on the first chunk.
"""
handler = VertexAILlama3StreamingHandler(
streaming_response=iter([]),
sync_stream=True,
)
chunk = {
"id": "test-id",
"object": "chat.completion.chunk",
"created": 123,
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
"choices": [
{
"index": 0,
"delta": {"content": None, "role": None},
"finish_reason": None,
}
],
}
result = handler.chunk_parser(chunk)
assert result.choices[0].delta.role == "assistant"
def test_subsequent_chunks_no_role_override(self):
"""
Only the first chunk should have role injected.
"""
handler = VertexAILlama3StreamingHandler(
streaming_response=iter([]),
sync_stream=True,
)
first_chunk = {
"id": "test-id",
"object": "chat.completion.chunk",
"created": 123,
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
"choices": [
{
"index": 0,
"delta": {"content": "Hello", "role": None},
"finish_reason": None,
}
],
}
second_chunk = {
"id": "test-id",
"object": "chat.completion.chunk",
"created": 123,
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
"choices": [
{
"index": 0,
"delta": {"content": " world", "role": None},
"finish_reason": None,
}
],
}
first_result = handler.chunk_parser(first_chunk)
second_result = handler.chunk_parser(second_chunk)
assert first_result.choices[0].delta.role == "assistant"
assert second_result.choices[0].delta.role is None
def test_first_chunk_preserves_existing_role(self):
"""
If the API already provides role, don't overwrite it.
"""
handler = VertexAILlama3StreamingHandler(
streaming_response=iter([]),
sync_stream=True,
)
chunk = {
"id": "test-id",
"object": "chat.completion.chunk",
"created": 123,
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
"choices": [
{
"index": 0,
"delta": {"content": None, "role": "assistant"},
"finish_reason": None,
}
],
}
result = handler.chunk_parser(chunk)
assert result.choices[0].delta.role == "assistant"