mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
ci/cd fixes - streaming role & bedrock model cost (#21200)
* fix(model_cost): add missing supports_system_messages and supports_tool_choice to bedrock/moonshotai.kimi-k2.5 * fix(streaming): ensure role=assistant is set on first streaming chunk via strip_role_from_delta * fix(vertex_ai): ensure role=assistant on first streaming chunk for Llama models Add VertexAILlama3StreamingHandler that injects role='assistant' into the first streaming chunk delta when the Vertex AI Llama API omits it.
This commit is contained in:
parent
c32544a284
commit
f8334dfeda
4 changed files with 133 additions and 4 deletions
|
|
@ -1,12 +1,15 @@
|
|||
import types
|
||||
from typing import Any, List, Optional
|
||||
from typing import Any, AsyncIterator, Iterator, List, Optional, Union
|
||||
|
||||
import httpx
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
|
||||
from litellm.llms.openai.chat.gpt_transformation import (
|
||||
OpenAIChatCompletionStreamingHandler,
|
||||
OpenAIGPTConfig,
|
||||
)
|
||||
from litellm.types.llms.openai import AllMessageValues, OpenAIChatCompletionResponse
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
from litellm.types.utils import ModelResponse, ModelResponseStream, Usage
|
||||
|
||||
from ...common_utils import VertexAIError
|
||||
|
||||
|
|
@ -79,6 +82,18 @@ class VertexAILlama3Config(OpenAIGPTConfig):
|
|||
drop_params=drop_params,
|
||||
)
|
||||
|
||||
def get_model_response_iterator(
|
||||
self,
|
||||
streaming_response: Union[Iterator[str], AsyncIterator[str], ModelResponse],
|
||||
sync_stream: bool,
|
||||
json_mode: Optional[bool] = False,
|
||||
) -> Any:
|
||||
return VertexAILlama3StreamingHandler(
|
||||
streaming_response=streaming_response,
|
||||
sync_stream=sync_stream,
|
||||
json_mode=json_mode,
|
||||
)
|
||||
|
||||
def transform_response(
|
||||
self,
|
||||
model: str,
|
||||
|
|
@ -124,3 +139,23 @@ class VertexAILlama3Config(OpenAIGPTConfig):
|
|||
)
|
||||
|
||||
return model_response
|
||||
|
||||
|
||||
class VertexAILlama3StreamingHandler(OpenAIChatCompletionStreamingHandler):
|
||||
"""
|
||||
Vertex AI Llama models may not include role in streaming chunk deltas.
|
||||
This handler ensures the first chunk always has role="assistant".
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.sent_role = False
|
||||
|
||||
def chunk_parser(self, chunk: dict) -> ModelResponseStream:
|
||||
result = super().chunk_parser(chunk)
|
||||
if not self.sent_role and result.choices:
|
||||
delta = result.choices[0].delta
|
||||
if delta.role is None:
|
||||
delta.role = "assistant"
|
||||
self.sent_role = True
|
||||
return result
|
||||
|
|
|
|||
|
|
@ -6191,6 +6191,8 @@
|
|||
"source": "https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_video_input": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
|
|
|
|||
|
|
@ -6191,6 +6191,8 @@
|
|||
"source": "https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart",
|
||||
"supports_function_calling": true,
|
||||
"supports_reasoning": true,
|
||||
"supports_system_messages": true,
|
||||
"supports_tool_choice": true,
|
||||
"supports_video_input": true,
|
||||
"supports_vision": true
|
||||
},
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ sys.path.insert(
|
|||
|
||||
from litellm.llms.vertex_ai.vertex_ai_partner_models.llama3.transformation import (
|
||||
VertexAILlama3Config,
|
||||
VertexAILlama3StreamingHandler,
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -59,4 +60,93 @@ class TestVertexAILlama3Config:
|
|||
)
|
||||
assert response[0].message.tool_calls is not None
|
||||
assert response[0].finish_reason == "tool_calls"
|
||||
# response = config.transform_response(
|
||||
|
||||
|
||||
class TestVertexAILlama3StreamingHandler:
|
||||
def test_first_chunk_has_role_assistant_when_missing(self):
|
||||
"""
|
||||
Vertex AI Llama streaming may return chunks without role in delta.
|
||||
The handler should inject role='assistant' on the first chunk.
|
||||
"""
|
||||
handler = VertexAILlama3StreamingHandler(
|
||||
streaming_response=iter([]),
|
||||
sync_stream=True,
|
||||
)
|
||||
chunk = {
|
||||
"id": "test-id",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 123,
|
||||
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"content": None, "role": None},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
}
|
||||
result = handler.chunk_parser(chunk)
|
||||
assert result.choices[0].delta.role == "assistant"
|
||||
|
||||
def test_subsequent_chunks_no_role_override(self):
|
||||
"""
|
||||
Only the first chunk should have role injected.
|
||||
"""
|
||||
handler = VertexAILlama3StreamingHandler(
|
||||
streaming_response=iter([]),
|
||||
sync_stream=True,
|
||||
)
|
||||
first_chunk = {
|
||||
"id": "test-id",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 123,
|
||||
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"content": "Hello", "role": None},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
}
|
||||
second_chunk = {
|
||||
"id": "test-id",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 123,
|
||||
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"content": " world", "role": None},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
}
|
||||
first_result = handler.chunk_parser(first_chunk)
|
||||
second_result = handler.chunk_parser(second_chunk)
|
||||
assert first_result.choices[0].delta.role == "assistant"
|
||||
assert second_result.choices[0].delta.role is None
|
||||
|
||||
def test_first_chunk_preserves_existing_role(self):
|
||||
"""
|
||||
If the API already provides role, don't overwrite it.
|
||||
"""
|
||||
handler = VertexAILlama3StreamingHandler(
|
||||
streaming_response=iter([]),
|
||||
sync_stream=True,
|
||||
)
|
||||
chunk = {
|
||||
"id": "test-id",
|
||||
"object": "chat.completion.chunk",
|
||||
"created": 123,
|
||||
"model": "meta/llama-4-scout-17b-16e-instruct-maas",
|
||||
"choices": [
|
||||
{
|
||||
"index": 0,
|
||||
"delta": {"content": None, "role": "assistant"},
|
||||
"finish_reason": None,
|
||||
}
|
||||
],
|
||||
}
|
||||
result = handler.chunk_parser(chunk)
|
||||
assert result.choices[0].delta.role == "assistant"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue