mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix(bedrock): map OpenAI video_url parts to Converse video blocks
`BedrockConverseMessagesProcessor` maps `image_url`, `file` and `document` content parts but had no branch for `video_url`, so an OpenAI-style video part was dropped without an error: only the text block survived into the request body, and the model answered about nothing while `usage.prompt_tokens` stayed at the text-only count. `BedrockImageProcessor` already picks the block type from the mime type (video/* -> VideoBlock, image/* -> ImageBlock), so routing `video_url` through the same processor yields the video block Converse expects. Both the sync and the async message-building paths are handled. Repro before/after (messages -> Converse content block kinds): video_url (mp4) ['text'] -> ['text', 'video'] image_url + mp4 ['text', 'video'] (unchanged) Fixes #40681
This commit is contained in:
parent
9a715df212
commit
02f06e0782
2 changed files with 196 additions and 0 deletions
|
|
@ -4371,6 +4371,21 @@ class BedrockConverseMessagesProcessor:
|
|||
image_url=image_url, format=format
|
||||
)
|
||||
_parts.append(_part)
|
||||
elif element["type"] == "video_url":
|
||||
# see the sync path: video shares the image
|
||||
# processor, which keys the block type off the
|
||||
# mime type.
|
||||
video_element = element["video_url"]
|
||||
if isinstance(video_element, dict):
|
||||
video_url = video_element["url"]
|
||||
video_format = video_element.get("format")
|
||||
else:
|
||||
video_url = video_element
|
||||
video_format = None
|
||||
_part = await BedrockImageProcessor.process_image_async(
|
||||
image_url=video_url, format=video_format
|
||||
)
|
||||
_parts.append(_part)
|
||||
elif element["type"] == "file":
|
||||
_part = await BedrockConverseMessagesProcessor._async_process_file_message(
|
||||
message=cast(ChatCompletionFileObject, element)
|
||||
|
|
@ -4744,6 +4759,23 @@ def _bedrock_converse_messages_pt(
|
|||
format=format,
|
||||
)
|
||||
_parts.append(_part)
|
||||
elif element["type"] == "video_url":
|
||||
# OpenAI `video_url` parts must reach Converse as
|
||||
# video blocks too. `process_image_sync` picks the
|
||||
# block type from the mime type (video/* -> video,
|
||||
# image/* -> image), so a video shares this path.
|
||||
video_element = element["video_url"]
|
||||
if isinstance(video_element, dict):
|
||||
video_url = video_element["url"]
|
||||
video_format = video_element.get("format")
|
||||
else:
|
||||
video_url = video_element
|
||||
video_format = None
|
||||
_part = BedrockImageProcessor.process_image_sync(
|
||||
image_url=video_url,
|
||||
format=video_format,
|
||||
)
|
||||
_parts.append(_part)
|
||||
elif element["type"] == "file":
|
||||
_part = BedrockConverseMessagesProcessor._process_file_message(
|
||||
message=cast(ChatCompletionFileObject, element)
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import asyncio
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
|
||||
|
|
@ -6953,3 +6954,166 @@ def test_transform_response_honors_json_mode_kwarg_when_optional_params_lack_it(
|
|||
)
|
||||
assert result.choices[0].message.tool_calls is None
|
||||
assert json.loads(result.choices[0].message.content) == {"city": "Paris", "population": 2100000}
|
||||
|
||||
|
||||
def _video_clip_b64() -> str:
|
||||
"""A minimal fake mp4 payload - the Converse path only inspects the mime type."""
|
||||
return base64.b64encode(b"\x00\x00\x00\x18ftypmp42" + b"\xab" * 32).decode()
|
||||
|
||||
|
||||
def test_bedrock_converse_user_video_url_becomes_video_block():
|
||||
"""
|
||||
An OpenAI `video_url` part used to be dropped on the Converse path: only
|
||||
the text block reached Bedrock, so the model answered about nothing while
|
||||
`usage.prompt_tokens` stayed at the text-only count.
|
||||
"""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
_bedrock_converse_messages_pt,
|
||||
)
|
||||
|
||||
clip_b64 = _video_clip_b64()
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Describe this video."},
|
||||
{
|
||||
"type": "video_url",
|
||||
"video_url": {"url": f"data:video/mp4;base64,{clip_b64}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
translated = _bedrock_converse_messages_pt(
|
||||
messages=messages, model="amazon.nova-pro-v1:0", llm_provider="bedrock"
|
||||
)
|
||||
|
||||
blocks = translated[0]["content"]
|
||||
assert [next(iter(block)) for block in blocks] == ["text", "video"]
|
||||
video_block = blocks[1]["video"]
|
||||
assert video_block["format"] == "mp4"
|
||||
assert video_block["source"]["bytes"] == clip_b64
|
||||
|
||||
|
||||
def test_bedrock_converse_user_video_url_str_form_becomes_video_block():
|
||||
"""`video_url` may also be a bare data uri instead of a mapping."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
_bedrock_converse_messages_pt,
|
||||
)
|
||||
|
||||
clip_b64 = _video_clip_b64()
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "video_url",
|
||||
"video_url": f"data:video/webm;base64,{clip_b64}",
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
translated = _bedrock_converse_messages_pt(
|
||||
messages=messages, model="amazon.nova-pro-v1:0", llm_provider="bedrock"
|
||||
)
|
||||
|
||||
blocks = translated[0]["content"]
|
||||
assert [next(iter(block)) for block in blocks] == ["video"]
|
||||
assert blocks[0]["video"]["format"] == "webm"
|
||||
assert blocks[0]["video"]["source"]["bytes"] == clip_b64
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bedrock_converse_user_video_url_becomes_video_block_async():
|
||||
"""The async (acompletion) message path must map video_url the same way."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
BedrockConverseMessagesProcessor,
|
||||
)
|
||||
|
||||
clip_b64 = _video_clip_b64()
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Describe this video."},
|
||||
{
|
||||
"type": "video_url",
|
||||
"video_url": {"url": f"data:video/mp4;base64,{clip_b64}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
translated = (
|
||||
await BedrockConverseMessagesProcessor._bedrock_converse_messages_pt_async(
|
||||
messages=messages,
|
||||
model="amazon.nova-pro-v1:0",
|
||||
llm_provider="bedrock",
|
||||
)
|
||||
)
|
||||
|
||||
blocks = translated[0]["content"]
|
||||
assert [next(iter(block)) for block in blocks] == ["text", "video"]
|
||||
assert blocks[1]["video"]["format"] == "mp4"
|
||||
|
||||
|
||||
def test_bedrock_converse_image_url_still_becomes_image_block():
|
||||
"""The video_url branch must not hijack ordinary image parts."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
_bedrock_converse_messages_pt,
|
||||
)
|
||||
|
||||
png_b64 = (
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGNgYGBgAAAABQABXvMqOgAAAABJRU5ErkJggg=="
|
||||
)
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Describe this image."},
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {"url": f"data:image/png;base64,{png_b64}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
translated = _bedrock_converse_messages_pt(
|
||||
messages=messages, model="amazon.nova-pro-v1:0", llm_provider="bedrock"
|
||||
)
|
||||
|
||||
blocks = translated[0]["content"]
|
||||
assert [next(iter(block)) for block in blocks] == ["text", "image"]
|
||||
assert blocks[1]["image"]["format"] == "png"
|
||||
|
||||
|
||||
def test_bedrock_converse_transform_request_keeps_video_url():
|
||||
"""End-to-end request build: the video block survives into the wire body."""
|
||||
clip_b64 = _video_clip_b64()
|
||||
messages = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Describe this video."},
|
||||
{
|
||||
"type": "video_url",
|
||||
"video_url": {"url": f"data:video/mp4;base64,{clip_b64}"},
|
||||
},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
body = AmazonConverseConfig().transform_request(
|
||||
model="amazon.nova-pro-v1:0",
|
||||
messages=messages,
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
blocks = body["messages"][0]["content"]
|
||||
assert [next(iter(block)) for block in blocks] == ["text", "video"]
|
||||
assert blocks[1]["video"]["source"]["bytes"] == clip_b64
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue