From 2c6577959ae5a5910a2f07038014418f74c216fc Mon Sep 17 00:00:00 2001 From: Darsh Joshi Date: Fri, 2 Oct 2026 19:03:45 -0400 Subject: [PATCH] fix(deepseek): keep tool result images instead of collapsing them to text The DeepSeek transform only kept image content lists on user messages, so a role=tool message carrying an image_url block was collapsed to its text before the shared OpenAI transform ran. That transform already moves tool message images into a user message, but by then the image was gone Keep tool message image lists so the image reaches the model Fixes #44211 --- litellm/llms/deepseek/chat/transformation.py | 14 ++++++++------ .../chat/test_deepseek_chat_transformation.py | 10 ++++++++++ 2 files changed, 18 insertions(+), 6 deletions(-) diff --git a/litellm/llms/deepseek/chat/transformation.py b/litellm/llms/deepseek/chat/transformation.py index 4e428a23392..653b5d66bf3 100644 --- a/litellm/llms/deepseek/chat/transformation.py +++ b/litellm/llms/deepseek/chat/transformation.py @@ -121,10 +121,12 @@ class DeepSeekChatConfig(OpenAIGPTConfig): DeepSeek vision models accept image_url content blocks in user messages (https://api-docs.deepseek.com/guides/vision), so those content lists are forwarded as-is, with any search_results text - appended as a trailing text block. Every other message keeps the - historical string collapse (which also folds search_results text - into string content); a list with no extractable text stays - unchanged, matching what DeepSeek historically received. + appended as a trailing text block. Tool message image lists are kept + too, so the parent transform can move their images into a user + message. Every other message keeps the historical string collapse + (which also folds search_results text into string content); a list + with no extractable text stays unchanged, matching what DeepSeek + historically received. """ forward_images: Final = any( isinstance(message.get("content"), list) for message in messages @@ -160,13 +162,13 @@ class DeepSeekChatConfig(OpenAIGPTConfig): def _is_vision_forwardable_content(self, message: AllMessageValues, content: Sequence[object]) -> bool: """ - True only for a user message whose content list holds well-formed + True only for a user or tool message whose content list holds well-formed text and image_url blocks with at least one image; a block missing its payload falls back to the string collapse instead of crashing or reaching the wire malformed. The model capability gate lives in the caller. """ - if message.get("role") != "user": + if message.get("role") not in ("user", "tool"): return False if not all(self._is_forwardable_block(block) for block in content): return False diff --git a/tests/unit/llms/deepseek/chat/test_deepseek_chat_transformation.py b/tests/unit/llms/deepseek/chat/test_deepseek_chat_transformation.py index 3f93264d0d0..9c584ffa605 100644 --- a/tests/unit/llms/deepseek/chat/test_deepseek_chat_transformation.py +++ b/tests/unit/llms/deepseek/chat/test_deepseek_chat_transformation.py @@ -169,6 +169,16 @@ class TestDeepSeekVisionMultimodalContent: assert result[0]["content"] == "what is in this image?" + def test_tool_result_image_reaches_vision_model(self): + tool_message = {**self._image_message(role="tool"), "tool_call_id": "call_1"} + image_block = tool_message["content"][1] + + result = self.config._transform_messages([tool_message], model=self.VISION_MODEL) + + assert result[0]["role"] == "tool" + assert result[0]["tool_call_id"] == "call_1" + assert any(isinstance(message["content"], list) and image_block in message["content"] for message in result) + def test_audio_block_collapsed_even_on_vision_model(self): messages = [ {