mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(responses): keep system messages at the front of the bridged request
instructions became a leading system message and the Responses input could carry one of its own, so an Anthropic /v1/messages conversation bridged to chat completions came out as system, user, system. Chat templates that require the system message first reject that with "System message must be at the beginning", which is how Claude Code against an OpenAI-compatible backend hits it. Gather the system prompts into one leading message instead. A conversation whose system message is already first is untouched. Fixes #40693
This commit is contained in:
parent
9a715df212
commit
206b74effa
2 changed files with 111 additions and 8 deletions
|
|
@ -433,22 +433,50 @@ class LiteLLMCompletionResponsesConfig:
|
|||
| ChatCompletionResponseMessage
|
||||
| Message
|
||||
] = []
|
||||
if responses_api_request.get("instructions"):
|
||||
messages.append(
|
||||
LiteLLMCompletionResponsesConfig.transform_instructions_to_system_message(
|
||||
responses_api_request.get("instructions")
|
||||
)
|
||||
)
|
||||
|
||||
messages.extend(
|
||||
input_messages: Final = tuple(
|
||||
LiteLLMCompletionResponsesConfig._transform_response_input_param_to_chat_completion_message(
|
||||
input=input,
|
||||
replay_reasoning=replay_reasoning,
|
||||
)
|
||||
)
|
||||
|
||||
# `instructions` and the input can each carry a system message, which left a
|
||||
# conversation reading system, user, system. Chat templates that require the
|
||||
# system message first reject that, so gather them into one leading message.
|
||||
instructions: Final = responses_api_request.get("instructions")
|
||||
system_contents: Final = tuple(
|
||||
content
|
||||
for content in (
|
||||
instructions,
|
||||
*(message.get("content") for message in input_messages if message.get("role") == "system"),
|
||||
)
|
||||
if content
|
||||
)
|
||||
|
||||
if system_contents:
|
||||
messages.append(
|
||||
LiteLLMCompletionResponsesConfig._merge_system_contents(system_contents)
|
||||
)
|
||||
|
||||
messages.extend(message for message in input_messages if message.get("role") != "system")
|
||||
|
||||
return messages
|
||||
|
||||
@staticmethod
|
||||
def _merge_system_contents(contents: tuple[Any, ...]) -> ChatCompletionSystemMessage:
|
||||
"""Join system prompts into one leading message.
|
||||
|
||||
Part lists collapse to their text: system content is conventionally a string, and
|
||||
the backends that reject a trailing system message are the same ones that expect one.
|
||||
"""
|
||||
texts: Final = tuple(
|
||||
content
|
||||
if isinstance(content, str)
|
||||
else "\n\n".join(part.get("text", "") for part in content if isinstance(part, dict))
|
||||
for content in contents
|
||||
)
|
||||
return ChatCompletionSystemMessage(role="system", content="\n\n".join(text for text in texts if text))
|
||||
|
||||
@staticmethod
|
||||
async def async_responses_api_session_handler(
|
||||
previous_response_id: str,
|
||||
|
|
|
|||
|
|
@ -0,0 +1,75 @@
|
|||
"""
|
||||
Unit tests for keeping system messages at the front of the bridged chat request.
|
||||
|
||||
``instructions`` becomes a leading system message and the Responses input can carry
|
||||
a system message of its own, so an Anthropic ``/v1/messages`` conversation bridged
|
||||
to chat completions could come out as system, user, system. Chat templates that
|
||||
require the system message to come first reject that with
|
||||
"System message must be at the beginning".
|
||||
|
||||
See: https://github.com/BerriAI/litellm/issues/40693
|
||||
"""
|
||||
|
||||
from litellm.responses.litellm_completion_transformation.transformation import (
|
||||
LiteLLMCompletionResponsesConfig,
|
||||
)
|
||||
|
||||
|
||||
def _roles(input, responses_api_request):
|
||||
messages = LiteLLMCompletionResponsesConfig.transform_responses_api_input_to_messages(
|
||||
input=input, responses_api_request=responses_api_request
|
||||
)
|
||||
return [message.get("role") for message in messages], messages
|
||||
|
||||
|
||||
def test_system_message_after_user_content_is_hoisted():
|
||||
roles, messages = _roles(
|
||||
[
|
||||
{"role": "user", "content": "first question"},
|
||||
{"role": "system", "content": "mid"},
|
||||
{"role": "user", "content": "second question"},
|
||||
],
|
||||
{"instructions": "lead"},
|
||||
)
|
||||
|
||||
assert roles == ["system", "user", "user"]
|
||||
assert "system" not in roles[1:]
|
||||
|
||||
|
||||
def test_both_system_prompts_survive_the_merge():
|
||||
_, messages = _roles(
|
||||
[
|
||||
{"role": "user", "content": "q"},
|
||||
{"role": "system", "content": "mid"},
|
||||
],
|
||||
{"instructions": "lead"},
|
||||
)
|
||||
|
||||
assert messages[0]["content"] == "lead\n\nmid"
|
||||
|
||||
|
||||
def test_part_list_content_collapses_to_its_text():
|
||||
"""A system message given as parts still contributes its text to the merged prompt."""
|
||||
_, messages = _roles(
|
||||
[
|
||||
{"role": "user", "content": "q"},
|
||||
{"role": "system", "content": [{"type": "text", "text": "B"}]},
|
||||
],
|
||||
{"instructions": "A"},
|
||||
)
|
||||
|
||||
assert messages[0]["content"] == "A\n\nB"
|
||||
|
||||
|
||||
def test_an_already_leading_system_message_is_left_alone():
|
||||
"""The common case must not be rewritten."""
|
||||
roles, messages = _roles([{"role": "user", "content": "q"}], {"instructions": "lead"})
|
||||
|
||||
assert roles == ["system", "user"]
|
||||
assert messages[0]["content"] == "lead"
|
||||
|
||||
|
||||
def test_a_conversation_without_a_system_message_is_unchanged():
|
||||
roles, _ = _roles([{"role": "user", "content": "q"}], {})
|
||||
|
||||
assert roles == ["user"]
|
||||
Loading…
Add table
Reference in a new issue