fix(anthropic-bridge): convert mid-conversation system turns to user turns on /v1/messages to chat completions

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
shivam 2026-09-16 21:47:46 +00:00
parent 930ec9643a
commit 8a41e10332
6 changed files with 297 additions and 92 deletions

View file

@ -118,6 +118,10 @@ from litellm.llms.anthropic.common_utils import (
from litellm.llms.anthropic.experimental_pass_through.context_management import (
PolyfillResult,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
convert_mid_conversation_system_turns,
is_system_role_message,
)
from litellm.llms.anthropic.experimental_pass_through.messages.utils import (
openai_chat_refusal_text,
refusal_stop_details,
@ -421,7 +425,15 @@ class LiteLLMAnthropicMessagesAdapter:
) -> list:
new_messages: Final[list[AllMessageValues]] = []
replayable_messages: Final = strip_encrypted_reasoning_blocks_from_anthropic_messages(messages)
for m in replayable_messages:
leading_count: Final = next(
(i for i, m in enumerate(replayable_messages) if not is_system_role_message(m)),
len(replayable_messages),
)
ordered_messages: Final = (
*replayable_messages[:leading_count],
*convert_mid_conversation_system_turns(replayable_messages[leading_count:]),
)
for m in ordered_messages:
user_message: ChatCompletionUserMessage | None = None
tool_message_list: list[ChatCompletionToolMessage] = []
new_user_content_list: list[ChatCompletionTextObject | ChatCompletionImageObject] = []
@ -494,7 +506,7 @@ class LiteLLMAnthropicMessagesAdapter:
if isinstance(m.get("content"), str):
assistant_message_str = str(m.get("content", ""))
elif isinstance(m.get("content"), list):
for content in m.get("content", []):
for content in cast(list, m.get("content", [])):
if isinstance(content, str):
assistant_message_str = str(content)
elif isinstance(content, dict):

View file

@ -0,0 +1,84 @@
from collections.abc import Mapping, Sequence
from typing import Final
CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
def as_system_content_blocks(value: object) -> list[object]:
if value is None:
return []
if isinstance(value, list):
return list(value)
if isinstance(value, str):
return [{"type": "text", "text": value}]
return [value]
def is_system_role_message(message: object) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
def system_role_message_as_user(message: Mapping[str, object]) -> Mapping[str, object]:
return {
"role": "user",
"content": as_system_content_blocks(CONVERTED_SYSTEM_NOTE) + as_system_content_blocks(message.get("content")),
}
def opens_with_tool_results(message: object) -> bool:
if not isinstance(message, dict) or message.get("role") != "user":
return False
content: Final = message.get("content")
return (
isinstance(content, list)
and len(content) > 0
and isinstance(content[0], dict)
and content[0].get("type") == "tool_result"
)
def system_run_before(messages: Sequence[Mapping[str, object]], index: int) -> Sequence[Mapping[str, object]]:
start: Final = next(
(j + 1 for j in range(index - 1, -1, -1) if not is_system_role_message(messages[j])),
0,
)
return messages[start:index]
def system_run_end(messages: Sequence[Mapping[str, object]], index: int) -> int:
return next(
(j for j in range(index, len(messages)) if not is_system_role_message(messages[j])),
len(messages),
)
def reordered_around_tool_results(
messages: Sequence[Mapping[str, object]], index: int
) -> tuple[Mapping[str, object], ...]:
message: Final = messages[index]
if opens_with_tool_results(message):
return (message, *system_run_before(messages, index))
if not is_system_role_message(message):
return (message,)
run_end: Final = system_run_end(messages, index)
follower: Final = messages[run_end] if run_end < len(messages) else None
return () if opens_with_tool_results(follower) else (message,)
def system_turns_after_tool_results(
messages: Sequence[Mapping[str, object]],
) -> tuple[Mapping[str, object], ...]:
return tuple(
message for index in range(len(messages)) for message in reordered_around_tool_results(messages, index)
)
def convert_mid_conversation_system_turns(
messages: Sequence[Mapping[str, object]],
) -> tuple[Mapping[str, object], ...]:
return tuple(
system_role_message_as_user(m) if is_system_role_message(m) else m
for m in system_turns_after_tool_results(messages)
)

View file

@ -27,6 +27,11 @@ from ...common_utils import (
strip_advisor_blocks_from_messages,
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
from .mid_conversation_system import (
as_system_content_blocks,
convert_mid_conversation_system_turns,
is_system_role_message,
)
DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01"
@ -151,73 +156,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
else:
return system_param
@staticmethod
def _as_system_content_blocks(value: object) -> list:
if value is None:
return []
if isinstance(value, list):
return list(value)
if isinstance(value, str):
return [{"type": "text", "text": value}]
return [value]
@staticmethod
def _is_system_role_message(message: object) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
_CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
def _system_role_message_as_user(self, message: Mapping) -> Mapping:
return {
"role": "user",
"content": self._as_system_content_blocks(self._CONVERTED_SYSTEM_NOTE)
+ self._as_system_content_blocks(message.get("content")),
}
@staticmethod
def _opens_with_tool_results(message: object) -> bool:
if not isinstance(message, dict) or message.get("role") != "user":
return False
content: Final = message.get("content")
return (
isinstance(content, list)
and len(content) > 0
and isinstance(content[0], dict)
and content[0].get("type") == "tool_result"
)
def _system_run_before(self, messages: Sequence, index: int) -> Sequence:
start: Final = next(
(j + 1 for j in range(index - 1, -1, -1) if not self._is_system_role_message(messages[j])),
0,
)
return messages[start:index]
def _system_run_end(self, messages: Sequence, index: int) -> int:
return next(
(j for j in range(index, len(messages)) if not self._is_system_role_message(messages[j])),
len(messages),
)
def _reordered_around_tool_results(self, messages: Sequence, index: int) -> tuple:
message: Final = messages[index]
if self._opens_with_tool_results(message):
return (message, *self._system_run_before(messages, index))
if not self._is_system_role_message(message):
return (message,)
run_end: Final = self._system_run_end(messages, index)
follower: Final = messages[run_end] if run_end < len(messages) else None
return () if self._opens_with_tool_results(follower) else (message,)
def _system_turns_after_tool_results(self, messages: Sequence) -> tuple:
return tuple(
message
for index in range(len(messages))
for message in self._reordered_around_tool_results(messages, index)
)
def _normalize_system_role_messages(self, anthropic_messages_request: dict, model: str) -> None:
"""Normalize ``role: "system"`` entries in ``messages`` per the Anthropic
``/v1/messages`` contract, which the first-party API, Bedrock Invoke,
@ -254,7 +192,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if not isinstance(messages, list):
return
leading_count: Final = next(
(i for i, m in enumerate(messages) if not self._is_system_role_message(m)),
(i for i, m in enumerate(messages) if not is_system_role_message(m)),
len(messages),
)
hoisted: Final = messages[:leading_count]
@ -265,10 +203,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
custom_llm_provider=self.custom_llm_provider,
key="supports_mid_conversation_system",
)
else [
self._system_role_message_as_user(m) if self._is_system_role_message(m) else m
for m in self._system_turns_after_tool_results(messages[leading_count:])
]
else list(convert_mid_conversation_system_turns(messages[leading_count:]))
)
if hoisted or remaining != messages:
anthropic_messages_request["messages"] = remaining
@ -278,7 +213,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
anthropic_messages_request.get("system"),
*(m.get("content") for m in hoisted),
)
for block in self._as_system_content_blocks(source)
for block in as_system_content_blocks(source)
]
filtered_system: Final = self._filter_billing_headers_from_system(system_content)
if filtered_system:

View file

@ -23,6 +23,9 @@ from litellm.llms.anthropic.experimental_pass_through.adapters.transformation im
create_tool_name_mapping,
truncate_tool_name,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
CONVERTED_SYSTEM_NOTE,
)
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from litellm.types.llms.anthropic import (
AnthopicMessagesAssistantMessageParam,
@ -563,10 +566,19 @@ def test_translate_anthropic_messages_to_openai_tool_message_placement():
@pytest.mark.parametrize(
("system_content", "expected_content"),
[
("Use the corrected result.", "Use the corrected result."),
(
"Use the corrected result.",
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Use the corrected result."},
],
),
(
[{"type": "text", "text": "Use the corrected result."}],
[{"type": "text", "text": "Use the corrected result."}],
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Use the corrected result."},
],
),
(
[
@ -576,7 +588,11 @@ def test_translate_anthropic_messages_to_openai_tool_message_placement():
},
{"type": "text", "text": "Use the corrected result."},
],
[{"type": "text", "text": "Use the corrected result."}],
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "image_url", "image_url": {"url": "https://example.com/a.png"}},
{"type": "text", "text": "Use the corrected result."},
],
),
(
[
@ -584,13 +600,14 @@ def test_translate_anthropic_messages_to_openai_tool_message_placement():
{"type": "text", "text": "Second correction."},
],
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "First correction."},
{"type": "text", "text": "Second correction."},
],
),
],
)
def test_translate_anthropic_messages_to_openai_preserves_midturn_system_correction(
def test_translate_anthropic_messages_to_openai_converts_midturn_system_correction(
system_content: object,
expected_content: object,
):
@ -646,7 +663,7 @@ def test_translate_anthropic_messages_to_openai_preserves_midturn_system_correct
"tool_call_id": "toolu_01234",
"content": "Rainy, 55°F",
},
{"role": "system", "content": expected_content},
{"role": "user", "content": expected_content},
{"role": "user", "content": "Continue."},
]
@ -752,8 +769,8 @@ def test_translate_anthropic_messages_to_openai_drops_empty_midturn_system(
def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
"""
Request level: the trusted top-level prompt is hoisted to index 0 exactly once and the
in-sequence correction keeps its own position and `role: "system"` -- no duplication of
either, and no reordering of the surrounding turns.
in-sequence correction keeps its own position as a user turn prefixed with the operator
note -- no duplication of either, and no reordering of the surrounding turns.
"""
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={
@ -773,11 +790,107 @@ def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
{"role": "system", "content": "Trusted top-level prompt."},
{"role": "user", "content": "First question."},
{"role": "assistant", "content": "First answer.", "thinking_blocks": None},
{"role": "system", "content": "Use the corrected result."},
{
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Use the corrected result."},
],
},
{"role": "user", "content": "Continue."},
]
def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn():
"""
Claude Code appends a system-role harness reminder after the user turn. On a
chat-completions target the outbound request must have exactly one system message,
at index 0, and the converted turn must carry the operator note first.
"""
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={
"model": "qwen3.8-27B",
"max_tokens": 128,
"system": [{"type": "text", "text": "You are Claude Code."}],
"messages": [
{"role": "user", "content": "say hi"},
{
"role": "system",
"content": [
{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}
],
},
{"role": "assistant", "content": "Hi."},
{"role": "user", "content": "say bye"},
],
}
)
roles = [m["role"] for m in openai_request["messages"]]
assert roles == ["system", "user", "user", "assistant", "user"]
converted = openai_request["messages"][2]
assert converted["content"][0]["text"] == CONVERTED_SYSTEM_NOTE
assert converted["content"][1]["text"] == "<system-reminder>Keep answers to one sentence.</system-reminder>"
def test_translate_anthropic_to_openai_moves_midturn_system_after_tool_result():
"""
A system entry wedged between an assistant tool_use turn and its tool_result turn is
emitted after the role: "tool" message, so the tool call stays paired with its result.
"""
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
messages=[
{
"role": "assistant",
"content": [
{
"type": "tool_use",
"id": "toolu_01234",
"name": "get_weather",
"input": {"location": "Boston"},
}
],
},
{"role": "system", "content": "Use the corrected result."},
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01234",
"content": "Rainy, 55°F",
}
],
},
],
model="claude-3-5-sonnet-20240620",
)
assert [m["role"] for m in result] == ["assistant", "tool", "user"]
assert result[2]["content"][0]["text"] == CONVERTED_SYSTEM_NOTE
def test_translate_anthropic_messages_to_openai_converts_string_midturn_system():
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
messages=[
{"role": "user", "content": "hi"},
{"role": "system", "content": "Keep it short."},
],
model="claude-3-5-sonnet-20240620",
)
assert result == [
{"role": "user", "content": "hi"},
{
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Keep it short."},
],
},
]
def _claude_code_user_id(session_id: str) -> str:
return json.dumps({"device_id": "d" * 64, "account_uuid": "", "session_id": session_id})

View file

@ -0,0 +1,62 @@
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
CONVERTED_SYSTEM_NOTE,
convert_mid_conversation_system_turns,
)
def test_convert_mid_conversation_system_turns_converts_system_to_user_in_place():
result = convert_mid_conversation_system_turns(
[
{"role": "user", "content": "hi"},
{"role": "system", "content": [{"type": "text", "text": "Keep it short."}]},
{"role": "assistant", "content": "Hi."},
]
)
assert result == (
{"role": "user", "content": "hi"},
{
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Keep it short."},
],
},
{"role": "assistant", "content": "Hi."},
)
def test_convert_mid_conversation_system_turns_wraps_string_content():
result = convert_mid_conversation_system_turns(
[
{"role": "user", "content": "hi"},
{"role": "system", "content": "Keep it short."},
]
)
assert result[1] == {
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Keep it short."},
],
}
def test_convert_mid_conversation_system_turns_moves_system_after_tool_result():
assistant_tool_use = {
"role": "assistant",
"content": [{"type": "tool_use", "id": "toolu_1", "name": "get_weather", "input": {}}],
}
wedged_system = {"role": "system", "content": "Use the corrected result."}
tool_result = {
"role": "user",
"content": [{"type": "tool_result", "tool_use_id": "toolu_1", "content": "Rainy"}],
}
result = convert_mid_conversation_system_turns([assistant_tool_use, wedged_system, tool_result])
assert result[0] is assistant_tool_use
assert result[1] is tool_result
assert result[2]["role"] == "user"
assert result[2]["content"][0]["text"] == CONVERTED_SYSTEM_NOTE

View file

@ -23,6 +23,9 @@ from litellm.constants import (
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
as_system_content_blocks,
)
from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_transformation import (
AmazonAnthropicClaudeMessagesConfig,
AmazonAnthropicClaudeMessagesStreamDecoder,
@ -2533,20 +2536,16 @@ def test_bedrock_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_
def test_as_system_content_blocks_handles_each_shape():
"""``_as_system_content_blocks`` normalizes every system shape: ``None`` -> empty,
"""``as_system_content_blocks`` normalizes every system shape: ``None`` -> empty,
a string -> a single text block, a list -> a shallow copy, and any other value
(e.g. a bare content-block dict) -> wrapped in a single-element list."""
block = {"type": "text", "text": "x"}
assert AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks(None) == []
assert AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks("hello") == [
{"type": "text", "text": "hello"}
]
assert as_system_content_blocks(None) == []
assert as_system_content_blocks("hello") == [{"type": "text", "text": "hello"}]
blocks = [block]
out = AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks(blocks)
out = as_system_content_blocks(blocks)
assert out == blocks and out is not blocks
assert AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks(block) == [
block
]
assert as_system_content_blocks(block) == [block]
@pytest.mark.parametrize(