Merge pull request #41493 from BerriAI/litellm_bridge_mid_conversation_system_turns

fix(anthropic-bridge): convert mid-conversation system turns to user turns on /v1/messages to chat completions
This commit is contained in:
Yassin Kortam 2026-09-17 14:25:36 -07:00 • committed by GitHub
commit 1b4739c415
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 385 additions and 95 deletions

View file

@ -508,7 +508,8 @@ class AnthropicMessagesHandler(BaseTranslation):
chat_completion_compatible_request,
_tool_name_mapping,
) = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request=cast(AnthropicMessagesRequest, data.copy())
anthropic_message_request=cast(AnthropicMessagesRequest, data.copy()),
preserve_midturn_system=True,
)
return chat_completion_compatible_request

View file

@ -118,6 +118,10 @@ from litellm.llms.anthropic.common_utils import (
from litellm.llms.anthropic.experimental_pass_through.context_management import (
PolyfillResult,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
convert_mid_conversation_system_turns,
is_system_role_message,
)
from litellm.llms.anthropic.experimental_pass_through.messages.utils import (
openai_chat_refusal_text,
refusal_stop_details,
@ -176,6 +180,7 @@ from litellm.types.llms.openai import (
ToolMessageContentPart,
)
from litellm.types.utils import Choices, ModelResponse, StreamingChoices, Usage
from litellm.utils import supports_mid_conversation_system
from .streaming_iterator import AnthropicStreamWrapper
@ -186,6 +191,12 @@ if TYPE_CHECKING:
ToolResultContent: TypeAlias = str | list[ToolMessageContentPart]
def target_supports_mid_conversation_system(model: str | None, custom_llm_provider: str | None) -> bool:
if not model:
return False
return supports_mid_conversation_system(model=model, custom_llm_provider=custom_llm_provider)
class AnthropicAdapter:
def __init__(self) -> None:
pass
@ -418,10 +429,28 @@ class LiteLLMAnthropicMessagesAdapter:
self,
messages: list[AllAnthropicPassThroughMessageValues],
model: str | None = None,
*,
custom_llm_provider: str | None = None,
preserve_midturn_system: bool = False,
) -> list:
new_messages: Final[list[AllMessageValues]] = []
replayable_messages: Final = strip_encrypted_reasoning_blocks_from_anthropic_messages(messages)
for m in replayable_messages:
leading_count: Final = next(
(i for i, m in enumerate(replayable_messages) if not is_system_role_message(m)),
len(replayable_messages),
)
trailing_messages: Final = replayable_messages[leading_count:]
keeps_midturn_system: Final = (
preserve_midturn_system
or not any(is_system_role_message(m) for m in trailing_messages)
or target_supports_mid_conversation_system(model, custom_llm_provider)
)
ordered_messages: Final = (
replayable_messages
if keeps_midturn_system
else (*replayable_messages[:leading_count], *convert_mid_conversation_system_turns(trailing_messages))
)
for m in ordered_messages:
user_message: ChatCompletionUserMessage | None = None
tool_message_list: list[ChatCompletionToolMessage] = []
new_user_content_list: list[ChatCompletionTextObject | ChatCompletionImageObject] = []
@ -494,7 +523,7 @@ class LiteLLMAnthropicMessagesAdapter:
if isinstance(m.get("content"), str):
assistant_message_str = str(m.get("content", ""))
elif isinstance(m.get("content"), list):
for content in m.get("content", []):
for content in cast(list, m.get("content", [])): # cast-ok: untrusted client payload
if isinstance(content, str):
assistant_message_str = str(content)
elif isinstance(content, dict):
@ -1154,6 +1183,7 @@ class LiteLLMAnthropicMessagesAdapter:
anthropic_message_request: AnthropicMessagesRequest,
*,
custom_llm_provider: str | None = None,
preserve_midturn_system: bool = False,
) -> tuple[ChatCompletionRequest, dict[str, str]]:
"""
This is used by the beta Anthropic Adapter, for translating anthropic `/v1/messages` requests to the openai format.
@ -1175,6 +1205,8 @@ class LiteLLMAnthropicMessagesAdapter:
new_messages = self.translate_anthropic_messages_to_openai(
messages=messages_list,
model=anthropic_message_request.get("model"),
custom_llm_provider=custom_llm_provider,
preserve_midturn_system=preserve_midturn_system,
)
## ADD SYSTEM MESSAGE TO MESSAGES
self._add_system_message_to_messages(new_messages, anthropic_message_request)

View file

@ -765,7 +765,8 @@ def _count_effective_tokens(
messages=cast(
"list[AllAnthropicPassThroughMessageValues]",
messages_without_compaction,
)
),
preserve_midturn_system=True,
)
except Exception as e:
verbose_logger.debug(
@ -920,7 +921,8 @@ def _build_summary_messages(
messages=cast(
"list[AllAnthropicPassThroughMessageValues]",
stripped,
)
),
preserve_midturn_system=True,
)
except Exception as e:
verbose_logger.warning(

View file

@ -0,0 +1,77 @@
from collections.abc import Mapping, Sequence
from itertools import groupby
from typing import Final
CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
def as_system_content_blocks(value: object) -> list[object]:
if value is None:
return []
if isinstance(value, list):
return list(value)
if isinstance(value, str):
return [{"type": "text", "text": value}]
return [value]
def is_system_role_message(message: object) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
def system_role_message_as_user(message: Mapping[str, object]) -> Mapping[str, object]:
return {
"role": "user",
"content": as_system_content_blocks(CONVERTED_SYSTEM_NOTE) + as_system_content_blocks(message.get("content")),
}
def opens_with_tool_results(message: object) -> bool:
if not isinstance(message, dict) or message.get("role") != "user":
return False
content: Final = message.get("content")
return (
isinstance(content, list)
and len(content) > 0
and isinstance(content[0], dict)
and content[0].get("type") == "tool_result"
)
def system_run_placed_after_tool_results(
system_run: Sequence[Mapping[str, object]], follower_run: Sequence[Mapping[str, object]]
) -> tuple[Mapping[str, object], ...]:
if follower_run and opens_with_tool_results(follower_run[0]):
return (follower_run[0], *system_run, *follower_run[1:])
return (*system_run, *follower_run)
def system_turns_after_tool_results(
messages: Sequence[Mapping[str, object]],
) -> tuple[Mapping[str, object], ...]:
runs: Final = tuple(tuple(run) for _, run in groupby(messages, key=is_system_role_message))
if not runs:
return ()
first_system_run: Final = 0 if is_system_role_message(runs[0][0]) else 1
paired_runs: Final = tuple(
(runs[i], runs[i + 1] if i + 1 < len(runs) else ()) for i in range(first_system_run, len(runs), 2)
)
return (
*(runs[0] if first_system_run else ()),
*(
m
for system_run, follower_run in paired_runs
for m in system_run_placed_after_tool_results(system_run, follower_run)
),
)
def convert_mid_conversation_system_turns(
messages: Sequence[Mapping[str, object]],
) -> tuple[Mapping[str, object], ...]:
return tuple(
system_role_message_as_user(m) if is_system_role_message(m) else m
for m in system_turns_after_tool_results(messages)
)

View file

@ -27,6 +27,11 @@ from ...common_utils import (
strip_advisor_blocks_from_messages,
strip_encrypted_reasoning_blocks_from_anthropic_messages,
)
from .mid_conversation_system import (
as_system_content_blocks,
convert_mid_conversation_system_turns,
is_system_role_message,
)
DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01"
@ -151,73 +156,6 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
else:
return system_param
@staticmethod
def _as_system_content_blocks(value: object) -> list:
if value is None:
return []
if isinstance(value, list):
return list(value)
if isinstance(value, str):
return [{"type": "text", "text": value}]
return [value]
@staticmethod
def _is_system_role_message(message: object) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
_CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
def _system_role_message_as_user(self, message: Mapping) -> Mapping:
return {
"role": "user",
"content": self._as_system_content_blocks(self._CONVERTED_SYSTEM_NOTE)
+ self._as_system_content_blocks(message.get("content")),
}
@staticmethod
def _opens_with_tool_results(message: object) -> bool:
if not isinstance(message, dict) or message.get("role") != "user":
return False
content: Final = message.get("content")
return (
isinstance(content, list)
and len(content) > 0
and isinstance(content[0], dict)
and content[0].get("type") == "tool_result"
)
def _system_run_before(self, messages: Sequence, index: int) -> Sequence:
start: Final = next(
(j + 1 for j in range(index - 1, -1, -1) if not self._is_system_role_message(messages[j])),
0,
)
return messages[start:index]
def _system_run_end(self, messages: Sequence, index: int) -> int:
return next(
(j for j in range(index, len(messages)) if not self._is_system_role_message(messages[j])),
len(messages),
)
def _reordered_around_tool_results(self, messages: Sequence, index: int) -> tuple:
message: Final = messages[index]
if self._opens_with_tool_results(message):
return (message, *self._system_run_before(messages, index))
if not self._is_system_role_message(message):
return (message,)
run_end: Final = self._system_run_end(messages, index)
follower: Final = messages[run_end] if run_end < len(messages) else None
return () if self._opens_with_tool_results(follower) else (message,)
def _system_turns_after_tool_results(self, messages: Sequence) -> tuple:
return tuple(
message
for index in range(len(messages))
for message in self._reordered_around_tool_results(messages, index)
)
def _normalize_system_role_messages(self, anthropic_messages_request: dict, model: str) -> None:
"""Normalize ``role: "system"`` entries in ``messages`` per the Anthropic
``/v1/messages`` contract, which the first-party API, Bedrock Invoke,
@ -254,7 +192,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
if not isinstance(messages, list):
return
leading_count: Final = next(
(i for i, m in enumerate(messages) if not self._is_system_role_message(m)),
(i for i, m in enumerate(messages) if not is_system_role_message(m)),
len(messages),
)
hoisted: Final = messages[:leading_count]
@ -265,10 +203,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
custom_llm_provider=self.custom_llm_provider,
key="supports_mid_conversation_system",
)
else [
self._system_role_message_as_user(m) if self._is_system_role_message(m) else m
for m in self._system_turns_after_tool_results(messages[leading_count:])
]
else list(convert_mid_conversation_system_turns(messages[leading_count:]))
)
if hoisted or remaining != messages:
anthropic_messages_request["messages"] = remaining
@ -278,7 +213,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
anthropic_messages_request.get("system"),
*(m.get("content") for m in hoisted),
)
for block in self._as_system_content_blocks(source)
for block in as_system_content_blocks(source)
]
filtered_system: Final = self._filter_billing_headers_from_system(system_content)
if filtered_system:

View file

@ -2916,6 +2916,15 @@ def supports_none_reasoning_effort(model: str, custom_llm_provider: str | None =
return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_none_reasoning_effort")
def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model accepts a system role message after the leading system block and return a boolean value.
"""
return _supports_factory(
model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system"
)
def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model supports native structured outputs and return a boolean value.

View file

@ -23,6 +23,9 @@ from litellm.llms.anthropic.experimental_pass_through.adapters.transformation im
create_tool_name_mapping,
truncate_tool_name,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
CONVERTED_SYSTEM_NOTE,
)
from litellm.llms.openai.chat.gpt_transformation import OpenAIGPTConfig
from litellm.types.llms.anthropic import (
AnthopicMessagesAssistantMessageParam,
@ -563,10 +566,19 @@ def test_translate_anthropic_messages_to_openai_tool_message_placement():
@pytest.mark.parametrize(
("system_content", "expected_content"),
[
("Use the corrected result.", "Use the corrected result."),
(
"Use the corrected result.",
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Use the corrected result."},
],
),
(
[{"type": "text", "text": "Use the corrected result."}],
[{"type": "text", "text": "Use the corrected result."}],
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Use the corrected result."},
],
),
(
[
@ -576,7 +588,11 @@ def test_translate_anthropic_messages_to_openai_tool_message_placement():
},
{"type": "text", "text": "Use the corrected result."},
],
[{"type": "text", "text": "Use the corrected result."}],
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "image_url", "image_url": {"url": "https://example.com/a.png"}},
{"type": "text", "text": "Use the corrected result."},
],
),
(
[
@ -584,13 +600,14 @@ def test_translate_anthropic_messages_to_openai_tool_message_placement():
{"type": "text", "text": "Second correction."},
],
[
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "First correction."},
{"type": "text", "text": "Second correction."},
],
),
],
)
def test_translate_anthropic_messages_to_openai_preserves_midturn_system_correction(
def test_translate_anthropic_messages_to_openai_converts_midturn_system_correction(
system_content: object,
expected_content: object,
):
@ -646,7 +663,7 @@ def test_translate_anthropic_messages_to_openai_preserves_midturn_system_correct
"tool_call_id": "toolu_01234",
"content": "Rainy, 55°F",
},
{"role": "system", "content": expected_content},
{"role": "user", "content": expected_content},
{"role": "user", "content": "Continue."},
]
@ -752,8 +769,8 @@ def test_translate_anthropic_messages_to_openai_drops_empty_midturn_system(
def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
"""
Request level: the trusted top-level prompt is hoisted to index 0 exactly once and the
in-sequence correction keeps its own position and `role: "system"` -- no duplication of
either, and no reordering of the surrounding turns.
in-sequence correction keeps its own position as a user turn prefixed with the operator
note -- no duplication of either, and no reordering of the surrounding turns.
"""
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={
@ -773,11 +790,140 @@ def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
{"role": "system", "content": "Trusted top-level prompt."},
{"role": "user", "content": "First question."},
{"role": "assistant", "content": "First answer.", "thinking_blocks": None},
{"role": "system", "content": "Use the corrected result."},
{
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Use the corrected result."},
],
},
{"role": "user", "content": "Continue."},
]
_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST: Final = {
"max_tokens": 128,
"system": [{"type": "text", "text": "You are Claude Code."}],
"messages": [
{"role": "user", "content": "say hi"},
{
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}],
},
{"role": "assistant", "content": "Hi."},
{"role": "user", "content": "say bye"},
],
}
@pytest.mark.parametrize("custom_llm_provider", [None, "hosted_vllm"])
def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(custom_llm_provider: str | None):
"""
Claude Code appends a system-role harness reminder after the user turn. On a chat-completions
target that does not declare ``supports_mid_conversation_system`` (a self-hosted model the cost
map knows nothing about) the outbound request must have exactly one system message, at index 0,
and the converted turn must carry the operator note first.
"""
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={"model": "qwen3.8-27B", **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
custom_llm_provider=custom_llm_provider,
)
roles = [m["role"] for m in openai_request["messages"]]
assert roles == ["system", "user", "user", "assistant", "user"]
converted = openai_request["messages"][2]
assert converted["content"][0]["text"] == CONVERTED_SYSTEM_NOTE
assert converted["content"][1]["text"] == "<system-reminder>Keep answers to one sentence.</system-reminder>"
def test_translate_anthropic_to_openai_keeps_midturn_system_when_target_declares_support(monkeypatch):
"""
A chat-completions target flagged ``supports_mid_conversation_system`` in the cost map accepts
the role anywhere, so the harness reminder is forwarded in place with its role and content
untouched, the same rule the native Anthropic Messages path applies.
"""
model: Final = "system-role-anywhere-chat-model"
monkeypatch.setitem(
litellm.model_cost,
model,
{"litellm_provider": "openai", "mode": "chat", "supports_mid_conversation_system": True},
)
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={"model": model, **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
custom_llm_provider="openai",
)
assert openai_request["messages"] == [
{"role": "system", "content": [{"type": "text", "text": "You are Claude Code."}]},
{"role": "user", "content": "say hi"},
{
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}],
},
{"role": "assistant", "content": "Hi.", "thinking_blocks": None},
{"role": "user", "content": "say bye"},
]
def test_translate_anthropic_to_openai_moves_midturn_system_after_tool_result():
"""
A system entry wedged between an assistant tool_use turn and its tool_result turn is
emitted after the role: "tool" message, so the tool call stays paired with its result.
"""
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
messages=[
{
"role": "assistant",
"content": [
{
"type": "tool_use",
"id": "toolu_01234",
"name": "get_weather",
"input": {"location": "Boston"},
}
],
},
{"role": "system", "content": "Use the corrected result."},
{
"role": "user",
"content": [
{
"type": "tool_result",
"tool_use_id": "toolu_01234",
"content": "Rainy, 55°F",
}
],
},
],
model="claude-3-5-sonnet-20240620",
)
assert [m["role"] for m in result] == ["assistant", "tool", "user"]
assert result[2]["content"][0]["text"] == CONVERTED_SYSTEM_NOTE
def test_translate_anthropic_messages_to_openai_converts_string_midturn_system():
result = LiteLLMAnthropicMessagesAdapter().translate_anthropic_messages_to_openai(
messages=[
{"role": "user", "content": "hi"},
{"role": "system", "content": "Keep it short."},
],
model="claude-3-5-sonnet-20240620",
)
assert result == [
{"role": "user", "content": "hi"},
{
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Keep it short."},
],
},
]
def _claude_code_user_id(session_id: str) -> str:
return json.dumps({"device_id": "d" * 64, "account_uuid": "", "session_id": session_id})

View file

@ -0,0 +1,89 @@
from collections import Counter
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
CONVERTED_SYSTEM_NOTE,
convert_mid_conversation_system_turns,
)
class RoleReadCountingMessage(dict):
def __init__(self, role: str, content: object, reads: Counter):
super().__init__(role=role, content=content)
self.reads = reads
def get(self, key, default=None):
self.reads[key] += 1
return super().get(key, default)
def test_convert_mid_conversation_system_turns_converts_system_to_user_in_place():
result = convert_mid_conversation_system_turns(
[
{"role": "user", "content": "hi"},
{"role": "system", "content": [{"type": "text", "text": "Keep it short."}]},
{"role": "assistant", "content": "Hi."},
]
)
assert result == (
{"role": "user", "content": "hi"},
{
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Keep it short."},
],
},
{"role": "assistant", "content": "Hi."},
)
def test_convert_mid_conversation_system_turns_wraps_string_content():
result = convert_mid_conversation_system_turns(
[
{"role": "user", "content": "hi"},
{"role": "system", "content": "Keep it short."},
]
)
assert result[1] == {
"role": "user",
"content": [
{"type": "text", "text": CONVERTED_SYSTEM_NOTE},
{"type": "text", "text": "Keep it short."},
],
}
def test_convert_mid_conversation_system_turns_moves_system_after_tool_result():
assistant_tool_use = {
"role": "assistant",
"content": [{"type": "tool_use", "id": "toolu_1", "name": "get_weather", "input": {}}],
}
wedged_system = {"role": "system", "content": "Use the corrected result."}
tool_result = {
"role": "user",
"content": [{"type": "tool_result", "tool_use_id": "toolu_1", "content": "Rainy"}],
}
result = convert_mid_conversation_system_turns([assistant_tool_use, wedged_system, tool_result])
assert result[0] is assistant_tool_use
assert result[1] is tool_result
assert result[2]["role"] == "user"
assert result[2]["content"][0]["text"] == CONVERTED_SYSTEM_NOTE
def test_convert_mid_conversation_system_turns_reads_each_role_a_bounded_number_of_times():
reads = Counter()
system_run = [RoleReadCountingMessage("system", f"reminder {i}", reads) for i in range(2_000)]
tool_result = RoleReadCountingMessage(
"user", [{"type": "tool_result", "tool_use_id": "toolu_1", "content": "Rainy"}], reads
)
messages = [RoleReadCountingMessage("user", "hi", reads), *system_run, tool_result]
result = convert_mid_conversation_system_turns(messages)
assert reads["role"] <= 3 * len(messages)
assert result[1] is tool_result
assert [m["content"][1]["text"] for m in result[2:]] == [m["content"] for m in system_run]

View file

@ -23,6 +23,9 @@ from litellm.constants import (
DEFAULT_REASONING_EFFORT_MEDIUM_THINKING_BUDGET,
DEFAULT_REASONING_EFFORT_XHIGH_THINKING_BUDGET,
)
from litellm.llms.anthropic.experimental_pass_through.messages.mid_conversation_system import (
as_system_content_blocks,
)
from litellm.llms.bedrock.messages.invoke_transformations.anthropic_claude3_transformation import (
AmazonAnthropicClaudeMessagesConfig,
AmazonAnthropicClaudeMessagesStreamDecoder,
@ -2533,20 +2536,16 @@ def test_bedrock_claude_4_8_plus_cost_map_entries_carry_mid_conversation_system_
def test_as_system_content_blocks_handles_each_shape():
"""``_as_system_content_blocks`` normalizes every system shape: ``None`` -> empty,
"""``as_system_content_blocks`` normalizes every system shape: ``None`` -> empty,
a string -> a single text block, a list -> a shallow copy, and any other value
(e.g. a bare content-block dict) -> wrapped in a single-element list."""
block = {"type": "text", "text": "x"}
assert AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks(None) == []
assert AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks("hello") == [
{"type": "text", "text": "hello"}
]
assert as_system_content_blocks(None) == []
assert as_system_content_blocks("hello") == [{"type": "text", "text": "hello"}]
blocks = [block]
out = AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks(blocks)
out = as_system_content_blocks(blocks)
assert out == blocks and out is not blocks
assert AmazonAnthropicClaudeMessagesConfig._as_system_content_blocks(block) == [
block
]
assert as_system_content_blocks(block) == [block]
@pytest.mark.parametrize(