fix(anthropic-bridge): keep mid-conversation system turns when the target declares supports_mid_conversation_system

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
yassin 2026-09-17 18:06:55 +00:00
parent 6c9f258658
commit 99667ad633
3 changed files with 79 additions and 25 deletions

View file

@ -180,6 +180,7 @@ from litellm.types.llms.openai import (
ToolMessageContentPart,
)
from litellm.types.utils import Choices, ModelResponse, StreamingChoices, Usage
from litellm.utils import supports_mid_conversation_system
from .streaming_iterator import AnthropicStreamWrapper
@ -190,6 +191,12 @@ if TYPE_CHECKING:
ToolResultContent: TypeAlias = str | list[ToolMessageContentPart]
def target_supports_mid_conversation_system(model: str | None, custom_llm_provider: str | None) -> bool:
if not model:
return False
return supports_mid_conversation_system(model=model, custom_llm_provider=custom_llm_provider)
class AnthropicAdapter:
def __init__(self) -> None:
pass
@ -423,6 +430,7 @@ class LiteLLMAnthropicMessagesAdapter:
messages: list[AllAnthropicPassThroughMessageValues],
model: str | None = None,
*,
custom_llm_provider: str | None = None,
preserve_midturn_system: bool = False,
) -> list:
new_messages: Final[list[AllMessageValues]] = []
@ -431,13 +439,16 @@ class LiteLLMAnthropicMessagesAdapter:
(i for i, m in enumerate(replayable_messages) if not is_system_role_message(m)),
len(replayable_messages),
)
trailing_messages: Final = replayable_messages[leading_count:]
keeps_midturn_system: Final = (
preserve_midturn_system
or not any(is_system_role_message(m) for m in trailing_messages)
or target_supports_mid_conversation_system(model, custom_llm_provider)
)
ordered_messages: Final = (
replayable_messages
if preserve_midturn_system
else (
*replayable_messages[:leading_count],
*convert_mid_conversation_system_turns(replayable_messages[leading_count:]),
)
if keeps_midturn_system
else (*replayable_messages[:leading_count], *convert_mid_conversation_system_turns(trailing_messages))
)
for m in ordered_messages:
user_message: ChatCompletionUserMessage | None = None
@ -1194,6 +1205,7 @@ class LiteLLMAnthropicMessagesAdapter:
new_messages = self.translate_anthropic_messages_to_openai(
messages=messages_list,
model=anthropic_message_request.get("model"),
custom_llm_provider=custom_llm_provider,
preserve_midturn_system=preserve_midturn_system,
)
## ADD SYSTEM MESSAGE TO MESSAGES

View file

@ -2885,6 +2885,15 @@ def supports_none_reasoning_effort(model: str, custom_llm_provider: str | None =
return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_none_reasoning_effort")
def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model accepts a system role message after the leading system block and return a boolean value.
"""
return _supports_factory(
model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system"
)
def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model supports native structured outputs and return a boolean value.

View file

@ -801,29 +801,32 @@ def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
]
def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn():
_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST: Final = {
"max_tokens": 128,
"system": [{"type": "text", "text": "You are Claude Code."}],
"messages": [
{"role": "user", "content": "say hi"},
{
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}],
},
{"role": "assistant", "content": "Hi."},
{"role": "user", "content": "say bye"},
],
}
@pytest.mark.parametrize("custom_llm_provider", [None, "hosted_vllm"])
def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(custom_llm_provider: str | None):
"""
Claude Code appends a system-role harness reminder after the user turn. On a
chat-completions target the outbound request must have exactly one system message,
at index 0, and the converted turn must carry the operator note first.
Claude Code appends a system-role harness reminder after the user turn. On a chat-completions
target that does not declare ``supports_mid_conversation_system`` (a self-hosted model the cost
map knows nothing about) the outbound request must have exactly one system message, at index 0,
and the converted turn must carry the operator note first.
"""
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={
"model": "qwen3.8-27B",
"max_tokens": 128,
"system": [{"type": "text", "text": "You are Claude Code."}],
"messages": [
{"role": "user", "content": "say hi"},
{
"role": "system",
"content": [
{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}
],
},
{"role": "assistant", "content": "Hi."},
{"role": "user", "content": "say bye"},
],
}
anthropic_message_request={"model": "qwen3.8-27B", **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
custom_llm_provider=custom_llm_provider,
)
roles = [m["role"] for m in openai_request["messages"]]
@ -833,6 +836,36 @@ def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(
assert converted["content"][1]["text"] == "<system-reminder>Keep answers to one sentence.</system-reminder>"
def test_translate_anthropic_to_openai_keeps_midturn_system_when_target_declares_support(monkeypatch):
"""
A chat-completions target flagged ``supports_mid_conversation_system`` in the cost map accepts
the role anywhere, so the harness reminder is forwarded in place with its role and content
untouched, the same rule the native Anthropic Messages path applies.
"""
model: Final = "system-role-anywhere-chat-model"
monkeypatch.setitem(
litellm.model_cost,
model,
{"litellm_provider": "openai", "mode": "chat", "supports_mid_conversation_system": True},
)
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
anthropic_message_request={"model": model, **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
custom_llm_provider="openai",
)
assert openai_request["messages"] == [
{"role": "system", "content": [{"type": "text", "text": "You are Claude Code."}]},
{"role": "user", "content": "say hi"},
{
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}],
},
{"role": "assistant", "content": "Hi.", "thinking_blocks": None},
{"role": "user", "content": "say bye"},
]
def test_translate_anthropic_to_openai_moves_midturn_system_after_tool_result():
"""
A system entry wedged between an assistant tool_use turn and its tool_result turn is