diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
index 10ba2431bcc..864bb9b99ee 100644
--- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
+++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py
@@ -180,6 +180,7 @@ from litellm.types.llms.openai import (
ToolMessageContentPart,
)
from litellm.types.utils import Choices, ModelResponse, StreamingChoices, Usage
+from litellm.utils import supports_mid_conversation_system
from .streaming_iterator import AnthropicStreamWrapper
@@ -190,6 +191,12 @@ if TYPE_CHECKING:
ToolResultContent: TypeAlias = str | list[ToolMessageContentPart]
+def target_supports_mid_conversation_system(model: str | None, custom_llm_provider: str | None) -> bool:
+ if not model:
+ return False
+ return supports_mid_conversation_system(model=model, custom_llm_provider=custom_llm_provider)
+
+
class AnthropicAdapter:
def __init__(self) -> None:
pass
@@ -423,6 +430,7 @@ class LiteLLMAnthropicMessagesAdapter:
messages: list[AllAnthropicPassThroughMessageValues],
model: str | None = None,
*,
+ custom_llm_provider: str | None = None,
preserve_midturn_system: bool = False,
) -> list:
new_messages: Final[list[AllMessageValues]] = []
@@ -431,13 +439,16 @@ class LiteLLMAnthropicMessagesAdapter:
(i for i, m in enumerate(replayable_messages) if not is_system_role_message(m)),
len(replayable_messages),
)
+ trailing_messages: Final = replayable_messages[leading_count:]
+ keeps_midturn_system: Final = (
+ preserve_midturn_system
+ or not any(is_system_role_message(m) for m in trailing_messages)
+ or target_supports_mid_conversation_system(model, custom_llm_provider)
+ )
ordered_messages: Final = (
replayable_messages
- if preserve_midturn_system
- else (
- *replayable_messages[:leading_count],
- *convert_mid_conversation_system_turns(replayable_messages[leading_count:]),
- )
+ if keeps_midturn_system
+ else (*replayable_messages[:leading_count], *convert_mid_conversation_system_turns(trailing_messages))
)
for m in ordered_messages:
user_message: ChatCompletionUserMessage | None = None
@@ -1194,6 +1205,7 @@ class LiteLLMAnthropicMessagesAdapter:
new_messages = self.translate_anthropic_messages_to_openai(
messages=messages_list,
model=anthropic_message_request.get("model"),
+ custom_llm_provider=custom_llm_provider,
preserve_midturn_system=preserve_midturn_system,
)
## ADD SYSTEM MESSAGE TO MESSAGES
diff --git a/litellm/utils.py b/litellm/utils.py
index 734522c0c6a..cfeb6f4d75f 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -2885,6 +2885,15 @@ def supports_none_reasoning_effort(model: str, custom_llm_provider: str | None =
return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_none_reasoning_effort")
+def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool:
+ """
+ Check if the given model accepts a system role message after the leading system block and return a boolean value.
+ """
+ return _supports_factory(
+ model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system"
+ )
+
+
def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model supports native structured outputs and return a boolean value.
diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py
index ad98a817a1a..e6782b70d3e 100644
--- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py
+++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py
@@ -801,29 +801,32 @@ def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
]
-def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn():
+_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST: Final = {
+ "max_tokens": 128,
+ "system": [{"type": "text", "text": "You are Claude Code."}],
+ "messages": [
+ {"role": "user", "content": "say hi"},
+ {
+ "role": "system",
+ "content": [{"type": "text", "text": "Keep answers to one sentence."}],
+ },
+ {"role": "assistant", "content": "Hi."},
+ {"role": "user", "content": "say bye"},
+ ],
+}
+
+
+@pytest.mark.parametrize("custom_llm_provider", [None, "hosted_vllm"])
+def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(custom_llm_provider: str | None):
"""
- Claude Code appends a system-role harness reminder after the user turn. On a
- chat-completions target the outbound request must have exactly one system message,
- at index 0, and the converted turn must carry the operator note first.
+ Claude Code appends a system-role harness reminder after the user turn. On a chat-completions
+ target that does not declare ``supports_mid_conversation_system`` (a self-hosted model the cost
+ map knows nothing about) the outbound request must have exactly one system message, at index 0,
+ and the converted turn must carry the operator note first.
"""
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
- anthropic_message_request={
- "model": "qwen3.8-27B",
- "max_tokens": 128,
- "system": [{"type": "text", "text": "You are Claude Code."}],
- "messages": [
- {"role": "user", "content": "say hi"},
- {
- "role": "system",
- "content": [
- {"type": "text", "text": "Keep answers to one sentence."}
- ],
- },
- {"role": "assistant", "content": "Hi."},
- {"role": "user", "content": "say bye"},
- ],
- }
+ anthropic_message_request={"model": "qwen3.8-27B", **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
+ custom_llm_provider=custom_llm_provider,
)
roles = [m["role"] for m in openai_request["messages"]]
@@ -833,6 +836,36 @@ def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(
assert converted["content"][1]["text"] == "Keep answers to one sentence."
+def test_translate_anthropic_to_openai_keeps_midturn_system_when_target_declares_support(monkeypatch):
+ """
+ A chat-completions target flagged ``supports_mid_conversation_system`` in the cost map accepts
+ the role anywhere, so the harness reminder is forwarded in place with its role and content
+ untouched, the same rule the native Anthropic Messages path applies.
+ """
+ model: Final = "system-role-anywhere-chat-model"
+ monkeypatch.setitem(
+ litellm.model_cost,
+ model,
+ {"litellm_provider": "openai", "mode": "chat", "supports_mid_conversation_system": True},
+ )
+
+ openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
+ anthropic_message_request={"model": model, **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
+ custom_llm_provider="openai",
+ )
+
+ assert openai_request["messages"] == [
+ {"role": "system", "content": [{"type": "text", "text": "You are Claude Code."}]},
+ {"role": "user", "content": "say hi"},
+ {
+ "role": "system",
+ "content": [{"type": "text", "text": "Keep answers to one sentence."}],
+ },
+ {"role": "assistant", "content": "Hi.", "thinking_blocks": None},
+ {"role": "user", "content": "say bye"},
+ ]
+
+
def test_translate_anthropic_to_openai_moves_midturn_system_after_tool_result():
"""
A system entry wedged between an assistant tool_use turn and its tool_result turn is