mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
fix(anthropic-bridge): keep mid-conversation system turns when the target declares supports_mid_conversation_system
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
6c9f258658
commit
99667ad633
3 changed files with 79 additions and 25 deletions
|
|
@ -180,6 +180,7 @@ from litellm.types.llms.openai import (
|
|||
ToolMessageContentPart,
|
||||
)
|
||||
from litellm.types.utils import Choices, ModelResponse, StreamingChoices, Usage
|
||||
from litellm.utils import supports_mid_conversation_system
|
||||
|
||||
from .streaming_iterator import AnthropicStreamWrapper
|
||||
|
||||
|
|
@ -190,6 +191,12 @@ if TYPE_CHECKING:
|
|||
ToolResultContent: TypeAlias = str | list[ToolMessageContentPart]
|
||||
|
||||
|
||||
def target_supports_mid_conversation_system(model: str | None, custom_llm_provider: str | None) -> bool:
|
||||
if not model:
|
||||
return False
|
||||
return supports_mid_conversation_system(model=model, custom_llm_provider=custom_llm_provider)
|
||||
|
||||
|
||||
class AnthropicAdapter:
|
||||
def __init__(self) -> None:
|
||||
pass
|
||||
|
|
@ -423,6 +430,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
messages: list[AllAnthropicPassThroughMessageValues],
|
||||
model: str | None = None,
|
||||
*,
|
||||
custom_llm_provider: str | None = None,
|
||||
preserve_midturn_system: bool = False,
|
||||
) -> list:
|
||||
new_messages: Final[list[AllMessageValues]] = []
|
||||
|
|
@ -431,13 +439,16 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
(i for i, m in enumerate(replayable_messages) if not is_system_role_message(m)),
|
||||
len(replayable_messages),
|
||||
)
|
||||
trailing_messages: Final = replayable_messages[leading_count:]
|
||||
keeps_midturn_system: Final = (
|
||||
preserve_midturn_system
|
||||
or not any(is_system_role_message(m) for m in trailing_messages)
|
||||
or target_supports_mid_conversation_system(model, custom_llm_provider)
|
||||
)
|
||||
ordered_messages: Final = (
|
||||
replayable_messages
|
||||
if preserve_midturn_system
|
||||
else (
|
||||
*replayable_messages[:leading_count],
|
||||
*convert_mid_conversation_system_turns(replayable_messages[leading_count:]),
|
||||
)
|
||||
if keeps_midturn_system
|
||||
else (*replayable_messages[:leading_count], *convert_mid_conversation_system_turns(trailing_messages))
|
||||
)
|
||||
for m in ordered_messages:
|
||||
user_message: ChatCompletionUserMessage | None = None
|
||||
|
|
@ -1194,6 +1205,7 @@ class LiteLLMAnthropicMessagesAdapter:
|
|||
new_messages = self.translate_anthropic_messages_to_openai(
|
||||
messages=messages_list,
|
||||
model=anthropic_message_request.get("model"),
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
preserve_midturn_system=preserve_midturn_system,
|
||||
)
|
||||
## ADD SYSTEM MESSAGE TO MESSAGES
|
||||
|
|
|
|||
|
|
@ -2885,6 +2885,15 @@ def supports_none_reasoning_effort(model: str, custom_llm_provider: str | None =
|
|||
return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_none_reasoning_effort")
|
||||
|
||||
|
||||
def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool:
|
||||
"""
|
||||
Check if the given model accepts a system role message after the leading system block and return a boolean value.
|
||||
"""
|
||||
return _supports_factory(
|
||||
model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system"
|
||||
)
|
||||
|
||||
|
||||
def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool:
|
||||
"""
|
||||
Check if the given model supports native structured outputs and return a boolean value.
|
||||
|
|
|
|||
|
|
@ -801,29 +801,32 @@ def test_translate_anthropic_to_openai_orders_top_level_and_midturn_system():
|
|||
]
|
||||
|
||||
|
||||
def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn():
|
||||
_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST: Final = {
|
||||
"max_tokens": 128,
|
||||
"system": [{"type": "text", "text": "You are Claude Code."}],
|
||||
"messages": [
|
||||
{"role": "user", "content": "say hi"},
|
||||
{
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}],
|
||||
},
|
||||
{"role": "assistant", "content": "Hi."},
|
||||
{"role": "user", "content": "say bye"},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("custom_llm_provider", [None, "hosted_vllm"])
|
||||
def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(custom_llm_provider: str | None):
|
||||
"""
|
||||
Claude Code appends a system-role harness reminder after the user turn. On a
|
||||
chat-completions target the outbound request must have exactly one system message,
|
||||
at index 0, and the converted turn must carry the operator note first.
|
||||
Claude Code appends a system-role harness reminder after the user turn. On a chat-completions
|
||||
target that does not declare ``supports_mid_conversation_system`` (a self-hosted model the cost
|
||||
map knows nothing about) the outbound request must have exactly one system message, at index 0,
|
||||
and the converted turn must carry the operator note first.
|
||||
"""
|
||||
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
|
||||
anthropic_message_request={
|
||||
"model": "qwen3.8-27B",
|
||||
"max_tokens": 128,
|
||||
"system": [{"type": "text", "text": "You are Claude Code."}],
|
||||
"messages": [
|
||||
{"role": "user", "content": "say hi"},
|
||||
{
|
||||
"role": "system",
|
||||
"content": [
|
||||
{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}
|
||||
],
|
||||
},
|
||||
{"role": "assistant", "content": "Hi."},
|
||||
{"role": "user", "content": "say bye"},
|
||||
],
|
||||
}
|
||||
anthropic_message_request={"model": "qwen3.8-27B", **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
roles = [m["role"] for m in openai_request["messages"]]
|
||||
|
|
@ -833,6 +836,36 @@ def test_translate_anthropic_to_openai_converts_claude_code_midturn_system_turn(
|
|||
assert converted["content"][1]["text"] == "<system-reminder>Keep answers to one sentence.</system-reminder>"
|
||||
|
||||
|
||||
def test_translate_anthropic_to_openai_keeps_midturn_system_when_target_declares_support(monkeypatch):
|
||||
"""
|
||||
A chat-completions target flagged ``supports_mid_conversation_system`` in the cost map accepts
|
||||
the role anywhere, so the harness reminder is forwarded in place with its role and content
|
||||
untouched, the same rule the native Anthropic Messages path applies.
|
||||
"""
|
||||
model: Final = "system-role-anywhere-chat-model"
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
model,
|
||||
{"litellm_provider": "openai", "mode": "chat", "supports_mid_conversation_system": True},
|
||||
)
|
||||
|
||||
openai_request, _ = LiteLLMAnthropicMessagesAdapter().translate_anthropic_to_openai(
|
||||
anthropic_message_request={"model": model, **_CLAUDE_CODE_MIDTURN_SYSTEM_REQUEST},
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
assert openai_request["messages"] == [
|
||||
{"role": "system", "content": [{"type": "text", "text": "You are Claude Code."}]},
|
||||
{"role": "user", "content": "say hi"},
|
||||
{
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "<system-reminder>Keep answers to one sentence.</system-reminder>"}],
|
||||
},
|
||||
{"role": "assistant", "content": "Hi.", "thinking_blocks": None},
|
||||
{"role": "user", "content": "say bye"},
|
||||
]
|
||||
|
||||
|
||||
def test_translate_anthropic_to_openai_moves_midturn_system_after_tool_result():
|
||||
"""
|
||||
A system entry wedged between an assistant tool_use turn and its tool_result turn is
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue