fix(responses-bridge): fold list-format system content into instructions again

convert_chat_completion_messages_to_responses_api only folded a leading
system message into `instructions` when its content was a plain str.
List-format content ([{"type": "text", "text": "..."}]), how a client
attaching cache_control sends a system prompt and how the Anthropic
/v1/messages adapter hands one over, fell through to a `role: "system"`
input item instead, which some Responses backends reject outright.

This is a regression of #21192: that PR added the list-content fold and
it was lost in a later merge, leaving only the str branch. Restore it
within the leading run of system messages, matching #40269's rule that
a system message after the first non-system message stays a positioned
input item so byte-identical prompts keep hitting the cache. A leading
list whose blocks are not all plain text (e.g. an image) still falls
through to an input item unchanged, rather than dropping the non-text
block silently.

Fixes #42171

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
poodle64 2026-09-21 12:56:32 +10:00
parent 3242bdfed2
commit a05cc57b4e
2 changed files with 122 additions and 7 deletions

View file

@ -388,13 +388,30 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
tool_call_id = msg.get("tool_call_id")
if role == "system":
# Extract system message as instructions
if isinstance(content, str) and index < leading_system_count:
# Extract system message as instructions, but only within the leading run of
# system messages: a later one stays a positioned input item so its bytes don't
# unsettle the prompt-cache-stable prefix (#40269).
extracted_instructions: str | None = None
if index < leading_system_count:
if isinstance(content, str):
extracted_instructions = content
elif isinstance(content, list) and all(
isinstance(block, str) or (isinstance(block, dict) and block.get("type") == "text")
for block in content
):
# Every block is plain text (a bare string or a `{"type": "text", ...}`
# dict), the same shape a client attaching cache_control sends; a
# non-text block (an image, say) fails the `all()` above and falls
# through to the input-item branch below instead of losing it silently.
extracted_instructions = " ".join(
block if isinstance(block, str) else block.get("text", "") for block in content
)
if extracted_instructions is not None:
if instructions:
# Concatenate multiple system prompts with a space
instructions = f"{instructions} {content}"
instructions = f"{instructions} {extracted_instructions}"
else:
instructions = content
instructions = extracted_instructions
else:
input_items.append(
{

View file

@ -4466,9 +4466,15 @@ def test_claude_code_shaped_history_keeps_a_byte_stable_input_prefix_across_requ
litellm_logging_obj=Mock(),
)
assert "instructions" not in first_request
assert "instructions" not in second_request
assert first_request["input"][0] == _system_input_item("You are Claude Code.")
# The leading, list-format top-level system prompt folds into `instructions` (#42171)
# and stays byte-identical across requests, rather than occupying `input[0]`.
assert first_request["instructions"] == "You are Claude Code."
assert second_request["instructions"] == first_request["instructions"]
assert first_request["input"][0] == {
"type": "message",
"role": "user",
"content": [{"type": "input_text", "text": "Read inventory.py."}],
}
assert json.dumps(second_request["input"][: len(first_request["input"])]) == json.dumps(first_request["input"])
assert second_request["input"][len(first_request["input"]) :] == [
{"type": "function_call", "call_id": "call_1", "name": "Read", "arguments": '{"file_path": "inventory.py"}'},
@ -4493,6 +4499,98 @@ def test_system_string_after_a_developer_message_stays_in_input_in_client_order(
assert input_items[1] == _system_input_item("Be brief.")
def test_leading_system_list_content_folds_into_instructions():
"""A leading system message whose content is a list of text blocks (how a client
attaching cache_control sends it) folds into `instructions` exactly like a plain string
one, instead of surfacing as a `system` input item that some Responses backends reject
outright (#42171)."""
handler: Final = LiteLLMResponsesTransformationHandler()
input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
[
{
"role": "system",
"content": [
{"type": "text", "text": "You are a helpful assistant.", "cache_control": {"type": "ephemeral"}}
],
},
{"role": "user", "content": "hi"},
]
)
assert instructions == "You are a helpful assistant."
assert input_items == [
{"type": "message", "role": "user", "content": [{"type": "input_text", "text": "hi"}]},
]
def test_leading_system_messages_mixed_str_and_list_concatenate_in_order():
"""Several leading system messages, some string and some list content, concatenate in
order the same way an all-string leading run does."""
handler: Final = LiteLLMResponsesTransformationHandler()
input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
[
{"role": "system", "content": "Be brief."},
{"role": "system", "content": [{"type": "text", "text": "Answer in French."}]},
{"role": "system", "content": "Never use emoji."},
{"role": "user", "content": "Bonjour"},
]
)
assert instructions == "Be brief. Answer in French. Never use emoji."
assert input_items == [
{"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Bonjour"}]},
]
def test_mid_conversation_system_list_content_stays_in_input_after_a_user_turn():
"""Only the leading run folds (#40269): a list-content system message after the first
non-system message stays a positioned input item."""
handler: Final = LiteLLMResponsesTransformationHandler()
input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
[
{"role": "user", "content": "Read the file."},
{
"role": "system",
"content": [{"type": "text", "text": "<total_tokens>14982391 tokens left</total_tokens>"}],
},
]
)
assert instructions is None
assert input_items == [
{"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Read the file."}]},
_system_input_item("<total_tokens>14982391 tokens left</total_tokens>"),
]
def test_leading_system_list_with_a_non_text_block_stays_an_input_item():
"""A non-text block (e.g. an image) in a leading system message's list content is never
silently dropped: the whole message stays an input item."""
handler: Final = LiteLLMResponsesTransformationHandler()
input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
[
{
"role": "system",
"content": [
{"type": "text", "text": "You are a helpful assistant."},
{"type": "image_url", "image_url": {"url": "https://example.com/logo.png"}},
],
},
{"role": "user", "content": "hi"},
]
)
assert instructions is None
assert len(input_items) == 2
assert input_items[0]["role"] == "system"
assert input_items[0]["content"][0] == {"type": "input_text", "text": "You are a helpful assistant."}
assert input_items[0]["content"][1]["type"] == "input_image"
def test_map_optional_params_verbosity_merges_into_text():
"""Chat verbosity must land on Responses text.verbosity alongside text.format regardless of key order."""
from litellm.completion_extras.litellm_responses_transformation.transformation import (