diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py
index 93d79bb3ac8..1d7b371caa3 100644
--- a/litellm/completion_extras/litellm_responses_transformation/transformation.py
+++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py
@@ -388,13 +388,30 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge):
tool_call_id = msg.get("tool_call_id")
if role == "system":
- # Extract system message as instructions
- if isinstance(content, str) and index < leading_system_count:
+ # Extract system message as instructions, but only within the leading run of
+ # system messages: a later one stays a positioned input item so its bytes don't
+ # unsettle the prompt-cache-stable prefix (#40269).
+ extracted_instructions: str | None = None
+ if index < leading_system_count:
+ if isinstance(content, str):
+ extracted_instructions = content
+ elif isinstance(content, list) and all(
+ isinstance(block, str) or (isinstance(block, dict) and block.get("type") == "text")
+ for block in content
+ ):
+ # Every block is plain text (a bare string or a `{"type": "text", ...}`
+ # dict), the same shape a client attaching cache_control sends; a
+ # non-text block (an image, say) fails the `all()` above and falls
+ # through to the input-item branch below instead of losing it silently.
+ extracted_instructions = " ".join(
+ block if isinstance(block, str) else block.get("text", "") for block in content
+ )
+ if extracted_instructions is not None:
if instructions:
# Concatenate multiple system prompts with a space
- instructions = f"{instructions} {content}"
+ instructions = f"{instructions} {extracted_instructions}"
else:
- instructions = content
+ instructions = extracted_instructions
else:
input_items.append(
{
diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py
index 25a3220792f..9b82f449eb7 100644
--- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py
+++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py
@@ -4466,9 +4466,15 @@ def test_claude_code_shaped_history_keeps_a_byte_stable_input_prefix_across_requ
litellm_logging_obj=Mock(),
)
- assert "instructions" not in first_request
- assert "instructions" not in second_request
- assert first_request["input"][0] == _system_input_item("You are Claude Code.")
+ # The leading, list-format top-level system prompt folds into `instructions` (#42171)
+ # and stays byte-identical across requests, rather than occupying `input[0]`.
+ assert first_request["instructions"] == "You are Claude Code."
+ assert second_request["instructions"] == first_request["instructions"]
+ assert first_request["input"][0] == {
+ "type": "message",
+ "role": "user",
+ "content": [{"type": "input_text", "text": "Read inventory.py."}],
+ }
assert json.dumps(second_request["input"][: len(first_request["input"])]) == json.dumps(first_request["input"])
assert second_request["input"][len(first_request["input"]) :] == [
{"type": "function_call", "call_id": "call_1", "name": "Read", "arguments": '{"file_path": "inventory.py"}'},
@@ -4493,6 +4499,98 @@ def test_system_string_after_a_developer_message_stays_in_input_in_client_order(
assert input_items[1] == _system_input_item("Be brief.")
+def test_leading_system_list_content_folds_into_instructions():
+ """A leading system message whose content is a list of text blocks (how a client
+ attaching cache_control sends it) folds into `instructions` exactly like a plain string
+ one, instead of surfacing as a `system` input item that some Responses backends reject
+ outright (#42171)."""
+ handler: Final = LiteLLMResponsesTransformationHandler()
+
+ input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
+ [
+ {
+ "role": "system",
+ "content": [
+ {"type": "text", "text": "You are a helpful assistant.", "cache_control": {"type": "ephemeral"}}
+ ],
+ },
+ {"role": "user", "content": "hi"},
+ ]
+ )
+
+ assert instructions == "You are a helpful assistant."
+ assert input_items == [
+ {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "hi"}]},
+ ]
+
+
+def test_leading_system_messages_mixed_str_and_list_concatenate_in_order():
+ """Several leading system messages, some string and some list content, concatenate in
+ order the same way an all-string leading run does."""
+ handler: Final = LiteLLMResponsesTransformationHandler()
+
+ input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
+ [
+ {"role": "system", "content": "Be brief."},
+ {"role": "system", "content": [{"type": "text", "text": "Answer in French."}]},
+ {"role": "system", "content": "Never use emoji."},
+ {"role": "user", "content": "Bonjour"},
+ ]
+ )
+
+ assert instructions == "Be brief. Answer in French. Never use emoji."
+ assert input_items == [
+ {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Bonjour"}]},
+ ]
+
+
+def test_mid_conversation_system_list_content_stays_in_input_after_a_user_turn():
+ """Only the leading run folds (#40269): a list-content system message after the first
+ non-system message stays a positioned input item."""
+ handler: Final = LiteLLMResponsesTransformationHandler()
+
+ input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
+ [
+ {"role": "user", "content": "Read the file."},
+ {
+ "role": "system",
+ "content": [{"type": "text", "text": "14982391 tokens left"}],
+ },
+ ]
+ )
+
+ assert instructions is None
+ assert input_items == [
+ {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Read the file."}]},
+ _system_input_item("14982391 tokens left"),
+ ]
+
+
+def test_leading_system_list_with_a_non_text_block_stays_an_input_item():
+ """A non-text block (e.g. an image) in a leading system message's list content is never
+ silently dropped: the whole message stays an input item."""
+ handler: Final = LiteLLMResponsesTransformationHandler()
+
+ input_items, instructions = handler.convert_chat_completion_messages_to_responses_api(
+ [
+ {
+ "role": "system",
+ "content": [
+ {"type": "text", "text": "You are a helpful assistant."},
+ {"type": "image_url", "image_url": {"url": "https://example.com/logo.png"}},
+ ],
+ },
+ {"role": "user", "content": "hi"},
+ ]
+ )
+
+ assert instructions is None
+ assert len(input_items) == 2
+ assert input_items[0]["role"] == "system"
+ assert input_items[0]["content"][0] == {"type": "input_text", "text": "You are a helpful assistant."}
+ assert input_items[0]["content"][1]["type"] == "input_image"
+
+
def test_map_optional_params_verbosity_merges_into_text():
"""Chat verbosity must land on Responses text.verbosity alongside text.format regardless of key order."""
from litellm.completion_extras.litellm_responses_transformation.transformation import (