diff --git a/litellm/completion_extras/litellm_responses_transformation/transformation.py b/litellm/completion_extras/litellm_responses_transformation/transformation.py index 93d79bb3ac8..1d7b371caa3 100644 --- a/litellm/completion_extras/litellm_responses_transformation/transformation.py +++ b/litellm/completion_extras/litellm_responses_transformation/transformation.py @@ -388,13 +388,30 @@ class LiteLLMResponsesTransformationHandler(CompletionTransformationBridge): tool_call_id = msg.get("tool_call_id") if role == "system": - # Extract system message as instructions - if isinstance(content, str) and index < leading_system_count: + # Extract system message as instructions, but only within the leading run of + # system messages: a later one stays a positioned input item so its bytes don't + # unsettle the prompt-cache-stable prefix (#40269). + extracted_instructions: str | None = None + if index < leading_system_count: + if isinstance(content, str): + extracted_instructions = content + elif isinstance(content, list) and all( + isinstance(block, str) or (isinstance(block, dict) and block.get("type") == "text") + for block in content + ): + # Every block is plain text (a bare string or a `{"type": "text", ...}` + # dict), the same shape a client attaching cache_control sends; a + # non-text block (an image, say) fails the `all()` above and falls + # through to the input-item branch below instead of losing it silently. + extracted_instructions = " ".join( + block if isinstance(block, str) else block.get("text", "") for block in content + ) + if extracted_instructions is not None: if instructions: # Concatenate multiple system prompts with a space - instructions = f"{instructions} {content}" + instructions = f"{instructions} {extracted_instructions}" else: - instructions = content + instructions = extracted_instructions else: input_items.append( { diff --git a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py index 25a3220792f..9b82f449eb7 100644 --- a/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py +++ b/tests/unit/completion_extras/litellm_responses_transformation/test_completion_extras_litellm_responses_transformation_transformation.py @@ -4466,9 +4466,15 @@ def test_claude_code_shaped_history_keeps_a_byte_stable_input_prefix_across_requ litellm_logging_obj=Mock(), ) - assert "instructions" not in first_request - assert "instructions" not in second_request - assert first_request["input"][0] == _system_input_item("You are Claude Code.") + # The leading, list-format top-level system prompt folds into `instructions` (#42171) + # and stays byte-identical across requests, rather than occupying `input[0]`. + assert first_request["instructions"] == "You are Claude Code." + assert second_request["instructions"] == first_request["instructions"] + assert first_request["input"][0] == { + "type": "message", + "role": "user", + "content": [{"type": "input_text", "text": "Read inventory.py."}], + } assert json.dumps(second_request["input"][: len(first_request["input"])]) == json.dumps(first_request["input"]) assert second_request["input"][len(first_request["input"]) :] == [ {"type": "function_call", "call_id": "call_1", "name": "Read", "arguments": '{"file_path": "inventory.py"}'}, @@ -4493,6 +4499,98 @@ def test_system_string_after_a_developer_message_stays_in_input_in_client_order( assert input_items[1] == _system_input_item("Be brief.") +def test_leading_system_list_content_folds_into_instructions(): + """A leading system message whose content is a list of text blocks (how a client + attaching cache_control sends it) folds into `instructions` exactly like a plain string + one, instead of surfacing as a `system` input item that some Responses backends reject + outright (#42171).""" + handler: Final = LiteLLMResponsesTransformationHandler() + + input_items, instructions = handler.convert_chat_completion_messages_to_responses_api( + [ + { + "role": "system", + "content": [ + {"type": "text", "text": "You are a helpful assistant.", "cache_control": {"type": "ephemeral"}} + ], + }, + {"role": "user", "content": "hi"}, + ] + ) + + assert instructions == "You are a helpful assistant." + assert input_items == [ + {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "hi"}]}, + ] + + +def test_leading_system_messages_mixed_str_and_list_concatenate_in_order(): + """Several leading system messages, some string and some list content, concatenate in + order the same way an all-string leading run does.""" + handler: Final = LiteLLMResponsesTransformationHandler() + + input_items, instructions = handler.convert_chat_completion_messages_to_responses_api( + [ + {"role": "system", "content": "Be brief."}, + {"role": "system", "content": [{"type": "text", "text": "Answer in French."}]}, + {"role": "system", "content": "Never use emoji."}, + {"role": "user", "content": "Bonjour"}, + ] + ) + + assert instructions == "Be brief. Answer in French. Never use emoji." + assert input_items == [ + {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Bonjour"}]}, + ] + + +def test_mid_conversation_system_list_content_stays_in_input_after_a_user_turn(): + """Only the leading run folds (#40269): a list-content system message after the first + non-system message stays a positioned input item.""" + handler: Final = LiteLLMResponsesTransformationHandler() + + input_items, instructions = handler.convert_chat_completion_messages_to_responses_api( + [ + {"role": "user", "content": "Read the file."}, + { + "role": "system", + "content": [{"type": "text", "text": "14982391 tokens left"}], + }, + ] + ) + + assert instructions is None + assert input_items == [ + {"type": "message", "role": "user", "content": [{"type": "input_text", "text": "Read the file."}]}, + _system_input_item("14982391 tokens left"), + ] + + +def test_leading_system_list_with_a_non_text_block_stays_an_input_item(): + """A non-text block (e.g. an image) in a leading system message's list content is never + silently dropped: the whole message stays an input item.""" + handler: Final = LiteLLMResponsesTransformationHandler() + + input_items, instructions = handler.convert_chat_completion_messages_to_responses_api( + [ + { + "role": "system", + "content": [ + {"type": "text", "text": "You are a helpful assistant."}, + {"type": "image_url", "image_url": {"url": "https://example.com/logo.png"}}, + ], + }, + {"role": "user", "content": "hi"}, + ] + ) + + assert instructions is None + assert len(input_items) == 2 + assert input_items[0]["role"] == "system" + assert input_items[0]["content"][0] == {"type": "input_text", "text": "You are a helpful assistant."} + assert input_items[0]["content"][1]["type"] == "input_image" + + def test_map_optional_params_verbosity_merges_into_text(): """Chat verbosity must land on Responses text.verbosity alongside text.format regardless of key order.""" from litellm.completion_extras.litellm_responses_transformation.transformation import (