From dedd37460cc4fdc2345a1577a56487e9f69ff9eb Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Sun, 23 Aug 2026 19:44:25 -0500 Subject: [PATCH 1/7] feat(anthropic): placement policy for mid-conversation system messages Pure functions over the OpenAI-format message list: split off the leading system run, keep later system messages as role=system at a placement Anthropic accepts on models flagged supports_mid_conversation_system (after a user turn, before an assistant turn or the end, never adjacent), and convert them to user turns in place elsewhere, keeping tool_result first in a merged user turn. --- .../llms/anthropic/mid_conversation_system.py | 290 ++++++++++++++++++ .../anthropic/test_mid_conversation_system.py | 147 +++++++++ 2 files changed, 437 insertions(+) create mode 100644 litellm/llms/anthropic/mid_conversation_system.py create mode 100644 tests/test_litellm/llms/anthropic/test_mid_conversation_system.py diff --git a/litellm/llms/anthropic/mid_conversation_system.py b/litellm/llms/anthropic/mid_conversation_system.py new file mode 100644 index 00000000000..766faf66312 --- /dev/null +++ b/litellm/llms/anthropic/mid_conversation_system.py @@ -0,0 +1,290 @@ +"""Placement policy for ``role: "system"`` messages that appear after the first turn +of an Anthropic-shaped chat completions request. + +Only the leading run of system messages belongs in the top-level ``system`` +parameter. Hoisting a later one there rewrites the cached prefix, so the provider +re-bills the whole conversation at cache-write pricing on every reminder (#36559). + +Models flagged ``supports_mid_conversation_system`` in the cost map accept the role +inside ``messages`` under Anthropic's placement rules: the message must directly +follow a user turn, must be the last entry or be followed by an assistant turn, and +must not sit next to another system message. OpenAI-shaped clients put system +messages anywhere, so this module moves each one to the nearest valid slot and +merges runs that land together. + +Models without the flag reject the role inside ``messages``. Their system messages +become user turns in place, prefixed with an operator note so the model can tell +the instruction apart from the user's own words. A run caught between a tool call +and its result moves to just after the result so the ``tool_result`` block stays +first in the merged user turn. + +Every transformation here is a pure function of the message sequence: turn N's +output stays a prefix of turn N+1's output, which is what keeps the provider-side +prompt cache readable across turns. Messages are handled in OpenAI format; the +Anthropic wire shape is built later by ``anthropic_messages_pt``. +""" + +from collections.abc import Iterator, Mapping, Sequence +from itertools import groupby +from typing import Final, Literal + +from litellm.types.llms.anthropic import AnthropicMessagesSystemMessageParam, AnthropicSystemMessageContent +from litellm.types.llms.openai import ( + AllMessageValues, + ChatCompletionCachedContent, + ChatCompletionSystemMessage, + ChatCompletionTextObject, + ChatCompletionUserMessage, +) + +CONVERTED_SYSTEM_NOTE: Final = ( + "Operator note (not from the user): the following was originally a mid-conversation system-role reminder." +) + +_USER_TYPE_ROLES: Final = frozenset({"user", "tool", "function"}) +_TOOL_ROLES: Final = frozenset({"tool", "function"}) + +_MessageKind = Literal["system", "tool", "user", "other"] +_TextPart = tuple[str, ChatCompletionCachedContent | None] + + +def _as_mapping(value: object) -> Mapping[str, object] | None: + return value if isinstance(value, Mapping) else None + + +def _as_items(value: object) -> tuple[object, ...]: + return tuple(value) if isinstance(value, Sequence) and not isinstance(value, str) else () + + +def _field(message: object, key: str) -> object: + """A message field, whether the message is a dict or a pydantic ``Message``.""" + mapping: Final = _as_mapping(message) + return mapping.get(key) if mapping is not None else getattr(message, key, None) + + +def is_system_message(message: object) -> bool: + return _field(message, "role") == "system" + + +def _is_user_type(message: object) -> bool: + return _field(message, "role") in _USER_TYPE_ROLES + + +def _kind(message: object) -> _MessageKind: + role: Final = _field(message, "role") + if role == "system": + return "system" + if role in _TOOL_ROLES: + return "tool" + if role == "user": + return "user" + return "other" + + +def split_leading_system_run( + messages: Sequence[AllMessageValues], +) -> tuple[tuple[AllMessageValues, ...], tuple[AllMessageValues, ...]]: + """Split ``messages`` into the leading run of system messages and everything after it.""" + leading_count: Final = next( + (index for index, message in enumerate(messages) if not is_system_message(message)), + len(messages), + ) + return tuple(messages[:leading_count]), tuple(messages[leading_count:]) + + +def _cache_control(holder: object) -> ChatCompletionCachedContent | None: + """The client's ``cache_control`` rebuilt in the only shape Anthropic accepts.""" + value: Final = _as_mapping(_field(holder, "cache_control")) + if value is None or value.get("type") != "ephemeral": + return None + ttl: Final = value.get("ttl") + if ttl == "1h": + one_hour: Final[ChatCompletionCachedContent] = {"type": "ephemeral", "ttl": "1h"} + return one_hour + if ttl == "5m": + five_minutes: Final[ChatCompletionCachedContent] = {"type": "ephemeral", "ttl": "5m"} + return five_minutes + ephemeral: Final[ChatCompletionCachedContent] = {"type": "ephemeral"} + return ephemeral + + +def _text_parts(message: object) -> tuple[_TextPart, ...]: + """``(text, cache_control)`` for each non-empty text part of a system message. + + Anthropic rejects empty text blocks and only accepts text in system content. A + ``cache_control`` on the message itself belongs to the block built from string + content; block-level ``cache_control`` stays with its block. + """ + content: Final = _field(message, "content") + if isinstance(content, str): + return ((content, _cache_control(message)),) if content else () + return tuple( + (text, _cache_control(part)) + for part in _as_items(content) + if _field(part, "type") == "text" + for text in (_field(part, "text"),) + if isinstance(text, str) and text + ) + + +def _openai_text_block(part: _TextPart) -> ChatCompletionTextObject: + text, cache_control = part + if cache_control is None: + plain: Final[ChatCompletionTextObject] = {"type": "text", "text": text} + return plain + cached: Final[ChatCompletionTextObject] = {"type": "text", "text": text, "cache_control": cache_control} + return cached + + +def _anthropic_text_block(part: _TextPart) -> AnthropicSystemMessageContent: + text, cache_control = part + if cache_control is None: + plain: Final[AnthropicSystemMessageContent] = {"type": "text", "text": text} + return plain + cached: Final[AnthropicSystemMessageContent] = {"type": "text", "text": text, "cache_control": cache_control} + return cached + + +def anthropic_system_messages(message: object) -> tuple[AnthropicMessagesSystemMessageParam, ...]: + """The Anthropic wire message for a system message, or nothing when it carries no text.""" + blocks: Final = tuple(_anthropic_text_block(part) for part in _text_parts(message)) + if not blocks: + return () + wire: Final[AnthropicMessagesSystemMessageParam] = { + "role": "system", + "content": list(blocks), # mutable-ok: wire payload; cache_control hooks edit content blocks in place + } + return (wire,) + + +def system_message_as_user(message: object) -> ChatCompletionUserMessage: + """A system message re-rolled as a user turn, prefixed with the operator note.""" + note: Final[ChatCompletionTextObject] = {"type": "text", "text": CONVERTED_SYSTEM_NOTE} + content: Final[list[ChatCompletionTextObject]] = [ # mutable-ok: anthropic_messages_pt only recognises list content + note, + *(_openai_text_block(part) for part in _text_parts(message)), + ] + turn: Final[ChatCompletionUserMessage] = {"role": "user", "content": content} + return turn + + +def _merged_system_message(run: Sequence[object]) -> tuple[ChatCompletionSystemMessage, ...]: + parts: Final = tuple(part for message in run for part in _text_parts(message)) + if not parts: + return () + content: Final[list[ChatCompletionTextObject]] = [ # mutable-ok: anthropic_messages_pt only recognises list content + _openai_text_block(part) for part in parts + ] + merged: Final[ChatCompletionSystemMessage] = {"role": "system", "content": content} + return (merged,) + + +def _converted_user_turns(run: Sequence[object]) -> tuple[ChatCompletionUserMessage, ...]: + return tuple(system_message_as_user(message) for message in run if _text_parts(message)) + + +def _runs(messages: Sequence[AllMessageValues]) -> tuple[tuple[_MessageKind, tuple[AllMessageValues, ...]], ...]: + return tuple((kind, tuple(group)) for kind, group in groupby(messages, key=_kind)) + + +def _converted_for_unflagged_model(messages: Sequence[AllMessageValues]) -> tuple[AllMessageValues, ...]: + """Convert every system message to a user turn in place. + + A system run whose follower is a tool message is emitted after that tool run: + ``tool_result`` blocks have to open the merged user turn. + """ + runs: Final = _runs(messages) + + def emit(index: int) -> tuple[AllMessageValues, ...]: + kind, run = runs[index] + follower: Final = runs[index + 1][0] if index + 1 < len(runs) else None + if kind == "system": + return () if follower == "tool" else _converted_user_turns(run) + if kind == "tool" and index > 0 and runs[index - 1][0] == "system": + return (*run, *_converted_user_turns(runs[index - 1][1])) + return run + + return tuple(message for index in range(len(runs)) for message in emit(index)) + + +def _user_type_blocks(messages: Sequence[AllMessageValues]) -> tuple[tuple[bool, tuple[int, ...]], ...]: + """Maximal groups of consecutive non-system messages, keyed by whether they are user-type. + + Consecutive user-type messages become one user turn on the wire, so a group is + the unit a system message can validly follow. + """ + indexed: Final = tuple((index, message) for index, message in enumerate(messages) if not is_system_message(message)) + return tuple( + (is_user, tuple(index for index, _ in group)) + for is_user, group in groupby(indexed, key=lambda pair: _is_user_type(pair[1])) + ) + + +def _system_runs(messages: Sequence[AllMessageValues]) -> tuple[tuple[int, ...], ...]: + """Index runs of consecutive system messages.""" + system_indices: Final = tuple(index for index, message in enumerate(messages) if is_system_message(message)) + return tuple( + tuple(index for _, index in group) + for _, group in groupby(enumerate(system_indices), key=lambda pair: pair[1] - pair[0]) + ) + + +def _anchor_block( + run_start: int, + messages: Sequence[AllMessageValues], + blocks: Sequence[tuple[bool, tuple[int, ...]]], +) -> int | None: + """The user-type block a system run must follow, or ``None`` when no user turn can host it. + + ``run_start`` is never 0 here: the leading system run was split off before this + policy runs, so the message before a run is always a non-system message. + """ + if _is_user_type(messages[run_start - 1]): + return next(index for index, (is_user, indices) in enumerate(blocks) if is_user and run_start - 1 in indices) + return next((index for index, (is_user, indices) in enumerate(blocks) if is_user and indices[0] > run_start), None) + + +def _placed_for_flagged_model(messages: Sequence[AllMessageValues]) -> tuple[AllMessageValues, ...]: + """Keep system messages as ``role: "system"`` at a placement Anthropic accepts. + + A run already sitting after a user-type message stays with that user turn. A + run after an assistant turn moves to just after the next user turn. Runs that + share a user turn merge into one system message. A run with no user turn left + to host it becomes user turns in place. + """ + blocks: Final = _user_type_blocks(messages) + anchors: Final = tuple((run, _anchor_block(run[0], messages, blocks)) for run in _system_runs(messages)) + + def anchored_to(block_index: int) -> tuple[AllMessageValues, ...]: + return tuple(messages[index] for run, anchor in anchors if anchor == block_index for index in run) + + def converted_after(message_index: int) -> tuple[ChatCompletionUserMessage, ...]: + return tuple( + turn + for run, anchor in anchors + if anchor is None and run[0] == message_index + 1 + for turn in _converted_user_turns(tuple(messages[index] for index in run)) + ) + + def emit(block_index: int) -> Iterator[AllMessageValues]: + is_user, indices = blocks[block_index] + for index in indices: + yield messages[index] + yield from converted_after(index) + if is_user: + yield from _merged_system_message(anchored_to(block_index)) + + return tuple(message for block_index in range(len(blocks)) for message in emit(block_index)) + + +def place_mid_conversation_system( + messages: Sequence[AllMessageValues], + *, + supports_mid_conversation_system: bool, +) -> tuple[AllMessageValues, ...]: + """Apply the placement policy to the messages after the leading system run.""" + if not any(is_system_message(message) for message in messages): + return tuple(messages) + if supports_mid_conversation_system: + return _placed_for_flagged_model(messages) + return _converted_for_unflagged_model(messages) diff --git a/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py b/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py new file mode 100644 index 00000000000..f1042a2e17e --- /dev/null +++ b/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py @@ -0,0 +1,147 @@ +"""Placement policy for mid-conversation ``role: "system"`` messages on the chat path. + +The provider-facing behaviour is covered through ``transform_request`` in the +Anthropic, Vertex, Azure AI and Bedrock Invoke transformation tests; these pin +the pure placement rules on the OpenAI-format message list. +""" + +import litellm +from litellm.llms.anthropic.mid_conversation_system import ( + CONVERTED_SYSTEM_NOTE, + place_mid_conversation_system, + split_leading_system_run, +) + + +def _roles(messages): + return [m["role"] if isinstance(m, dict) else m.role for m in messages] + + +def _texts(message): + return [block["text"] for block in message["content"]] + + +def test_split_leading_system_run_keeps_later_system_messages_in_the_conversation(): + messages = [ + {"role": "system", "content": "one"}, + {"role": "system", "content": "two"}, + {"role": "user", "content": "q"}, + {"role": "system", "content": "reminder"}, + ] + + leading, later = split_leading_system_run(messages) + + assert [m["content"] for m in leading] == ["one", "two"] + assert _roles(later) == ["user", "system"] + + +def test_flagged_placement_moves_a_system_run_after_the_user_turn_that_follows_it(): + messages = [ + {"role": "user", "content": "q1"}, + {"role": "assistant", "content": "a1"}, + {"role": "system", "content": "reminder"}, + {"role": "user", "content": "q2"}, + {"role": "assistant", "content": "a2"}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True) + + assert _roles(placed) == ["user", "assistant", "user", "system", "assistant"] + + +def test_flagged_placement_pushes_a_system_between_two_user_turns_after_both(): + """Two user turns collapse into one on the wire, and a system message must + be followed by an assistant turn or nothing.""" + messages = [ + {"role": "user", "content": "q1"}, + {"role": "system", "content": "reminder"}, + {"role": "user", "content": "q2"}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True) + + assert _roles(placed) == ["user", "user", "system"] + + +def test_flagged_placement_keeps_a_system_after_tool_results(): + messages = [ + {"role": "user", "content": "q1"}, + { + "role": "assistant", + "content": None, + "tool_calls": [{"id": "c1", "type": "function", "function": {"name": "f", "arguments": "{}"}}], + }, + {"role": "tool", "tool_call_id": "c1", "content": "r"}, + {"role": "system", "content": "reminder"}, + {"role": "assistant", "content": "a2"}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True) + + assert _roles(placed) == ["user", "assistant", "tool", "system", "assistant"] + + +def test_flagged_placement_drops_a_system_message_with_no_text(): + messages = [ + {"role": "user", "content": "q1"}, + {"role": "system", "content": ""}, + {"role": "assistant", "content": "a1"}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True) + + assert _roles(placed) == ["user", "assistant"] + + +def test_placement_reads_roles_off_pydantic_messages_in_the_history(): + """Callers routinely append the previous ``litellm.Message`` object straight + into the history; placement must read its role without assuming a dict and + hand the object through untouched.""" + assistant = litellm.Message(role="assistant", content="a1") + messages = [ + {"role": "user", "content": "q1"}, + assistant, + {"role": "system", "content": "reminder"}, + {"role": "user", "content": "q2"}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True) + + assert _roles(placed) == ["user", "assistant", "user", "system"] + assert placed[1] is assistant + + +def test_unflagged_conversion_keeps_the_client_order_when_no_tool_result_follows(): + messages = [ + {"role": "user", "content": "q1"}, + {"role": "assistant", "content": "a1"}, + {"role": "system", "content": "reminder"}, + {"role": "user", "content": "q2"}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False) + + assert _roles(placed) == ["user", "assistant", "user", "user"] + assert _texts(placed[2]) == [CONVERTED_SYSTEM_NOTE, "reminder"] + + +def test_unflagged_conversion_rebuilds_cache_control_on_the_converted_block(): + messages = [ + {"role": "user", "content": "q1"}, + {"role": "system", "content": "reminder", "cache_control": {"type": "ephemeral", "ttl": "1h"}}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False) + + assert placed[1]["content"][1] == { + "type": "text", + "text": "reminder", + "cache_control": {"type": "ephemeral", "ttl": "1h"}, + } + + +def test_placement_is_a_no_op_without_later_system_messages(): + messages = [{"role": "user", "content": "q1"}, {"role": "assistant", "content": "a1"}] + + assert place_mid_conversation_system(messages, supports_mid_conversation_system=False) == tuple(messages) + assert place_mid_conversation_system(messages, supports_mid_conversation_system=True) == tuple(messages) From 986a50513103321f7c0bcb30d688d3bfbc335bf1 Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Sun, 23 Aug 2026 19:44:25 -0500 Subject: [PATCH 2/7] fix(anthropic): keep mid-conversation system out of the chat completions system prompt translate_system_message hoisted every role=system message, at any index, into the top-level system block. On a conversation carrying a mid-session reminder that rewrites the cached prefix, so the provider re-bills the whole history at cache-write pricing on every turn (#36559). #36968 fixed this on /v1/messages; the chat completions path, shared by first-party Anthropic, Vertex, Azure AI and Bedrock Invoke, still hoisted. Only the leading system run becomes the system prompt now. Later system messages go through the placement policy, and anthropic_messages_pt emits a system message instead of rejecting the role. The caller's message list is no longer mutated. Tests pin the two-turn prefix invariant across all four chat configs and both flag states. --- .../prompt_templates/factory.py | 18 +- litellm/llms/anthropic/chat/transformation.py | 21 +- litellm/types/llms/anthropic.py | 3 +- ...llm_core_utils_prompt_templates_factory.py | 35 +++ .../test_anthropic_chat_transformation.py | 218 ++++++++++++++++++ .../test_azure_anthropic_transformation.py | 48 ++++ ...ations_anthropic_claude3_transformation.py | 48 ++++ ...partner_models_anthropic_transformation.py | 48 ++++ 8 files changed, 432 insertions(+), 7 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 86cfbf70255..4cd8a689afa 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -18,6 +18,7 @@ from litellm import verbose_logger from litellm._uuid import uuid from litellm.constants import REDACTED_BY_LITELLM from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get +from litellm.llms.anthropic.mid_conversation_system import anthropic_system_messages from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client from litellm.types.files import get_file_extension_from_mime_type from litellm.types.llms.anthropic import * @@ -2305,14 +2306,20 @@ def _drop_unsignable_thinking_blocks( return [block for block in thinking_blocks if not _is_unsignable_thinking_block(block)] +# mutable-ok: anthropic_messages_pt's callers have always appended to the list it returns +_AnthropicMessageList = list[AllAnthropicPassThroughMessageValues] + + def anthropic_messages_pt( messages: list[AllMessageValues], model: str, llm_provider: str, -) -> list[AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam]: +) -> _AnthropicMessageList: """ format messages for anthropic - 1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately) + 1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately). + Models flagged ``supports_mid_conversation_system`` also accept "system" inside + messages after a user turn; the caller decides placement, this keeps such messages. 2. The first message always needs to be of role "user" 3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm) 4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise) @@ -2337,7 +2344,7 @@ def anthropic_messages_pt( # add role=tool support to allow function call result/error submission user_message_types: Final = {"user", "tool", "function"} # reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them. - new_messages: Final[list[AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam]] = [] + new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract if len(messages) == 0: if not litellm.modify_params: @@ -2730,6 +2737,11 @@ def anthropic_messages_pt( if assistant_content: new_messages.append({"role": "assistant", "content": assistant_content}) + ## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ## + while msg_i < len(messages) and messages[msg_i]["role"] == "system": + new_messages.extend(anthropic_system_messages(messages[msg_i])) + msg_i += 1 + if msg_i == init_msg_i: # prevent infinite loops raise litellm.BadRequestError( message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}", diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index 15fc482b34e..b442d27c91e 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -75,6 +75,7 @@ from litellm.types.utils import Message as LitellmMessage from litellm.utils import ( ModelResponse, Usage, + _supports_factory, add_dummy_tool, any_assistant_message_has_thinking_blocks, get_max_tokens, @@ -90,6 +91,7 @@ from ..common_utils import ( process_anthropic_headers, strip_advisor_blocks_from_messages, ) +from ..mid_conversation_system import place_mid_conversation_system, split_leading_system_run if TYPE_CHECKING: from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj @@ -1897,16 +1899,29 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): if _name_reverse_map and isinstance(litellm_params, dict): litellm_params[ANTHROPIC_TOOL_NAME_REVERSE_MAP_KEY] = _name_reverse_map - # Separate system prompt from rest of message - anthropic_system_message_list: Final = self.translate_system_message(messages=messages) + # Only the leading system run becomes the top-level system prompt. A later + # system message stays in the conversation: hoisting it rewrites the cached + # prefix and re-bills the whole history at cache-write pricing (#36559). + leading_system_run, later_messages = split_leading_system_run(messages) + anthropic_system_message_list: Final = self.translate_system_message( + messages=list(leading_system_run) # mutable-ok: translate_system_message pops from the list it is given + ) # Handling anthropic API Prompt Caching if len(anthropic_system_message_list) > 0: optional_params["system"] = anthropic_system_message_list + conversation: Final = place_mid_conversation_system( + later_messages, + supports_mid_conversation_system=_supports_factory( + model=model, + custom_llm_provider=self.custom_llm_provider, + key="supports_mid_conversation_system", + ), + ) # Format rest of message according to anthropic guidelines try: anthropic_messages = anthropic_messages_pt( model=model, - messages=messages, + messages=list(conversation), # mutable-ok: anthropic_messages_pt rewrites entries in place llm_provider=self._resolved_provider, ) except Exception as e: diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py index f127366cc21..a1cd2fa9ca1 100644 --- a/litellm/types/llms/anthropic.py +++ b/litellm/types/llms/anthropic.py @@ -361,7 +361,8 @@ class AnthropicMessagesSystemMessageParam(TypedDict, total=False): AllAnthropicMessageValues = AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam -# System is not a native Anthropic message role; only pass-through adapters use this union. +# role=system inside messages is accepted after a user turn on models flagged +# supports_mid_conversation_system; pass-through adapters and the chat translator both emit it. AllAnthropicPassThroughMessageValues: TypeAlias = ( AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam | AnthropicMessagesSystemMessageParam ) diff --git a/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index 6265779b90d..cf85329aa12 100644 --- a/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -3578,3 +3578,38 @@ async def test_bedrock_converse_pdf_only_user_message_gets_text_block_async(): assert len(result) == 1 assert any("document" in block for block in result[0]["content"]) assert _text_blocks(result[0]) == [BEDROCK_DOCUMENT_PLACEHOLDER_TEXT] + + +def test_anthropic_messages_pt_keeps_system_role_after_user_turn(): + """Models flagged supports_mid_conversation_system accept role=system inside + messages; the converter must emit it as a system message with its text + blocks and cache_control intact instead of rejecting the role.""" + messages = [ + {"role": "user", "content": "First question"}, + { + "role": "system", + "content": [{"type": "text", "text": "Answer in one word.", "cache_control": {"type": "ephemeral"}}], + }, + {"role": "assistant", "content": "Yes"}, + {"role": "user", "content": "Second question"}, + ] + + result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic") + + assert [m["role"] for m in result] == ["user", "system", "assistant", "user"] + assert result[1] == { + "role": "system", + "content": [{"type": "text", "text": "Answer in one word.", "cache_control": {"type": "ephemeral"}}], + } + + +def test_anthropic_messages_pt_system_string_content_becomes_text_block(): + messages = [ + {"role": "user", "content": "First question"}, + {"role": "system", "content": "Answer in one word."}, + ] + + result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic") + + assert result[1] == {"role": "system", "content": [{"type": "text", "text": "Answer in one word."}]} + diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 25e2c3cda80..74d2c22bdf3 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -1,4 +1,5 @@ +import copy import pytest from unittest.mock import MagicMock, patch @@ -17,6 +18,13 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig from litellm.llms.anthropic.experimental_pass_through.messages.transformation import ( AnthropicMessagesConfig, ) +from litellm.llms.azure_ai.anthropic.transformation import AzureAnthropicConfig +from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import ( + AmazonAnthropicClaudeConfig, +) +from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import ( + VertexAIAnthropicConfig, +) from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES from litellm.types.utils import ServerToolUse, Usage @@ -6207,3 +6215,213 @@ def test_disabled_thinking_omitted_only_for_always_on_models( assert "thinking" not in request else: assert request["thinking"] == {"type": "disabled"} + + +# --------------------------------------------------------------------------- +# Mid-conversation ``role: "system"`` on the chat completions path. +# +# Hoisting a later system message into the top-level ``system`` block rewrites +# the cached prefix and re-bills the whole conversation at cache-write pricing +# on every reminder (#36559). The chat path must keep the prefix stable: leading +# system messages still become the ``system`` param, later ones stay in place as +# ``role: "system"`` on models flagged ``supports_mid_conversation_system`` and +# become a user turn on models that reject the role inside ``messages``. +# --------------------------------------------------------------------------- + +UNFLAGGED_CLAUDE = "claude-opus-4-7" +FLAGGED_CLAUDE = "claude-opus-4-8" +CONVERTED_SYSTEM_NOTE = ( + "Operator note (not from the user): the following was originally a mid-conversation system-role reminder." +) +REMINDER_TEXT = "Answer with exactly one word." +CACHED_SYSTEM_BLOCK = {"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}} + + +def _chat_request(config, model, messages): + return config.transform_request( + model=model, + messages=messages, + optional_params={}, + litellm_params={}, + headers={}, + ) + + +def _reminder_conversation(): + """The shape Claude Code sends mid-session: cached system prompt, turns, a + reminder right after a user turn, an assistant turn, a fresh user turn.""" + return [ + {"role": "system", "content": [dict(CACHED_SYSTEM_BLOCK)]}, + {"role": "user", "content": "First question"}, + {"role": "assistant", "content": "First answer"}, + {"role": "user", "content": "Second question"}, + {"role": "system", "content": REMINDER_TEXT}, + {"role": "assistant", "content": "Second answer"}, + {"role": "user", "content": "Third question"}, + ] + + +def _texts(message): + return [block["text"] for block in message["content"] if block.get("type") == "text"] + + +def test_chat_unflagged_model_converts_mid_conversation_system_to_user_turn(local_model_cost_map): + result = _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, _reminder_conversation()) + + assert result["system"] == [CACHED_SYSTEM_BLOCK] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"] + assert _texts(result["messages"][2]) == ["Second question", CONVERTED_SYSTEM_NOTE, REMINDER_TEXT] + + +def test_chat_flagged_model_keeps_mid_conversation_system_in_messages(local_model_cost_map): + result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, _reminder_conversation()) + + assert result["system"] == [CACHED_SYSTEM_BLOCK] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"] + assert result["messages"][3] == {"role": "system", "content": [{"type": "text", "text": REMINDER_TEXT}]} + + +def test_chat_flagged_model_keeps_cache_control_on_mid_conversation_system(local_model_cost_map): + messages = _reminder_conversation() + messages[4] = { + "role": "system", + "content": [{"type": "text", "text": REMINDER_TEXT, "cache_control": {"type": "ephemeral"}}], + } + + result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages) + + assert result["messages"][3]["content"] == [ + {"type": "text", "text": REMINDER_TEXT, "cache_control": {"type": "ephemeral"}} + ] + + +def test_chat_flagged_model_moves_system_after_the_user_turn_it_precedes(local_model_cost_map): + """Anthropic only accepts role=system directly after a user turn; an + OpenAI-shaped client that puts the reminder before its next question gets a + placement-valid request without the reminder leaving ``messages``.""" + messages = [ + {"role": "system", "content": "You are terse."}, + {"role": "user", "content": "First question"}, + {"role": "assistant", "content": "First answer"}, + {"role": "system", "content": REMINDER_TEXT}, + {"role": "user", "content": "Second question"}, + ] + + result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages) + + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system"] + assert _texts(result["messages"][2]) == ["Second question"] + assert _texts(result["messages"][3]) == [REMINDER_TEXT] + + +def test_chat_flagged_model_converts_system_with_no_following_user_turn(local_model_cost_map): + messages = [ + {"role": "system", "content": "You are terse."}, + {"role": "user", "content": "First question"}, + {"role": "assistant", "content": "First answer"}, + {"role": "system", "content": REMINDER_TEXT}, + ] + + result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages) + + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user"] + assert _texts(result["messages"][2]) == [CONVERTED_SYSTEM_NOTE, REMINDER_TEXT] + + +def test_chat_flagged_model_merges_adjacent_system_messages(local_model_cost_map): + messages = [ + {"role": "system", "content": "You are terse."}, + {"role": "user", "content": "First question"}, + {"role": "system", "content": "Reminder one."}, + {"role": "system", "content": "Reminder two."}, + {"role": "assistant", "content": "First answer"}, + {"role": "user", "content": "Second question"}, + ] + + result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages) + + assert [m["role"] for m in result["messages"]] == ["user", "system", "assistant", "user"] + assert _texts(result["messages"][1]) == ["Reminder one.", "Reminder two."] + + +def test_chat_unflagged_model_keeps_tool_result_first_when_system_precedes_tool_message(local_model_cost_map): + messages = [ + {"role": "system", "content": "You are terse."}, + {"role": "user", "content": "Weather?"}, + { + "role": "assistant", + "content": None, + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": {"name": "get_weather", "arguments": "{}"}, + } + ], + }, + {"role": "system", "content": REMINDER_TEXT}, + {"role": "tool", "tool_call_id": "call_1", "content": "sunny"}, + {"role": "user", "content": "Thanks"}, + ] + + result = _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, messages) + + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user"] + blocks = result["messages"][2]["content"] + assert blocks[0]["type"] == "tool_result" + assert blocks[0]["tool_use_id"] == "call_1" + assert _texts(result["messages"][2]) == [CONVERTED_SYSTEM_NOTE, REMINDER_TEXT, "Thanks"] + + +def test_chat_transform_request_does_not_mutate_caller_messages(local_model_cost_map): + messages = _reminder_conversation() + snapshot = copy.deepcopy(messages) + + _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, messages) + + assert messages == snapshot + + +_CHAT_CONFIGS = [ + pytest.param(AnthropicConfig, UNFLAGGED_CLAUDE, id="anthropic-unflagged"), + pytest.param(AnthropicConfig, FLAGGED_CLAUDE, id="anthropic-flagged"), + pytest.param(VertexAIAnthropicConfig, UNFLAGGED_CLAUDE, id="vertex_ai-unflagged"), + pytest.param(VertexAIAnthropicConfig, FLAGGED_CLAUDE, id="vertex_ai-flagged"), + pytest.param(AzureAnthropicConfig, UNFLAGGED_CLAUDE, id="azure_ai-unflagged"), + pytest.param(AzureAnthropicConfig, FLAGGED_CLAUDE, id="azure_ai-flagged"), + pytest.param(AmazonAnthropicClaudeConfig, "invoke/us.anthropic.claude-opus-4-7", id="bedrock_invoke-unflagged"), + pytest.param(AmazonAnthropicClaudeConfig, "invoke/us.anthropic.claude-opus-4-8", id="bedrock_invoke-flagged"), +] + + +@pytest.mark.parametrize("config_cls, model", _CHAT_CONFIGS) +def test_chat_mid_conversation_system_keeps_earlier_turns_a_prefix_of_the_next_request( + local_model_cost_map, config_cls, model +): + """The provider-side prompt cache is a prefix match over ``system`` + + ``messages``. Whatever the policy for the reminder, turn N's request must + stay a prefix of turn N+1's request or the whole conversation is re-billed. + + Anthropic combines consecutive same-role messages into one turn, so the + cache-relevant sequence is ``(role, content block)`` pairs, not the message + list: a reminder that joins the preceding user turn still extends the prefix. + """ + conversation = _reminder_conversation() + + earlier = _chat_request(config_cls(), model, copy.deepcopy(conversation[:4])) + later = _chat_request(config_cls(), model, copy.deepcopy(conversation)) + + assert later["system"] == earlier["system"] + earlier_blocks = _role_block_pairs(earlier["messages"]) + later_blocks = _role_block_pairs(later["messages"]) + assert later_blocks[: len(earlier_blocks)] == earlier_blocks + assert len(later_blocks) > len(earlier_blocks) + + +def _role_block_pairs(messages): + return [ + (message["role"], block) + for message in messages + for block in (message["content"] if isinstance(message["content"], list) else [message["content"]]) + ] + diff --git a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py index 9dac914ca4d..257d5f92ee5 100644 --- a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py +++ b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py @@ -413,3 +413,51 @@ class TestAzureAnthropicConfig: assert "anthropic-beta" in headers assert "compact-2026-01-12" in headers["anthropic-beta"] assert "context-management-2025-06-27" in headers["anthropic-beta"] + + +def _mid_conversation_system_conversation(): + return [ + {"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]}, + {"role": "user", "content": "First question"}, + {"role": "assistant", "content": "First answer"}, + {"role": "user", "content": "Second question"}, + {"role": "system", "content": "Answer with exactly one word."}, + {"role": "assistant", "content": "Second answer"}, + {"role": "user", "content": "Third question"}, + ] + + +def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map): + """A hoisted reminder rewrites the top-level system block and invalidates the + prompt cache for the whole conversation (#36559).""" + result = AzureAnthropicConfig().transform_request( + model="claude-opus-4-7", + messages=_mid_conversation_system_conversation(), + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"] + texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"] + assert texts[0] == "Second question" + assert texts[-1] == "Answer with exactly one word." + + +def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map): + result = AzureAnthropicConfig().transform_request( + model="claude-opus-4-8", + messages=_mid_conversation_system_conversation(), + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"] + assert result["messages"][3] == { + "role": "system", + "content": [{"type": "text", "text": "Answer with exactly one word."}], + } + diff --git a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py index cea299280f8..080050c7621 100644 --- a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py @@ -542,3 +542,51 @@ def test_output_format_removed_from_bedrock_invoke_request(): assert ( "output_format" not in result ), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}" + + +def _mid_conversation_system_conversation(): + return [ + {"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]}, + {"role": "user", "content": "First question"}, + {"role": "assistant", "content": "First answer"}, + {"role": "user", "content": "Second question"}, + {"role": "system", "content": "Answer with exactly one word."}, + {"role": "assistant", "content": "Second answer"}, + {"role": "user", "content": "Third question"}, + ] + + +def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map): + """A hoisted reminder rewrites the top-level system block and invalidates the + prompt cache for the whole conversation (#36559).""" + result = AmazonAnthropicClaudeConfig().transform_request( + model="invoke/us.anthropic.claude-opus-4-7", + messages=_mid_conversation_system_conversation(), + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"] + texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"] + assert texts[0] == "Second question" + assert texts[-1] == "Answer with exactly one word." + + +def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map): + result = AmazonAnthropicClaudeConfig().transform_request( + model="invoke/us.anthropic.claude-opus-4-8", + messages=_mid_conversation_system_conversation(), + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"] + assert result["messages"][3] == { + "role": "system", + "content": [{"type": "text", "text": "Answer with exactly one word."}], + } + diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 552ca98441f..65a77364bc7 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -727,3 +727,51 @@ def test_sanitize_strips_effort_for_haiku_45(): data = {"output_config": {"effort": "high"}} sanitize_vertex_anthropic_output_params(data, "vertex_ai/claude-opus-4-6") assert data["output_config"] == {"effort": "high"} + + +def _mid_conversation_system_conversation(): + return [ + {"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]}, + {"role": "user", "content": "First question"}, + {"role": "assistant", "content": "First answer"}, + {"role": "user", "content": "Second question"}, + {"role": "system", "content": "Answer with exactly one word."}, + {"role": "assistant", "content": "Second answer"}, + {"role": "user", "content": "Third question"}, + ] + + +def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map): + """A hoisted reminder rewrites the top-level system block and invalidates the + prompt cache for the whole conversation (#36559).""" + result = VertexAIAnthropicConfig().transform_request( + model="claude-opus-4-7", + messages=_mid_conversation_system_conversation(), + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"] + texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"] + assert texts[0] == "Second question" + assert texts[-1] == "Answer with exactly one word." + + +def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map): + result = VertexAIAnthropicConfig().transform_request( + model="claude-opus-4-8", + messages=_mid_conversation_system_conversation(), + optional_params={}, + litellm_params={}, + headers={}, + ) + + assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}] + assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"] + assert result["messages"][3] == { + "role": "system", + "content": [{"type": "text", "text": "Answer with exactly one word."}], + } + From c79f3ac19552f82ab5f6fa5b365ddd21fb239a35 Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Sun, 23 Aug 2026 19:44:25 -0500 Subject: [PATCH 3/7] refactor(anthropic): single-source the converted system note The /v1/messages pass-through and the chat completions path must prefix a converted system turn with the same operator note. --- .../experimental_pass_through/messages/transformation.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index ebd514c2605..f0dc5ecce3c 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -31,6 +31,7 @@ from ...common_utils import ( optionally_handle_anthropic_oauth, strip_advisor_blocks_from_messages, ) +from ...mid_conversation_system import CONVERTED_SYSTEM_NOTE DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01" @@ -160,9 +161,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): def _is_system_role_message(message: Any) -> bool: return isinstance(message, dict) and message.get("role") == "system" - _CONVERTED_SYSTEM_NOTE: Final = ( - "Operator note (not from the user): the following was originally a mid-conversation system-role reminder." - ) + _CONVERTED_SYSTEM_NOTE: Final = CONVERTED_SYSTEM_NOTE def _system_role_message_as_user(self, message: Mapping) -> Mapping: return { From 87591e95d942cf9eddec07e2d70a4f322750e8e4 Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Sun, 23 Aug 2026 19:48:46 -0500 Subject: [PATCH 4/7] test(e2e): prove the prompt cache survives a mid-conversation system reminder on chat completions Same priming and assertions as the /v1/messages cases, through /v1/chat/completions with OpenAI-format messages, for first-party Anthropic and Bedrock Invoke on a flagged (Opus 4.8) and an unflagged (Haiku 4.5) model. The reminder sits between the assistant turn and the next user turn, the shape OpenAI-style agent frameworks send, which is the placement the chat path has to translate. --- .../coverage_registry/llm_conversational.yaml | 4 + .../test_chat_mid_conversation_system_e2e.py | 323 ++++++++++++++++++ 2 files changed, 327 insertions(+) create mode 100644 tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml index 5662bdadb9c..10cd00298d7 100644 --- a/tests/e2e/coverage_registry/llm_conversational.yaml +++ b/tests/e2e/coverage_registry/llm_conversational.yaml @@ -23,6 +23,8 @@ - {id: llm.chat_completions.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude prompt caching"} - {id: llm.chat_completions.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude extended thinking"} - {id: llm.chat_completions.anthropic.structured_output.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: structured_output, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude response_schema"} +- {id: llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to first-party Anthropic: flagged Claude 4.8+/5 must keep a mid-conversation role system reminder in messages; hoisting it into the top-level system field mutates the cached prefix and re-bills the conversation at cache-write pricing (#36559)", fail_before_fix: proven} +- {id: llm.chat_completions.anthropic.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to first-party Anthropic: Claude <= 4.7 and Haiku 4.5 reject role system inside messages, so unflagged models must convert a mid-conversation reminder to a user turn in place (hoisting collapses the prompt cache) and still answer (#36559)", fail_before_fix: proven} - {id: llm.chat_completions.bedrock_converse.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Bedrock Converse unified"} - {id: llm.chat_completions.bedrock_converse.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Converse"} - {id: llm.chat_completions.bedrock_converse.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Converse function_calling; AWS adoption"} @@ -33,6 +35,8 @@ - {id: llm.chat_completions.bedrock_converse.response_headers.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: response_headers, streaming: stream, assertions: [works], source: "llms/bedrock/chat/converse_handler.py:154", rationale: "The llm_provider-* headers must also surface on streaming /chat/completions, where CustomStreamWrapper carries them instead of the nonstream setter"} - {id: llm.chat_completions.bedrock_invoke.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Regional inference-profile ids (us.anthropic.*) over the invoke route, the deployment shape behind a customer timeout report on v1.90.0"} - {id: llm.chat_completions.bedrock_invoke.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming with regional inference-profile ids over the invoke route"} +- {id: llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to Bedrock Invoke builds the Anthropic request through AnthropicConfig.transform_request, so flagged Claude 4.8+/5 must keep a mid-conversation role system reminder in messages; hoisting mutates the cached prefix and collapses the prompt cache (#36559)", fail_before_fix: proven} +- {id: llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to Bedrock Invoke: Claude <= 4.7 and Haiku 4.5 reject role system inside messages, so unflagged models must convert a mid-conversation reminder to a user turn in place (hoisting collapses the prompt cache) and still answer (#36559)", fail_before_fix: proven} - {id: llm.chat_completions.vertex.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Vertex AI"} - {id: llm.chat_completions.gemini.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: gemini, capability: basic, streaming: nonstream, assertions: [works], source: "test_chat_completions_regression_e2e.py", rationale: "Gemini OpenAI-compatible chat translation"} - {id: llm.chat_completions.gemini.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: gemini, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "test_chat_completions_regression_e2e.py", rationale: "Gemini chat cost lands in SpendLogs"} diff --git a/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py b/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py new file mode 100644 index 00000000000..84e147fd09e --- /dev/null +++ b/tests/e2e/llm_translation/test_chat_mid_conversation_system_e2e.py @@ -0,0 +1,323 @@ +"""Live e2e: mid-conversation ``role: "system"`` handling on the OpenAI-format +/v1/chat/completions path is model-aware for first-party Anthropic and Bedrock +Invoke, both of which build the Anthropic request through +``AnthropicConfig.transform_request`` (#36559). + +Only the leading run of system messages becomes the top-level ``system`` +parameter. A ``role: "system"`` entry that appears later in ``messages`` used to +be hoisted into that same field, which rewrote the cached prefix and re-billed +the whole conversation at cache-write pricing on every reminder. Models flagged +``supports_mid_conversation_system`` in the cost map (Claude 4.8+ and the 5 +family) must keep the reminder in ``messages`` as ``role: "system"``; models +without the flag (Claude 4.7 and older, Haiku 4.5) reject that role inside +``messages``, so the proxy must convert the reminder to a user turn in place, +prefixed with an operator note. Either way the prompt cache written on turn one +must be read back in full on turn two. + +The conversation shape mirrors what an OpenAI-SDK client sends mid-session: a +cached system prompt, a user turn carrying its own ``cache_control`` breakpoint, +an assistant turn, a ``role: "system"`` reminder, and a fresh user turn. The +message-turn breakpoint is what makes the cache assertion able to fail: a cache +entry whose prefix spans ``system`` plus message turns is invalidated when the +reminder is hoisted (the ``system`` field mutates and a turn disappears from +``messages``), while an entry ending at the system block itself would survive +the hoist and mask the regression. + +The provider-native ``cache_control`` request shape is not expressible with the +shared ``ChatBody`` (whose content parts carry no cache_control), so the body is +built from the typed content blocks shared in ``endpoints_client.py``. +""" + +from __future__ import annotations + +import time + +import pytest +from pydantic import BaseModel + +from e2e_config import unique_marker +from e2e_http import Result, unwrap +from endpoints_client import CacheControl, RichMessage, TextBlock +from lifecycle import ResourceManager +from models import ChatResponse, LiteLLMParamsBody, Usage +from passthrough_client import PassthroughClient + +pytestmark = pytest.mark.e2e + +CACHE_PRIMING_DEADLINE_SECONDS = 60.0 +CACHE_PRIMING_INTERVAL_SECONDS = 3.0 +CACHE_WARM_CONSECUTIVE_READS = 3 + + +class CacheChatRequest(BaseModel): + """OpenAI-format chat body whose content blocks carry ``cache_control``.""" + + model: str + messages: list[RichMessage] + max_tokens: int = 64 + cache: dict[str, bool] = {"no-cache": True} + + +def _anthropic_params(model: str) -> LiteLLMParamsBody: + return LiteLLMParamsBody(model=model, api_key="os.environ/ANTHROPIC_API_KEY") + + +def _invoke_params(model: str, region: str) -> LiteLLMParamsBody: + return LiteLLMParamsBody(model=model, aws_region_name=region) + + +def _cacheable_system_turn(marker: str) -> RichMessage: + """A system prompt comfortably above the 4096-token minimum cacheable size + of Haiku 4.5 (the smallest model here), unique per run so no other run's + cache entry can satisfy the read.""" + text = " ".join(f"Reference paragraph {index} for run {marker}." for index in range(300)) + return RichMessage(role="system", content=[TextBlock(text=text, cache_control=CacheControl())]) + + +def _user_turn(text: str, *, cached: bool = False) -> RichMessage: + block = TextBlock(text=text, cache_control=CacheControl() if cached else None) + return RichMessage(role="user", content=[block]) + + +def _assistant_turn(text: str) -> RichMessage: + return RichMessage(role="assistant", content=[TextBlock(text=text)]) + + +def _system_reminder_turn() -> RichMessage: + return RichMessage( + role="system", + content=[TextBlock(text="Answer with exactly one word.")], + ) + + +def _post_chat(client: PassthroughClient, key: str, body: CacheChatRequest) -> Result[ChatResponse]: + return client.proxy.transport.post( + "/v1/chat/completions", + headers=client.proxy.transport.bearer(key), + json=body, + response_type=ChatResponse, + ) + + +def _register_deployment(client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody) -> str: + model = f"e2e-chat-midsys-{unique_marker()}" + model_id = client.proxy.create_model(model, params) + resources.defer(lambda: client.proxy.delete_model(model_id)) + return model + + +def _first_turn_user_text(marker: str) -> str: + """A first user turn heavy enough (hundreds of tokens) that losing its cache + entry is unambiguous in the usage numbers, unique per attempt so priming + retries never depend on the proxy's response cache behavior.""" + notes = " ".join(f"Session note {index} for attempt {marker}." for index in range(100)) + return f"Reply with one word.\n{notes}" + + +def _cache_read_tokens(usage: Usage | None) -> int: + """Cache-read tokens however the chat usage reports them: the Anthropic-style + ``cache_read_input_tokens`` litellm forwards, or the OpenAI-style + ``prompt_tokens_details.cached_tokens`` it mirrors them into.""" + if usage is None: + return 0 + if usage.cache_read_input_tokens: + return usage.cache_read_input_tokens + if usage.prompt_tokens_details and usage.prompt_tokens_details.cached_tokens: + return usage.prompt_tokens_details.cached_tokens + return 0 + + +def _cache_creation_tokens(usage: Usage | None) -> int: + if usage is None: + return 0 + return usage.cache_creation_input_tokens or 0 + + +def _response_text(response: ChatResponse) -> str: + return "".join(choice.message.content or "" for choice in response.choices if choice.message) + + +def _response_role(response: ChatResponse) -> str | None: + first = response.choices[0].message if response.choices else None + return first.role if first else None + + +class PrimedCache(BaseModel): + first_user_text: str + prefix_read_tokens: int + first_turn_creation_tokens: int + + @property + def full_prefix_tokens(self) -> int: + return self.prefix_read_tokens + self.first_turn_creation_tokens + + +def _prime_prompt_cache(client: PassthroughClient, key: str, model: str, system_turn: RichMessage) -> PrimedCache: + """Send first-turn calls (fresh cache-marked user turn each attempt, + identical system prefix) until one both reads the system prefix back from + cache and writes its own user-turn chunk, then re-send that exact turn until + its own chunk reads back on three sends in a row, proving the cache is live + in both directions before the reminder turn goes out (a freshly written entry + can take a few seconds to become readable). Only the pre-reminder turn is + ever retried here, so retries can never warm a mutated-prefix cache entry and + mask the regression the second turn asserts on.""" + deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS + while True: + user_text = _first_turn_user_text(unique_marker()) + body = CacheChatRequest(model=model, messages=[system_turn, _user_turn(user_text, cached=True)]) + usage = unwrap(_post_chat(client, key, body)).usage + read_tokens = _cache_read_tokens(usage) + creation_tokens = _cache_creation_tokens(usage) + if read_tokens > 0 and creation_tokens > 0: + primed = PrimedCache( + first_user_text=user_text, + prefix_read_tokens=read_tokens, + first_turn_creation_tokens=creation_tokens, + ) + if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline): + return primed + if time.monotonic() >= deadline: + pytest.fail( + f"{model}: prompt cache never became readable in full within " + f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})" + ) + time.sleep(CACHE_PRIMING_INTERVAL_SECONDS) + + +def _reads_full_prefix(client: PassthroughClient, key: str, body: CacheChatRequest, full_prefix_tokens: int) -> bool: + return _cache_read_tokens(unwrap(_post_chat(client, key, body)).usage) >= full_prefix_tokens + + +def _first_turn_reads_back( + client: PassthroughClient, + key: str, + body: CacheChatRequest, + full_prefix_tokens: int, + deadline: float, +) -> bool: + """True once the full prefix reads back on CACHE_WARM_CONSECUTIVE_READS sends in + a row. Some providers' global endpoints serve the prompt cache per region, so a + fresh entry can be missing from the region the next request lands on; each miss + re-creates the entry there, so the streak converges as the regions warm up.""" + while time.monotonic() < deadline: + if all(_reads_full_prefix(client, key, body, full_prefix_tokens) for _ in range(CACHE_WARM_CONSECUTIVE_READS)): + return True + time.sleep(CACHE_PRIMING_INTERVAL_SECONDS) + return False + + +def _reminder_turn_body(model: str, system_turn: RichMessage, primed: PrimedCache) -> CacheChatRequest: + """Turn two in OpenAI shape: the primed prefix, an assistant reply, the + mid-conversation system reminder, and a fresh cache-marked user turn.""" + return CacheChatRequest( + model=model, + messages=[ + system_turn, + _user_turn(primed.first_user_text, cached=True), + _assistant_turn("OK."), + _system_reminder_turn(), + _user_turn("Reply with one word again.", cached=True), + ], + ) + + +def _assert_flagged_model_keeps_cache( + client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody +) -> None: + model = _register_deployment(client, resources, params) + key = resources.key(models=[model]) + system_turn = _cacheable_system_turn(unique_marker()) + + primed = _prime_prompt_cache(client, key, model, system_turn) + + second = unwrap(_post_chat(client, key, _reminder_turn_body(model, system_turn, primed))) + read_tokens = _cache_read_tokens(second.usage) + + assert _response_role(second) == "assistant", f"{model}: unexpected role {_response_role(second)!r}" + assert _response_text(second).strip(), f"{model}: reminder turn returned no completion text" + assert read_tokens >= primed.full_prefix_tokens, ( + f"{model}: turn with a mid-conversation system reminder read {read_tokens} " + f"cached tokens, expected at least the {primed.full_prefix_tokens} cached on " + f"turn one ({primed.prefix_read_tokens} system prefix + " + f"{primed.first_turn_creation_tokens} first user turn); the reminder was " + f"hoisted into the top-level system field, which mutates the cached prefix " + f"and re-bills the conversation at cache-write pricing" + ) + + +def _assert_unflagged_model_converts_and_succeeds( + client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody +) -> None: + model = _register_deployment(client, resources, params) + key = resources.key(models=[model]) + system_turn = _cacheable_system_turn(unique_marker()) + + primed = _prime_prompt_cache(client, key, model, system_turn) + + second = unwrap(_post_chat(client, key, _reminder_turn_body(model, system_turn, primed))) + read_tokens = _cache_read_tokens(second.usage) + + assert _response_role(second) == "assistant", f"{model}: unexpected role {_response_role(second)!r}" + assert _response_text(second).strip(), ( + f"{model}: conversation with a mid-conversation system reminder returned " + f"no text; the reminder was forwarded in place to a model that rejects " + f"role 'system' inside messages instead of being converted to a user turn" + ) + assert read_tokens >= primed.full_prefix_tokens, ( + f"{model}: reminder turn read {read_tokens} cached tokens, expected at least " + f"the {primed.full_prefix_tokens} cached on turn one " + f"({primed.prefix_read_tokens} system prefix + " + f"{primed.first_turn_creation_tokens} first user turn); the reminder was " + f"hoisted into the top-level system field instead of being converted to a " + f"user turn in place, mutating the cached prefix and re-billing the " + f"conversation at cache-write pricing" + ) + + +class TestAnthropicChatMidConversationSystem: + FLAGGED_MODEL = "anthropic/claude-opus-4-8" + UNFLAGGED_MODEL = "anthropic/claude-haiku-4-5-20251001" + + @pytest.mark.covers( + "llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit", + exercised_on=[], + ) + def test_flagged_model_keeps_prompt_cache_across_system_reminder( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + _assert_flagged_model_keeps_cache(client, resources, _anthropic_params(self.FLAGGED_MODEL)) + + @pytest.mark.covers( + "llm.chat_completions.anthropic.mid_conversation_system.nonstream.works", + exercised_on=[], + ) + def test_unflagged_model_converts_system_reminder_and_succeeds( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + _assert_unflagged_model_converts_and_succeeds(client, resources, _anthropic_params(self.UNFLAGGED_MODEL)) + + +class TestBedrockInvokeChatMidConversationSystem: + FLAGGED_MODEL = "bedrock/invoke/us.anthropic.claude-sonnet-5" + UNFLAGGED_MODEL = "bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0" + AWS_REGION = "us-east-1" + + @pytest.mark.covers( + "llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit", + exercised_on=[], + ) + def test_flagged_model_keeps_prompt_cache_across_system_reminder( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + _assert_flagged_model_keeps_cache(client, resources, _invoke_params(self.FLAGGED_MODEL, self.AWS_REGION)) + + @pytest.mark.covers( + "llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works", + exercised_on=[], + ) + def test_unflagged_model_converts_system_reminder_and_succeeds( + self, client: PassthroughClient, resources: ResourceManager + ) -> None: + _assert_unflagged_model_converts_and_succeeds( + client, resources, _invoke_params(self.UNFLAGGED_MODEL, self.AWS_REGION) + ) From 2bd7771fb988182d14ef5238dbfeb67b1397814e Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Sun, 23 Aug 2026 21:21:53 -0500 Subject: [PATCH 5/7] test(anthropic): cover the cache_control rebuild shapes and type the test helpers Codecov flagged the 5m ttl branch and the empty-system path of the wire builder; both now have a test. Greptile asked for full typing on the new test helpers. --- ...llm_core_utils_prompt_templates_factory.py | 13 +++++++ .../test_anthropic_chat_transformation.py | 8 ++--- .../anthropic/test_mid_conversation_system.py | 36 ++++++++++++++----- .../test_azure_anthropic_transformation.py | 2 +- ...ations_anthropic_claude3_transformation.py | 2 +- ...partner_models_anthropic_transformation.py | 2 +- 6 files changed, 47 insertions(+), 16 deletions(-) diff --git a/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py b/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py index cf85329aa12..6c9d274d101 100644 --- a/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py +++ b/tests/test_litellm/litellm_core_utils/prompt_templates/test_litellm_core_utils_prompt_templates_factory.py @@ -3613,3 +3613,16 @@ def test_anthropic_messages_pt_system_string_content_becomes_text_block(): assert result[1] == {"role": "system", "content": [{"type": "text", "text": "Answer in one word."}]} + +def test_anthropic_messages_pt_drops_a_system_message_with_no_text(): + """Anthropic rejects empty text blocks, so a text-less system message must + vanish rather than reach the wire as an empty system turn.""" + messages = [ + {"role": "user", "content": "First question"}, + {"role": "system", "content": ""}, + {"role": "assistant", "content": "Yes"}, + ] + + result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic") + + assert [m["role"] for m in result] == ["user", "assistant"] diff --git a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py index 74d2c22bdf3..cbde7c3cd79 100644 --- a/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/test_litellm/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -6237,7 +6237,7 @@ REMINDER_TEXT = "Answer with exactly one word. dict: return config.transform_request( model=model, messages=messages, @@ -6247,7 +6247,7 @@ def _chat_request(config, model, messages): ) -def _reminder_conversation(): +def _reminder_conversation() -> list[dict]: """The shape Claude Code sends mid-session: cached system prompt, turns, a reminder right after a user turn, an assistant turn, a fresh user turn.""" return [ @@ -6261,7 +6261,7 @@ def _reminder_conversation(): ] -def _texts(message): +def _texts(message: dict) -> list[str]: return [block["text"] for block in message["content"] if block.get("type") == "text"] @@ -6418,7 +6418,7 @@ def test_chat_mid_conversation_system_keeps_earlier_turns_a_prefix_of_the_next_r assert len(later_blocks) > len(earlier_blocks) -def _role_block_pairs(messages): +def _role_block_pairs(messages: list[dict]) -> list[tuple[str, object]]: return [ (message["role"], block) for message in messages diff --git a/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py b/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py index f1042a2e17e..a3a6dd50a50 100644 --- a/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py +++ b/tests/test_litellm/llms/anthropic/test_mid_conversation_system.py @@ -5,6 +5,8 @@ Anthropic, Vertex, Azure AI and Bedrock Invoke transformation tests; these pin the pure placement rules on the OpenAI-format message list. """ +import pytest + import litellm from litellm.llms.anthropic.mid_conversation_system import ( CONVERTED_SYSTEM_NOTE, @@ -13,11 +15,11 @@ from litellm.llms.anthropic.mid_conversation_system import ( ) -def _roles(messages): +def _roles(messages: object) -> list[str]: return [m["role"] if isinstance(m, dict) else m.role for m in messages] -def _texts(message): +def _texts(message: dict) -> list[str]: return [block["text"] for block in message["content"]] @@ -125,19 +127,35 @@ def test_unflagged_conversion_keeps_the_client_order_when_no_tool_result_follows assert _texts(placed[2]) == [CONVERTED_SYSTEM_NOTE, "reminder"] -def test_unflagged_conversion_rebuilds_cache_control_on_the_converted_block(): +@pytest.mark.parametrize( + "cache_control, expected", + [ + ({"type": "ephemeral", "ttl": "1h"}, {"type": "ephemeral", "ttl": "1h"}), + ({"type": "ephemeral", "ttl": "5m"}, {"type": "ephemeral", "ttl": "5m"}), + ({"type": "ephemeral", "ttl": "2h"}, {"type": "ephemeral"}), + ], +) +def test_unflagged_conversion_rebuilds_cache_control_on_the_converted_block(cache_control, expected): + """Only the shapes Anthropic accepts survive: ephemeral with a 5m or 1h ttl, or no ttl.""" messages = [ {"role": "user", "content": "q1"}, - {"role": "system", "content": "reminder", "cache_control": {"type": "ephemeral", "ttl": "1h"}}, + {"role": "system", "content": "reminder", "cache_control": cache_control}, ] placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False) - assert placed[1]["content"][1] == { - "type": "text", - "text": "reminder", - "cache_control": {"type": "ephemeral", "ttl": "1h"}, - } + assert placed[1]["content"][1] == {"type": "text", "text": "reminder", "cache_control": expected} + + +def test_unflagged_conversion_drops_a_cache_control_that_is_not_ephemeral(): + messages = [ + {"role": "user", "content": "q1"}, + {"role": "system", "content": "reminder", "cache_control": {"type": "persistent"}}, + ] + + placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False) + + assert placed[1]["content"][1] == {"type": "text", "text": "reminder"} def test_placement_is_a_no_op_without_later_system_messages(): diff --git a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py index 257d5f92ee5..234eeae8722 100644 --- a/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py +++ b/tests/test_litellm/llms/azure_ai/claude/test_azure_anthropic_transformation.py @@ -415,7 +415,7 @@ class TestAzureAnthropicConfig: assert "context-management-2025-06-27" in headers["anthropic-beta"] -def _mid_conversation_system_conversation(): +def _mid_conversation_system_conversation() -> list[dict]: return [ {"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]}, {"role": "user", "content": "First question"}, diff --git a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py index 080050c7621..62313c4f562 100644 --- a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py @@ -544,7 +544,7 @@ def test_output_format_removed_from_bedrock_invoke_request(): ), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}" -def _mid_conversation_system_conversation(): +def _mid_conversation_system_conversation() -> list[dict]: return [ {"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]}, {"role": "user", "content": "First question"}, diff --git a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py index 65a77364bc7..d8f8f2170d7 100644 --- a/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py +++ b/tests/test_litellm/llms/vertex_ai/vertex_ai_partner_models/anthropic/test_vertex_ai_partner_models_anthropic_transformation.py @@ -729,7 +729,7 @@ def test_sanitize_strips_effort_for_haiku_45(): assert data["output_config"] == {"effort": "high"} -def _mid_conversation_system_conversation(): +def _mid_conversation_system_conversation() -> list[dict]: return [ {"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]}, {"role": "user", "content": "First question"}, From d2c3d07267fb66e8c45415f756dad0f7f2741883 Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Tue, 25 Aug 2026 01:27:39 -0500 Subject: [PATCH 6/7] refactor(anthropic): read the mid-conversation flag through a public supports_ helper supports_mid_conversation_system joins the other supports_* helpers in litellm.utils, so the chat transformation stops importing the private _supports_factory. --- litellm/llms/anthropic/chat/transformation.py | 8 +++----- litellm/utils.py | 9 +++++++++ 2 files changed, 12 insertions(+), 5 deletions(-) diff --git a/litellm/llms/anthropic/chat/transformation.py b/litellm/llms/anthropic/chat/transformation.py index b442d27c91e..c1b2cb2985e 100644 --- a/litellm/llms/anthropic/chat/transformation.py +++ b/litellm/llms/anthropic/chat/transformation.py @@ -75,12 +75,12 @@ from litellm.types.utils import Message as LitellmMessage from litellm.utils import ( ModelResponse, Usage, - _supports_factory, add_dummy_tool, any_assistant_message_has_thinking_blocks, get_max_tokens, has_tool_call_blocks, last_assistant_with_tool_calls_has_no_thinking_blocks, + supports_mid_conversation_system, supports_reasoning, token_counter, ) @@ -1911,10 +1911,8 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig): optional_params["system"] = anthropic_system_message_list conversation: Final = place_mid_conversation_system( later_messages, - supports_mid_conversation_system=_supports_factory( - model=model, - custom_llm_provider=self.custom_llm_provider, - key="supports_mid_conversation_system", + supports_mid_conversation_system=supports_mid_conversation_system( + model=model, custom_llm_provider=self.custom_llm_provider ), ) # Format rest of message according to anthropic guidelines diff --git a/litellm/utils.py b/litellm/utils.py index c2770a1a26d..727ea140ee4 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -2722,6 +2722,15 @@ def supports_reasoning(model: str, custom_llm_provider: str | None = None) -> bo return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_reasoning") +def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool: + """ + Check if the given model accepts role=system messages after the first turn and return a boolean value. + """ + return _supports_factory( + model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system" + ) + + def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool: """ Check if the given model supports native structured outputs and return a boolean value. From 6e1e44ce41ddaf52cc5b68f65b560ea8e06fa5f0 Mon Sep 17 00:00:00 2001 From: Shifat Islam Santo Date: Wed, 26 Aug 2026 16:49:48 -0500 Subject: [PATCH 7/7] chore(typing): declare the mid-conversation type aliases with TypeAlias The Final sweep tightened LIT010, which exempts TypeAlias declarations but counts a bare alias assignment as an unannotated binding. --- litellm/litellm_core_utils/prompt_templates/factory.py | 4 ++-- litellm/llms/anthropic/mid_conversation_system.py | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/factory.py b/litellm/litellm_core_utils/prompt_templates/factory.py index 4cd8a689afa..9949639ffc6 100644 --- a/litellm/litellm_core_utils/prompt_templates/factory.py +++ b/litellm/litellm_core_utils/prompt_templates/factory.py @@ -7,7 +7,7 @@ import re import xml.etree.ElementTree as ET from collections.abc import Iterator, Mapping, Sequence from enum import Enum -from typing import Any, Final, TypedDict, cast, overload +from typing import Any, Final, TypeAlias, TypedDict, cast, overload from jinja2.sandbox import ImmutableSandboxedEnvironment @@ -2307,7 +2307,7 @@ def _drop_unsignable_thinking_blocks( # mutable-ok: anthropic_messages_pt's callers have always appended to the list it returns -_AnthropicMessageList = list[AllAnthropicPassThroughMessageValues] +_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues] def anthropic_messages_pt( diff --git a/litellm/llms/anthropic/mid_conversation_system.py b/litellm/llms/anthropic/mid_conversation_system.py index 766faf66312..8b1c2b0c6d2 100644 --- a/litellm/llms/anthropic/mid_conversation_system.py +++ b/litellm/llms/anthropic/mid_conversation_system.py @@ -26,7 +26,7 @@ Anthropic wire shape is built later by ``anthropic_messages_pt``. from collections.abc import Iterator, Mapping, Sequence from itertools import groupby -from typing import Final, Literal +from typing import Final, Literal, TypeAlias from litellm.types.llms.anthropic import AnthropicMessagesSystemMessageParam, AnthropicSystemMessageContent from litellm.types.llms.openai import ( @@ -44,8 +44,8 @@ CONVERTED_SYSTEM_NOTE: Final = ( _USER_TYPE_ROLES: Final = frozenset({"user", "tool", "function"}) _TOOL_ROLES: Final = frozenset({"tool", "function"}) -_MessageKind = Literal["system", "tool", "user", "other"] -_TextPart = tuple[str, ChatCompletionCachedContent | None] +_MessageKind: TypeAlias = Literal["system", "tool", "user", "other"] +_TextPart: TypeAlias = tuple[str, ChatCompletionCachedContent | None] def _as_mapping(value: object) -> Mapping[str, object] | None: