This commit is contained in:
Shifat Islam Santo 2026-08-26 21:50:06 +00:00 • committed by GitHub
commit e71b4af5a5
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
14 changed files with 1237 additions and 11 deletions

View file

@ -7,7 +7,7 @@ import re
import xml.etree.ElementTree as ET
from collections.abc import Iterator, Mapping, Sequence
from enum import Enum
from typing import Any, Final, TypedDict, cast, overload
from typing import Any, Final, TypeAlias, TypedDict, cast, overload
from jinja2.sandbox import ImmutableSandboxedEnvironment
@ -18,6 +18,7 @@ from litellm import verbose_logger
from litellm._uuid import uuid
from litellm.constants import REDACTED_BY_LITELLM
from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get
from litellm.llms.anthropic.mid_conversation_system import anthropic_system_messages
from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client
from litellm.types.files import get_file_extension_from_mime_type
from litellm.types.llms.anthropic import *
@ -2305,14 +2306,20 @@ def _drop_unsignable_thinking_blocks(
return [block for block in thinking_blocks if not _is_unsignable_thinking_block(block)]
# mutable-ok: anthropic_messages_pt's callers have always appended to the list it returns
_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues]
def anthropic_messages_pt(
messages: list[AllMessageValues],
model: str,
llm_provider: str,
) -> list[AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam]:
) -> _AnthropicMessageList:
"""
format messages for anthropic
1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately)
1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately).
Models flagged ``supports_mid_conversation_system`` also accept "system" inside
messages after a user turn; the caller decides placement, this keeps such messages.
2. The first message always needs to be of role "user"
3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm)
4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise)
@ -2337,7 +2344,7 @@ def anthropic_messages_pt(
# add role=tool support to allow function call result/error submission
user_message_types: Final = {"user", "tool", "function"}
# reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them.
new_messages: Final[list[AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam]] = []
new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract
if len(messages) == 0:
if not litellm.modify_params:
@ -2730,6 +2737,11 @@ def anthropic_messages_pt(
if assistant_content:
new_messages.append({"role": "assistant", "content": assistant_content})
## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ##
while msg_i < len(messages) and messages[msg_i]["role"] == "system":
new_messages.extend(anthropic_system_messages(messages[msg_i]))
msg_i += 1
if msg_i == init_msg_i: # prevent infinite loops
raise litellm.BadRequestError(
message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}",

View file

@ -80,6 +80,7 @@ from litellm.utils import (
get_max_tokens,
has_tool_call_blocks,
last_assistant_with_tool_calls_has_no_thinking_blocks,
supports_mid_conversation_system,
supports_reasoning,
token_counter,
)
@ -90,6 +91,7 @@ from ..common_utils import (
process_anthropic_headers,
strip_advisor_blocks_from_messages,
)
from ..mid_conversation_system import place_mid_conversation_system, split_leading_system_run
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
@ -1897,16 +1899,27 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
if _name_reverse_map and isinstance(litellm_params, dict):
litellm_params[ANTHROPIC_TOOL_NAME_REVERSE_MAP_KEY] = _name_reverse_map
# Separate system prompt from rest of message
anthropic_system_message_list: Final = self.translate_system_message(messages=messages)
# Only the leading system run becomes the top-level system prompt. A later
# system message stays in the conversation: hoisting it rewrites the cached
# prefix and re-bills the whole history at cache-write pricing (#36559).
leading_system_run, later_messages = split_leading_system_run(messages)
anthropic_system_message_list: Final = self.translate_system_message(
messages=list(leading_system_run) # mutable-ok: translate_system_message pops from the list it is given
)
# Handling anthropic API Prompt Caching
if len(anthropic_system_message_list) > 0:
optional_params["system"] = anthropic_system_message_list
conversation: Final = place_mid_conversation_system(
later_messages,
supports_mid_conversation_system=supports_mid_conversation_system(
model=model, custom_llm_provider=self.custom_llm_provider
),
)
# Format rest of message according to anthropic guidelines
try:
anthropic_messages = anthropic_messages_pt(
model=model,
messages=messages,
messages=list(conversation), # mutable-ok: anthropic_messages_pt rewrites entries in place
llm_provider=self._resolved_provider,
)
except Exception as e:

View file

@ -31,6 +31,7 @@ from ...common_utils import (
optionally_handle_anthropic_oauth,
strip_advisor_blocks_from_messages,
)
from ...mid_conversation_system import CONVERTED_SYSTEM_NOTE
DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01"
@ -160,9 +161,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
def _is_system_role_message(message: Any) -> bool:
return isinstance(message, dict) and message.get("role") == "system"
_CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
_CONVERTED_SYSTEM_NOTE: Final = CONVERTED_SYSTEM_NOTE
def _system_role_message_as_user(self, message: Mapping) -> Mapping:
return {

View file

@ -0,0 +1,290 @@
"""Placement policy for ``role: "system"`` messages that appear after the first turn
of an Anthropic-shaped chat completions request.
Only the leading run of system messages belongs in the top-level ``system``
parameter. Hoisting a later one there rewrites the cached prefix, so the provider
re-bills the whole conversation at cache-write pricing on every reminder (#36559).
Models flagged ``supports_mid_conversation_system`` in the cost map accept the role
inside ``messages`` under Anthropic's placement rules: the message must directly
follow a user turn, must be the last entry or be followed by an assistant turn, and
must not sit next to another system message. OpenAI-shaped clients put system
messages anywhere, so this module moves each one to the nearest valid slot and
merges runs that land together.
Models without the flag reject the role inside ``messages``. Their system messages
become user turns in place, prefixed with an operator note so the model can tell
the instruction apart from the user's own words. A run caught between a tool call
and its result moves to just after the result so the ``tool_result`` block stays
first in the merged user turn.
Every transformation here is a pure function of the message sequence: turn N's
output stays a prefix of turn N+1's output, which is what keeps the provider-side
prompt cache readable across turns. Messages are handled in OpenAI format; the
Anthropic wire shape is built later by ``anthropic_messages_pt``.
"""
from collections.abc import Iterator, Mapping, Sequence
from itertools import groupby
from typing import Final, Literal, TypeAlias
from litellm.types.llms.anthropic import AnthropicMessagesSystemMessageParam, AnthropicSystemMessageContent
from litellm.types.llms.openai import (
AllMessageValues,
ChatCompletionCachedContent,
ChatCompletionSystemMessage,
ChatCompletionTextObject,
ChatCompletionUserMessage,
)
CONVERTED_SYSTEM_NOTE: Final = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
_USER_TYPE_ROLES: Final = frozenset({"user", "tool", "function"})
_TOOL_ROLES: Final = frozenset({"tool", "function"})
_MessageKind: TypeAlias = Literal["system", "tool", "user", "other"]
_TextPart: TypeAlias = tuple[str, ChatCompletionCachedContent | None]
def _as_mapping(value: object) -> Mapping[str, object] | None:
return value if isinstance(value, Mapping) else None
def _as_items(value: object) -> tuple[object, ...]:
return tuple(value) if isinstance(value, Sequence) and not isinstance(value, str) else ()
def _field(message: object, key: str) -> object:
"""A message field, whether the message is a dict or a pydantic ``Message``."""
mapping: Final = _as_mapping(message)
return mapping.get(key) if mapping is not None else getattr(message, key, None)
def is_system_message(message: object) -> bool:
return _field(message, "role") == "system"
def _is_user_type(message: object) -> bool:
return _field(message, "role") in _USER_TYPE_ROLES
def _kind(message: object) -> _MessageKind:
role: Final = _field(message, "role")
if role == "system":
return "system"
if role in _TOOL_ROLES:
return "tool"
if role == "user":
return "user"
return "other"
def split_leading_system_run(
messages: Sequence[AllMessageValues],
) -> tuple[tuple[AllMessageValues, ...], tuple[AllMessageValues, ...]]:
"""Split ``messages`` into the leading run of system messages and everything after it."""
leading_count: Final = next(
(index for index, message in enumerate(messages) if not is_system_message(message)),
len(messages),
)
return tuple(messages[:leading_count]), tuple(messages[leading_count:])
def _cache_control(holder: object) -> ChatCompletionCachedContent | None:
"""The client's ``cache_control`` rebuilt in the only shape Anthropic accepts."""
value: Final = _as_mapping(_field(holder, "cache_control"))
if value is None or value.get("type") != "ephemeral":
return None
ttl: Final = value.get("ttl")
if ttl == "1h":
one_hour: Final[ChatCompletionCachedContent] = {"type": "ephemeral", "ttl": "1h"}
return one_hour
if ttl == "5m":
five_minutes: Final[ChatCompletionCachedContent] = {"type": "ephemeral", "ttl": "5m"}
return five_minutes
ephemeral: Final[ChatCompletionCachedContent] = {"type": "ephemeral"}
return ephemeral
def _text_parts(message: object) -> tuple[_TextPart, ...]:
"""``(text, cache_control)`` for each non-empty text part of a system message.
Anthropic rejects empty text blocks and only accepts text in system content. A
``cache_control`` on the message itself belongs to the block built from string
content; block-level ``cache_control`` stays with its block.
"""
content: Final = _field(message, "content")
if isinstance(content, str):
return ((content, _cache_control(message)),) if content else ()
return tuple(
(text, _cache_control(part))
for part in _as_items(content)
if _field(part, "type") == "text"
for text in (_field(part, "text"),)
if isinstance(text, str) and text
)
def _openai_text_block(part: _TextPart) -> ChatCompletionTextObject:
text, cache_control = part
if cache_control is None:
plain: Final[ChatCompletionTextObject] = {"type": "text", "text": text}
return plain
cached: Final[ChatCompletionTextObject] = {"type": "text", "text": text, "cache_control": cache_control}
return cached
def _anthropic_text_block(part: _TextPart) -> AnthropicSystemMessageContent:
text, cache_control = part
if cache_control is None:
plain: Final[AnthropicSystemMessageContent] = {"type": "text", "text": text}
return plain
cached: Final[AnthropicSystemMessageContent] = {"type": "text", "text": text, "cache_control": cache_control}
return cached
def anthropic_system_messages(message: object) -> tuple[AnthropicMessagesSystemMessageParam, ...]:
"""The Anthropic wire message for a system message, or nothing when it carries no text."""
blocks: Final = tuple(_anthropic_text_block(part) for part in _text_parts(message))
if not blocks:
return ()
wire: Final[AnthropicMessagesSystemMessageParam] = {
"role": "system",
"content": list(blocks), # mutable-ok: wire payload; cache_control hooks edit content blocks in place
}
return (wire,)
def system_message_as_user(message: object) -> ChatCompletionUserMessage:
"""A system message re-rolled as a user turn, prefixed with the operator note."""
note: Final[ChatCompletionTextObject] = {"type": "text", "text": CONVERTED_SYSTEM_NOTE}
content: Final[list[ChatCompletionTextObject]] = [ # mutable-ok: anthropic_messages_pt only recognises list content
note,
*(_openai_text_block(part) for part in _text_parts(message)),
]
turn: Final[ChatCompletionUserMessage] = {"role": "user", "content": content}
return turn
def _merged_system_message(run: Sequence[object]) -> tuple[ChatCompletionSystemMessage, ...]:
parts: Final = tuple(part for message in run for part in _text_parts(message))
if not parts:
return ()
content: Final[list[ChatCompletionTextObject]] = [ # mutable-ok: anthropic_messages_pt only recognises list content
_openai_text_block(part) for part in parts
]
merged: Final[ChatCompletionSystemMessage] = {"role": "system", "content": content}
return (merged,)
def _converted_user_turns(run: Sequence[object]) -> tuple[ChatCompletionUserMessage, ...]:
return tuple(system_message_as_user(message) for message in run if _text_parts(message))
def _runs(messages: Sequence[AllMessageValues]) -> tuple[tuple[_MessageKind, tuple[AllMessageValues, ...]], ...]:
return tuple((kind, tuple(group)) for kind, group in groupby(messages, key=_kind))
def _converted_for_unflagged_model(messages: Sequence[AllMessageValues]) -> tuple[AllMessageValues, ...]:
"""Convert every system message to a user turn in place.
A system run whose follower is a tool message is emitted after that tool run:
``tool_result`` blocks have to open the merged user turn.
"""
runs: Final = _runs(messages)
def emit(index: int) -> tuple[AllMessageValues, ...]:
kind, run = runs[index]
follower: Final = runs[index + 1][0] if index + 1 < len(runs) else None
if kind == "system":
return () if follower == "tool" else _converted_user_turns(run)
if kind == "tool" and index > 0 and runs[index - 1][0] == "system":
return (*run, *_converted_user_turns(runs[index - 1][1]))
return run
return tuple(message for index in range(len(runs)) for message in emit(index))
def _user_type_blocks(messages: Sequence[AllMessageValues]) -> tuple[tuple[bool, tuple[int, ...]], ...]:
"""Maximal groups of consecutive non-system messages, keyed by whether they are user-type.
Consecutive user-type messages become one user turn on the wire, so a group is
the unit a system message can validly follow.
"""
indexed: Final = tuple((index, message) for index, message in enumerate(messages) if not is_system_message(message))
return tuple(
(is_user, tuple(index for index, _ in group))
for is_user, group in groupby(indexed, key=lambda pair: _is_user_type(pair[1]))
)
def _system_runs(messages: Sequence[AllMessageValues]) -> tuple[tuple[int, ...], ...]:
"""Index runs of consecutive system messages."""
system_indices: Final = tuple(index for index, message in enumerate(messages) if is_system_message(message))
return tuple(
tuple(index for _, index in group)
for _, group in groupby(enumerate(system_indices), key=lambda pair: pair[1] - pair[0])
)
def _anchor_block(
run_start: int,
messages: Sequence[AllMessageValues],
blocks: Sequence[tuple[bool, tuple[int, ...]]],
) -> int | None:
"""The user-type block a system run must follow, or ``None`` when no user turn can host it.
``run_start`` is never 0 here: the leading system run was split off before this
policy runs, so the message before a run is always a non-system message.
"""
if _is_user_type(messages[run_start - 1]):
return next(index for index, (is_user, indices) in enumerate(blocks) if is_user and run_start - 1 in indices)
return next((index for index, (is_user, indices) in enumerate(blocks) if is_user and indices[0] > run_start), None)
def _placed_for_flagged_model(messages: Sequence[AllMessageValues]) -> tuple[AllMessageValues, ...]:
"""Keep system messages as ``role: "system"`` at a placement Anthropic accepts.
A run already sitting after a user-type message stays with that user turn. A
run after an assistant turn moves to just after the next user turn. Runs that
share a user turn merge into one system message. A run with no user turn left
to host it becomes user turns in place.
"""
blocks: Final = _user_type_blocks(messages)
anchors: Final = tuple((run, _anchor_block(run[0], messages, blocks)) for run in _system_runs(messages))
def anchored_to(block_index: int) -> tuple[AllMessageValues, ...]:
return tuple(messages[index] for run, anchor in anchors if anchor == block_index for index in run)
def converted_after(message_index: int) -> tuple[ChatCompletionUserMessage, ...]:
return tuple(
turn
for run, anchor in anchors
if anchor is None and run[0] == message_index + 1
for turn in _converted_user_turns(tuple(messages[index] for index in run))
)
def emit(block_index: int) -> Iterator[AllMessageValues]:
is_user, indices = blocks[block_index]
for index in indices:
yield messages[index]
yield from converted_after(index)
if is_user:
yield from _merged_system_message(anchored_to(block_index))
return tuple(message for block_index in range(len(blocks)) for message in emit(block_index))
def place_mid_conversation_system(
messages: Sequence[AllMessageValues],
*,
supports_mid_conversation_system: bool,
) -> tuple[AllMessageValues, ...]:
"""Apply the placement policy to the messages after the leading system run."""
if not any(is_system_message(message) for message in messages):
return tuple(messages)
if supports_mid_conversation_system:
return _placed_for_flagged_model(messages)
return _converted_for_unflagged_model(messages)

View file

@ -361,7 +361,8 @@ class AnthropicMessagesSystemMessageParam(TypedDict, total=False):
AllAnthropicMessageValues = AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam
# System is not a native Anthropic message role; only pass-through adapters use this union.
# role=system inside messages is accepted after a user turn on models flagged
# supports_mid_conversation_system; pass-through adapters and the chat translator both emit it.
AllAnthropicPassThroughMessageValues: TypeAlias = (
AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam | AnthropicMessagesSystemMessageParam
)

View file

@ -2722,6 +2722,15 @@ def supports_reasoning(model: str, custom_llm_provider: str | None = None) -> bo
return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_reasoning")
def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model accepts role=system messages after the first turn and return a boolean value.
"""
return _supports_factory(
model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system"
)
def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool:
"""
Check if the given model supports native structured outputs and return a boolean value.

View file

@ -23,6 +23,8 @@
- {id: llm.chat_completions.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude prompt caching"}
- {id: llm.chat_completions.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude extended thinking"}
- {id: llm.chat_completions.anthropic.structured_output.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: structured_output, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude response_schema"}
- {id: llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to first-party Anthropic: flagged Claude 4.8+/5 must keep a mid-conversation role system reminder in messages; hoisting it into the top-level system field mutates the cached prefix and re-bills the conversation at cache-write pricing (#36559)", fail_before_fix: proven}
- {id: llm.chat_completions.anthropic.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to first-party Anthropic: Claude <= 4.7 and Haiku 4.5 reject role system inside messages, so unflagged models must convert a mid-conversation reminder to a user turn in place (hoisting collapses the prompt cache) and still answer (#36559)", fail_before_fix: proven}
- {id: llm.chat_completions.bedrock_converse.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Bedrock Converse unified"}
- {id: llm.chat_completions.bedrock_converse.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Converse"}
- {id: llm.chat_completions.bedrock_converse.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Converse function_calling; AWS adoption"}
@ -33,6 +35,8 @@
- {id: llm.chat_completions.bedrock_converse.response_headers.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: response_headers, streaming: stream, assertions: [works], source: "llms/bedrock/chat/converse_handler.py:154", rationale: "The llm_provider-* headers must also surface on streaming /chat/completions, where CustomStreamWrapper carries them instead of the nonstream setter"}
- {id: llm.chat_completions.bedrock_invoke.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Regional inference-profile ids (us.anthropic.*) over the invoke route, the deployment shape behind a customer timeout report on v1.90.0"}
- {id: llm.chat_completions.bedrock_invoke.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming with regional inference-profile ids over the invoke route"}
- {id: llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to Bedrock Invoke builds the Anthropic request through AnthropicConfig.transform_request, so flagged Claude 4.8+/5 must keep a mid-conversation role system reminder in messages; hoisting mutates the cached prefix and collapses the prompt cache (#36559)", fail_before_fix: proven}
- {id: llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to Bedrock Invoke: Claude <= 4.7 and Haiku 4.5 reject role system inside messages, so unflagged models must convert a mid-conversation reminder to a user turn in place (hoisting collapses the prompt cache) and still answer (#36559)", fail_before_fix: proven}
- {id: llm.chat_completions.vertex.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Vertex AI"}
- {id: llm.chat_completions.gemini.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: gemini, capability: basic, streaming: nonstream, assertions: [works], source: "test_chat_completions_regression_e2e.py", rationale: "Gemini OpenAI-compatible chat translation"}
- {id: llm.chat_completions.gemini.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: gemini, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "test_chat_completions_regression_e2e.py", rationale: "Gemini chat cost lands in SpendLogs"}

View file

@ -0,0 +1,323 @@
"""Live e2e: mid-conversation ``role: "system"`` handling on the OpenAI-format
/v1/chat/completions path is model-aware for first-party Anthropic and Bedrock
Invoke, both of which build the Anthropic request through
``AnthropicConfig.transform_request`` (#36559).
Only the leading run of system messages becomes the top-level ``system``
parameter. A ``role: "system"`` entry that appears later in ``messages`` used to
be hoisted into that same field, which rewrote the cached prefix and re-billed
the whole conversation at cache-write pricing on every reminder. Models flagged
``supports_mid_conversation_system`` in the cost map (Claude 4.8+ and the 5
family) must keep the reminder in ``messages`` as ``role: "system"``; models
without the flag (Claude 4.7 and older, Haiku 4.5) reject that role inside
``messages``, so the proxy must convert the reminder to a user turn in place,
prefixed with an operator note. Either way the prompt cache written on turn one
must be read back in full on turn two.
The conversation shape mirrors what an OpenAI-SDK client sends mid-session: a
cached system prompt, a user turn carrying its own ``cache_control`` breakpoint,
an assistant turn, a ``role: "system"`` reminder, and a fresh user turn. The
message-turn breakpoint is what makes the cache assertion able to fail: a cache
entry whose prefix spans ``system`` plus message turns is invalidated when the
reminder is hoisted (the ``system`` field mutates and a turn disappears from
``messages``), while an entry ending at the system block itself would survive
the hoist and mask the regression.
The provider-native ``cache_control`` request shape is not expressible with the
shared ``ChatBody`` (whose content parts carry no cache_control), so the body is
built from the typed content blocks shared in ``endpoints_client.py``.
"""
from __future__ import annotations
import time
import pytest
from pydantic import BaseModel
from e2e_config import unique_marker
from e2e_http import Result, unwrap
from endpoints_client import CacheControl, RichMessage, TextBlock
from lifecycle import ResourceManager
from models import ChatResponse, LiteLLMParamsBody, Usage
from passthrough_client import PassthroughClient
pytestmark = pytest.mark.e2e
CACHE_PRIMING_DEADLINE_SECONDS = 60.0
CACHE_PRIMING_INTERVAL_SECONDS = 3.0
CACHE_WARM_CONSECUTIVE_READS = 3
class CacheChatRequest(BaseModel):
"""OpenAI-format chat body whose content blocks carry ``cache_control``."""
model: str
messages: list[RichMessage]
max_tokens: int = 64
cache: dict[str, bool] = {"no-cache": True}
def _anthropic_params(model: str) -> LiteLLMParamsBody:
return LiteLLMParamsBody(model=model, api_key="os.environ/ANTHROPIC_API_KEY")
def _invoke_params(model: str, region: str) -> LiteLLMParamsBody:
return LiteLLMParamsBody(model=model, aws_region_name=region)
def _cacheable_system_turn(marker: str) -> RichMessage:
"""A system prompt comfortably above the 4096-token minimum cacheable size
of Haiku 4.5 (the smallest model here), unique per run so no other run's
cache entry can satisfy the read."""
text = " ".join(f"Reference paragraph {index} for run {marker}." for index in range(300))
return RichMessage(role="system", content=[TextBlock(text=text, cache_control=CacheControl())])
def _user_turn(text: str, *, cached: bool = False) -> RichMessage:
block = TextBlock(text=text, cache_control=CacheControl() if cached else None)
return RichMessage(role="user", content=[block])
def _assistant_turn(text: str) -> RichMessage:
return RichMessage(role="assistant", content=[TextBlock(text=text)])
def _system_reminder_turn() -> RichMessage:
return RichMessage(
role="system",
content=[TextBlock(text="<system-reminder>Answer with exactly one word.</system-reminder>")],
)
def _post_chat(client: PassthroughClient, key: str, body: CacheChatRequest) -> Result[ChatResponse]:
return client.proxy.transport.post(
"/v1/chat/completions",
headers=client.proxy.transport.bearer(key),
json=body,
response_type=ChatResponse,
)
def _register_deployment(client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody) -> str:
model = f"e2e-chat-midsys-{unique_marker()}"
model_id = client.proxy.create_model(model, params)
resources.defer(lambda: client.proxy.delete_model(model_id))
return model
def _first_turn_user_text(marker: str) -> str:
"""A first user turn heavy enough (hundreds of tokens) that losing its cache
entry is unambiguous in the usage numbers, unique per attempt so priming
retries never depend on the proxy's response cache behavior."""
notes = " ".join(f"Session note {index} for attempt {marker}." for index in range(100))
return f"Reply with one word.\n{notes}"
def _cache_read_tokens(usage: Usage | None) -> int:
"""Cache-read tokens however the chat usage reports them: the Anthropic-style
``cache_read_input_tokens`` litellm forwards, or the OpenAI-style
``prompt_tokens_details.cached_tokens`` it mirrors them into."""
if usage is None:
return 0
if usage.cache_read_input_tokens:
return usage.cache_read_input_tokens
if usage.prompt_tokens_details and usage.prompt_tokens_details.cached_tokens:
return usage.prompt_tokens_details.cached_tokens
return 0
def _cache_creation_tokens(usage: Usage | None) -> int:
if usage is None:
return 0
return usage.cache_creation_input_tokens or 0
def _response_text(response: ChatResponse) -> str:
return "".join(choice.message.content or "" for choice in response.choices if choice.message)
def _response_role(response: ChatResponse) -> str | None:
first = response.choices[0].message if response.choices else None
return first.role if first else None
class PrimedCache(BaseModel):
first_user_text: str
prefix_read_tokens: int
first_turn_creation_tokens: int
@property
def full_prefix_tokens(self) -> int:
return self.prefix_read_tokens + self.first_turn_creation_tokens
def _prime_prompt_cache(client: PassthroughClient, key: str, model: str, system_turn: RichMessage) -> PrimedCache:
"""Send first-turn calls (fresh cache-marked user turn each attempt,
identical system prefix) until one both reads the system prefix back from
cache and writes its own user-turn chunk, then re-send that exact turn until
its own chunk reads back on three sends in a row, proving the cache is live
in both directions before the reminder turn goes out (a freshly written entry
can take a few seconds to become readable). Only the pre-reminder turn is
ever retried here, so retries can never warm a mutated-prefix cache entry and
mask the regression the second turn asserts on."""
deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS
while True:
user_text = _first_turn_user_text(unique_marker())
body = CacheChatRequest(model=model, messages=[system_turn, _user_turn(user_text, cached=True)])
usage = unwrap(_post_chat(client, key, body)).usage
read_tokens = _cache_read_tokens(usage)
creation_tokens = _cache_creation_tokens(usage)
if read_tokens > 0 and creation_tokens > 0:
primed = PrimedCache(
first_user_text=user_text,
prefix_read_tokens=read_tokens,
first_turn_creation_tokens=creation_tokens,
)
if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline):
return primed
if time.monotonic() >= deadline:
pytest.fail(
f"{model}: prompt cache never became readable in full within "
f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})"
)
time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
def _reads_full_prefix(client: PassthroughClient, key: str, body: CacheChatRequest, full_prefix_tokens: int) -> bool:
return _cache_read_tokens(unwrap(_post_chat(client, key, body)).usage) >= full_prefix_tokens
def _first_turn_reads_back(
client: PassthroughClient,
key: str,
body: CacheChatRequest,
full_prefix_tokens: int,
deadline: float,
) -> bool:
"""True once the full prefix reads back on CACHE_WARM_CONSECUTIVE_READS sends in
a row. Some providers' global endpoints serve the prompt cache per region, so a
fresh entry can be missing from the region the next request lands on; each miss
re-creates the entry there, so the streak converges as the regions warm up."""
while time.monotonic() < deadline:
if all(_reads_full_prefix(client, key, body, full_prefix_tokens) for _ in range(CACHE_WARM_CONSECUTIVE_READS)):
return True
time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
return False
def _reminder_turn_body(model: str, system_turn: RichMessage, primed: PrimedCache) -> CacheChatRequest:
"""Turn two in OpenAI shape: the primed prefix, an assistant reply, the
mid-conversation system reminder, and a fresh cache-marked user turn."""
return CacheChatRequest(
model=model,
messages=[
system_turn,
_user_turn(primed.first_user_text, cached=True),
_assistant_turn("OK."),
_system_reminder_turn(),
_user_turn("Reply with one word again.", cached=True),
],
)
def _assert_flagged_model_keeps_cache(
client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody
) -> None:
model = _register_deployment(client, resources, params)
key = resources.key(models=[model])
system_turn = _cacheable_system_turn(unique_marker())
primed = _prime_prompt_cache(client, key, model, system_turn)
second = unwrap(_post_chat(client, key, _reminder_turn_body(model, system_turn, primed)))
read_tokens = _cache_read_tokens(second.usage)
assert _response_role(second) == "assistant", f"{model}: unexpected role {_response_role(second)!r}"
assert _response_text(second).strip(), f"{model}: reminder turn returned no completion text"
assert read_tokens >= primed.full_prefix_tokens, (
f"{model}: turn with a mid-conversation system reminder read {read_tokens} "
f"cached tokens, expected at least the {primed.full_prefix_tokens} cached on "
f"turn one ({primed.prefix_read_tokens} system prefix + "
f"{primed.first_turn_creation_tokens} first user turn); the reminder was "
f"hoisted into the top-level system field, which mutates the cached prefix "
f"and re-bills the conversation at cache-write pricing"
)
def _assert_unflagged_model_converts_and_succeeds(
client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody
) -> None:
model = _register_deployment(client, resources, params)
key = resources.key(models=[model])
system_turn = _cacheable_system_turn(unique_marker())
primed = _prime_prompt_cache(client, key, model, system_turn)
second = unwrap(_post_chat(client, key, _reminder_turn_body(model, system_turn, primed)))
read_tokens = _cache_read_tokens(second.usage)
assert _response_role(second) == "assistant", f"{model}: unexpected role {_response_role(second)!r}"
assert _response_text(second).strip(), (
f"{model}: conversation with a mid-conversation system reminder returned "
f"no text; the reminder was forwarded in place to a model that rejects "
f"role 'system' inside messages instead of being converted to a user turn"
)
assert read_tokens >= primed.full_prefix_tokens, (
f"{model}: reminder turn read {read_tokens} cached tokens, expected at least "
f"the {primed.full_prefix_tokens} cached on turn one "
f"({primed.prefix_read_tokens} system prefix + "
f"{primed.first_turn_creation_tokens} first user turn); the reminder was "
f"hoisted into the top-level system field instead of being converted to a "
f"user turn in place, mutating the cached prefix and re-billing the "
f"conversation at cache-write pricing"
)
class TestAnthropicChatMidConversationSystem:
FLAGGED_MODEL = "anthropic/claude-opus-4-8"
UNFLAGGED_MODEL = "anthropic/claude-haiku-4-5-20251001"
@pytest.mark.covers(
"llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit",
exercised_on=[],
)
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
_assert_flagged_model_keeps_cache(client, resources, _anthropic_params(self.FLAGGED_MODEL))
@pytest.mark.covers(
"llm.chat_completions.anthropic.mid_conversation_system.nonstream.works",
exercised_on=[],
)
def test_unflagged_model_converts_system_reminder_and_succeeds(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
_assert_unflagged_model_converts_and_succeeds(client, resources, _anthropic_params(self.UNFLAGGED_MODEL))
class TestBedrockInvokeChatMidConversationSystem:
FLAGGED_MODEL = "bedrock/invoke/us.anthropic.claude-sonnet-5"
UNFLAGGED_MODEL = "bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0"
AWS_REGION = "us-east-1"
@pytest.mark.covers(
"llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit",
exercised_on=[],
)
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
_assert_flagged_model_keeps_cache(client, resources, _invoke_params(self.FLAGGED_MODEL, self.AWS_REGION))
@pytest.mark.covers(
"llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works",
exercised_on=[],
)
def test_unflagged_model_converts_system_reminder_and_succeeds(
self, client: PassthroughClient, resources: ResourceManager
) -> None:
_assert_unflagged_model_converts_and_succeeds(
client, resources, _invoke_params(self.UNFLAGGED_MODEL, self.AWS_REGION)
)

View file

@ -3578,3 +3578,51 @@ async def test_bedrock_converse_pdf_only_user_message_gets_text_block_async():
assert len(result) == 1
assert any("document" in block for block in result[0]["content"])
assert _text_blocks(result[0]) == [BEDROCK_DOCUMENT_PLACEHOLDER_TEXT]
def test_anthropic_messages_pt_keeps_system_role_after_user_turn():
"""Models flagged supports_mid_conversation_system accept role=system inside
messages; the converter must emit it as a system message with its text
blocks and cache_control intact instead of rejecting the role."""
messages = [
{"role": "user", "content": "First question"},
{
"role": "system",
"content": [{"type": "text", "text": "Answer in one word.", "cache_control": {"type": "ephemeral"}}],
},
{"role": "assistant", "content": "Yes"},
{"role": "user", "content": "Second question"},
]
result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic")
assert [m["role"] for m in result] == ["user", "system", "assistant", "user"]
assert result[1] == {
"role": "system",
"content": [{"type": "text", "text": "Answer in one word.", "cache_control": {"type": "ephemeral"}}],
}
def test_anthropic_messages_pt_system_string_content_becomes_text_block():
messages = [
{"role": "user", "content": "First question"},
{"role": "system", "content": "Answer in one word."},
]
result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic")
assert result[1] == {"role": "system", "content": [{"type": "text", "text": "Answer in one word."}]}
def test_anthropic_messages_pt_drops_a_system_message_with_no_text():
"""Anthropic rejects empty text blocks, so a text-less system message must
vanish rather than reach the wire as an empty system turn."""
messages = [
{"role": "user", "content": "First question"},
{"role": "system", "content": ""},
{"role": "assistant", "content": "Yes"},
]
result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic")
assert [m["role"] for m in result] == ["user", "assistant"]

View file

@ -1,4 +1,5 @@
import copy
import pytest
from unittest.mock import MagicMock, patch
@ -17,6 +18,13 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
AnthropicMessagesConfig,
)
from litellm.llms.azure_ai.anthropic.transformation import AzureAnthropicConfig
from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import (
AmazonAnthropicClaudeConfig,
)
from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import (
VertexAIAnthropicConfig,
)
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES
from litellm.types.utils import ServerToolUse, Usage
@ -6207,3 +6215,213 @@ def test_disabled_thinking_omitted_only_for_always_on_models(
assert "thinking" not in request
else:
assert request["thinking"] == {"type": "disabled"}
# ---------------------------------------------------------------------------
# Mid-conversation ``role: "system"`` on the chat completions path.
#
# Hoisting a later system message into the top-level ``system`` block rewrites
# the cached prefix and re-bills the whole conversation at cache-write pricing
# on every reminder (#36559). The chat path must keep the prefix stable: leading
# system messages still become the ``system`` param, later ones stay in place as
# ``role: "system"`` on models flagged ``supports_mid_conversation_system`` and
# become a user turn on models that reject the role inside ``messages``.
# ---------------------------------------------------------------------------
UNFLAGGED_CLAUDE = "claude-opus-4-7"
FLAGGED_CLAUDE = "claude-opus-4-8"
CONVERTED_SYSTEM_NOTE = (
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
REMINDER_TEXT = "<system-reminder>Answer with exactly one word.</system-reminder>"
CACHED_SYSTEM_BLOCK = {"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}
def _chat_request(config: AnthropicConfig, model: str, messages: list[dict]) -> dict:
return config.transform_request(
model=model,
messages=messages,
optional_params={},
litellm_params={},
headers={},
)
def _reminder_conversation() -> list[dict]:
"""The shape Claude Code sends mid-session: cached system prompt, turns, a
reminder right after a user turn, an assistant turn, a fresh user turn."""
return [
{"role": "system", "content": [dict(CACHED_SYSTEM_BLOCK)]},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "user", "content": "Second question"},
{"role": "system", "content": REMINDER_TEXT},
{"role": "assistant", "content": "Second answer"},
{"role": "user", "content": "Third question"},
]
def _texts(message: dict) -> list[str]:
return [block["text"] for block in message["content"] if block.get("type") == "text"]
def test_chat_unflagged_model_converts_mid_conversation_system_to_user_turn(local_model_cost_map):
result = _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, _reminder_conversation())
assert result["system"] == [CACHED_SYSTEM_BLOCK]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
assert _texts(result["messages"][2]) == ["Second question", CONVERTED_SYSTEM_NOTE, REMINDER_TEXT]
def test_chat_flagged_model_keeps_mid_conversation_system_in_messages(local_model_cost_map):
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, _reminder_conversation())
assert result["system"] == [CACHED_SYSTEM_BLOCK]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
assert result["messages"][3] == {"role": "system", "content": [{"type": "text", "text": REMINDER_TEXT}]}
def test_chat_flagged_model_keeps_cache_control_on_mid_conversation_system(local_model_cost_map):
messages = _reminder_conversation()
messages[4] = {
"role": "system",
"content": [{"type": "text", "text": REMINDER_TEXT, "cache_control": {"type": "ephemeral"}}],
}
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
assert result["messages"][3]["content"] == [
{"type": "text", "text": REMINDER_TEXT, "cache_control": {"type": "ephemeral"}}
]
def test_chat_flagged_model_moves_system_after_the_user_turn_it_precedes(local_model_cost_map):
"""Anthropic only accepts role=system directly after a user turn; an
OpenAI-shaped client that puts the reminder before its next question gets a
placement-valid request without the reminder leaving ``messages``."""
messages = [
{"role": "system", "content": "You are terse."},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "system", "content": REMINDER_TEXT},
{"role": "user", "content": "Second question"},
]
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system"]
assert _texts(result["messages"][2]) == ["Second question"]
assert _texts(result["messages"][3]) == [REMINDER_TEXT]
def test_chat_flagged_model_converts_system_with_no_following_user_turn(local_model_cost_map):
messages = [
{"role": "system", "content": "You are terse."},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "system", "content": REMINDER_TEXT},
]
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user"]
assert _texts(result["messages"][2]) == [CONVERTED_SYSTEM_NOTE, REMINDER_TEXT]
def test_chat_flagged_model_merges_adjacent_system_messages(local_model_cost_map):
messages = [
{"role": "system", "content": "You are terse."},
{"role": "user", "content": "First question"},
{"role": "system", "content": "Reminder one."},
{"role": "system", "content": "Reminder two."},
{"role": "assistant", "content": "First answer"},
{"role": "user", "content": "Second question"},
]
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
assert [m["role"] for m in result["messages"]] == ["user", "system", "assistant", "user"]
assert _texts(result["messages"][1]) == ["Reminder one.", "Reminder two."]
def test_chat_unflagged_model_keeps_tool_result_first_when_system_precedes_tool_message(local_model_cost_map):
messages = [
{"role": "system", "content": "You are terse."},
{"role": "user", "content": "Weather?"},
{
"role": "assistant",
"content": None,
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {"name": "get_weather", "arguments": "{}"},
}
],
},
{"role": "system", "content": REMINDER_TEXT},
{"role": "tool", "tool_call_id": "call_1", "content": "sunny"},
{"role": "user", "content": "Thanks"},
]
result = _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, messages)
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user"]
blocks = result["messages"][2]["content"]
assert blocks[0]["type"] == "tool_result"
assert blocks[0]["tool_use_id"] == "call_1"
assert _texts(result["messages"][2]) == [CONVERTED_SYSTEM_NOTE, REMINDER_TEXT, "Thanks"]
def test_chat_transform_request_does_not_mutate_caller_messages(local_model_cost_map):
messages = _reminder_conversation()
snapshot = copy.deepcopy(messages)
_chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, messages)
assert messages == snapshot
_CHAT_CONFIGS = [
pytest.param(AnthropicConfig, UNFLAGGED_CLAUDE, id="anthropic-unflagged"),
pytest.param(AnthropicConfig, FLAGGED_CLAUDE, id="anthropic-flagged"),
pytest.param(VertexAIAnthropicConfig, UNFLAGGED_CLAUDE, id="vertex_ai-unflagged"),
pytest.param(VertexAIAnthropicConfig, FLAGGED_CLAUDE, id="vertex_ai-flagged"),
pytest.param(AzureAnthropicConfig, UNFLAGGED_CLAUDE, id="azure_ai-unflagged"),
pytest.param(AzureAnthropicConfig, FLAGGED_CLAUDE, id="azure_ai-flagged"),
pytest.param(AmazonAnthropicClaudeConfig, "invoke/us.anthropic.claude-opus-4-7", id="bedrock_invoke-unflagged"),
pytest.param(AmazonAnthropicClaudeConfig, "invoke/us.anthropic.claude-opus-4-8", id="bedrock_invoke-flagged"),
]
@pytest.mark.parametrize("config_cls, model", _CHAT_CONFIGS)
def test_chat_mid_conversation_system_keeps_earlier_turns_a_prefix_of_the_next_request(
local_model_cost_map, config_cls, model
):
"""The provider-side prompt cache is a prefix match over ``system`` +
``messages``. Whatever the policy for the reminder, turn N's request must
stay a prefix of turn N+1's request or the whole conversation is re-billed.
Anthropic combines consecutive same-role messages into one turn, so the
cache-relevant sequence is ``(role, content block)`` pairs, not the message
list: a reminder that joins the preceding user turn still extends the prefix.
"""
conversation = _reminder_conversation()
earlier = _chat_request(config_cls(), model, copy.deepcopy(conversation[:4]))
later = _chat_request(config_cls(), model, copy.deepcopy(conversation))
assert later["system"] == earlier["system"]
earlier_blocks = _role_block_pairs(earlier["messages"])
later_blocks = _role_block_pairs(later["messages"])
assert later_blocks[: len(earlier_blocks)] == earlier_blocks
assert len(later_blocks) > len(earlier_blocks)
def _role_block_pairs(messages: list[dict]) -> list[tuple[str, object]]:
return [
(message["role"], block)
for message in messages
for block in (message["content"] if isinstance(message["content"], list) else [message["content"]])
]

View file

@ -0,0 +1,165 @@
"""Placement policy for mid-conversation ``role: "system"`` messages on the chat path.
The provider-facing behaviour is covered through ``transform_request`` in the
Anthropic, Vertex, Azure AI and Bedrock Invoke transformation tests; these pin
the pure placement rules on the OpenAI-format message list.
"""
import pytest
import litellm
from litellm.llms.anthropic.mid_conversation_system import (
CONVERTED_SYSTEM_NOTE,
place_mid_conversation_system,
split_leading_system_run,
)
def _roles(messages: object) -> list[str]:
return [m["role"] if isinstance(m, dict) else m.role for m in messages]
def _texts(message: dict) -> list[str]:
return [block["text"] for block in message["content"]]
def test_split_leading_system_run_keeps_later_system_messages_in_the_conversation():
messages = [
{"role": "system", "content": "one"},
{"role": "system", "content": "two"},
{"role": "user", "content": "q"},
{"role": "system", "content": "reminder"},
]
leading, later = split_leading_system_run(messages)
assert [m["content"] for m in leading] == ["one", "two"]
assert _roles(later) == ["user", "system"]
def test_flagged_placement_moves_a_system_run_after_the_user_turn_that_follows_it():
messages = [
{"role": "user", "content": "q1"},
{"role": "assistant", "content": "a1"},
{"role": "system", "content": "reminder"},
{"role": "user", "content": "q2"},
{"role": "assistant", "content": "a2"},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
assert _roles(placed) == ["user", "assistant", "user", "system", "assistant"]
def test_flagged_placement_pushes_a_system_between_two_user_turns_after_both():
"""Two user turns collapse into one on the wire, and a system message must
be followed by an assistant turn or nothing."""
messages = [
{"role": "user", "content": "q1"},
{"role": "system", "content": "reminder"},
{"role": "user", "content": "q2"},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
assert _roles(placed) == ["user", "user", "system"]
def test_flagged_placement_keeps_a_system_after_tool_results():
messages = [
{"role": "user", "content": "q1"},
{
"role": "assistant",
"content": None,
"tool_calls": [{"id": "c1", "type": "function", "function": {"name": "f", "arguments": "{}"}}],
},
{"role": "tool", "tool_call_id": "c1", "content": "r"},
{"role": "system", "content": "reminder"},
{"role": "assistant", "content": "a2"},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
assert _roles(placed) == ["user", "assistant", "tool", "system", "assistant"]
def test_flagged_placement_drops_a_system_message_with_no_text():
messages = [
{"role": "user", "content": "q1"},
{"role": "system", "content": ""},
{"role": "assistant", "content": "a1"},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
assert _roles(placed) == ["user", "assistant"]
def test_placement_reads_roles_off_pydantic_messages_in_the_history():
"""Callers routinely append the previous ``litellm.Message`` object straight
into the history; placement must read its role without assuming a dict and
hand the object through untouched."""
assistant = litellm.Message(role="assistant", content="a1")
messages = [
{"role": "user", "content": "q1"},
assistant,
{"role": "system", "content": "reminder"},
{"role": "user", "content": "q2"},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
assert _roles(placed) == ["user", "assistant", "user", "system"]
assert placed[1] is assistant
def test_unflagged_conversion_keeps_the_client_order_when_no_tool_result_follows():
messages = [
{"role": "user", "content": "q1"},
{"role": "assistant", "content": "a1"},
{"role": "system", "content": "reminder"},
{"role": "user", "content": "q2"},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False)
assert _roles(placed) == ["user", "assistant", "user", "user"]
assert _texts(placed[2]) == [CONVERTED_SYSTEM_NOTE, "reminder"]
@pytest.mark.parametrize(
"cache_control, expected",
[
({"type": "ephemeral", "ttl": "1h"}, {"type": "ephemeral", "ttl": "1h"}),
({"type": "ephemeral", "ttl": "5m"}, {"type": "ephemeral", "ttl": "5m"}),
({"type": "ephemeral", "ttl": "2h"}, {"type": "ephemeral"}),
],
)
def test_unflagged_conversion_rebuilds_cache_control_on_the_converted_block(cache_control, expected):
"""Only the shapes Anthropic accepts survive: ephemeral with a 5m or 1h ttl, or no ttl."""
messages = [
{"role": "user", "content": "q1"},
{"role": "system", "content": "reminder", "cache_control": cache_control},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False)
assert placed[1]["content"][1] == {"type": "text", "text": "reminder", "cache_control": expected}
def test_unflagged_conversion_drops_a_cache_control_that_is_not_ephemeral():
messages = [
{"role": "user", "content": "q1"},
{"role": "system", "content": "reminder", "cache_control": {"type": "persistent"}},
]
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False)
assert placed[1]["content"][1] == {"type": "text", "text": "reminder"}
def test_placement_is_a_no_op_without_later_system_messages():
messages = [{"role": "user", "content": "q1"}, {"role": "assistant", "content": "a1"}]
assert place_mid_conversation_system(messages, supports_mid_conversation_system=False) == tuple(messages)
assert place_mid_conversation_system(messages, supports_mid_conversation_system=True) == tuple(messages)

View file

@ -413,3 +413,51 @@ class TestAzureAnthropicConfig:
assert "anthropic-beta" in headers
assert "compact-2026-01-12" in headers["anthropic-beta"]
assert "context-management-2025-06-27" in headers["anthropic-beta"]
def _mid_conversation_system_conversation() -> list[dict]:
return [
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "user", "content": "Second question"},
{"role": "system", "content": "<system-reminder>Answer with exactly one word.</system-reminder>"},
{"role": "assistant", "content": "Second answer"},
{"role": "user", "content": "Third question"},
]
def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map):
"""A hoisted reminder rewrites the top-level system block and invalidates the
prompt cache for the whole conversation (#36559)."""
result = AzureAnthropicConfig().transform_request(
model="claude-opus-4-7",
messages=_mid_conversation_system_conversation(),
optional_params={},
litellm_params={},
headers={},
)
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"]
assert texts[0] == "Second question"
assert texts[-1] == "<system-reminder>Answer with exactly one word.</system-reminder>"
def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map):
result = AzureAnthropicConfig().transform_request(
model="claude-opus-4-8",
messages=_mid_conversation_system_conversation(),
optional_params={},
litellm_params={},
headers={},
)
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
assert result["messages"][3] == {
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Answer with exactly one word.</system-reminder>"}],
}

View file

@ -542,3 +542,51 @@ def test_output_format_removed_from_bedrock_invoke_request():
assert (
"output_format" not in result
), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}"
def _mid_conversation_system_conversation() -> list[dict]:
return [
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "user", "content": "Second question"},
{"role": "system", "content": "<system-reminder>Answer with exactly one word.</system-reminder>"},
{"role": "assistant", "content": "Second answer"},
{"role": "user", "content": "Third question"},
]
def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map):
"""A hoisted reminder rewrites the top-level system block and invalidates the
prompt cache for the whole conversation (#36559)."""
result = AmazonAnthropicClaudeConfig().transform_request(
model="invoke/us.anthropic.claude-opus-4-7",
messages=_mid_conversation_system_conversation(),
optional_params={},
litellm_params={},
headers={},
)
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"]
assert texts[0] == "Second question"
assert texts[-1] == "<system-reminder>Answer with exactly one word.</system-reminder>"
def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map):
result = AmazonAnthropicClaudeConfig().transform_request(
model="invoke/us.anthropic.claude-opus-4-8",
messages=_mid_conversation_system_conversation(),
optional_params={},
litellm_params={},
headers={},
)
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
assert result["messages"][3] == {
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Answer with exactly one word.</system-reminder>"}],
}

View file

@ -727,3 +727,51 @@ def test_sanitize_strips_effort_for_haiku_45():
data = {"output_config": {"effort": "high"}}
sanitize_vertex_anthropic_output_params(data, "vertex_ai/claude-opus-4-6")
assert data["output_config"] == {"effort": "high"}
def _mid_conversation_system_conversation() -> list[dict]:
return [
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
{"role": "user", "content": "First question"},
{"role": "assistant", "content": "First answer"},
{"role": "user", "content": "Second question"},
{"role": "system", "content": "<system-reminder>Answer with exactly one word.</system-reminder>"},
{"role": "assistant", "content": "Second answer"},
{"role": "user", "content": "Third question"},
]
def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map):
"""A hoisted reminder rewrites the top-level system block and invalidates the
prompt cache for the whole conversation (#36559)."""
result = VertexAIAnthropicConfig().transform_request(
model="claude-opus-4-7",
messages=_mid_conversation_system_conversation(),
optional_params={},
litellm_params={},
headers={},
)
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"]
assert texts[0] == "Second question"
assert texts[-1] == "<system-reminder>Answer with exactly one word.</system-reminder>"
def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map):
result = VertexAIAnthropicConfig().transform_request(
model="claude-opus-4-8",
messages=_mid_conversation_system_conversation(),
optional_params={},
litellm_params={},
headers={},
)
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
assert result["messages"][3] == {
"role": "system",
"content": [{"type": "text", "text": "<system-reminder>Answer with exactly one word.</system-reminder>"}],
}