mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Merge 6e1e44ce41 into 16e9efccaf
This commit is contained in:
commit
e71b4af5a5
14 changed files with 1237 additions and 11 deletions
|
|
@ -7,7 +7,7 @@ import re
|
|||
import xml.etree.ElementTree as ET
|
||||
from collections.abc import Iterator, Mapping, Sequence
|
||||
from enum import Enum
|
||||
from typing import Any, Final, TypedDict, cast, overload
|
||||
from typing import Any, Final, TypeAlias, TypedDict, cast, overload
|
||||
|
||||
from jinja2.sandbox import ImmutableSandboxedEnvironment
|
||||
|
||||
|
|
@ -18,6 +18,7 @@ from litellm import verbose_logger
|
|||
from litellm._uuid import uuid
|
||||
from litellm.constants import REDACTED_BY_LITELLM
|
||||
from litellm.litellm_core_utils.url_utils import async_safe_get, safe_get
|
||||
from litellm.llms.anthropic.mid_conversation_system import anthropic_system_messages
|
||||
from litellm.llms.custom_httpx.http_handler import HTTPHandler, get_async_httpx_client
|
||||
from litellm.types.files import get_file_extension_from_mime_type
|
||||
from litellm.types.llms.anthropic import *
|
||||
|
|
@ -2305,14 +2306,20 @@ def _drop_unsignable_thinking_blocks(
|
|||
return [block for block in thinking_blocks if not _is_unsignable_thinking_block(block)]
|
||||
|
||||
|
||||
# mutable-ok: anthropic_messages_pt's callers have always appended to the list it returns
|
||||
_AnthropicMessageList: TypeAlias = list[AllAnthropicPassThroughMessageValues]
|
||||
|
||||
|
||||
def anthropic_messages_pt(
|
||||
messages: list[AllMessageValues],
|
||||
model: str,
|
||||
llm_provider: str,
|
||||
) -> list[AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam]:
|
||||
) -> _AnthropicMessageList:
|
||||
"""
|
||||
format messages for anthropic
|
||||
1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately)
|
||||
1. Anthropic supports roles like "user" and "assistant" (system prompt sent separately).
|
||||
Models flagged ``supports_mid_conversation_system`` also accept "system" inside
|
||||
messages after a user turn; the caller decides placement, this keeps such messages.
|
||||
2. The first message always needs to be of role "user"
|
||||
3. Each message must alternate between "user" and "assistant" (this is not addressed as now by litellm)
|
||||
4. final assistant content cannot end with trailing whitespace (anthropic raises an error otherwise)
|
||||
|
|
@ -2337,7 +2344,7 @@ def anthropic_messages_pt(
|
|||
# add role=tool support to allow function call result/error submission
|
||||
user_message_types: Final = {"user", "tool", "function"}
|
||||
# reformat messages to ensure user/assistant are alternating, if there's either 2 consecutive 'user' messages or 2 consecutive 'assistant' message, merge them.
|
||||
new_messages: Final[list[AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam]] = []
|
||||
new_messages: Final[_AnthropicMessageList] = [] # mutable-ok: accumulator behind the mutable return contract
|
||||
|
||||
if len(messages) == 0:
|
||||
if not litellm.modify_params:
|
||||
|
|
@ -2730,6 +2737,11 @@ def anthropic_messages_pt(
|
|||
if assistant_content:
|
||||
new_messages.append({"role": "assistant", "content": assistant_content})
|
||||
|
||||
## MID-CONVERSATION SYSTEM MESSAGES (placement is the caller's job) ##
|
||||
while msg_i < len(messages) and messages[msg_i]["role"] == "system":
|
||||
new_messages.extend(anthropic_system_messages(messages[msg_i]))
|
||||
msg_i += 1
|
||||
|
||||
if msg_i == init_msg_i: # prevent infinite loops
|
||||
raise litellm.BadRequestError(
|
||||
message=BAD_MESSAGE_ERROR_STR + f"passed in {messages[msg_i]}",
|
||||
|
|
|
|||
|
|
@ -80,6 +80,7 @@ from litellm.utils import (
|
|||
get_max_tokens,
|
||||
has_tool_call_blocks,
|
||||
last_assistant_with_tool_calls_has_no_thinking_blocks,
|
||||
supports_mid_conversation_system,
|
||||
supports_reasoning,
|
||||
token_counter,
|
||||
)
|
||||
|
|
@ -90,6 +91,7 @@ from ..common_utils import (
|
|||
process_anthropic_headers,
|
||||
strip_advisor_blocks_from_messages,
|
||||
)
|
||||
from ..mid_conversation_system import place_mid_conversation_system, split_leading_system_run
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
|
@ -1897,16 +1899,27 @@ class AnthropicConfig(AnthropicModelInfo, BaseConfig):
|
|||
if _name_reverse_map and isinstance(litellm_params, dict):
|
||||
litellm_params[ANTHROPIC_TOOL_NAME_REVERSE_MAP_KEY] = _name_reverse_map
|
||||
|
||||
# Separate system prompt from rest of message
|
||||
anthropic_system_message_list: Final = self.translate_system_message(messages=messages)
|
||||
# Only the leading system run becomes the top-level system prompt. A later
|
||||
# system message stays in the conversation: hoisting it rewrites the cached
|
||||
# prefix and re-bills the whole history at cache-write pricing (#36559).
|
||||
leading_system_run, later_messages = split_leading_system_run(messages)
|
||||
anthropic_system_message_list: Final = self.translate_system_message(
|
||||
messages=list(leading_system_run) # mutable-ok: translate_system_message pops from the list it is given
|
||||
)
|
||||
# Handling anthropic API Prompt Caching
|
||||
if len(anthropic_system_message_list) > 0:
|
||||
optional_params["system"] = anthropic_system_message_list
|
||||
conversation: Final = place_mid_conversation_system(
|
||||
later_messages,
|
||||
supports_mid_conversation_system=supports_mid_conversation_system(
|
||||
model=model, custom_llm_provider=self.custom_llm_provider
|
||||
),
|
||||
)
|
||||
# Format rest of message according to anthropic guidelines
|
||||
try:
|
||||
anthropic_messages = anthropic_messages_pt(
|
||||
model=model,
|
||||
messages=messages,
|
||||
messages=list(conversation), # mutable-ok: anthropic_messages_pt rewrites entries in place
|
||||
llm_provider=self._resolved_provider,
|
||||
)
|
||||
except Exception as e:
|
||||
|
|
|
|||
|
|
@ -31,6 +31,7 @@ from ...common_utils import (
|
|||
optionally_handle_anthropic_oauth,
|
||||
strip_advisor_blocks_from_messages,
|
||||
)
|
||||
from ...mid_conversation_system import CONVERTED_SYSTEM_NOTE
|
||||
|
||||
DEFAULT_ANTHROPIC_API_VERSION: Final = "2023-06-01"
|
||||
|
||||
|
|
@ -160,9 +161,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
|
|||
def _is_system_role_message(message: Any) -> bool:
|
||||
return isinstance(message, dict) and message.get("role") == "system"
|
||||
|
||||
_CONVERTED_SYSTEM_NOTE: Final = (
|
||||
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
|
||||
)
|
||||
_CONVERTED_SYSTEM_NOTE: Final = CONVERTED_SYSTEM_NOTE
|
||||
|
||||
def _system_role_message_as_user(self, message: Mapping) -> Mapping:
|
||||
return {
|
||||
|
|
|
|||
290
litellm/llms/anthropic/mid_conversation_system.py
Normal file
290
litellm/llms/anthropic/mid_conversation_system.py
Normal file
|
|
@ -0,0 +1,290 @@
|
|||
"""Placement policy for ``role: "system"`` messages that appear after the first turn
|
||||
of an Anthropic-shaped chat completions request.
|
||||
|
||||
Only the leading run of system messages belongs in the top-level ``system``
|
||||
parameter. Hoisting a later one there rewrites the cached prefix, so the provider
|
||||
re-bills the whole conversation at cache-write pricing on every reminder (#36559).
|
||||
|
||||
Models flagged ``supports_mid_conversation_system`` in the cost map accept the role
|
||||
inside ``messages`` under Anthropic's placement rules: the message must directly
|
||||
follow a user turn, must be the last entry or be followed by an assistant turn, and
|
||||
must not sit next to another system message. OpenAI-shaped clients put system
|
||||
messages anywhere, so this module moves each one to the nearest valid slot and
|
||||
merges runs that land together.
|
||||
|
||||
Models without the flag reject the role inside ``messages``. Their system messages
|
||||
become user turns in place, prefixed with an operator note so the model can tell
|
||||
the instruction apart from the user's own words. A run caught between a tool call
|
||||
and its result moves to just after the result so the ``tool_result`` block stays
|
||||
first in the merged user turn.
|
||||
|
||||
Every transformation here is a pure function of the message sequence: turn N's
|
||||
output stays a prefix of turn N+1's output, which is what keeps the provider-side
|
||||
prompt cache readable across turns. Messages are handled in OpenAI format; the
|
||||
Anthropic wire shape is built later by ``anthropic_messages_pt``.
|
||||
"""
|
||||
|
||||
from collections.abc import Iterator, Mapping, Sequence
|
||||
from itertools import groupby
|
||||
from typing import Final, Literal, TypeAlias
|
||||
|
||||
from litellm.types.llms.anthropic import AnthropicMessagesSystemMessageParam, AnthropicSystemMessageContent
|
||||
from litellm.types.llms.openai import (
|
||||
AllMessageValues,
|
||||
ChatCompletionCachedContent,
|
||||
ChatCompletionSystemMessage,
|
||||
ChatCompletionTextObject,
|
||||
ChatCompletionUserMessage,
|
||||
)
|
||||
|
||||
CONVERTED_SYSTEM_NOTE: Final = (
|
||||
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
|
||||
)
|
||||
|
||||
_USER_TYPE_ROLES: Final = frozenset({"user", "tool", "function"})
|
||||
_TOOL_ROLES: Final = frozenset({"tool", "function"})
|
||||
|
||||
_MessageKind: TypeAlias = Literal["system", "tool", "user", "other"]
|
||||
_TextPart: TypeAlias = tuple[str, ChatCompletionCachedContent | None]
|
||||
|
||||
|
||||
def _as_mapping(value: object) -> Mapping[str, object] | None:
|
||||
return value if isinstance(value, Mapping) else None
|
||||
|
||||
|
||||
def _as_items(value: object) -> tuple[object, ...]:
|
||||
return tuple(value) if isinstance(value, Sequence) and not isinstance(value, str) else ()
|
||||
|
||||
|
||||
def _field(message: object, key: str) -> object:
|
||||
"""A message field, whether the message is a dict or a pydantic ``Message``."""
|
||||
mapping: Final = _as_mapping(message)
|
||||
return mapping.get(key) if mapping is not None else getattr(message, key, None)
|
||||
|
||||
|
||||
def is_system_message(message: object) -> bool:
|
||||
return _field(message, "role") == "system"
|
||||
|
||||
|
||||
def _is_user_type(message: object) -> bool:
|
||||
return _field(message, "role") in _USER_TYPE_ROLES
|
||||
|
||||
|
||||
def _kind(message: object) -> _MessageKind:
|
||||
role: Final = _field(message, "role")
|
||||
if role == "system":
|
||||
return "system"
|
||||
if role in _TOOL_ROLES:
|
||||
return "tool"
|
||||
if role == "user":
|
||||
return "user"
|
||||
return "other"
|
||||
|
||||
|
||||
def split_leading_system_run(
|
||||
messages: Sequence[AllMessageValues],
|
||||
) -> tuple[tuple[AllMessageValues, ...], tuple[AllMessageValues, ...]]:
|
||||
"""Split ``messages`` into the leading run of system messages and everything after it."""
|
||||
leading_count: Final = next(
|
||||
(index for index, message in enumerate(messages) if not is_system_message(message)),
|
||||
len(messages),
|
||||
)
|
||||
return tuple(messages[:leading_count]), tuple(messages[leading_count:])
|
||||
|
||||
|
||||
def _cache_control(holder: object) -> ChatCompletionCachedContent | None:
|
||||
"""The client's ``cache_control`` rebuilt in the only shape Anthropic accepts."""
|
||||
value: Final = _as_mapping(_field(holder, "cache_control"))
|
||||
if value is None or value.get("type") != "ephemeral":
|
||||
return None
|
||||
ttl: Final = value.get("ttl")
|
||||
if ttl == "1h":
|
||||
one_hour: Final[ChatCompletionCachedContent] = {"type": "ephemeral", "ttl": "1h"}
|
||||
return one_hour
|
||||
if ttl == "5m":
|
||||
five_minutes: Final[ChatCompletionCachedContent] = {"type": "ephemeral", "ttl": "5m"}
|
||||
return five_minutes
|
||||
ephemeral: Final[ChatCompletionCachedContent] = {"type": "ephemeral"}
|
||||
return ephemeral
|
||||
|
||||
|
||||
def _text_parts(message: object) -> tuple[_TextPart, ...]:
|
||||
"""``(text, cache_control)`` for each non-empty text part of a system message.
|
||||
|
||||
Anthropic rejects empty text blocks and only accepts text in system content. A
|
||||
``cache_control`` on the message itself belongs to the block built from string
|
||||
content; block-level ``cache_control`` stays with its block.
|
||||
"""
|
||||
content: Final = _field(message, "content")
|
||||
if isinstance(content, str):
|
||||
return ((content, _cache_control(message)),) if content else ()
|
||||
return tuple(
|
||||
(text, _cache_control(part))
|
||||
for part in _as_items(content)
|
||||
if _field(part, "type") == "text"
|
||||
for text in (_field(part, "text"),)
|
||||
if isinstance(text, str) and text
|
||||
)
|
||||
|
||||
|
||||
def _openai_text_block(part: _TextPart) -> ChatCompletionTextObject:
|
||||
text, cache_control = part
|
||||
if cache_control is None:
|
||||
plain: Final[ChatCompletionTextObject] = {"type": "text", "text": text}
|
||||
return plain
|
||||
cached: Final[ChatCompletionTextObject] = {"type": "text", "text": text, "cache_control": cache_control}
|
||||
return cached
|
||||
|
||||
|
||||
def _anthropic_text_block(part: _TextPart) -> AnthropicSystemMessageContent:
|
||||
text, cache_control = part
|
||||
if cache_control is None:
|
||||
plain: Final[AnthropicSystemMessageContent] = {"type": "text", "text": text}
|
||||
return plain
|
||||
cached: Final[AnthropicSystemMessageContent] = {"type": "text", "text": text, "cache_control": cache_control}
|
||||
return cached
|
||||
|
||||
|
||||
def anthropic_system_messages(message: object) -> tuple[AnthropicMessagesSystemMessageParam, ...]:
|
||||
"""The Anthropic wire message for a system message, or nothing when it carries no text."""
|
||||
blocks: Final = tuple(_anthropic_text_block(part) for part in _text_parts(message))
|
||||
if not blocks:
|
||||
return ()
|
||||
wire: Final[AnthropicMessagesSystemMessageParam] = {
|
||||
"role": "system",
|
||||
"content": list(blocks), # mutable-ok: wire payload; cache_control hooks edit content blocks in place
|
||||
}
|
||||
return (wire,)
|
||||
|
||||
|
||||
def system_message_as_user(message: object) -> ChatCompletionUserMessage:
|
||||
"""A system message re-rolled as a user turn, prefixed with the operator note."""
|
||||
note: Final[ChatCompletionTextObject] = {"type": "text", "text": CONVERTED_SYSTEM_NOTE}
|
||||
content: Final[list[ChatCompletionTextObject]] = [ # mutable-ok: anthropic_messages_pt only recognises list content
|
||||
note,
|
||||
*(_openai_text_block(part) for part in _text_parts(message)),
|
||||
]
|
||||
turn: Final[ChatCompletionUserMessage] = {"role": "user", "content": content}
|
||||
return turn
|
||||
|
||||
|
||||
def _merged_system_message(run: Sequence[object]) -> tuple[ChatCompletionSystemMessage, ...]:
|
||||
parts: Final = tuple(part for message in run for part in _text_parts(message))
|
||||
if not parts:
|
||||
return ()
|
||||
content: Final[list[ChatCompletionTextObject]] = [ # mutable-ok: anthropic_messages_pt only recognises list content
|
||||
_openai_text_block(part) for part in parts
|
||||
]
|
||||
merged: Final[ChatCompletionSystemMessage] = {"role": "system", "content": content}
|
||||
return (merged,)
|
||||
|
||||
|
||||
def _converted_user_turns(run: Sequence[object]) -> tuple[ChatCompletionUserMessage, ...]:
|
||||
return tuple(system_message_as_user(message) for message in run if _text_parts(message))
|
||||
|
||||
|
||||
def _runs(messages: Sequence[AllMessageValues]) -> tuple[tuple[_MessageKind, tuple[AllMessageValues, ...]], ...]:
|
||||
return tuple((kind, tuple(group)) for kind, group in groupby(messages, key=_kind))
|
||||
|
||||
|
||||
def _converted_for_unflagged_model(messages: Sequence[AllMessageValues]) -> tuple[AllMessageValues, ...]:
|
||||
"""Convert every system message to a user turn in place.
|
||||
|
||||
A system run whose follower is a tool message is emitted after that tool run:
|
||||
``tool_result`` blocks have to open the merged user turn.
|
||||
"""
|
||||
runs: Final = _runs(messages)
|
||||
|
||||
def emit(index: int) -> tuple[AllMessageValues, ...]:
|
||||
kind, run = runs[index]
|
||||
follower: Final = runs[index + 1][0] if index + 1 < len(runs) else None
|
||||
if kind == "system":
|
||||
return () if follower == "tool" else _converted_user_turns(run)
|
||||
if kind == "tool" and index > 0 and runs[index - 1][0] == "system":
|
||||
return (*run, *_converted_user_turns(runs[index - 1][1]))
|
||||
return run
|
||||
|
||||
return tuple(message for index in range(len(runs)) for message in emit(index))
|
||||
|
||||
|
||||
def _user_type_blocks(messages: Sequence[AllMessageValues]) -> tuple[tuple[bool, tuple[int, ...]], ...]:
|
||||
"""Maximal groups of consecutive non-system messages, keyed by whether they are user-type.
|
||||
|
||||
Consecutive user-type messages become one user turn on the wire, so a group is
|
||||
the unit a system message can validly follow.
|
||||
"""
|
||||
indexed: Final = tuple((index, message) for index, message in enumerate(messages) if not is_system_message(message))
|
||||
return tuple(
|
||||
(is_user, tuple(index for index, _ in group))
|
||||
for is_user, group in groupby(indexed, key=lambda pair: _is_user_type(pair[1]))
|
||||
)
|
||||
|
||||
|
||||
def _system_runs(messages: Sequence[AllMessageValues]) -> tuple[tuple[int, ...], ...]:
|
||||
"""Index runs of consecutive system messages."""
|
||||
system_indices: Final = tuple(index for index, message in enumerate(messages) if is_system_message(message))
|
||||
return tuple(
|
||||
tuple(index for _, index in group)
|
||||
for _, group in groupby(enumerate(system_indices), key=lambda pair: pair[1] - pair[0])
|
||||
)
|
||||
|
||||
|
||||
def _anchor_block(
|
||||
run_start: int,
|
||||
messages: Sequence[AllMessageValues],
|
||||
blocks: Sequence[tuple[bool, tuple[int, ...]]],
|
||||
) -> int | None:
|
||||
"""The user-type block a system run must follow, or ``None`` when no user turn can host it.
|
||||
|
||||
``run_start`` is never 0 here: the leading system run was split off before this
|
||||
policy runs, so the message before a run is always a non-system message.
|
||||
"""
|
||||
if _is_user_type(messages[run_start - 1]):
|
||||
return next(index for index, (is_user, indices) in enumerate(blocks) if is_user and run_start - 1 in indices)
|
||||
return next((index for index, (is_user, indices) in enumerate(blocks) if is_user and indices[0] > run_start), None)
|
||||
|
||||
|
||||
def _placed_for_flagged_model(messages: Sequence[AllMessageValues]) -> tuple[AllMessageValues, ...]:
|
||||
"""Keep system messages as ``role: "system"`` at a placement Anthropic accepts.
|
||||
|
||||
A run already sitting after a user-type message stays with that user turn. A
|
||||
run after an assistant turn moves to just after the next user turn. Runs that
|
||||
share a user turn merge into one system message. A run with no user turn left
|
||||
to host it becomes user turns in place.
|
||||
"""
|
||||
blocks: Final = _user_type_blocks(messages)
|
||||
anchors: Final = tuple((run, _anchor_block(run[0], messages, blocks)) for run in _system_runs(messages))
|
||||
|
||||
def anchored_to(block_index: int) -> tuple[AllMessageValues, ...]:
|
||||
return tuple(messages[index] for run, anchor in anchors if anchor == block_index for index in run)
|
||||
|
||||
def converted_after(message_index: int) -> tuple[ChatCompletionUserMessage, ...]:
|
||||
return tuple(
|
||||
turn
|
||||
for run, anchor in anchors
|
||||
if anchor is None and run[0] == message_index + 1
|
||||
for turn in _converted_user_turns(tuple(messages[index] for index in run))
|
||||
)
|
||||
|
||||
def emit(block_index: int) -> Iterator[AllMessageValues]:
|
||||
is_user, indices = blocks[block_index]
|
||||
for index in indices:
|
||||
yield messages[index]
|
||||
yield from converted_after(index)
|
||||
if is_user:
|
||||
yield from _merged_system_message(anchored_to(block_index))
|
||||
|
||||
return tuple(message for block_index in range(len(blocks)) for message in emit(block_index))
|
||||
|
||||
|
||||
def place_mid_conversation_system(
|
||||
messages: Sequence[AllMessageValues],
|
||||
*,
|
||||
supports_mid_conversation_system: bool,
|
||||
) -> tuple[AllMessageValues, ...]:
|
||||
"""Apply the placement policy to the messages after the leading system run."""
|
||||
if not any(is_system_message(message) for message in messages):
|
||||
return tuple(messages)
|
||||
if supports_mid_conversation_system:
|
||||
return _placed_for_flagged_model(messages)
|
||||
return _converted_for_unflagged_model(messages)
|
||||
|
|
@ -361,7 +361,8 @@ class AnthropicMessagesSystemMessageParam(TypedDict, total=False):
|
|||
|
||||
AllAnthropicMessageValues = AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam
|
||||
|
||||
# System is not a native Anthropic message role; only pass-through adapters use this union.
|
||||
# role=system inside messages is accepted after a user turn on models flagged
|
||||
# supports_mid_conversation_system; pass-through adapters and the chat translator both emit it.
|
||||
AllAnthropicPassThroughMessageValues: TypeAlias = (
|
||||
AnthropicMessagesUserMessageParam | AnthopicMessagesAssistantMessageParam | AnthropicMessagesSystemMessageParam
|
||||
)
|
||||
|
|
|
|||
|
|
@ -2722,6 +2722,15 @@ def supports_reasoning(model: str, custom_llm_provider: str | None = None) -> bo
|
|||
return _supports_factory(model=model, custom_llm_provider=custom_llm_provider, key="supports_reasoning")
|
||||
|
||||
|
||||
def supports_mid_conversation_system(model: str, custom_llm_provider: str | None = None) -> bool:
|
||||
"""
|
||||
Check if the given model accepts role=system messages after the first turn and return a boolean value.
|
||||
"""
|
||||
return _supports_factory(
|
||||
model=model, custom_llm_provider=custom_llm_provider, key="supports_mid_conversation_system"
|
||||
)
|
||||
|
||||
|
||||
def supports_native_structured_output(model: str, custom_llm_provider: str | None = None) -> bool:
|
||||
"""
|
||||
Check if the given model supports native structured outputs and return a boolean value.
|
||||
|
|
|
|||
|
|
@ -23,6 +23,8 @@
|
|||
- {id: llm.chat_completions.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude prompt caching"}
|
||||
- {id: llm.chat_completions.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude extended thinking"}
|
||||
- {id: llm.chat_completions.anthropic.structured_output.nonstream.works, module: llm, tier: P1, subject_endpoint: chat_completions, route: anthropic, capability: structured_output, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Claude response_schema"}
|
||||
- {id: llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to first-party Anthropic: flagged Claude 4.8+/5 must keep a mid-conversation role system reminder in messages; hoisting it into the top-level system field mutates the cached prefix and re-bills the conversation at cache-write pricing (#36559)", fail_before_fix: proven}
|
||||
- {id: llm.chat_completions.anthropic.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: anthropic, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to first-party Anthropic: Claude <= 4.7 and Haiku 4.5 reject role system inside messages, so unflagged models must convert a mid-conversation reminder to a user turn in place (hoisting collapses the prompt cache) and still answer (#36559)", fail_before_fix: proven}
|
||||
- {id: llm.chat_completions.bedrock_converse.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Bedrock Converse unified"}
|
||||
- {id: llm.chat_completions.bedrock_converse.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming over Converse"}
|
||||
- {id: llm.chat_completions.bedrock_converse.tool_use.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: tool_use, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Converse function_calling; AWS adoption"}
|
||||
|
|
@ -33,6 +35,8 @@
|
|||
- {id: llm.chat_completions.bedrock_converse.response_headers.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_converse, capability: response_headers, streaming: stream, assertions: [works], source: "llms/bedrock/chat/converse_handler.py:154", rationale: "The llm_provider-* headers must also surface on streaming /chat/completions, where CustomStreamWrapper carries them instead of the nonstream setter"}
|
||||
- {id: llm.chat_completions.bedrock_invoke.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "Regional inference-profile ids (us.anthropic.*) over the invoke route, the deployment shape behind a customer timeout report on v1.90.0"}
|
||||
- {id: llm.chat_completions.bedrock_invoke.basic.stream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: basic, streaming: stream, assertions: [works], source: "proxy_server.py:8455", rationale: "Streaming with regional inference-profile ids over the invoke route"}
|
||||
- {id: llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to Bedrock Invoke builds the Anthropic request through AnthropicConfig.transform_request, so flagged Claude 4.8+/5 must keep a mid-conversation role system reminder in messages; hoisting mutates the cached prefix and collapses the prompt cache (#36559)", fail_before_fix: proven}
|
||||
- {id: llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/anthropic/chat/transformation.py", rationale: "OpenAI-format chat to Bedrock Invoke: Claude <= 4.7 and Haiku 4.5 reject role system inside messages, so unflagged models must convert a mid-conversation reminder to a user turn in place (hoisting collapses the prompt cache) and still answer (#36559)", fail_before_fix: proven}
|
||||
- {id: llm.chat_completions.vertex.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: vertex, capability: basic, streaming: nonstream, assertions: [works], source: "proxy_server.py:8455", rationale: "P0 route; Vertex AI"}
|
||||
- {id: llm.chat_completions.gemini.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: chat_completions, route: gemini, capability: basic, streaming: nonstream, assertions: [works], source: "test_chat_completions_regression_e2e.py", rationale: "Gemini OpenAI-compatible chat translation"}
|
||||
- {id: llm.chat_completions.gemini.basic.nonstream.cost_logged, module: llm, tier: P0, subject_endpoint: chat_completions, route: gemini, capability: basic, streaming: nonstream, assertions: [works, cost_logged], source: "test_chat_completions_regression_e2e.py", rationale: "Gemini chat cost lands in SpendLogs"}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,323 @@
|
|||
"""Live e2e: mid-conversation ``role: "system"`` handling on the OpenAI-format
|
||||
/v1/chat/completions path is model-aware for first-party Anthropic and Bedrock
|
||||
Invoke, both of which build the Anthropic request through
|
||||
``AnthropicConfig.transform_request`` (#36559).
|
||||
|
||||
Only the leading run of system messages becomes the top-level ``system``
|
||||
parameter. A ``role: "system"`` entry that appears later in ``messages`` used to
|
||||
be hoisted into that same field, which rewrote the cached prefix and re-billed
|
||||
the whole conversation at cache-write pricing on every reminder. Models flagged
|
||||
``supports_mid_conversation_system`` in the cost map (Claude 4.8+ and the 5
|
||||
family) must keep the reminder in ``messages`` as ``role: "system"``; models
|
||||
without the flag (Claude 4.7 and older, Haiku 4.5) reject that role inside
|
||||
``messages``, so the proxy must convert the reminder to a user turn in place,
|
||||
prefixed with an operator note. Either way the prompt cache written on turn one
|
||||
must be read back in full on turn two.
|
||||
|
||||
The conversation shape mirrors what an OpenAI-SDK client sends mid-session: a
|
||||
cached system prompt, a user turn carrying its own ``cache_control`` breakpoint,
|
||||
an assistant turn, a ``role: "system"`` reminder, and a fresh user turn. The
|
||||
message-turn breakpoint is what makes the cache assertion able to fail: a cache
|
||||
entry whose prefix spans ``system`` plus message turns is invalidated when the
|
||||
reminder is hoisted (the ``system`` field mutates and a turn disappears from
|
||||
``messages``), while an entry ending at the system block itself would survive
|
||||
the hoist and mask the regression.
|
||||
|
||||
The provider-native ``cache_control`` request shape is not expressible with the
|
||||
shared ``ChatBody`` (whose content parts carry no cache_control), so the body is
|
||||
built from the typed content blocks shared in ``endpoints_client.py``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
|
||||
from e2e_config import unique_marker
|
||||
from e2e_http import Result, unwrap
|
||||
from endpoints_client import CacheControl, RichMessage, TextBlock
|
||||
from lifecycle import ResourceManager
|
||||
from models import ChatResponse, LiteLLMParamsBody, Usage
|
||||
from passthrough_client import PassthroughClient
|
||||
|
||||
pytestmark = pytest.mark.e2e
|
||||
|
||||
CACHE_PRIMING_DEADLINE_SECONDS = 60.0
|
||||
CACHE_PRIMING_INTERVAL_SECONDS = 3.0
|
||||
CACHE_WARM_CONSECUTIVE_READS = 3
|
||||
|
||||
|
||||
class CacheChatRequest(BaseModel):
|
||||
"""OpenAI-format chat body whose content blocks carry ``cache_control``."""
|
||||
|
||||
model: str
|
||||
messages: list[RichMessage]
|
||||
max_tokens: int = 64
|
||||
cache: dict[str, bool] = {"no-cache": True}
|
||||
|
||||
|
||||
def _anthropic_params(model: str) -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=model, api_key="os.environ/ANTHROPIC_API_KEY")
|
||||
|
||||
|
||||
def _invoke_params(model: str, region: str) -> LiteLLMParamsBody:
|
||||
return LiteLLMParamsBody(model=model, aws_region_name=region)
|
||||
|
||||
|
||||
def _cacheable_system_turn(marker: str) -> RichMessage:
|
||||
"""A system prompt comfortably above the 4096-token minimum cacheable size
|
||||
of Haiku 4.5 (the smallest model here), unique per run so no other run's
|
||||
cache entry can satisfy the read."""
|
||||
text = " ".join(f"Reference paragraph {index} for run {marker}." for index in range(300))
|
||||
return RichMessage(role="system", content=[TextBlock(text=text, cache_control=CacheControl())])
|
||||
|
||||
|
||||
def _user_turn(text: str, *, cached: bool = False) -> RichMessage:
|
||||
block = TextBlock(text=text, cache_control=CacheControl() if cached else None)
|
||||
return RichMessage(role="user", content=[block])
|
||||
|
||||
|
||||
def _assistant_turn(text: str) -> RichMessage:
|
||||
return RichMessage(role="assistant", content=[TextBlock(text=text)])
|
||||
|
||||
|
||||
def _system_reminder_turn() -> RichMessage:
|
||||
return RichMessage(
|
||||
role="system",
|
||||
content=[TextBlock(text="<system-reminder>Answer with exactly one word.</system-reminder>")],
|
||||
)
|
||||
|
||||
|
||||
def _post_chat(client: PassthroughClient, key: str, body: CacheChatRequest) -> Result[ChatResponse]:
|
||||
return client.proxy.transport.post(
|
||||
"/v1/chat/completions",
|
||||
headers=client.proxy.transport.bearer(key),
|
||||
json=body,
|
||||
response_type=ChatResponse,
|
||||
)
|
||||
|
||||
|
||||
def _register_deployment(client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody) -> str:
|
||||
model = f"e2e-chat-midsys-{unique_marker()}"
|
||||
model_id = client.proxy.create_model(model, params)
|
||||
resources.defer(lambda: client.proxy.delete_model(model_id))
|
||||
return model
|
||||
|
||||
|
||||
def _first_turn_user_text(marker: str) -> str:
|
||||
"""A first user turn heavy enough (hundreds of tokens) that losing its cache
|
||||
entry is unambiguous in the usage numbers, unique per attempt so priming
|
||||
retries never depend on the proxy's response cache behavior."""
|
||||
notes = " ".join(f"Session note {index} for attempt {marker}." for index in range(100))
|
||||
return f"Reply with one word.\n{notes}"
|
||||
|
||||
|
||||
def _cache_read_tokens(usage: Usage | None) -> int:
|
||||
"""Cache-read tokens however the chat usage reports them: the Anthropic-style
|
||||
``cache_read_input_tokens`` litellm forwards, or the OpenAI-style
|
||||
``prompt_tokens_details.cached_tokens`` it mirrors them into."""
|
||||
if usage is None:
|
||||
return 0
|
||||
if usage.cache_read_input_tokens:
|
||||
return usage.cache_read_input_tokens
|
||||
if usage.prompt_tokens_details and usage.prompt_tokens_details.cached_tokens:
|
||||
return usage.prompt_tokens_details.cached_tokens
|
||||
return 0
|
||||
|
||||
|
||||
def _cache_creation_tokens(usage: Usage | None) -> int:
|
||||
if usage is None:
|
||||
return 0
|
||||
return usage.cache_creation_input_tokens or 0
|
||||
|
||||
|
||||
def _response_text(response: ChatResponse) -> str:
|
||||
return "".join(choice.message.content or "" for choice in response.choices if choice.message)
|
||||
|
||||
|
||||
def _response_role(response: ChatResponse) -> str | None:
|
||||
first = response.choices[0].message if response.choices else None
|
||||
return first.role if first else None
|
||||
|
||||
|
||||
class PrimedCache(BaseModel):
|
||||
first_user_text: str
|
||||
prefix_read_tokens: int
|
||||
first_turn_creation_tokens: int
|
||||
|
||||
@property
|
||||
def full_prefix_tokens(self) -> int:
|
||||
return self.prefix_read_tokens + self.first_turn_creation_tokens
|
||||
|
||||
|
||||
def _prime_prompt_cache(client: PassthroughClient, key: str, model: str, system_turn: RichMessage) -> PrimedCache:
|
||||
"""Send first-turn calls (fresh cache-marked user turn each attempt,
|
||||
identical system prefix) until one both reads the system prefix back from
|
||||
cache and writes its own user-turn chunk, then re-send that exact turn until
|
||||
its own chunk reads back on three sends in a row, proving the cache is live
|
||||
in both directions before the reminder turn goes out (a freshly written entry
|
||||
can take a few seconds to become readable). Only the pre-reminder turn is
|
||||
ever retried here, so retries can never warm a mutated-prefix cache entry and
|
||||
mask the regression the second turn asserts on."""
|
||||
deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS
|
||||
while True:
|
||||
user_text = _first_turn_user_text(unique_marker())
|
||||
body = CacheChatRequest(model=model, messages=[system_turn, _user_turn(user_text, cached=True)])
|
||||
usage = unwrap(_post_chat(client, key, body)).usage
|
||||
read_tokens = _cache_read_tokens(usage)
|
||||
creation_tokens = _cache_creation_tokens(usage)
|
||||
if read_tokens > 0 and creation_tokens > 0:
|
||||
primed = PrimedCache(
|
||||
first_user_text=user_text,
|
||||
prefix_read_tokens=read_tokens,
|
||||
first_turn_creation_tokens=creation_tokens,
|
||||
)
|
||||
if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline):
|
||||
return primed
|
||||
if time.monotonic() >= deadline:
|
||||
pytest.fail(
|
||||
f"{model}: prompt cache never became readable in full within "
|
||||
f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})"
|
||||
)
|
||||
time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
|
||||
|
||||
|
||||
def _reads_full_prefix(client: PassthroughClient, key: str, body: CacheChatRequest, full_prefix_tokens: int) -> bool:
|
||||
return _cache_read_tokens(unwrap(_post_chat(client, key, body)).usage) >= full_prefix_tokens
|
||||
|
||||
|
||||
def _first_turn_reads_back(
|
||||
client: PassthroughClient,
|
||||
key: str,
|
||||
body: CacheChatRequest,
|
||||
full_prefix_tokens: int,
|
||||
deadline: float,
|
||||
) -> bool:
|
||||
"""True once the full prefix reads back on CACHE_WARM_CONSECUTIVE_READS sends in
|
||||
a row. Some providers' global endpoints serve the prompt cache per region, so a
|
||||
fresh entry can be missing from the region the next request lands on; each miss
|
||||
re-creates the entry there, so the streak converges as the regions warm up."""
|
||||
while time.monotonic() < deadline:
|
||||
if all(_reads_full_prefix(client, key, body, full_prefix_tokens) for _ in range(CACHE_WARM_CONSECUTIVE_READS)):
|
||||
return True
|
||||
time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
|
||||
return False
|
||||
|
||||
|
||||
def _reminder_turn_body(model: str, system_turn: RichMessage, primed: PrimedCache) -> CacheChatRequest:
|
||||
"""Turn two in OpenAI shape: the primed prefix, an assistant reply, the
|
||||
mid-conversation system reminder, and a fresh cache-marked user turn."""
|
||||
return CacheChatRequest(
|
||||
model=model,
|
||||
messages=[
|
||||
system_turn,
|
||||
_user_turn(primed.first_user_text, cached=True),
|
||||
_assistant_turn("OK."),
|
||||
_system_reminder_turn(),
|
||||
_user_turn("Reply with one word again.", cached=True),
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def _assert_flagged_model_keeps_cache(
|
||||
client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody
|
||||
) -> None:
|
||||
model = _register_deployment(client, resources, params)
|
||||
key = resources.key(models=[model])
|
||||
system_turn = _cacheable_system_turn(unique_marker())
|
||||
|
||||
primed = _prime_prompt_cache(client, key, model, system_turn)
|
||||
|
||||
second = unwrap(_post_chat(client, key, _reminder_turn_body(model, system_turn, primed)))
|
||||
read_tokens = _cache_read_tokens(second.usage)
|
||||
|
||||
assert _response_role(second) == "assistant", f"{model}: unexpected role {_response_role(second)!r}"
|
||||
assert _response_text(second).strip(), f"{model}: reminder turn returned no completion text"
|
||||
assert read_tokens >= primed.full_prefix_tokens, (
|
||||
f"{model}: turn with a mid-conversation system reminder read {read_tokens} "
|
||||
f"cached tokens, expected at least the {primed.full_prefix_tokens} cached on "
|
||||
f"turn one ({primed.prefix_read_tokens} system prefix + "
|
||||
f"{primed.first_turn_creation_tokens} first user turn); the reminder was "
|
||||
f"hoisted into the top-level system field, which mutates the cached prefix "
|
||||
f"and re-bills the conversation at cache-write pricing"
|
||||
)
|
||||
|
||||
|
||||
def _assert_unflagged_model_converts_and_succeeds(
|
||||
client: PassthroughClient, resources: ResourceManager, params: LiteLLMParamsBody
|
||||
) -> None:
|
||||
model = _register_deployment(client, resources, params)
|
||||
key = resources.key(models=[model])
|
||||
system_turn = _cacheable_system_turn(unique_marker())
|
||||
|
||||
primed = _prime_prompt_cache(client, key, model, system_turn)
|
||||
|
||||
second = unwrap(_post_chat(client, key, _reminder_turn_body(model, system_turn, primed)))
|
||||
read_tokens = _cache_read_tokens(second.usage)
|
||||
|
||||
assert _response_role(second) == "assistant", f"{model}: unexpected role {_response_role(second)!r}"
|
||||
assert _response_text(second).strip(), (
|
||||
f"{model}: conversation with a mid-conversation system reminder returned "
|
||||
f"no text; the reminder was forwarded in place to a model that rejects "
|
||||
f"role 'system' inside messages instead of being converted to a user turn"
|
||||
)
|
||||
assert read_tokens >= primed.full_prefix_tokens, (
|
||||
f"{model}: reminder turn read {read_tokens} cached tokens, expected at least "
|
||||
f"the {primed.full_prefix_tokens} cached on turn one "
|
||||
f"({primed.prefix_read_tokens} system prefix + "
|
||||
f"{primed.first_turn_creation_tokens} first user turn); the reminder was "
|
||||
f"hoisted into the top-level system field instead of being converted to a "
|
||||
f"user turn in place, mutating the cached prefix and re-billing the "
|
||||
f"conversation at cache-write pricing"
|
||||
)
|
||||
|
||||
|
||||
class TestAnthropicChatMidConversationSystem:
|
||||
FLAGGED_MODEL = "anthropic/claude-opus-4-8"
|
||||
UNFLAGGED_MODEL = "anthropic/claude-haiku-4-5-20251001"
|
||||
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.anthropic.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
_assert_flagged_model_keeps_cache(client, resources, _anthropic_params(self.FLAGGED_MODEL))
|
||||
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.anthropic.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
_assert_unflagged_model_converts_and_succeeds(client, resources, _anthropic_params(self.UNFLAGGED_MODEL))
|
||||
|
||||
|
||||
class TestBedrockInvokeChatMidConversationSystem:
|
||||
FLAGGED_MODEL = "bedrock/invoke/us.anthropic.claude-sonnet-5"
|
||||
UNFLAGGED_MODEL = "bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0"
|
||||
AWS_REGION = "us-east-1"
|
||||
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.cache_hit",
|
||||
exercised_on=[],
|
||||
)
|
||||
def test_flagged_model_keeps_prompt_cache_across_system_reminder(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
_assert_flagged_model_keeps_cache(client, resources, _invoke_params(self.FLAGGED_MODEL, self.AWS_REGION))
|
||||
|
||||
@pytest.mark.covers(
|
||||
"llm.chat_completions.bedrock_invoke.mid_conversation_system.nonstream.works",
|
||||
exercised_on=[],
|
||||
)
|
||||
def test_unflagged_model_converts_system_reminder_and_succeeds(
|
||||
self, client: PassthroughClient, resources: ResourceManager
|
||||
) -> None:
|
||||
_assert_unflagged_model_converts_and_succeeds(
|
||||
client, resources, _invoke_params(self.UNFLAGGED_MODEL, self.AWS_REGION)
|
||||
)
|
||||
|
|
@ -3578,3 +3578,51 @@ async def test_bedrock_converse_pdf_only_user_message_gets_text_block_async():
|
|||
assert len(result) == 1
|
||||
assert any("document" in block for block in result[0]["content"])
|
||||
assert _text_blocks(result[0]) == [BEDROCK_DOCUMENT_PLACEHOLDER_TEXT]
|
||||
|
||||
|
||||
def test_anthropic_messages_pt_keeps_system_role_after_user_turn():
|
||||
"""Models flagged supports_mid_conversation_system accept role=system inside
|
||||
messages; the converter must emit it as a system message with its text
|
||||
blocks and cache_control intact instead of rejecting the role."""
|
||||
messages = [
|
||||
{"role": "user", "content": "First question"},
|
||||
{
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "Answer in one word.", "cache_control": {"type": "ephemeral"}}],
|
||||
},
|
||||
{"role": "assistant", "content": "Yes"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
]
|
||||
|
||||
result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic")
|
||||
|
||||
assert [m["role"] for m in result] == ["user", "system", "assistant", "user"]
|
||||
assert result[1] == {
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "Answer in one word.", "cache_control": {"type": "ephemeral"}}],
|
||||
}
|
||||
|
||||
|
||||
def test_anthropic_messages_pt_system_string_content_becomes_text_block():
|
||||
messages = [
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "system", "content": "Answer in one word."},
|
||||
]
|
||||
|
||||
result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic")
|
||||
|
||||
assert result[1] == {"role": "system", "content": [{"type": "text", "text": "Answer in one word."}]}
|
||||
|
||||
|
||||
def test_anthropic_messages_pt_drops_a_system_message_with_no_text():
|
||||
"""Anthropic rejects empty text blocks, so a text-less system message must
|
||||
vanish rather than reach the wire as an empty system turn."""
|
||||
messages = [
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "system", "content": ""},
|
||||
{"role": "assistant", "content": "Yes"},
|
||||
]
|
||||
|
||||
result = anthropic_messages_pt(messages=messages, model="claude-opus-4-8", llm_provider="anthropic")
|
||||
|
||||
assert [m["role"] for m in result] == ["user", "assistant"]
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
|
||||
import copy
|
||||
import pytest
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
|
@ -17,6 +18,13 @@ from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
|||
from litellm.llms.anthropic.experimental_pass_through.messages.transformation import (
|
||||
AnthropicMessagesConfig,
|
||||
)
|
||||
from litellm.llms.azure_ai.anthropic.transformation import AzureAnthropicConfig
|
||||
from litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation import (
|
||||
AmazonAnthropicClaudeConfig,
|
||||
)
|
||||
from litellm.llms.vertex_ai.vertex_ai_partner_models.anthropic.transformation import (
|
||||
VertexAIAnthropicConfig,
|
||||
)
|
||||
from litellm.types.llms.anthropic import ANTHROPIC_BETA_HEADER_VALUES
|
||||
from litellm.types.utils import ServerToolUse, Usage
|
||||
|
||||
|
|
@ -6207,3 +6215,213 @@ def test_disabled_thinking_omitted_only_for_always_on_models(
|
|||
assert "thinking" not in request
|
||||
else:
|
||||
assert request["thinking"] == {"type": "disabled"}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mid-conversation ``role: "system"`` on the chat completions path.
|
||||
#
|
||||
# Hoisting a later system message into the top-level ``system`` block rewrites
|
||||
# the cached prefix and re-bills the whole conversation at cache-write pricing
|
||||
# on every reminder (#36559). The chat path must keep the prefix stable: leading
|
||||
# system messages still become the ``system`` param, later ones stay in place as
|
||||
# ``role: "system"`` on models flagged ``supports_mid_conversation_system`` and
|
||||
# become a user turn on models that reject the role inside ``messages``.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
UNFLAGGED_CLAUDE = "claude-opus-4-7"
|
||||
FLAGGED_CLAUDE = "claude-opus-4-8"
|
||||
CONVERTED_SYSTEM_NOTE = (
|
||||
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
|
||||
)
|
||||
REMINDER_TEXT = "<system-reminder>Answer with exactly one word.</system-reminder>"
|
||||
CACHED_SYSTEM_BLOCK = {"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}
|
||||
|
||||
|
||||
def _chat_request(config: AnthropicConfig, model: str, messages: list[dict]) -> dict:
|
||||
return config.transform_request(
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
|
||||
def _reminder_conversation() -> list[dict]:
|
||||
"""The shape Claude Code sends mid-session: cached system prompt, turns, a
|
||||
reminder right after a user turn, an assistant turn, a fresh user turn."""
|
||||
return [
|
||||
{"role": "system", "content": [dict(CACHED_SYSTEM_BLOCK)]},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
{"role": "system", "content": REMINDER_TEXT},
|
||||
{"role": "assistant", "content": "Second answer"},
|
||||
{"role": "user", "content": "Third question"},
|
||||
]
|
||||
|
||||
|
||||
def _texts(message: dict) -> list[str]:
|
||||
return [block["text"] for block in message["content"] if block.get("type") == "text"]
|
||||
|
||||
|
||||
def test_chat_unflagged_model_converts_mid_conversation_system_to_user_turn(local_model_cost_map):
|
||||
result = _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, _reminder_conversation())
|
||||
|
||||
assert result["system"] == [CACHED_SYSTEM_BLOCK]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
|
||||
assert _texts(result["messages"][2]) == ["Second question", CONVERTED_SYSTEM_NOTE, REMINDER_TEXT]
|
||||
|
||||
|
||||
def test_chat_flagged_model_keeps_mid_conversation_system_in_messages(local_model_cost_map):
|
||||
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, _reminder_conversation())
|
||||
|
||||
assert result["system"] == [CACHED_SYSTEM_BLOCK]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
|
||||
assert result["messages"][3] == {"role": "system", "content": [{"type": "text", "text": REMINDER_TEXT}]}
|
||||
|
||||
|
||||
def test_chat_flagged_model_keeps_cache_control_on_mid_conversation_system(local_model_cost_map):
|
||||
messages = _reminder_conversation()
|
||||
messages[4] = {
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": REMINDER_TEXT, "cache_control": {"type": "ephemeral"}}],
|
||||
}
|
||||
|
||||
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
|
||||
|
||||
assert result["messages"][3]["content"] == [
|
||||
{"type": "text", "text": REMINDER_TEXT, "cache_control": {"type": "ephemeral"}}
|
||||
]
|
||||
|
||||
|
||||
def test_chat_flagged_model_moves_system_after_the_user_turn_it_precedes(local_model_cost_map):
|
||||
"""Anthropic only accepts role=system directly after a user turn; an
|
||||
OpenAI-shaped client that puts the reminder before its next question gets a
|
||||
placement-valid request without the reminder leaving ``messages``."""
|
||||
messages = [
|
||||
{"role": "system", "content": "You are terse."},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "system", "content": REMINDER_TEXT},
|
||||
{"role": "user", "content": "Second question"},
|
||||
]
|
||||
|
||||
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
|
||||
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system"]
|
||||
assert _texts(result["messages"][2]) == ["Second question"]
|
||||
assert _texts(result["messages"][3]) == [REMINDER_TEXT]
|
||||
|
||||
|
||||
def test_chat_flagged_model_converts_system_with_no_following_user_turn(local_model_cost_map):
|
||||
messages = [
|
||||
{"role": "system", "content": "You are terse."},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "system", "content": REMINDER_TEXT},
|
||||
]
|
||||
|
||||
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
|
||||
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user"]
|
||||
assert _texts(result["messages"][2]) == [CONVERTED_SYSTEM_NOTE, REMINDER_TEXT]
|
||||
|
||||
|
||||
def test_chat_flagged_model_merges_adjacent_system_messages(local_model_cost_map):
|
||||
messages = [
|
||||
{"role": "system", "content": "You are terse."},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "system", "content": "Reminder one."},
|
||||
{"role": "system", "content": "Reminder two."},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
]
|
||||
|
||||
result = _chat_request(AnthropicConfig(), FLAGGED_CLAUDE, messages)
|
||||
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "system", "assistant", "user"]
|
||||
assert _texts(result["messages"][1]) == ["Reminder one.", "Reminder two."]
|
||||
|
||||
|
||||
def test_chat_unflagged_model_keeps_tool_result_first_when_system_precedes_tool_message(local_model_cost_map):
|
||||
messages = [
|
||||
{"role": "system", "content": "You are terse."},
|
||||
{"role": "user", "content": "Weather?"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [
|
||||
{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {"name": "get_weather", "arguments": "{}"},
|
||||
}
|
||||
],
|
||||
},
|
||||
{"role": "system", "content": REMINDER_TEXT},
|
||||
{"role": "tool", "tool_call_id": "call_1", "content": "sunny"},
|
||||
{"role": "user", "content": "Thanks"},
|
||||
]
|
||||
|
||||
result = _chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, messages)
|
||||
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user"]
|
||||
blocks = result["messages"][2]["content"]
|
||||
assert blocks[0]["type"] == "tool_result"
|
||||
assert blocks[0]["tool_use_id"] == "call_1"
|
||||
assert _texts(result["messages"][2]) == [CONVERTED_SYSTEM_NOTE, REMINDER_TEXT, "Thanks"]
|
||||
|
||||
|
||||
def test_chat_transform_request_does_not_mutate_caller_messages(local_model_cost_map):
|
||||
messages = _reminder_conversation()
|
||||
snapshot = copy.deepcopy(messages)
|
||||
|
||||
_chat_request(AnthropicConfig(), UNFLAGGED_CLAUDE, messages)
|
||||
|
||||
assert messages == snapshot
|
||||
|
||||
|
||||
_CHAT_CONFIGS = [
|
||||
pytest.param(AnthropicConfig, UNFLAGGED_CLAUDE, id="anthropic-unflagged"),
|
||||
pytest.param(AnthropicConfig, FLAGGED_CLAUDE, id="anthropic-flagged"),
|
||||
pytest.param(VertexAIAnthropicConfig, UNFLAGGED_CLAUDE, id="vertex_ai-unflagged"),
|
||||
pytest.param(VertexAIAnthropicConfig, FLAGGED_CLAUDE, id="vertex_ai-flagged"),
|
||||
pytest.param(AzureAnthropicConfig, UNFLAGGED_CLAUDE, id="azure_ai-unflagged"),
|
||||
pytest.param(AzureAnthropicConfig, FLAGGED_CLAUDE, id="azure_ai-flagged"),
|
||||
pytest.param(AmazonAnthropicClaudeConfig, "invoke/us.anthropic.claude-opus-4-7", id="bedrock_invoke-unflagged"),
|
||||
pytest.param(AmazonAnthropicClaudeConfig, "invoke/us.anthropic.claude-opus-4-8", id="bedrock_invoke-flagged"),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("config_cls, model", _CHAT_CONFIGS)
|
||||
def test_chat_mid_conversation_system_keeps_earlier_turns_a_prefix_of_the_next_request(
|
||||
local_model_cost_map, config_cls, model
|
||||
):
|
||||
"""The provider-side prompt cache is a prefix match over ``system`` +
|
||||
``messages``. Whatever the policy for the reminder, turn N's request must
|
||||
stay a prefix of turn N+1's request or the whole conversation is re-billed.
|
||||
|
||||
Anthropic combines consecutive same-role messages into one turn, so the
|
||||
cache-relevant sequence is ``(role, content block)`` pairs, not the message
|
||||
list: a reminder that joins the preceding user turn still extends the prefix.
|
||||
"""
|
||||
conversation = _reminder_conversation()
|
||||
|
||||
earlier = _chat_request(config_cls(), model, copy.deepcopy(conversation[:4]))
|
||||
later = _chat_request(config_cls(), model, copy.deepcopy(conversation))
|
||||
|
||||
assert later["system"] == earlier["system"]
|
||||
earlier_blocks = _role_block_pairs(earlier["messages"])
|
||||
later_blocks = _role_block_pairs(later["messages"])
|
||||
assert later_blocks[: len(earlier_blocks)] == earlier_blocks
|
||||
assert len(later_blocks) > len(earlier_blocks)
|
||||
|
||||
|
||||
def _role_block_pairs(messages: list[dict]) -> list[tuple[str, object]]:
|
||||
return [
|
||||
(message["role"], block)
|
||||
for message in messages
|
||||
for block in (message["content"] if isinstance(message["content"], list) else [message["content"]])
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,165 @@
|
|||
"""Placement policy for mid-conversation ``role: "system"`` messages on the chat path.
|
||||
|
||||
The provider-facing behaviour is covered through ``transform_request`` in the
|
||||
Anthropic, Vertex, Azure AI and Bedrock Invoke transformation tests; these pin
|
||||
the pure placement rules on the OpenAI-format message list.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.anthropic.mid_conversation_system import (
|
||||
CONVERTED_SYSTEM_NOTE,
|
||||
place_mid_conversation_system,
|
||||
split_leading_system_run,
|
||||
)
|
||||
|
||||
|
||||
def _roles(messages: object) -> list[str]:
|
||||
return [m["role"] if isinstance(m, dict) else m.role for m in messages]
|
||||
|
||||
|
||||
def _texts(message: dict) -> list[str]:
|
||||
return [block["text"] for block in message["content"]]
|
||||
|
||||
|
||||
def test_split_leading_system_run_keeps_later_system_messages_in_the_conversation():
|
||||
messages = [
|
||||
{"role": "system", "content": "one"},
|
||||
{"role": "system", "content": "two"},
|
||||
{"role": "user", "content": "q"},
|
||||
{"role": "system", "content": "reminder"},
|
||||
]
|
||||
|
||||
leading, later = split_leading_system_run(messages)
|
||||
|
||||
assert [m["content"] for m in leading] == ["one", "two"]
|
||||
assert _roles(later) == ["user", "system"]
|
||||
|
||||
|
||||
def test_flagged_placement_moves_a_system_run_after_the_user_turn_that_follows_it():
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{"role": "assistant", "content": "a1"},
|
||||
{"role": "system", "content": "reminder"},
|
||||
{"role": "user", "content": "q2"},
|
||||
{"role": "assistant", "content": "a2"},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
|
||||
|
||||
assert _roles(placed) == ["user", "assistant", "user", "system", "assistant"]
|
||||
|
||||
|
||||
def test_flagged_placement_pushes_a_system_between_two_user_turns_after_both():
|
||||
"""Two user turns collapse into one on the wire, and a system message must
|
||||
be followed by an assistant turn or nothing."""
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{"role": "system", "content": "reminder"},
|
||||
{"role": "user", "content": "q2"},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
|
||||
|
||||
assert _roles(placed) == ["user", "user", "system"]
|
||||
|
||||
|
||||
def test_flagged_placement_keeps_a_system_after_tool_results():
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": None,
|
||||
"tool_calls": [{"id": "c1", "type": "function", "function": {"name": "f", "arguments": "{}"}}],
|
||||
},
|
||||
{"role": "tool", "tool_call_id": "c1", "content": "r"},
|
||||
{"role": "system", "content": "reminder"},
|
||||
{"role": "assistant", "content": "a2"},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
|
||||
|
||||
assert _roles(placed) == ["user", "assistant", "tool", "system", "assistant"]
|
||||
|
||||
|
||||
def test_flagged_placement_drops_a_system_message_with_no_text():
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{"role": "system", "content": ""},
|
||||
{"role": "assistant", "content": "a1"},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
|
||||
|
||||
assert _roles(placed) == ["user", "assistant"]
|
||||
|
||||
|
||||
def test_placement_reads_roles_off_pydantic_messages_in_the_history():
|
||||
"""Callers routinely append the previous ``litellm.Message`` object straight
|
||||
into the history; placement must read its role without assuming a dict and
|
||||
hand the object through untouched."""
|
||||
assistant = litellm.Message(role="assistant", content="a1")
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
assistant,
|
||||
{"role": "system", "content": "reminder"},
|
||||
{"role": "user", "content": "q2"},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=True)
|
||||
|
||||
assert _roles(placed) == ["user", "assistant", "user", "system"]
|
||||
assert placed[1] is assistant
|
||||
|
||||
|
||||
def test_unflagged_conversion_keeps_the_client_order_when_no_tool_result_follows():
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{"role": "assistant", "content": "a1"},
|
||||
{"role": "system", "content": "reminder"},
|
||||
{"role": "user", "content": "q2"},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False)
|
||||
|
||||
assert _roles(placed) == ["user", "assistant", "user", "user"]
|
||||
assert _texts(placed[2]) == [CONVERTED_SYSTEM_NOTE, "reminder"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cache_control, expected",
|
||||
[
|
||||
({"type": "ephemeral", "ttl": "1h"}, {"type": "ephemeral", "ttl": "1h"}),
|
||||
({"type": "ephemeral", "ttl": "5m"}, {"type": "ephemeral", "ttl": "5m"}),
|
||||
({"type": "ephemeral", "ttl": "2h"}, {"type": "ephemeral"}),
|
||||
],
|
||||
)
|
||||
def test_unflagged_conversion_rebuilds_cache_control_on_the_converted_block(cache_control, expected):
|
||||
"""Only the shapes Anthropic accepts survive: ephemeral with a 5m or 1h ttl, or no ttl."""
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{"role": "system", "content": "reminder", "cache_control": cache_control},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False)
|
||||
|
||||
assert placed[1]["content"][1] == {"type": "text", "text": "reminder", "cache_control": expected}
|
||||
|
||||
|
||||
def test_unflagged_conversion_drops_a_cache_control_that_is_not_ephemeral():
|
||||
messages = [
|
||||
{"role": "user", "content": "q1"},
|
||||
{"role": "system", "content": "reminder", "cache_control": {"type": "persistent"}},
|
||||
]
|
||||
|
||||
placed = place_mid_conversation_system(messages, supports_mid_conversation_system=False)
|
||||
|
||||
assert placed[1]["content"][1] == {"type": "text", "text": "reminder"}
|
||||
|
||||
|
||||
def test_placement_is_a_no_op_without_later_system_messages():
|
||||
messages = [{"role": "user", "content": "q1"}, {"role": "assistant", "content": "a1"}]
|
||||
|
||||
assert place_mid_conversation_system(messages, supports_mid_conversation_system=False) == tuple(messages)
|
||||
assert place_mid_conversation_system(messages, supports_mid_conversation_system=True) == tuple(messages)
|
||||
|
|
@ -413,3 +413,51 @@ class TestAzureAnthropicConfig:
|
|||
assert "anthropic-beta" in headers
|
||||
assert "compact-2026-01-12" in headers["anthropic-beta"]
|
||||
assert "context-management-2025-06-27" in headers["anthropic-beta"]
|
||||
|
||||
|
||||
def _mid_conversation_system_conversation() -> list[dict]:
|
||||
return [
|
||||
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
{"role": "system", "content": "<system-reminder>Answer with exactly one word.</system-reminder>"},
|
||||
{"role": "assistant", "content": "Second answer"},
|
||||
{"role": "user", "content": "Third question"},
|
||||
]
|
||||
|
||||
|
||||
def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map):
|
||||
"""A hoisted reminder rewrites the top-level system block and invalidates the
|
||||
prompt cache for the whole conversation (#36559)."""
|
||||
result = AzureAnthropicConfig().transform_request(
|
||||
model="claude-opus-4-7",
|
||||
messages=_mid_conversation_system_conversation(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
|
||||
texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"]
|
||||
assert texts[0] == "Second question"
|
||||
assert texts[-1] == "<system-reminder>Answer with exactly one word.</system-reminder>"
|
||||
|
||||
|
||||
def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map):
|
||||
result = AzureAnthropicConfig().transform_request(
|
||||
model="claude-opus-4-8",
|
||||
messages=_mid_conversation_system_conversation(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
|
||||
assert result["messages"][3] == {
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "<system-reminder>Answer with exactly one word.</system-reminder>"}],
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -542,3 +542,51 @@ def test_output_format_removed_from_bedrock_invoke_request():
|
|||
assert (
|
||||
"output_format" not in result
|
||||
), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}"
|
||||
|
||||
|
||||
def _mid_conversation_system_conversation() -> list[dict]:
|
||||
return [
|
||||
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
{"role": "system", "content": "<system-reminder>Answer with exactly one word.</system-reminder>"},
|
||||
{"role": "assistant", "content": "Second answer"},
|
||||
{"role": "user", "content": "Third question"},
|
||||
]
|
||||
|
||||
|
||||
def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map):
|
||||
"""A hoisted reminder rewrites the top-level system block and invalidates the
|
||||
prompt cache for the whole conversation (#36559)."""
|
||||
result = AmazonAnthropicClaudeConfig().transform_request(
|
||||
model="invoke/us.anthropic.claude-opus-4-7",
|
||||
messages=_mid_conversation_system_conversation(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
|
||||
texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"]
|
||||
assert texts[0] == "Second question"
|
||||
assert texts[-1] == "<system-reminder>Answer with exactly one word.</system-reminder>"
|
||||
|
||||
|
||||
def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map):
|
||||
result = AmazonAnthropicClaudeConfig().transform_request(
|
||||
model="invoke/us.anthropic.claude-opus-4-8",
|
||||
messages=_mid_conversation_system_conversation(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
|
||||
assert result["messages"][3] == {
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "<system-reminder>Answer with exactly one word.</system-reminder>"}],
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -727,3 +727,51 @@ def test_sanitize_strips_effort_for_haiku_45():
|
|||
data = {"output_config": {"effort": "high"}}
|
||||
sanitize_vertex_anthropic_output_params(data, "vertex_ai/claude-opus-4-6")
|
||||
assert data["output_config"] == {"effort": "high"}
|
||||
|
||||
|
||||
def _mid_conversation_system_conversation() -> list[dict]:
|
||||
return [
|
||||
{"role": "system", "content": [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]},
|
||||
{"role": "user", "content": "First question"},
|
||||
{"role": "assistant", "content": "First answer"},
|
||||
{"role": "user", "content": "Second question"},
|
||||
{"role": "system", "content": "<system-reminder>Answer with exactly one word.</system-reminder>"},
|
||||
{"role": "assistant", "content": "Second answer"},
|
||||
{"role": "user", "content": "Third question"},
|
||||
]
|
||||
|
||||
|
||||
def test_chat_unflagged_model_converts_mid_conversation_system_instead_of_hoisting(local_model_cost_map):
|
||||
"""A hoisted reminder rewrites the top-level system block and invalidates the
|
||||
prompt cache for the whole conversation (#36559)."""
|
||||
result = VertexAIAnthropicConfig().transform_request(
|
||||
model="claude-opus-4-7",
|
||||
messages=_mid_conversation_system_conversation(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "assistant", "user"]
|
||||
texts = [b["text"] for b in result["messages"][2]["content"] if b.get("type") == "text"]
|
||||
assert texts[0] == "Second question"
|
||||
assert texts[-1] == "<system-reminder>Answer with exactly one word.</system-reminder>"
|
||||
|
||||
|
||||
def test_chat_flagged_model_keeps_mid_conversation_system_role_in_place(local_model_cost_map):
|
||||
result = VertexAIAnthropicConfig().transform_request(
|
||||
model="claude-opus-4-8",
|
||||
messages=_mid_conversation_system_conversation(),
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["system"] == [{"type": "text", "text": "You are terse.", "cache_control": {"type": "ephemeral"}}]
|
||||
assert [m["role"] for m in result["messages"]] == ["user", "assistant", "user", "system", "assistant", "user"]
|
||||
assert result["messages"][3] == {
|
||||
"role": "system",
|
||||
"content": [{"type": "text", "text": "<system-reminder>Answer with exactly one word.</system-reminder>"}],
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue