mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
refactor: address Greptile P2 review items for cache_control /v1/messages fix
- Clarify apply_to_anthropic_messages_request() Returns: docstring: the no-injection early-return path returns original references (no deep copy), while the injection path returns deep-copied values; the previous wording overstated the no-copy path's mutation safety. - Add explanatory note in _supports_native_anthropic_messages() docstring explaining why get_llm_provider() is called unconditionally even when custom_llm_provider is set: routing prefixes like converse/ make naive prefix-stripping unsafe (wrong native-support answer for Bedrock Converse models), so the optimisation is deferred to a follow-up. - NON_CACHEABLE_TOOL_TYPES in anthropic_cache_control_hook is already imported from litellm.constants.ANTHROPIC_NON_CACHEABLE_TOOL_TYPES (shared with transformation.py), so no divergence risk.
This commit is contained in:
parent
cdf68def52
commit
b8f38829af
2 changed files with 15 additions and 9 deletions
|
|
@ -10,6 +10,7 @@ import copy
|
|||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union, cast
|
||||
|
||||
from litellm._logging import verbose_logger
|
||||
from litellm.constants import ANTHROPIC_NON_CACHEABLE_TOOL_TYPES
|
||||
from litellm.integrations.custom_logger import CustomLogger
|
||||
from litellm.integrations.custom_prompt_management import CustomPromptManagement
|
||||
from litellm.integrations.prompt_management_base import PromptManagementClient
|
||||
|
|
@ -31,13 +32,9 @@ else:
|
|||
# breakpoints: "A maximum of 4 blocks with cache_control may be provided."
|
||||
MAX_CACHE_CONTROL_BLOCKS = 4
|
||||
|
||||
# Tool types Anthropic rejects `cache_control` on. Mirrors the exclusion in
|
||||
# litellm/llms/anthropic/chat/transformation.py so the native /v1/messages path
|
||||
# never marks a tool the API would 400 on.
|
||||
NON_CACHEABLE_TOOL_TYPES = (
|
||||
"tool_search_tool_regex_20251119",
|
||||
"tool_search_tool_bm25_20251119",
|
||||
)
|
||||
# Re-export for any existing internal callers while the canonical definition
|
||||
# lives in litellm.constants (shared with transformation.py).
|
||||
NON_CACHEABLE_TOOL_TYPES = ANTHROPIC_NON_CACHEABLE_TOOL_TYPES
|
||||
|
||||
|
||||
class AnthropicCacheControlHook(CustomPromptManagement):
|
||||
|
|
@ -287,8 +284,10 @@ class AnthropicCacheControlHook(CustomPromptManagement):
|
|||
downstream provider transforms still receive them.
|
||||
|
||||
Returns:
|
||||
A tuple of (messages, system, tools) with cache control applied. The
|
||||
inputs are deep-copied, so the caller's originals are never mutated.
|
||||
A tuple of (messages, system, tools). When there are no injection
|
||||
points, the original references are returned as-is (no copy is
|
||||
made). When injection occurs, the inputs are deep-copied before
|
||||
modification so the caller's originals are not mutated.
|
||||
"""
|
||||
injection_points: list[CacheControlInjectionPoint] = non_default_params.pop(
|
||||
"cache_control_injection_points", []
|
||||
|
|
|
|||
|
|
@ -197,6 +197,13 @@ def _supports_native_anthropic_messages(
|
|||
provider selected purely by endpoint (e.g. an ``/anthropic`` ``api_base``)
|
||||
resolves the same way it will in the handler, instead of being misread as a
|
||||
fallback and having its cache injection skipped.
|
||||
|
||||
Note: ``get_llm_provider`` is called unconditionally here even when
|
||||
``custom_llm_provider`` is already set. Skipping it is non-trivial because
|
||||
model strings may carry routing prefixes (e.g. ``converse/``) that affect
|
||||
both the resolved model name and whether ``get_provider_anthropic_messages_config``
|
||||
returns a native config — plain prefix-stripping produces wrong answers for
|
||||
those cases. Left as a potential optimisation for a follow-up.
|
||||
"""
|
||||
from litellm.types.utils import LlmProviders
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue