mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
refactor: remove cc cost optimization template
can't deliver on promise right now
This commit is contained in:
parent
a05b7a7747
commit
cd33993f49
4 changed files with 2 additions and 158 deletions
|
|
@ -1,10 +1,9 @@
|
|||
"""
|
||||
Claude Code guardrail integrations for LiteLLM.
|
||||
|
||||
Two focused policy-enforcement guardrails for Claude Code deployments:
|
||||
Focused policy-enforcement guardrail for Claude Code deployments:
|
||||
|
||||
1. claude_code_prompt_cache — auto-inject Anthropic prompt-caching headers
|
||||
2. claude_code_block_expensive_flags — block expensive API flags (fast mode, etc.)
|
||||
- claude_code_block_expensive_flags — block expensive API flags (fast mode, etc.)
|
||||
|
||||
Hosted tool blocking is handled by the provider-agnostic 'block_hosted_tools'
|
||||
guardrail (see guardrail_hooks/block_hosted_tools/).
|
||||
|
|
@ -16,7 +15,6 @@ import litellm
|
|||
from litellm.types.guardrails import SupportedGuardrailIntegrations
|
||||
|
||||
from .block_expensive_flags import ClaudeCodeBlockExpensiveFlagsGuardrail
|
||||
from .prompt_cache import ClaudeCodePromptCacheGuardrail
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.types.guardrails import Guardrail, LitellmParams
|
||||
|
|
@ -27,21 +25,6 @@ if TYPE_CHECKING:
|
|||
# ------------------------------------------------------------------ #
|
||||
|
||||
|
||||
def _init_prompt_cache(
|
||||
litellm_params: "LitellmParams", guardrail: "Guardrail"
|
||||
) -> ClaudeCodePromptCacheGuardrail:
|
||||
guardrail_name = guardrail.get("guardrail_name")
|
||||
if not guardrail_name:
|
||||
raise ValueError("ClaudeCodePromptCacheGuardrail requires a guardrail_name")
|
||||
instance = ClaudeCodePromptCacheGuardrail(
|
||||
guardrail_name=guardrail_name,
|
||||
event_hook=litellm_params.mode,
|
||||
default_on=litellm_params.default_on,
|
||||
)
|
||||
litellm.logging_callback_manager.add_litellm_callback(instance)
|
||||
return instance
|
||||
|
||||
|
||||
def _init_block_expensive_flags(
|
||||
litellm_params: "LitellmParams", guardrail: "Guardrail"
|
||||
) -> ClaudeCodeBlockExpensiveFlagsGuardrail:
|
||||
|
|
@ -62,16 +45,13 @@ def _init_block_expensive_flags(
|
|||
# ------------------------------------------------------------------ #
|
||||
|
||||
guardrail_initializer_registry = {
|
||||
SupportedGuardrailIntegrations.CLAUDE_CODE_PROMPT_CACHE.value: _init_prompt_cache,
|
||||
SupportedGuardrailIntegrations.CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS.value: _init_block_expensive_flags,
|
||||
}
|
||||
|
||||
guardrail_class_registry = {
|
||||
SupportedGuardrailIntegrations.CLAUDE_CODE_PROMPT_CACHE.value: ClaudeCodePromptCacheGuardrail,
|
||||
SupportedGuardrailIntegrations.CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS.value: ClaudeCodeBlockExpensiveFlagsGuardrail,
|
||||
}
|
||||
|
||||
__all__ = [
|
||||
"ClaudeCodePromptCacheGuardrail",
|
||||
"ClaudeCodeBlockExpensiveFlagsGuardrail",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -1,91 +0,0 @@
|
|||
"""
|
||||
Claude Code - Prompt Cache Injection Guardrail
|
||||
|
||||
Automatically injects cache_control: {type: ephemeral} into system messages
|
||||
so Anthropic can cache the prompt prefix, reducing costs on repeated calls.
|
||||
|
||||
Only runs when the request targets an Anthropic API model. Uses the existing
|
||||
AnthropicCacheControlHook._safe_insert_cache_control_in_message utility so
|
||||
the injection logic is not duplicated.
|
||||
"""
|
||||
|
||||
from typing import TYPE_CHECKING, Literal, Optional
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.integrations.anthropic_cache_control_hook import \
|
||||
AnthropicCacheControlHook
|
||||
from litellm.integrations.custom_guardrail import (CustomGuardrail,
|
||||
log_guardrail_information)
|
||||
from litellm.types.guardrails import GuardrailEventHooks
|
||||
from litellm.types.llms.openai import ChatCompletionCachedContent
|
||||
from litellm.types.utils import GenericGuardrailAPIInputs
|
||||
from litellm.utils import get_llm_provider
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.litellm_core_utils.litellm_logging import \
|
||||
Logging as LiteLLMLoggingObj
|
||||
|
||||
|
||||
def _is_anthropic_model(model: Optional[str]) -> bool:
|
||||
"""Return True when the model resolves to the Anthropic provider."""
|
||||
if not model:
|
||||
return False
|
||||
try:
|
||||
_, custom_llm_provider, _, _ = get_llm_provider(model=model)
|
||||
return custom_llm_provider == "anthropic"
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
class ClaudeCodePromptCacheGuardrail(CustomGuardrail):
|
||||
"""
|
||||
Guardrail that injects Anthropic prompt-caching headers into system messages.
|
||||
|
||||
Targets only Anthropic API models — the provider is detected from the
|
||||
model string so this guardrail is safe to apply globally without
|
||||
restricting other providers.
|
||||
"""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
if "supported_event_hooks" not in kwargs:
|
||||
kwargs["supported_event_hooks"] = [GuardrailEventHooks.pre_call]
|
||||
super().__init__(**kwargs)
|
||||
verbose_proxy_logger.debug("ClaudeCodePromptCacheGuardrail initialized")
|
||||
|
||||
@log_guardrail_information
|
||||
async def apply_guardrail(
|
||||
self,
|
||||
inputs: GenericGuardrailAPIInputs,
|
||||
request_data: dict,
|
||||
input_type: Literal["request", "response"],
|
||||
logging_obj: Optional["LiteLLMLoggingObj"] = None,
|
||||
) -> GenericGuardrailAPIInputs:
|
||||
if input_type != "request":
|
||||
return inputs
|
||||
|
||||
model: Optional[str] = request_data.get("model") or inputs.get("model") # type: ignore[assignment]
|
||||
if not _is_anthropic_model(model):
|
||||
verbose_proxy_logger.debug(
|
||||
f"ClaudeCodePromptCacheGuardrail: skipping non-Anthropic model '{model}'"
|
||||
)
|
||||
return inputs
|
||||
|
||||
messages = request_data.get("messages") or []
|
||||
if not messages:
|
||||
return inputs
|
||||
|
||||
control = ChatCompletionCachedContent(type="ephemeral")
|
||||
modified = []
|
||||
for msg in messages:
|
||||
if msg.get("role") == "system":
|
||||
msg = AnthropicCacheControlHook._safe_insert_cache_control_in_message(
|
||||
message=msg, # type: ignore[arg-type]
|
||||
control=control,
|
||||
)
|
||||
modified.append(msg)
|
||||
|
||||
request_data["messages"] = modified
|
||||
verbose_proxy_logger.debug(
|
||||
"ClaudeCodePromptCacheGuardrail: injected cache_control into system messages"
|
||||
)
|
||||
return inputs
|
||||
|
|
@ -65,7 +65,6 @@ class SupportedGuardrailIntegrations(Enum):
|
|||
QUALIFIRE = "qualifire"
|
||||
CUSTOM_CODE = "custom_code"
|
||||
BLOCK_HOSTED_TOOLS = "block_hosted_tools"
|
||||
CLAUDE_CODE_PROMPT_CACHE = "claude_code_prompt_cache"
|
||||
CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS = "claude_code_block_expensive_flags"
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1106,49 +1106,5 @@
|
|||
"guardrails_add": ["mcp-security-block"],
|
||||
"guardrails_remove": []
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": "claude-code-cost-optimization",
|
||||
"title": "Claude Code Cost Optimization",
|
||||
"description": "Reduces Claude Code API spend by blocking expensive inference modes (fast/turbo, inference_geo, and extended thinking) and automatically injecting prompt caching headers into system messages to maximize cache hit rates.",
|
||||
"icon": "CurrencyDollarIcon",
|
||||
"iconColor": "text-green-500",
|
||||
"iconBg": "bg-green-50",
|
||||
"guardrails": [
|
||||
"claude-code-inject-prompt-cache",
|
||||
"claude-code-block-expensive-flags"
|
||||
],
|
||||
"complexity": "Low",
|
||||
"guardrailDefinitions": [
|
||||
{
|
||||
"guardrail_name": "claude-code-inject-prompt-cache",
|
||||
"litellm_params": {
|
||||
"guardrail": "claude_code_prompt_cache",
|
||||
"mode": "pre_call"
|
||||
},
|
||||
"guardrail_info": {
|
||||
"description": "Automatically adds cache_control: {type: ephemeral} to system messages so Anthropic can cache the system prompt prefix. Only applies to Anthropic API models. Reduces cost on repeated calls that share the same system prompt."
|
||||
}
|
||||
},
|
||||
{
|
||||
"guardrail_name": "claude-code-block-expensive-flags",
|
||||
"litellm_params": {
|
||||
"guardrail": "claude_code_block_expensive_flags",
|
||||
"mode": "pre_call"
|
||||
},
|
||||
"guardrail_info": {
|
||||
"description": "Blocks expensive Anthropic API flags: fast/turbo inference (speed=fast, ~6x pricing), inference geo-routing (inference_geo, 1.1x pricing), and extended thinking (thinking.type=enabled). Also blocks Anthropic-hosted tools inherited from the hosted tools policy."
|
||||
}
|
||||
}
|
||||
],
|
||||
"templateData": {
|
||||
"policy_name": "claude-code-cost-optimization",
|
||||
"description": "Reduces Claude Code API spend by blocking fast mode, inference_geo, extended thinking, and Anthropic-hosted tools, while auto-injecting prompt caching into system messages.",
|
||||
"guardrails_add": [
|
||||
"claude-code-inject-prompt-cache",
|
||||
"claude-code-block-expensive-flags"
|
||||
],
|
||||
"guardrails_remove": []
|
||||
}
|
||||
}
|
||||
]
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue