refactor: remove cc cost optimization template

can't deliver on promise right now
This commit is contained in:
Krrish Dholakia 2026-02-17 22:41:09 -08:00
parent a05b7a7747
commit cd33993f49
4 changed files with 2 additions and 158 deletions

View file

@ -1,10 +1,9 @@
"""
Claude Code guardrail integrations for LiteLLM.
Two focused policy-enforcement guardrails for Claude Code deployments:
Focused policy-enforcement guardrail for Claude Code deployments:
1. claude_code_prompt_cache — auto-inject Anthropic prompt-caching headers
2. claude_code_block_expensive_flags — block expensive API flags (fast mode, etc.)
- claude_code_block_expensive_flags — block expensive API flags (fast mode, etc.)
Hosted tool blocking is handled by the provider-agnostic 'block_hosted_tools'
guardrail (see guardrail_hooks/block_hosted_tools/).
@ -16,7 +15,6 @@ import litellm
from litellm.types.guardrails import SupportedGuardrailIntegrations
from .block_expensive_flags import ClaudeCodeBlockExpensiveFlagsGuardrail
from .prompt_cache import ClaudeCodePromptCacheGuardrail
if TYPE_CHECKING:
from litellm.types.guardrails import Guardrail, LitellmParams
@ -27,21 +25,6 @@ if TYPE_CHECKING:
# ------------------------------------------------------------------ #
def _init_prompt_cache(
litellm_params: "LitellmParams", guardrail: "Guardrail"
) -> ClaudeCodePromptCacheGuardrail:
guardrail_name = guardrail.get("guardrail_name")
if not guardrail_name:
raise ValueError("ClaudeCodePromptCacheGuardrail requires a guardrail_name")
instance = ClaudeCodePromptCacheGuardrail(
guardrail_name=guardrail_name,
event_hook=litellm_params.mode,
default_on=litellm_params.default_on,
)
litellm.logging_callback_manager.add_litellm_callback(instance)
return instance
def _init_block_expensive_flags(
litellm_params: "LitellmParams", guardrail: "Guardrail"
) -> ClaudeCodeBlockExpensiveFlagsGuardrail:
@ -62,16 +45,13 @@ def _init_block_expensive_flags(
# ------------------------------------------------------------------ #
guardrail_initializer_registry = {
SupportedGuardrailIntegrations.CLAUDE_CODE_PROMPT_CACHE.value: _init_prompt_cache,
SupportedGuardrailIntegrations.CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS.value: _init_block_expensive_flags,
}
guardrail_class_registry = {
SupportedGuardrailIntegrations.CLAUDE_CODE_PROMPT_CACHE.value: ClaudeCodePromptCacheGuardrail,
SupportedGuardrailIntegrations.CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS.value: ClaudeCodeBlockExpensiveFlagsGuardrail,
}
__all__ = [
"ClaudeCodePromptCacheGuardrail",
"ClaudeCodeBlockExpensiveFlagsGuardrail",
]

View file

@ -1,91 +0,0 @@
"""
Claude Code - Prompt Cache Injection Guardrail
Automatically injects cache_control: {type: ephemeral} into system messages
so Anthropic can cache the prompt prefix, reducing costs on repeated calls.
Only runs when the request targets an Anthropic API model. Uses the existing
AnthropicCacheControlHook._safe_insert_cache_control_in_message utility so
the injection logic is not duplicated.
"""
from typing import TYPE_CHECKING, Literal, Optional
from litellm._logging import verbose_proxy_logger
from litellm.integrations.anthropic_cache_control_hook import \
AnthropicCacheControlHook
from litellm.integrations.custom_guardrail import (CustomGuardrail,
log_guardrail_information)
from litellm.types.guardrails import GuardrailEventHooks
from litellm.types.llms.openai import ChatCompletionCachedContent
from litellm.types.utils import GenericGuardrailAPIInputs
from litellm.utils import get_llm_provider
if TYPE_CHECKING:
from litellm.litellm_core_utils.litellm_logging import \
Logging as LiteLLMLoggingObj
def _is_anthropic_model(model: Optional[str]) -> bool:
"""Return True when the model resolves to the Anthropic provider."""
if not model:
return False
try:
_, custom_llm_provider, _, _ = get_llm_provider(model=model)
return custom_llm_provider == "anthropic"
except Exception:
return False
class ClaudeCodePromptCacheGuardrail(CustomGuardrail):
"""
Guardrail that injects Anthropic prompt-caching headers into system messages.
Targets only Anthropic API models — the provider is detected from the
model string so this guardrail is safe to apply globally without
restricting other providers.
"""
def __init__(self, **kwargs):
if "supported_event_hooks" not in kwargs:
kwargs["supported_event_hooks"] = [GuardrailEventHooks.pre_call]
super().__init__(**kwargs)
verbose_proxy_logger.debug("ClaudeCodePromptCacheGuardrail initialized")
@log_guardrail_information
async def apply_guardrail(
self,
inputs: GenericGuardrailAPIInputs,
request_data: dict,
input_type: Literal["request", "response"],
logging_obj: Optional["LiteLLMLoggingObj"] = None,
) -> GenericGuardrailAPIInputs:
if input_type != "request":
return inputs
model: Optional[str] = request_data.get("model") or inputs.get("model") # type: ignore[assignment]
if not _is_anthropic_model(model):
verbose_proxy_logger.debug(
f"ClaudeCodePromptCacheGuardrail: skipping non-Anthropic model '{model}'"
)
return inputs
messages = request_data.get("messages") or []
if not messages:
return inputs
control = ChatCompletionCachedContent(type="ephemeral")
modified = []
for msg in messages:
if msg.get("role") == "system":
msg = AnthropicCacheControlHook._safe_insert_cache_control_in_message(
message=msg, # type: ignore[arg-type]
control=control,
)
modified.append(msg)
request_data["messages"] = modified
verbose_proxy_logger.debug(
"ClaudeCodePromptCacheGuardrail: injected cache_control into system messages"
)
return inputs

View file

@ -65,7 +65,6 @@ class SupportedGuardrailIntegrations(Enum):
QUALIFIRE = "qualifire"
CUSTOM_CODE = "custom_code"
BLOCK_HOSTED_TOOLS = "block_hosted_tools"
CLAUDE_CODE_PROMPT_CACHE = "claude_code_prompt_cache"
CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS = "claude_code_block_expensive_flags"

View file

@ -1106,49 +1106,5 @@
"guardrails_add": ["mcp-security-block"],
"guardrails_remove": []
}
},
{
"id": "claude-code-cost-optimization",
"title": "Claude Code Cost Optimization",
"description": "Reduces Claude Code API spend by blocking expensive inference modes (fast/turbo, inference_geo, and extended thinking) and automatically injecting prompt caching headers into system messages to maximize cache hit rates.",
"icon": "CurrencyDollarIcon",
"iconColor": "text-green-500",
"iconBg": "bg-green-50",
"guardrails": [
"claude-code-inject-prompt-cache",
"claude-code-block-expensive-flags"
],
"complexity": "Low",
"guardrailDefinitions": [
{
"guardrail_name": "claude-code-inject-prompt-cache",
"litellm_params": {
"guardrail": "claude_code_prompt_cache",
"mode": "pre_call"
},
"guardrail_info": {
"description": "Automatically adds cache_control: {type: ephemeral} to system messages so Anthropic can cache the system prompt prefix. Only applies to Anthropic API models. Reduces cost on repeated calls that share the same system prompt."
}
},
{
"guardrail_name": "claude-code-block-expensive-flags",
"litellm_params": {
"guardrail": "claude_code_block_expensive_flags",
"mode": "pre_call"
},
"guardrail_info": {
"description": "Blocks expensive Anthropic API flags: fast/turbo inference (speed=fast, ~6x pricing), inference geo-routing (inference_geo, 1.1x pricing), and extended thinking (thinking.type=enabled). Also blocks Anthropic-hosted tools inherited from the hosted tools policy."
}
}
],
"templateData": {
"policy_name": "claude-code-cost-optimization",
"description": "Reduces Claude Code API spend by blocking fast mode, inference_geo, extended thinking, and Anthropic-hosted tools, while auto-injecting prompt caching into system messages.",
"guardrails_add": [
"claude-code-inject-prompt-cache",
"claude-code-block-expensive-flags"
],
"guardrails_remove": []
}
}
]