mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
Guardrails - add built in guardrails for harmful content, bias, etc. (#18029)
* feat(litellm_content_filter.py): add support for content filtering categories make it easy for proxy admin to prevent messages about violence, self harm or illegal weapons going through litellm * feat: initial commit adding bias detection allows admin to block inappropriate content about sexual orientation, etc. * refactor: simplify content_filter.py use a more exhaustive set of keywords, instead of guessing at potential phrases user can use * feat(content_filter.py): add new denied topics for in-built content filter guardrails allow user to automatically block content relating to certain categories from being sent to the LLML * refactor(content-filter): document new params to litellm content filter * feat(ui/): litellm content filter - select content categories on ui * docs: update documentation * docs(litellm_content_filter.md): document new content filters
This commit is contained in:
parent
630f3d828e
commit
26cd2c4473
51 changed files with 3228 additions and 40 deletions
|
|
@ -3,10 +3,12 @@ import TabItem from '@theme/TabItem';
|
|||
import Image from '@theme/IdealImage';
|
||||
|
||||
|
||||
# LiteLLM Content Filter
|
||||
# LiteLLM Content Filter (Built-in Guardrails)
|
||||
|
||||
**Built-in guardrail** for detecting and filtering sensitive information using regex patterns and keyword matching. No external dependencies required.
|
||||
|
||||
**When to use?** Good for cases which do not require an ML model to detect sensitive information.
|
||||
|
||||
## Overview
|
||||
|
||||
| Property | Details |
|
||||
|
|
@ -56,6 +58,44 @@ Test examples:
|
|||
|
||||
### Step 1: Define Guardrails in config.yaml
|
||||
|
||||
<Tabs>
|
||||
<TabItem label="Harmful Content Detection" value="harmful">
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "harmful-content-filter"
|
||||
litellm_params:
|
||||
guardrail: litellm_content_filter
|
||||
mode: "pre_call"
|
||||
|
||||
# Enable harmful content categories
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
- category: "harmful_illegal_weapons"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem label="PII Protection" value="pii">
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
|
|
@ -86,6 +126,48 @@ guardrails:
|
|||
description: "Sensitive internal information"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
|
||||
<TabItem label="Combined" value="combined">
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
model_list:
|
||||
- model_name: gpt-3.5-turbo
|
||||
litellm_params:
|
||||
model: openai/gpt-3.5-turbo
|
||||
api_key: os.environ/OPENAI_API_KEY
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "comprehensive-filter"
|
||||
litellm_params:
|
||||
guardrail: litellm_content_filter
|
||||
mode: "pre_call"
|
||||
|
||||
# Harmful content categories
|
||||
categories:
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high"
|
||||
|
||||
# PII patterns
|
||||
patterns:
|
||||
- pattern_type: "prebuilt"
|
||||
pattern_name: "us_ssn"
|
||||
action: "BLOCK"
|
||||
- pattern_type: "prebuilt"
|
||||
pattern_name: "email"
|
||||
action: "MASK"
|
||||
|
||||
# Custom keywords
|
||||
blocked_words:
|
||||
- keyword: "confidential"
|
||||
action: "BLOCK"
|
||||
```
|
||||
|
||||
</TabItem>
|
||||
</Tabs>
|
||||
|
||||
### Step 2: Start LiteLLM Gateway
|
||||
|
||||
```shell
|
||||
|
|
@ -363,9 +445,171 @@ Output: "Email ***EMAIL***, SSN ***US_SSN***, ***REDACTED*** data"
|
|||
- Pattern names are automatically uppercased (e.g., `email` → `EMAIL`)
|
||||
- `keyword_redaction_tag` is a fixed string (no placeholders)
|
||||
|
||||
## Content Categories
|
||||
|
||||
Prebuilt categories use **keyword matching** to detect harmful content, bias, and inappropriate advice. Keywords are matched with word boundaries (single words) or as substrings (multi-word phrases), case-insensitive.
|
||||
|
||||
### Available Categories
|
||||
|
||||
| Category | Description |
|
||||
|----------|-------------|
|
||||
| **Harmful Content** | |
|
||||
| `harmful_self_harm` | Self-harm, suicide, eating disorders |
|
||||
| `harmful_violence` | Violence, criminal planning, attacks |
|
||||
| `harmful_illegal_weapons` | Illegal weapons, explosives, dangerous materials |
|
||||
| **Bias Detection** | |
|
||||
| `bias_gender` | Gender-based discrimination, stereotypes |
|
||||
| `bias_sexual_orientation` | LGBTQ+ discrimination, homophobia, transphobia |
|
||||
| `bias_racial` | Racial/ethnic discrimination, stereotypes |
|
||||
| `bias_religious` | Religious discrimination, stereotypes |
|
||||
| **Denied Advice** | |
|
||||
| `denied_financial_advice` | Personalized financial advice, investment recommendations |
|
||||
| `denied_medical_advice` | Medical advice, diagnosis, treatment recommendations |
|
||||
| `denied_legal_advice` | Legal advice, representation, legal strategy |
|
||||
|
||||
:::info Bias Detection Considerations
|
||||
|
||||
Bias detection is **complex and context-dependent**. Rule-based systems catch explicit discriminatory language but may generate false positives on legitimate discussions. Start with **high severity thresholds** and test thoroughly. For mission-critical bias detection, consider combining with AI-based guardrails (e.g., HiddenLayer, Lakera).
|
||||
|
||||
:::
|
||||
|
||||
### Configuration
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
guardrails:
|
||||
- guardrail_name: "content-filter"
|
||||
litellm_params:
|
||||
guardrail: litellm_content_filter
|
||||
mode: "pre_call"
|
||||
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium" # Blocks medium+ severity
|
||||
|
||||
- category: "bias_gender"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit discrimination
|
||||
|
||||
- category: "denied_financial_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
```
|
||||
|
||||
**Severity Thresholds:**
|
||||
- `"high"` - Only blocks high severity items
|
||||
- `"medium"` - Blocks medium and high severity (default)
|
||||
- `"low"` - Blocks all severity levels
|
||||
|
||||
### Custom Category Files
|
||||
|
||||
Override default categories with custom keyword lists:
|
||||
|
||||
```yaml showLineNumbers title="config.yaml"
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
category_file: "/path/to/custom.yaml"
|
||||
```
|
||||
|
||||
```yaml showLineNumbers title="custom.yaml"
|
||||
category_name: "harmful_self_harm"
|
||||
description: "Custom self-harm detection"
|
||||
default_action: "BLOCK"
|
||||
|
||||
keywords:
|
||||
- keyword: "suicide"
|
||||
severity: "high"
|
||||
- keyword: "harm myself"
|
||||
severity: "high"
|
||||
|
||||
exceptions:
|
||||
- "suicide prevention"
|
||||
- "mental health"
|
||||
```
|
||||
|
||||
## Use Cases
|
||||
|
||||
### 1. PII Protection
|
||||
### 1. Harmful Content Detection
|
||||
|
||||
Block or detect requests containing harmful, illegal, or dangerous content:
|
||||
|
||||
```yaml
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high"
|
||||
- category: "harmful_illegal_weapons"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
```
|
||||
|
||||
### 2. Bias and Discrimination Detection
|
||||
|
||||
Detect and block biased, discriminatory, or hateful content across multiple dimensions:
|
||||
|
||||
```yaml
|
||||
categories:
|
||||
# Gender-based discrimination
|
||||
- category: "bias_gender"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
# LGBTQ+ discrimination
|
||||
- category: "bias_sexual_orientation"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
# Racial/ethnic discrimination
|
||||
- category: "bias_racial"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
# Religious discrimination
|
||||
- category: "bias_religious"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
```
|
||||
|
||||
**Sensitivity Tuning:**
|
||||
|
||||
For bias detection, severity thresholds are critical to balance safety and legitimate discourse:
|
||||
|
||||
```yaml
|
||||
# Conservative (low false positives, may miss subtle bias)
|
||||
categories:
|
||||
- category: "bias_racial"
|
||||
severity_threshold: "high" # Only blocks explicit discriminatory language
|
||||
|
||||
# Balanced (recommended)
|
||||
categories:
|
||||
- category: "bias_gender"
|
||||
severity_threshold: "medium" # Blocks stereotypes and explicit discrimination
|
||||
|
||||
# Strict (high safety, may have more false positives)
|
||||
categories:
|
||||
- category: "bias_sexual_orientation"
|
||||
severity_threshold: "low" # Blocks all potentially problematic content
|
||||
```
|
||||
|
||||
|
||||
|
||||
### 3. PII Protection
|
||||
Block or mask personally identifiable information before sending to LLMs:
|
||||
|
||||
```yaml
|
||||
|
|
@ -409,7 +653,54 @@ For large lists of sensitive terms, use a file:
|
|||
blocked_words_file: "/path/to/sensitive_terms.yaml"
|
||||
```
|
||||
|
||||
### 4. Compliance
|
||||
### 4. Safe AI for Consumer Applications
|
||||
|
||||
Combining harmful content and bias detection for consumer-facing AI:
|
||||
|
||||
```yaml
|
||||
guardrails:
|
||||
- guardrail_name: "safe-consumer-ai"
|
||||
litellm_params:
|
||||
guardrail: litellm_content_filter
|
||||
mode: "pre_call"
|
||||
|
||||
categories:
|
||||
# Harmful content - strict
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
# Bias detection - balanced
|
||||
- category: "bias_gender"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Avoid blocking legitimate gender discussions
|
||||
|
||||
- category: "bias_sexual_orientation"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
|
||||
- category: "bias_racial"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Education and news may discuss race
|
||||
```
|
||||
|
||||
**Perfect for:**
|
||||
- Chatbots and virtual assistants
|
||||
- Educational AI tools
|
||||
- Customer service AI
|
||||
- Content generation platforms
|
||||
- Public-facing AI applications
|
||||
|
||||
### 5. Compliance
|
||||
Ensure regulatory compliance by filtering sensitive data types:
|
||||
|
||||
```yaml
|
||||
|
|
@ -422,8 +713,184 @@ patterns:
|
|||
action: "BLOCK"
|
||||
```
|
||||
|
||||
## Best Practices for Bias Detection
|
||||
|
||||
### Choosing the Right Severity Threshold
|
||||
|
||||
Bias detection requires careful tuning to avoid blocking legitimate content:
|
||||
|
||||
**High Threshold (Recommended for most use cases)**
|
||||
- Blocks only explicit discriminatory language
|
||||
- Lower false positives
|
||||
- Allows nuanced discussions about identity, diversity, and social issues
|
||||
- Good for: Public-facing applications, education, research
|
||||
|
||||
**Medium Threshold**
|
||||
- Blocks stereotypes and generalizations
|
||||
- Balanced approach
|
||||
- May catch some edge cases in legitimate discourse
|
||||
- Good for: Consumer applications, internal tools, moderated environments
|
||||
|
||||
**Low Threshold**
|
||||
- Strictest filtering
|
||||
- Blocks even borderline language
|
||||
- Higher false positives but maximum safety
|
||||
- Good for: Youth-focused applications, highly controlled environments
|
||||
|
||||
### Testing Your Bias Filters
|
||||
|
||||
Always test with realistic use cases:
|
||||
|
||||
```yaml
|
||||
# Test legitimate discussions (should NOT be blocked)
|
||||
- "Our company has a gender diversity initiative"
|
||||
- "Research shows racial disparities in healthcare"
|
||||
- "We support LGBTQ+ rights and equality"
|
||||
- "Religious freedom is a fundamental right"
|
||||
|
||||
# Test discriminatory content (SHOULD be blocked)
|
||||
- "Women are too emotional to lead"
|
||||
- "All [group] are [negative stereotype]"
|
||||
- "Being gay is unnatural"
|
||||
- "[Religious group] are all extremists"
|
||||
```
|
||||
|
||||
### Monitoring and Iteration
|
||||
|
||||
1. **Log blocked requests** to review false positives
|
||||
2. **Add exceptions** for legitimate terms in your domain
|
||||
3. **Adjust severity thresholds** based on your audience
|
||||
4. **Use custom category files** for domain-specific bias patterns
|
||||
|
||||
### Cultural and Linguistic Considerations
|
||||
|
||||
The prebuilt categories focus on English and common patterns. For other languages or cultural contexts:
|
||||
|
||||
1. Create custom category files with region-specific terms
|
||||
2. Consult with native speakers and cultural experts
|
||||
3. Include local slurs and stereotypes
|
||||
4. Adjust severity based on regional norms
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### False Positives with Bias Detection
|
||||
|
||||
**Issue:** Legitimate discussions about diversity, identity, or social issues are being blocked
|
||||
|
||||
**Solutions:**
|
||||
|
||||
1. **Raise severity threshold:**
|
||||
```yaml
|
||||
categories:
|
||||
- category: "bias_racial"
|
||||
severity_threshold: "high" # Only explicit discrimination
|
||||
```
|
||||
|
||||
2. **Add domain-specific exceptions:**
|
||||
```yaml
|
||||
# my_custom_bias_gender.yaml
|
||||
exceptions:
|
||||
- "gender pay gap"
|
||||
- "gender diversity"
|
||||
- "women in tech"
|
||||
- "gender equality"
|
||||
- "dei initiative"
|
||||
- "inclusion program"
|
||||
```
|
||||
|
||||
3. **Review what was blocked:**
|
||||
Check error details to understand what triggered the block:
|
||||
```json
|
||||
{
|
||||
"error": "Content blocked: bias_gender category keyword 'women' detected (severity: medium)",
|
||||
"category": "bias_gender",
|
||||
"keyword": "women",
|
||||
"severity": "medium"
|
||||
}
|
||||
```
|
||||
|
||||
If this is a false positive, add "women in leadership" or other legitimate phrases to exceptions in your custom category file.
|
||||
|
||||
### False Positives with Categories
|
||||
|
||||
**Issue:** Legitimate content is being blocked by category filters
|
||||
|
||||
**Solution 1:** Adjust severity threshold to only block high-severity items:
|
||||
```yaml
|
||||
categories:
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only block explicit harmful content
|
||||
```
|
||||
|
||||
**Solution 2:** Add exceptions to your custom category file:
|
||||
```yaml
|
||||
# my_custom_violence.yaml
|
||||
exceptions:
|
||||
- "crime statistics"
|
||||
- "documentary"
|
||||
- "news report"
|
||||
- "historical context"
|
||||
```
|
||||
|
||||
**Solution 3:** Use a custom category file with your own curated keyword list:
|
||||
```yaml
|
||||
categories:
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
category_file: "/path/to/my_violence_keywords.yaml"
|
||||
```
|
||||
|
||||
### Category Not Loading
|
||||
|
||||
**Issue:** Category is not being applied
|
||||
|
||||
**Checklist:**
|
||||
1. Verify category is enabled: `enabled: true`
|
||||
2. Check category name matches file: `harmful_self_harm.yaml`
|
||||
3. Check file exists in `litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/`
|
||||
4. Review logs for loading errors: `litellm --config config.yaml --detailed_debug`
|
||||
|
||||
### Keyword Not Matching
|
||||
|
||||
**Issue:** Expected keyword is not being detected
|
||||
|
||||
**Solutions:**
|
||||
|
||||
1. **For single words:** Ensure the keyword appears as a whole word. The system uses word boundary matching, so "men" won't match "recommend".
|
||||
|
||||
2. **For multi-word phrases:** Use the exact phrase as it should appear. Multi-word keywords are matched as substrings (case-insensitive), so "harm myself" will match "I want to harm myself" or "harming myself".
|
||||
|
||||
3. **Check exceptions:** If your keyword is in the exceptions list, it won't be detected. Review the category file's exceptions section.
|
||||
|
||||
4. **Verify severity threshold:** Lower severity keywords won't match if your threshold is set too high. For example, if a keyword has `severity: "low"` but your `severity_threshold: "high"`, it won't match.
|
||||
|
||||
### Too Many False Negatives
|
||||
|
||||
**Issue:** Harmful content is not being caught
|
||||
|
||||
**Solution 1:** Lower severity threshold:
|
||||
```yaml
|
||||
severity_threshold: "low" # Catch more but may increase false positives
|
||||
```
|
||||
|
||||
**Solution 2:** Add custom keywords for your specific use case:
|
||||
```yaml
|
||||
categories:
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
category_file: "/path/to/enhanced_violence.yaml"
|
||||
|
||||
# In enhanced_violence.yaml, add domain-specific keywords
|
||||
keywords:
|
||||
- keyword: "your specific harmful phrase"
|
||||
severity: "high"
|
||||
- keyword: "another harmful term"
|
||||
severity: "medium"
|
||||
```
|
||||
|
||||
### Pattern Not Matching
|
||||
|
||||
**Issue:** Regex pattern isn't detecting expected content
|
||||
|
|
@ -440,16 +907,23 @@ print(re.search(pattern, test_text)) # Should match
|
|||
|
||||
**Issue:** Text contains multiple sensitive patterns
|
||||
|
||||
**Solution:** First matching pattern/keyword is processed. Order patterns by priority:
|
||||
**Solution:** Guardrail checks in this order: categories (keywords), regex patterns, then blocked words. Order by priority:
|
||||
```yaml
|
||||
# Categories checked first (high priority)
|
||||
# Category keywords are matched first
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
severity_threshold: "high"
|
||||
|
||||
# Then regex patterns
|
||||
patterns:
|
||||
# Most critical first
|
||||
- pattern_type: "prebuilt"
|
||||
pattern_name: "us_ssn"
|
||||
action: "BLOCK"
|
||||
# Less critical
|
||||
- pattern_type: "prebuilt"
|
||||
pattern_name: "email"
|
||||
|
||||
# Then simple blocked keywords
|
||||
blocked_words:
|
||||
- keyword: "confidential"
|
||||
action: "MASK"
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -52,6 +52,7 @@ const sidebars = {
|
|||
]
|
||||
},
|
||||
"proxy/guardrails/test_playground",
|
||||
"proxy/guardrails/litellm_content_filter",
|
||||
...[
|
||||
"proxy/guardrails/aim_security",
|
||||
"proxy/guardrails/onyx_security",
|
||||
|
|
@ -63,7 +64,6 @@ const sidebars = {
|
|||
"proxy/guardrails/grayswan",
|
||||
"proxy/guardrails/hiddenlayer",
|
||||
"proxy/guardrails/lasso_security",
|
||||
"proxy/guardrails/litellm_content_filter",
|
||||
"proxy/guardrails/guardrails_ai",
|
||||
"proxy/guardrails/lakera_ai",
|
||||
"proxy/guardrails/model_armor",
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
|
|
@ -21,6 +21,61 @@ model_list:
|
|||
# api_base: http://localhost:8080
|
||||
# default_on: true
|
||||
|
||||
guardrails:
|
||||
- guardrail_name: "harmful-content-filter"
|
||||
litellm_params:
|
||||
guardrail: litellm_content_filter
|
||||
mode: "pre_call"
|
||||
default_on: true
|
||||
# Model configuration
|
||||
guardrail_model:
|
||||
- model: "gpt-4o" # From model_list
|
||||
supported_multimodal_content: # Supported content types = images, documents, audio, video, text
|
||||
- images
|
||||
- documents
|
||||
|
||||
categories:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium" # Block medium+
|
||||
|
||||
- category: "harmful_violence"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit
|
||||
|
||||
- category: "harmful_illegal_weapons"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "low" # Strictest
|
||||
|
||||
- category: "bias_gender"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "bias_sexual_orientation"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "denied_medical_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "denied_legal_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
- category: "denied_financial_advice"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "high" # Only explicit to reduce false positives
|
||||
|
||||
|
||||
prompts:
|
||||
- prompt_id: "simple_prompt"
|
||||
litellm_params:
|
||||
|
|
|
|||
|
|
@ -699,6 +699,7 @@ async def get_guardrail_ui_settings():
|
|||
"""
|
||||
from litellm.proxy.guardrails.guardrail_hooks.litellm_content_filter.patterns import (
|
||||
PATTERN_CATEGORIES,
|
||||
get_available_content_categories,
|
||||
get_pattern_metadata,
|
||||
)
|
||||
|
||||
|
|
@ -721,10 +722,56 @@ async def get_guardrail_ui_settings():
|
|||
"prebuilt_patterns": get_pattern_metadata(),
|
||||
"pattern_categories": list(PATTERN_CATEGORIES.keys()),
|
||||
"supported_actions": ["BLOCK", "MASK"],
|
||||
"content_categories": get_available_content_categories(),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/guardrails/ui/category_yaml/{category_name}",
|
||||
tags=["Guardrails"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
)
|
||||
async def get_category_yaml(category_name: str):
|
||||
"""
|
||||
Get the YAML content for a specific content filter category.
|
||||
|
||||
Args:
|
||||
category_name: The name of the category (e.g., "bias_gender", "harmful_self_harm")
|
||||
|
||||
Returns:
|
||||
The raw YAML content of the category file
|
||||
"""
|
||||
import os
|
||||
|
||||
# Get the categories directory path
|
||||
categories_dir = os.path.join(
|
||||
os.path.dirname(__file__),
|
||||
"guardrail_hooks",
|
||||
"litellm_content_filter",
|
||||
"categories",
|
||||
)
|
||||
|
||||
# Construct the file path
|
||||
category_file_path = os.path.join(categories_dir, f"{category_name}.yaml")
|
||||
|
||||
if not os.path.exists(category_file_path):
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"Category file not found: {category_name}"
|
||||
)
|
||||
|
||||
try:
|
||||
# Read and return the raw YAML content
|
||||
with open(category_file_path, "r") as f:
|
||||
yaml_content = f.read()
|
||||
|
||||
return {"category_name": category_name, "yaml_content": yaml_content}
|
||||
except Exception as e:
|
||||
raise HTTPException(
|
||||
status_code=500, detail=f"Error reading category file: {str(e)}"
|
||||
)
|
||||
|
||||
|
||||
@router.post(
|
||||
"/guardrails/validate_blocked_words_file",
|
||||
tags=["Guardrails"],
|
||||
|
|
|
|||
|
|
@ -13,18 +13,18 @@ if TYPE_CHECKING:
|
|||
def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"):
|
||||
"""
|
||||
Initialize the Content Filter Guardrail.
|
||||
|
||||
|
||||
Args:
|
||||
litellm_params: Guardrail configuration parameters
|
||||
guardrail: Guardrail metadata
|
||||
|
||||
|
||||
Returns:
|
||||
Initialized ContentFilterGuardrail instance
|
||||
"""
|
||||
guardrail_name = guardrail.get("guardrail_name")
|
||||
if not guardrail_name:
|
||||
raise ValueError("Content Filter: guardrail_name is required")
|
||||
|
||||
|
||||
content_filter_guardrail = ContentFilterGuardrail(
|
||||
guardrail_name=guardrail_name,
|
||||
patterns=litellm_params.patterns,
|
||||
|
|
@ -32,12 +32,12 @@ def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"
|
|||
blocked_words_file=litellm_params.blocked_words_file,
|
||||
event_hook=litellm_params.mode, # type: ignore
|
||||
default_on=litellm_params.default_on or False,
|
||||
categories=getattr(litellm_params, "categories", None),
|
||||
severity_threshold=getattr(litellm_params, "severity_threshold", "medium"),
|
||||
)
|
||||
|
||||
litellm.logging_callback_manager.add_litellm_callback(
|
||||
content_filter_guardrail
|
||||
)
|
||||
|
||||
|
||||
litellm.logging_callback_manager.add_litellm_callback(content_filter_guardrail)
|
||||
|
||||
return content_filter_guardrail
|
||||
|
||||
|
||||
|
|
@ -49,4 +49,3 @@ guardrail_initializer_registry = {
|
|||
guardrail_class_registry = {
|
||||
SupportedGuardrailIntegrations.LITELLM_CONTENT_FILTER.value: ContentFilterGuardrail,
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,53 @@
|
|||
# Gender-based bias and discrimination detection
|
||||
category_name: "bias_gender"
|
||||
description: "Detects gender-based discriminatory language, stereotypes, and biased content"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - gender identity terms
|
||||
- keyword: "women"
|
||||
severity: "high"
|
||||
- keyword: "woman"
|
||||
severity: "high"
|
||||
- keyword: "men"
|
||||
severity: "high"
|
||||
- keyword: "man"
|
||||
severity: "high"
|
||||
- keyword: "female"
|
||||
severity: "high"
|
||||
- keyword: "females"
|
||||
severity: "high"
|
||||
- keyword: "male"
|
||||
severity: "high"
|
||||
- keyword: "males"
|
||||
severity: "high"
|
||||
- keyword: "girl"
|
||||
severity: "high"
|
||||
- keyword: "girls"
|
||||
severity: "high"
|
||||
- keyword: "boy"
|
||||
severity: "high"
|
||||
- keyword: "boys"
|
||||
severity: "high"
|
||||
|
||||
# Exceptions - legitimate discussions about gender
|
||||
exceptions:
|
||||
- "gender equality"
|
||||
- "gender diversity"
|
||||
- "gender studies"
|
||||
- "gender gap"
|
||||
- "gender discrimination"
|
||||
- "combat gender"
|
||||
- "address gender"
|
||||
- "research shows"
|
||||
- "study found"
|
||||
- "gender identity"
|
||||
- "gender expression"
|
||||
- "transgender"
|
||||
- "gender neutral"
|
||||
- "women's rights"
|
||||
- "women's health"
|
||||
- "men's health"
|
||||
- "gender bias"
|
||||
- "gender equity"
|
||||
|
|
@ -0,0 +1,148 @@
|
|||
# Racial and ethnic bias detection
|
||||
category_name: "bias_racial"
|
||||
description: "Detects racial and ethnic discrimination, stereotypes, and biased content"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - racial/ethnic identity terms
|
||||
- keyword: "black"
|
||||
severity: "high"
|
||||
- keyword: "blacks"
|
||||
severity: "high"
|
||||
- keyword: "white"
|
||||
severity: "high"
|
||||
- keyword: "whites"
|
||||
severity: "high"
|
||||
- keyword: "asian"
|
||||
severity: "high"
|
||||
- keyword: "asians"
|
||||
severity: "high"
|
||||
- keyword: "hispanic"
|
||||
severity: "high"
|
||||
- keyword: "hispanics"
|
||||
severity: "high"
|
||||
- keyword: "latino"
|
||||
severity: "high"
|
||||
- keyword: "latina"
|
||||
severity: "high"
|
||||
- keyword: "latinos"
|
||||
severity: "high"
|
||||
- keyword: "mexican"
|
||||
severity: "high"
|
||||
- keyword: "mexicans"
|
||||
severity: "high"
|
||||
- keyword: "chinese"
|
||||
severity: "high"
|
||||
- keyword: "indian"
|
||||
severity: "high"
|
||||
- keyword: "indians"
|
||||
severity: "high"
|
||||
- keyword: "arab"
|
||||
severity: "high"
|
||||
- keyword: "arabs"
|
||||
severity: "high"
|
||||
- keyword: "african"
|
||||
severity: "high"
|
||||
- keyword: "africans"
|
||||
severity: "high"
|
||||
- keyword: "immigrant"
|
||||
severity: "high"
|
||||
- keyword: "immigrants"
|
||||
severity: "high"
|
||||
- keyword: "refugee"
|
||||
severity: "high"
|
||||
- keyword: "refugees"
|
||||
severity: "high"
|
||||
- keyword: "minority"
|
||||
severity: "high"
|
||||
- keyword: "minorities"
|
||||
severity: "high"
|
||||
|
||||
# High severity - discriminatory terms
|
||||
- keyword: "inferior race"
|
||||
severity: "high"
|
||||
- keyword: "superior race"
|
||||
severity: "high"
|
||||
- keyword: "racial purity"
|
||||
severity: "high"
|
||||
- keyword: "master race"
|
||||
severity: "high"
|
||||
- keyword: "white supremacy"
|
||||
severity: "high"
|
||||
- keyword: "white genocide"
|
||||
severity: "high"
|
||||
- keyword: "great replacement"
|
||||
severity: "high"
|
||||
- keyword: "race traitor"
|
||||
severity: "high"
|
||||
- keyword: "race mixing"
|
||||
severity: "high"
|
||||
- keyword: "model minority"
|
||||
severity: "high"
|
||||
- keyword: "ghetto culture"
|
||||
severity: "high"
|
||||
- keyword: "thug culture"
|
||||
severity: "high"
|
||||
- keyword: "diversity hire"
|
||||
severity: "high"
|
||||
- keyword: "black crime"
|
||||
severity: "high"
|
||||
- keyword: "immigrant crime"
|
||||
severity: "high"
|
||||
- keyword: "minority lazy"
|
||||
severity: "high"
|
||||
- keyword: "stealing jobs"
|
||||
severity: "high"
|
||||
- keyword: "go back"
|
||||
severity: "high"
|
||||
- keyword: "you people"
|
||||
severity: "medium"
|
||||
- keyword: "those people"
|
||||
severity: "medium"
|
||||
- keyword: "all blacks"
|
||||
severity: "high"
|
||||
- keyword: "all whites"
|
||||
severity: "high"
|
||||
- keyword: "all asians"
|
||||
severity: "high"
|
||||
- keyword: "all hispanics"
|
||||
severity: "high"
|
||||
- keyword: "all latinos"
|
||||
severity: "high"
|
||||
- keyword: "all mexicans"
|
||||
severity: "high"
|
||||
- keyword: "all immigrants"
|
||||
severity: "high"
|
||||
|
||||
# Exceptions - legitimate discussions about race, diversity, anti-racism
|
||||
exceptions:
|
||||
- "racial equality"
|
||||
- "racial justice"
|
||||
- "racial discrimination"
|
||||
- "anti-racism"
|
||||
- "combat racism"
|
||||
- "racial bias"
|
||||
- "systemic racism"
|
||||
- "structural racism"
|
||||
- "racial equity"
|
||||
- "diversity and inclusion"
|
||||
- "black lives matter"
|
||||
- "civil rights"
|
||||
- "fight racism"
|
||||
- "address racism"
|
||||
- "racial disparities"
|
||||
- "racism is"
|
||||
- "racist"
|
||||
- "study shows"
|
||||
- "research indicates"
|
||||
- "historical"
|
||||
- "black history"
|
||||
- "black culture"
|
||||
- "asian culture"
|
||||
- "hispanic culture"
|
||||
- "latino culture"
|
||||
- "immigrant rights"
|
||||
- "immigrant support"
|
||||
- "protect immigrants"
|
||||
- "minority rights"
|
||||
|
|
@ -0,0 +1,118 @@
|
|||
# Religious bias and discrimination detection
|
||||
category_name: "bias_religious"
|
||||
description: "Detects religious discrimination, stereotypes, and biased content"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - religion terms that could be used discriminatorily
|
||||
- keyword: "muslim"
|
||||
severity: "high"
|
||||
- keyword: "muslims"
|
||||
severity: "high"
|
||||
- keyword: "islam"
|
||||
severity: "high"
|
||||
- keyword: "islamic"
|
||||
severity: "high"
|
||||
- keyword: "jew"
|
||||
severity: "high"
|
||||
- keyword: "jews"
|
||||
severity: "high"
|
||||
- keyword: "jewish"
|
||||
severity: "high"
|
||||
- keyword: "christian"
|
||||
severity: "high"
|
||||
- keyword: "christians"
|
||||
severity: "high"
|
||||
- keyword: "christianity"
|
||||
severity: "high"
|
||||
- keyword: "atheist"
|
||||
severity: "high"
|
||||
- keyword: "atheists"
|
||||
severity: "high"
|
||||
- keyword: "hindu"
|
||||
severity: "high"
|
||||
- keyword: "hindus"
|
||||
severity: "high"
|
||||
- keyword: "hinduism"
|
||||
severity: "high"
|
||||
- keyword: "buddhist"
|
||||
severity: "high"
|
||||
- keyword: "buddhists"
|
||||
severity: "high"
|
||||
- keyword: "buddhism"
|
||||
severity: "high"
|
||||
- keyword: "sikh"
|
||||
severity: "high"
|
||||
- keyword: "sikhs"
|
||||
severity: "high"
|
||||
- keyword: "catholic"
|
||||
severity: "high"
|
||||
- keyword: "catholics"
|
||||
severity: "high"
|
||||
- keyword: "protestant"
|
||||
severity: "high"
|
||||
- keyword: "protestants"
|
||||
severity: "high"
|
||||
- keyword: "mormon"
|
||||
severity: "high"
|
||||
- keyword: "mormons"
|
||||
severity: "high"
|
||||
|
||||
|
||||
# Medium severity - context-dependent
|
||||
- keyword: "religious extremist"
|
||||
severity: "medium"
|
||||
- keyword: "religious fanatic"
|
||||
severity: "medium"
|
||||
- keyword: "infidel"
|
||||
severity: "medium"
|
||||
- keyword: "infidels"
|
||||
severity: "medium"
|
||||
- keyword: "jihad"
|
||||
severity: "medium"
|
||||
- keyword: "crusade"
|
||||
severity: "medium"
|
||||
|
||||
# Exceptions - legitimate religious discussion, education, interfaith work
|
||||
exceptions:
|
||||
- "religious freedom"
|
||||
- "religious tolerance"
|
||||
- "religious discrimination"
|
||||
- "religious persecution"
|
||||
- "interfaith"
|
||||
- "religious studies"
|
||||
- "religious history"
|
||||
- "combat islamophobia"
|
||||
- "fight antisemitism"
|
||||
- "religious diversity"
|
||||
- "freedom of religion"
|
||||
- "religious equality"
|
||||
- "anti-muslim bias"
|
||||
- "anti-semitism"
|
||||
- "antisemitism"
|
||||
- "religious bigotry"
|
||||
- "theological"
|
||||
- "scripture"
|
||||
- "comparative religion"
|
||||
- "protect muslims"
|
||||
- "protect jews"
|
||||
- "protect christians"
|
||||
- "muslim rights"
|
||||
- "jewish rights"
|
||||
- "christian rights"
|
||||
- "muslim community"
|
||||
- "jewish community"
|
||||
- "christian community"
|
||||
- "muslim culture"
|
||||
- "jewish culture"
|
||||
- "christian culture"
|
||||
- "islam teaches"
|
||||
- "judaism teaches"
|
||||
- "christianity teaches"
|
||||
- "muslim faith"
|
||||
- "jewish faith"
|
||||
- "christian faith"
|
||||
- "muslim americans"
|
||||
- "jewish americans"
|
||||
- "christian americans"
|
||||
|
|
@ -0,0 +1,251 @@
|
|||
# Sexual orientation bias and discrimination detection
|
||||
category_name: "bias_sexual_orientation"
|
||||
description: "Detects discriminatory language targeting LGBTQ+ individuals and communities"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# HIGH SEVERITY - Explicit slurs (always block regardless of context)
|
||||
- keyword: "faggot"
|
||||
severity: "high"
|
||||
- keyword: "fag"
|
||||
severity: "high"
|
||||
- keyword: "dyke"
|
||||
severity: "high"
|
||||
- keyword: "tranny"
|
||||
severity: "high"
|
||||
- keyword: "shemale"
|
||||
severity: "high"
|
||||
- keyword: "sodomite"
|
||||
severity: "high"
|
||||
- keyword: "homo"
|
||||
severity: "high"
|
||||
- keyword: "f*ggot"
|
||||
severity: "high"
|
||||
- keyword: "f@ggot"
|
||||
severity: "high"
|
||||
- keyword: "gay"
|
||||
severity: "high"
|
||||
|
||||
# HIGH SEVERITY - Always discriminatory phrases
|
||||
- keyword: "gay agenda"
|
||||
severity: "high"
|
||||
- keyword: "homosexual agenda"
|
||||
severity: "high"
|
||||
- keyword: "conversion therapy"
|
||||
severity: "high"
|
||||
- keyword: "reparative therapy"
|
||||
severity: "high"
|
||||
- keyword: "groomer"
|
||||
severity: "high"
|
||||
- keyword: "lifestyle choice"
|
||||
severity: "high"
|
||||
- keyword: "gay lifestyle"
|
||||
severity: "high"
|
||||
- keyword: "homosexual lifestyle"
|
||||
severity: "high"
|
||||
- keyword: "choose to be gay"
|
||||
severity: "high"
|
||||
- keyword: "gay is a choice"
|
||||
severity: "high"
|
||||
- keyword: "homosexuality is a choice"
|
||||
severity: "high"
|
||||
- keyword: "gay is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "gay is wrong"
|
||||
severity: "high"
|
||||
- keyword: "gay is a sin"
|
||||
severity: "high"
|
||||
- keyword: "gay is sinful"
|
||||
severity: "high"
|
||||
- keyword: "gay is a disease"
|
||||
severity: "high"
|
||||
- keyword: "gay is an illness"
|
||||
severity: "high"
|
||||
- keyword: "gay is a disorder"
|
||||
severity: "high"
|
||||
- keyword: "gay is abnormal"
|
||||
severity: "high"
|
||||
- keyword: "gay is deviant"
|
||||
severity: "high"
|
||||
- keyword: "gay is perverted"
|
||||
severity: "high"
|
||||
- keyword: "gay is a perversion"
|
||||
severity: "high"
|
||||
- keyword: "gay is degenerate"
|
||||
severity: "high"
|
||||
- keyword: "gay is an abomination"
|
||||
severity: "high"
|
||||
- keyword: "gay is disgusting"
|
||||
severity: "high"
|
||||
- keyword: "gay is immoral"
|
||||
severity: "high"
|
||||
- keyword: "gay is evil"
|
||||
severity: "high"
|
||||
- keyword: "gay is sick"
|
||||
severity: "high"
|
||||
- keyword: "gay is twisted"
|
||||
severity: "high"
|
||||
- keyword: "lesbian is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "lesbian is wrong"
|
||||
severity: "high"
|
||||
- keyword: "lesbian is a sin"
|
||||
severity: "high"
|
||||
- keyword: "lesbian is abnormal"
|
||||
severity: "high"
|
||||
- keyword: "bisexual is confused"
|
||||
severity: "high"
|
||||
- keyword: "bisexual is fake"
|
||||
severity: "high"
|
||||
- keyword: "transgender is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "transgender is wrong"
|
||||
severity: "high"
|
||||
- keyword: "transgender is a sin"
|
||||
severity: "high"
|
||||
- keyword: "transgender is a disease"
|
||||
severity: "high"
|
||||
- keyword: "transgender is a disorder"
|
||||
severity: "high"
|
||||
- keyword: "transgender is abnormal"
|
||||
severity: "high"
|
||||
- keyword: "transgender is fake"
|
||||
severity: "high"
|
||||
- keyword: "transgender is confused"
|
||||
severity: "high"
|
||||
- keyword: "transgender is a delusion"
|
||||
severity: "high"
|
||||
- keyword: "transgender is mental illness"
|
||||
severity: "high"
|
||||
- keyword: "trans is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "trans is wrong"
|
||||
severity: "high"
|
||||
- keyword: "trans is fake"
|
||||
severity: "high"
|
||||
- keyword: "trans is confused"
|
||||
severity: "high"
|
||||
- keyword: "trans is a delusion"
|
||||
severity: "high"
|
||||
- keyword: "trans is mental"
|
||||
severity: "high"
|
||||
- keyword: "queer is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "queer is wrong"
|
||||
severity: "high"
|
||||
- keyword: "homosexual is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "homosexual is wrong"
|
||||
severity: "high"
|
||||
- keyword: "homosexual is a sin"
|
||||
severity: "high"
|
||||
- keyword: "homosexual is a disease"
|
||||
severity: "high"
|
||||
- keyword: "homosexual is abnormal"
|
||||
severity: "high"
|
||||
- keyword: "lgbtq is unnatural"
|
||||
severity: "high"
|
||||
- keyword: "lgbtq is wrong"
|
||||
severity: "high"
|
||||
- keyword: "lgbtq is a sin"
|
||||
severity: "high"
|
||||
- keyword: "cure gay"
|
||||
severity: "high"
|
||||
- keyword: "cure homosexuality"
|
||||
severity: "high"
|
||||
- keyword: "cure transgender"
|
||||
severity: "high"
|
||||
- keyword: "fix gay"
|
||||
severity: "high"
|
||||
- keyword: "change gay"
|
||||
severity: "high"
|
||||
- keyword: "convert gay"
|
||||
severity: "high"
|
||||
- keyword: "stop being gay"
|
||||
severity: "high"
|
||||
- keyword: "prevent gay"
|
||||
severity: "high"
|
||||
- keyword: "eliminate gay"
|
||||
severity: "high"
|
||||
- keyword: "eradicate gay"
|
||||
severity: "high"
|
||||
- keyword: "ban gay"
|
||||
severity: "high"
|
||||
- keyword: "punish gay"
|
||||
severity: "high"
|
||||
- keyword: "cure lesbian"
|
||||
severity: "high"
|
||||
- keyword: "fix lesbian"
|
||||
severity: "high"
|
||||
- keyword: "change lesbian"
|
||||
severity: "high"
|
||||
- keyword: "cure trans"
|
||||
severity: "high"
|
||||
- keyword: "fix trans"
|
||||
severity: "high"
|
||||
- keyword: "change trans"
|
||||
severity: "high"
|
||||
- keyword: "cure lgbtq"
|
||||
severity: "high"
|
||||
- keyword: "fix lgbtq"
|
||||
severity: "high"
|
||||
- keyword: "all gays are"
|
||||
severity: "high"
|
||||
- keyword: "all lesbians are"
|
||||
severity: "high"
|
||||
- keyword: "all transgender are"
|
||||
severity: "high"
|
||||
- keyword: "all trans are"
|
||||
severity: "high"
|
||||
- keyword: "all lgbtq are"
|
||||
severity: "high"
|
||||
- keyword: "gays are pedophiles"
|
||||
severity: "high"
|
||||
- keyword: "gays are groomers"
|
||||
severity: "high"
|
||||
- keyword: "gays are perverts"
|
||||
severity: "high"
|
||||
- keyword: "gays are sinners"
|
||||
severity: "high"
|
||||
- keyword: "lesbians are perverts"
|
||||
severity: "high"
|
||||
- keyword: "transgender are predators"
|
||||
severity: "high"
|
||||
- keyword: "trans are predators"
|
||||
severity: "high"
|
||||
- keyword: "trans are groomers"
|
||||
severity: "high"
|
||||
- keyword: "trans women are men"
|
||||
severity: "high"
|
||||
- keyword: "trans men are women"
|
||||
severity: "high"
|
||||
- keyword: "not real women"
|
||||
severity: "medium"
|
||||
- keyword: "not real men"
|
||||
severity: "medium"
|
||||
|
||||
# Exceptions - legitimate discussions
|
||||
exceptions:
|
||||
- "lgbtq rights"
|
||||
- "lgbtq equality"
|
||||
- "lgbtq community"
|
||||
- "lgbtq support"
|
||||
- "lgbtq advocacy"
|
||||
- "sexual orientation"
|
||||
- "coming out"
|
||||
- "pride"
|
||||
- "marriage equality"
|
||||
- "anti-lgbtq"
|
||||
- "discrimination against"
|
||||
- "protect lgbtq"
|
||||
- "lgbtq history"
|
||||
- "gender identity"
|
||||
- "conversion therapy ban"
|
||||
- "harmful practice"
|
||||
- "fight discrimination"
|
||||
- "lgbtq acceptance"
|
||||
- "lgbtq inclusion"
|
||||
- "support lgbtq"
|
||||
- "lgbtq youth"
|
||||
- "lgbtq healthcare"
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
# Financial advice and investment guidance detection
|
||||
category_name: "denied_financial_advice"
|
||||
description: "Detects requests for personalized financial advice, investment recommendations, or financial planning that should be provided by licensed financial advisors"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - core financial terms
|
||||
- keyword: "invest"
|
||||
severity: "high"
|
||||
- keyword: "investing"
|
||||
severity: "high"
|
||||
- keyword: "investment"
|
||||
severity: "high"
|
||||
- keyword: "investments"
|
||||
severity: "high"
|
||||
- keyword: "stock"
|
||||
severity: "high"
|
||||
- keyword: "stocks"
|
||||
severity: "high"
|
||||
- keyword: "portfolio"
|
||||
severity: "high"
|
||||
- keyword: "crypto"
|
||||
severity: "high"
|
||||
- keyword: "cryptocurrency"
|
||||
severity: "high"
|
||||
- keyword: "bitcoin"
|
||||
severity: "high"
|
||||
- keyword: "ethereum"
|
||||
severity: "high"
|
||||
- keyword: "trading"
|
||||
severity: "high"
|
||||
- keyword: "trade"
|
||||
severity: "high"
|
||||
- keyword: "trader"
|
||||
severity: "high"
|
||||
- keyword: "retirement"
|
||||
severity: "high"
|
||||
- keyword: "401k"
|
||||
severity: "high"
|
||||
- keyword: "ira"
|
||||
severity: "high"
|
||||
- keyword: "roth"
|
||||
severity: "high"
|
||||
- keyword: "mortgage"
|
||||
severity: "high"
|
||||
- keyword: "refinance"
|
||||
severity: "high"
|
||||
- keyword: "loan"
|
||||
severity: "high"
|
||||
- keyword: "loans"
|
||||
severity: "high"
|
||||
- keyword: "debt"
|
||||
severity: "high"
|
||||
- keyword: "tax"
|
||||
severity: "high"
|
||||
- keyword: "taxes"
|
||||
severity: "high"
|
||||
- keyword: "etf"
|
||||
severity: "high"
|
||||
- keyword: "bond"
|
||||
severity: "high"
|
||||
- keyword: "bonds"
|
||||
severity: "high"
|
||||
- keyword: "mutual"
|
||||
severity: "high"
|
||||
- keyword: "forex"
|
||||
severity: "high"
|
||||
- keyword: "futures"
|
||||
severity: "high"
|
||||
- keyword: "diversify"
|
||||
severity: "high"
|
||||
- keyword: "diversification"
|
||||
severity: "high"
|
||||
|
||||
# Exceptions - legitimate financial discussions
|
||||
exceptions:
|
||||
- "consult a financial advisor"
|
||||
- "consult your financial advisor"
|
||||
- "speak with financial advisor"
|
||||
- "hire financial advisor"
|
||||
- "seek financial advice"
|
||||
- "financial professional"
|
||||
- "licensed financial advisor"
|
||||
- "certified financial planner"
|
||||
- "financial consultant"
|
||||
- "investment professional"
|
||||
- "tax professional"
|
||||
- "certified public accountant"
|
||||
- "speak to a professional"
|
||||
- "talk to a professional"
|
||||
- "cpa"
|
||||
- "tax preparer"
|
||||
- "financial education"
|
||||
- "financial literacy"
|
||||
- "personal finance education"
|
||||
- "investment education"
|
||||
- "general financial information"
|
||||
- "general information"
|
||||
- "educational purposes"
|
||||
- "for educational purposes"
|
||||
- "not financial advice"
|
||||
- "not investment advice"
|
||||
- "this is not financial advice"
|
||||
- "this is not investment advice"
|
||||
- "not a substitute for"
|
||||
- "financial disclaimer"
|
||||
- "investment disclaimer"
|
||||
- "financial research"
|
||||
- "market research"
|
||||
- "economic research"
|
||||
- "financial analysis"
|
||||
- "market analysis"
|
||||
- "financial news"
|
||||
- "market news"
|
||||
- "economic news"
|
||||
- "financial history"
|
||||
- "investment history"
|
||||
- "market trends"
|
||||
- "economic trends"
|
||||
- "financial concepts"
|
||||
- "investment concepts"
|
||||
- "financial terminology"
|
||||
- "investment terminology"
|
||||
- "stock market basics"
|
||||
- "investment basics"
|
||||
- "finance 101"
|
||||
- "budgeting basics"
|
||||
- "saving tips"
|
||||
- "general tips"
|
||||
- "debt reduction strategies"
|
||||
- "credit score information"
|
||||
- "how does"
|
||||
- "what is"
|
||||
- "what are"
|
||||
- "explain"
|
||||
- "definition of"
|
||||
- "means"
|
||||
|
||||
|
|
@ -0,0 +1,137 @@
|
|||
# Legal advice and representation detection
|
||||
category_name: "denied_legal_advice"
|
||||
description: "Detects requests for legal advice, representation, or legal strategy that should be provided by licensed attorneys"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - core legal terms
|
||||
- keyword: "lawyer"
|
||||
severity: "high"
|
||||
- keyword: "attorney"
|
||||
severity: "high"
|
||||
- keyword: "lawsuit"
|
||||
severity: "high"
|
||||
- keyword: "sue"
|
||||
severity: "high"
|
||||
- keyword: "suing"
|
||||
severity: "high"
|
||||
- keyword: "court"
|
||||
severity: "high"
|
||||
- keyword: "trial"
|
||||
severity: "high"
|
||||
- keyword: "case"
|
||||
severity: "high"
|
||||
- keyword: "contract"
|
||||
severity: "high"
|
||||
- keyword: "litigation"
|
||||
severity: "high"
|
||||
- keyword: "plead"
|
||||
severity: "high"
|
||||
- keyword: "guilty"
|
||||
severity: "high"
|
||||
- keyword: "divorce"
|
||||
severity: "high"
|
||||
- keyword: "custody"
|
||||
severity: "high"
|
||||
- keyword: "immigration"
|
||||
severity: "high"
|
||||
- keyword: "visa"
|
||||
severity: "high"
|
||||
- keyword: "asylum"
|
||||
severity: "high"
|
||||
- keyword: "deportation"
|
||||
severity: "high"
|
||||
- keyword: "criminal"
|
||||
severity: "high"
|
||||
- keyword: "charges"
|
||||
severity: "high"
|
||||
- keyword: "arrest"
|
||||
severity: "high"
|
||||
- keyword: "warrant"
|
||||
severity: "high"
|
||||
- keyword: "sentence"
|
||||
severity: "high"
|
||||
- keyword: "prosecution"
|
||||
severity: "high"
|
||||
- keyword: "bankruptcy"
|
||||
severity: "high"
|
||||
- keyword: "patent"
|
||||
severity: "high"
|
||||
- keyword: "trademark"
|
||||
severity: "high"
|
||||
- keyword: "copyright"
|
||||
severity: "high"
|
||||
- keyword: "settlement"
|
||||
severity: "high"
|
||||
- keyword: "defendant"
|
||||
severity: "high"
|
||||
- keyword: "plaintiff"
|
||||
severity: "high"
|
||||
- keyword: "testimony"
|
||||
severity: "high"
|
||||
|
||||
# Exceptions - legitimate legal discussions
|
||||
exceptions:
|
||||
- "consult a lawyer"
|
||||
- "consult an attorney"
|
||||
- "consult your lawyer"
|
||||
- "consult your attorney"
|
||||
- "hire a lawyer"
|
||||
- "hire an attorney"
|
||||
- "find a lawyer"
|
||||
- "find an attorney"
|
||||
- "seek legal counsel"
|
||||
- "seek legal advice"
|
||||
- "get legal advice"
|
||||
- "legal professional"
|
||||
- "qualified attorney"
|
||||
- "licensed lawyer"
|
||||
- "licensed attorney"
|
||||
- "legal representation"
|
||||
- "retain counsel"
|
||||
- "contact a lawyer"
|
||||
- "contact an attorney"
|
||||
- "speak with attorney"
|
||||
- "speak with lawyer"
|
||||
- "talk to a lawyer"
|
||||
- "talk to an attorney"
|
||||
- "legal consultation"
|
||||
- "attorney consultation"
|
||||
- "legal education"
|
||||
- "legal studies"
|
||||
- "law school"
|
||||
- "legal research"
|
||||
- "legal terminology"
|
||||
- "legal terms"
|
||||
- "legal system"
|
||||
- "court system"
|
||||
- "legal process"
|
||||
- "legal procedure"
|
||||
- "general legal information"
|
||||
- "general information"
|
||||
- "educational purposes"
|
||||
- "for educational purposes"
|
||||
- "not legal advice"
|
||||
- "this is not legal advice"
|
||||
- "not a substitute for"
|
||||
- "legal disclaimer"
|
||||
- "legal history"
|
||||
- "legal precedent"
|
||||
- "case law"
|
||||
- "supreme court"
|
||||
- "constitutional law"
|
||||
- "legal rights awareness"
|
||||
- "know your rights"
|
||||
- "civil rights"
|
||||
- "human rights"
|
||||
- "legal framework"
|
||||
- "how does"
|
||||
- "what is"
|
||||
- "what are"
|
||||
- "explain"
|
||||
- "definition of"
|
||||
- "means"
|
||||
- "criminal justice system"
|
||||
- "immigration system"
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
# Medical advice and diagnosis detection
|
||||
category_name: "denied_medical_advice"
|
||||
description: "Detects requests for medical advice, diagnosis, or treatment recommendations that should be provided by licensed healthcare professionals"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - core medical terms
|
||||
- keyword: "diagnose"
|
||||
severity: "high"
|
||||
- keyword: "diagnosis"
|
||||
severity: "high"
|
||||
- keyword: "doctor"
|
||||
severity: "high"
|
||||
- keyword: "physician"
|
||||
severity: "high"
|
||||
- keyword: "medication"
|
||||
severity: "high"
|
||||
- keyword: "medicine"
|
||||
severity: "high"
|
||||
- keyword: "prescription"
|
||||
severity: "high"
|
||||
- keyword: "prescribe"
|
||||
severity: "high"
|
||||
- keyword: "drug"
|
||||
severity: "high"
|
||||
- keyword: "drugs"
|
||||
severity: "high"
|
||||
- keyword: "treatment"
|
||||
severity: "high"
|
||||
- keyword: "treat"
|
||||
severity: "high"
|
||||
- keyword: "cure"
|
||||
severity: "high"
|
||||
- keyword: "surgery"
|
||||
severity: "high"
|
||||
- keyword: "symptoms"
|
||||
severity: "high"
|
||||
- keyword: "symptom"
|
||||
severity: "high"
|
||||
- keyword: "disease"
|
||||
severity: "high"
|
||||
- keyword: "illness"
|
||||
severity: "high"
|
||||
- keyword: "condition"
|
||||
severity: "high"
|
||||
- keyword: "cancer"
|
||||
severity: "high"
|
||||
- keyword: "diabetes"
|
||||
severity: "high"
|
||||
- keyword: "depression"
|
||||
severity: "high"
|
||||
- keyword: "anxiety"
|
||||
severity: "high"
|
||||
- keyword: "adhd"
|
||||
severity: "high"
|
||||
- keyword: "bipolar"
|
||||
severity: "high"
|
||||
- keyword: "psychiatric"
|
||||
severity: "high"
|
||||
- keyword: "vaccine"
|
||||
severity: "high"
|
||||
- keyword: "vaccination"
|
||||
severity: "high"
|
||||
- keyword: "dosage"
|
||||
severity: "high"
|
||||
- keyword: "dose"
|
||||
severity: "high"
|
||||
- keyword: "injury"
|
||||
severity: "high"
|
||||
- keyword: "treatment"
|
||||
severity: "high"
|
||||
- keyword: "injection"
|
||||
severity: "high"
|
||||
|
||||
# Exceptions - legitimate medical discussions
|
||||
exceptions:
|
||||
- "medical history"
|
||||
- "medical research"
|
||||
- "medical studies"
|
||||
- "according to research"
|
||||
- "research shows"
|
||||
- "studies show"
|
||||
- "consult a doctor"
|
||||
- "consult your doctor"
|
||||
- "see a doctor"
|
||||
- "see your doctor"
|
||||
- "talk to your doctor"
|
||||
- "speak with your doctor"
|
||||
- "ask your doctor"
|
||||
- "visit your doctor"
|
||||
- "contact your doctor"
|
||||
- "call your doctor"
|
||||
- "medical professional"
|
||||
- "healthcare provider"
|
||||
- "healthcare professional"
|
||||
- "seek medical attention"
|
||||
- "seek medical advice"
|
||||
- "emergency medical"
|
||||
- "call emergency"
|
||||
- "call 911"
|
||||
- "go to hospital"
|
||||
- "visit hospital"
|
||||
- "go to the emergency"
|
||||
- "medical emergency"
|
||||
- "general information"
|
||||
- "general health information"
|
||||
- "educational purposes"
|
||||
- "for educational purposes"
|
||||
- "not medical advice"
|
||||
- "this is not medical advice"
|
||||
- "not a substitute for"
|
||||
- "medical documentation"
|
||||
- "medical terminology"
|
||||
- "medical terms"
|
||||
- "anatomy and physiology"
|
||||
- "medical science"
|
||||
- "public health"
|
||||
- "health education"
|
||||
- "wellness information"
|
||||
- "preventive care"
|
||||
- "healthy lifestyle"
|
||||
- "medical disclaimer"
|
||||
- "how does"
|
||||
- "what is"
|
||||
- "what are"
|
||||
- "explain"
|
||||
- "definition of"
|
||||
- "means"
|
||||
- "cancer research"
|
||||
- "diabetes research"
|
||||
- "mental health awareness"
|
||||
|
||||
|
|
@ -0,0 +1,299 @@
|
|||
# Illegal weapons and explosives content detection
|
||||
category_name: "harmful_illegal_weapons"
|
||||
description: "Detects content related to illegal weapons, explosives, and dangerous materials"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - explicit illegal weapons/explosives instructions
|
||||
- keyword: "make a bomb"
|
||||
severity: "high"
|
||||
- keyword: "build a bomb"
|
||||
severity: "high"
|
||||
- keyword: "create explosive"
|
||||
severity: "high"
|
||||
- keyword: "homemade bomb"
|
||||
severity: "high"
|
||||
- keyword: "pipe bomb"
|
||||
severity: "high"
|
||||
- keyword: "pressure cooker bomb"
|
||||
severity: "high"
|
||||
- keyword: "fertilizer bomb"
|
||||
severity: "high"
|
||||
- keyword: "car bomb"
|
||||
severity: "high"
|
||||
- keyword: "suicide bomb"
|
||||
severity: "high"
|
||||
- keyword: "explosive device"
|
||||
severity: "high"
|
||||
- keyword: "bomb instructions"
|
||||
severity: "high"
|
||||
- keyword: "bomb recipe"
|
||||
severity: "high"
|
||||
- keyword: "bomb tutorial"
|
||||
severity: "high"
|
||||
- keyword: "homemade explosive"
|
||||
severity: "high"
|
||||
- keyword: "illegal firearm"
|
||||
severity: "high"
|
||||
- keyword: "unregistered gun"
|
||||
severity: "high"
|
||||
- keyword: "make poison"
|
||||
severity: "high"
|
||||
- keyword: "create poison"
|
||||
severity: "high"
|
||||
- keyword: "homemade poison"
|
||||
severity: "high"
|
||||
- keyword: "lethal poison"
|
||||
severity: "high"
|
||||
- keyword: "chemical weapon"
|
||||
severity: "high"
|
||||
- keyword: "biological weapon"
|
||||
severity: "high"
|
||||
- keyword: "ricin"
|
||||
severity: "high"
|
||||
- keyword: "cyanide poison"
|
||||
severity: "high"
|
||||
- keyword: "how to make a bomb"
|
||||
severity: "high"
|
||||
- keyword: "how to build a bomb"
|
||||
severity: "high"
|
||||
- keyword: "how to create explosive"
|
||||
severity: "high"
|
||||
- keyword: "bomb making instructions"
|
||||
severity: "high"
|
||||
- keyword: "bomb making tutorial"
|
||||
severity: "high"
|
||||
- keyword: "homemade explosive recipe"
|
||||
severity: "high"
|
||||
- keyword: "how to make poison"
|
||||
severity: "high"
|
||||
- keyword: "how to create poison"
|
||||
severity: "high"
|
||||
- keyword: "undetectable poison"
|
||||
severity: "high"
|
||||
- keyword: "untraceable weapon"
|
||||
severity: "high"
|
||||
- keyword: "3d print gun"
|
||||
severity: "high"
|
||||
- keyword: "3d printed gun"
|
||||
severity: "high"
|
||||
- keyword: "3d printing gun"
|
||||
severity: "high"
|
||||
- keyword: "make bomb"
|
||||
severity: "high"
|
||||
- keyword: "making bomb"
|
||||
severity: "high"
|
||||
- keyword: "build explosive"
|
||||
severity: "high"
|
||||
- keyword: "building explosive"
|
||||
severity: "high"
|
||||
- keyword: "create weapon"
|
||||
severity: "high"
|
||||
- keyword: "creating weapon"
|
||||
severity: "high"
|
||||
- keyword: "manufacture explosive"
|
||||
severity: "high"
|
||||
- keyword: "manufacturing explosive"
|
||||
severity: "high"
|
||||
- keyword: "manufactured explosive"
|
||||
severity: "high"
|
||||
- keyword: "construct bomb"
|
||||
severity: "high"
|
||||
- keyword: "constructing bomb"
|
||||
severity: "high"
|
||||
- keyword: "constructed bomb"
|
||||
severity: "high"
|
||||
- keyword: "fertilizer explosive"
|
||||
severity: "high"
|
||||
- keyword: "pressure cooker explosive"
|
||||
severity: "high"
|
||||
- keyword: "pipe explosive"
|
||||
severity: "high"
|
||||
- keyword: "homemade weapon"
|
||||
severity: "high"
|
||||
- keyword: "ingredients bomb"
|
||||
severity: "high"
|
||||
- keyword: "bomb ingredients"
|
||||
severity: "high"
|
||||
- keyword: "instructions explosive"
|
||||
severity: "high"
|
||||
- keyword: "explosive instructions"
|
||||
severity: "high"
|
||||
- keyword: "improvised explosive device"
|
||||
severity: "high"
|
||||
- keyword: "ied instructions"
|
||||
severity: "high"
|
||||
- keyword: "make ied"
|
||||
severity: "high"
|
||||
- keyword: "build ied"
|
||||
severity: "high"
|
||||
- keyword: "nail bomb"
|
||||
severity: "high"
|
||||
- keyword: "shrapnel bomb"
|
||||
severity: "high"
|
||||
- keyword: "dirty bomb"
|
||||
severity: "high"
|
||||
- keyword: "radiological weapon"
|
||||
severity: "high"
|
||||
- keyword: "nuclear weapon"
|
||||
severity: "high"
|
||||
- keyword: "molotov cocktail"
|
||||
severity: "high"
|
||||
- keyword: "incendiary device"
|
||||
severity: "high"
|
||||
- keyword: "fire bomb"
|
||||
severity: "high"
|
||||
- keyword: "napalm recipe"
|
||||
severity: "high"
|
||||
- keyword: "thermite recipe"
|
||||
severity: "high"
|
||||
- keyword: "ammonium nitrate bomb"
|
||||
severity: "high"
|
||||
- keyword: "anfo explosive"
|
||||
severity: "high"
|
||||
- keyword: "c4 explosive"
|
||||
severity: "high"
|
||||
- keyword: "semtex explosive"
|
||||
severity: "high"
|
||||
- keyword: "tnt recipe"
|
||||
severity: "high"
|
||||
- keyword: "dynamite recipe"
|
||||
severity: "high"
|
||||
- keyword: "plastic explosive"
|
||||
severity: "high"
|
||||
- keyword: "det cord"
|
||||
severity: "high"
|
||||
- keyword: "detonator instructions"
|
||||
severity: "high"
|
||||
- keyword: "remote detonator"
|
||||
severity: "high"
|
||||
- keyword: "timer bomb"
|
||||
severity: "high"
|
||||
- keyword: "time bomb"
|
||||
severity: "high"
|
||||
- keyword: "letter bomb"
|
||||
severity: "high"
|
||||
- keyword: "package bomb"
|
||||
severity: "high"
|
||||
- keyword: "mail bomb"
|
||||
severity: "high"
|
||||
- keyword: "anthrax weapon"
|
||||
severity: "high"
|
||||
- keyword: "bioweapon"
|
||||
severity: "high"
|
||||
- keyword: "weaponized anthrax"
|
||||
severity: "high"
|
||||
- keyword: "weaponized smallpox"
|
||||
severity: "high"
|
||||
- keyword: "nerve agent"
|
||||
severity: "high"
|
||||
- keyword: "sarin gas"
|
||||
severity: "high"
|
||||
- keyword: "vx gas"
|
||||
severity: "high"
|
||||
- keyword: "mustard gas"
|
||||
severity: "high"
|
||||
- keyword: "chlorine gas weapon"
|
||||
severity: "high"
|
||||
- keyword: "poison gas"
|
||||
severity: "high"
|
||||
- keyword: "ricin poison"
|
||||
severity: "high"
|
||||
- keyword: "ricin extraction"
|
||||
severity: "high"
|
||||
- keyword: "ricin recipe"
|
||||
severity: "high"
|
||||
- keyword: "botulinum toxin weapon"
|
||||
severity: "high"
|
||||
- keyword: "arsenic poison"
|
||||
severity: "high"
|
||||
- keyword: "thallium poison"
|
||||
severity: "high"
|
||||
- keyword: "polonium poison"
|
||||
severity: "high"
|
||||
- keyword: "undetectable poison recipe"
|
||||
severity: "high"
|
||||
- keyword: "untraceable poison"
|
||||
severity: "high"
|
||||
- keyword: "silent weapon"
|
||||
severity: "high"
|
||||
- keyword: "suppressor gun"
|
||||
severity: "high"
|
||||
- keyword: "silencer gun"
|
||||
severity: "high"
|
||||
- keyword: "homemade silencer"
|
||||
severity: "high"
|
||||
- keyword: "homemade suppressor"
|
||||
severity: "high"
|
||||
- keyword: "ghost gun"
|
||||
severity: "high"
|
||||
- keyword: "unregistered weapon"
|
||||
severity: "high"
|
||||
- keyword: "untraceable gun"
|
||||
severity: "high"
|
||||
- keyword: "illegal gun"
|
||||
severity: "high"
|
||||
- keyword: "black market gun"
|
||||
severity: "high"
|
||||
- keyword: "black market weapons"
|
||||
severity: "high"
|
||||
- keyword: "buy illegal gun"
|
||||
severity: "high"
|
||||
- keyword: "buy illegal weapon"
|
||||
severity: "high"
|
||||
- keyword: "automatic weapon conversion"
|
||||
severity: "high"
|
||||
- keyword: "full auto conversion"
|
||||
severity: "high"
|
||||
- keyword: "machine gun conversion"
|
||||
severity: "high"
|
||||
- keyword: "bump stock"
|
||||
severity: "medium"
|
||||
- keyword: "binary trigger"
|
||||
severity: "medium"
|
||||
- keyword: "explosive ammunition"
|
||||
severity: "high"
|
||||
- keyword: "armor piercing rounds"
|
||||
severity: "high"
|
||||
- keyword: "cop killer bullets"
|
||||
severity: "high"
|
||||
- keyword: "illegal ammunition"
|
||||
severity: "high"
|
||||
|
||||
# Medium severity - context-dependent
|
||||
- keyword: "bomb"
|
||||
severity: "medium"
|
||||
- keyword: "explosive"
|
||||
severity: "medium"
|
||||
- keyword: "grenade"
|
||||
severity: "medium"
|
||||
- keyword: "landmine"
|
||||
severity: "medium"
|
||||
- keyword: "ammunition"
|
||||
severity: "medium"
|
||||
- keyword: "firearm"
|
||||
severity: "medium"
|
||||
- keyword: "gun"
|
||||
severity: "medium"
|
||||
- keyword: "rifle"
|
||||
severity: "medium"
|
||||
- keyword: "poison"
|
||||
severity: "medium"
|
||||
|
||||
# Exception patterns (legitimate uses that should NOT be blocked)
|
||||
exceptions:
|
||||
- "bomb disposal"
|
||||
- "bomb squad"
|
||||
- "explosive ordnance disposal"
|
||||
- "eod"
|
||||
- "historical"
|
||||
- "educational"
|
||||
- "safety training"
|
||||
- "weapons history"
|
||||
- "weapons education"
|
||||
- "explosive safety"
|
||||
- "bomb detection"
|
||||
- "bomb threat"
|
||||
- "report bomb"
|
||||
- "reporting bomb"
|
||||
|
|
@ -0,0 +1,184 @@
|
|||
# Self-harm and suicide-related content detection
|
||||
category_name: "harmful_self_harm"
|
||||
description: "Detects content related to self-harm, suicide, and eating disorders"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - explicit self-harm intent
|
||||
- keyword: "kill myself"
|
||||
severity: "high"
|
||||
- keyword: "end my life"
|
||||
severity: "high"
|
||||
- keyword: "commit suicide"
|
||||
severity: "high"
|
||||
- keyword: "want to die"
|
||||
severity: "high"
|
||||
- keyword: "how to die"
|
||||
severity: "high"
|
||||
- keyword: "suicide method"
|
||||
severity: "high"
|
||||
- keyword: "painless death"
|
||||
severity: "high"
|
||||
- keyword: "overdose on"
|
||||
severity: "high"
|
||||
- keyword: "hang myself"
|
||||
severity: "high"
|
||||
- keyword: "cut myself"
|
||||
severity: "high"
|
||||
- keyword: "self harm"
|
||||
severity: "high"
|
||||
- keyword: "hurt myself"
|
||||
severity: "high"
|
||||
- keyword: "how to commit suicide"
|
||||
severity: "high"
|
||||
- keyword: "ways to kill myself"
|
||||
severity: "high"
|
||||
- keyword: "easiest way to die"
|
||||
severity: "high"
|
||||
- keyword: "suicide instructions"
|
||||
severity: "high"
|
||||
- keyword: "how to end my life"
|
||||
severity: "high"
|
||||
- keyword: "plan suicide"
|
||||
severity: "high"
|
||||
- keyword: "planning suicide"
|
||||
severity: "high"
|
||||
- keyword: "planned suicide"
|
||||
severity: "high"
|
||||
- keyword: "attempt suicide"
|
||||
severity: "high"
|
||||
- keyword: "attempting suicide"
|
||||
severity: "high"
|
||||
- keyword: "attempted suicide"
|
||||
severity: "high"
|
||||
- keyword: "self harm method"
|
||||
severity: "high"
|
||||
- keyword: "self harm methods"
|
||||
severity: "high"
|
||||
- keyword: "suicide note"
|
||||
severity: "high"
|
||||
- keyword: "end life"
|
||||
severity: "high"
|
||||
- keyword: "ending my life"
|
||||
severity: "high"
|
||||
- keyword: "take my life"
|
||||
severity: "high"
|
||||
- keyword: "taking my life"
|
||||
severity: "high"
|
||||
- keyword: "slash my wrists"
|
||||
severity: "high"
|
||||
- keyword: "slit my wrists"
|
||||
severity: "high"
|
||||
- keyword: "jump off bridge"
|
||||
severity: "high"
|
||||
- keyword: "jump off building"
|
||||
severity: "high"
|
||||
- keyword: "fatal overdose"
|
||||
severity: "high"
|
||||
- keyword: "lethal dose"
|
||||
severity: "high"
|
||||
- keyword: "suicide plan"
|
||||
severity: "high"
|
||||
- keyword: "suicide pact"
|
||||
severity: "high"
|
||||
- keyword: "want to end it"
|
||||
severity: "high"
|
||||
- keyword: "can't go on"
|
||||
severity: "high"
|
||||
- keyword: "better off dead"
|
||||
severity: "high"
|
||||
- keyword: "wish i was dead"
|
||||
severity: "high"
|
||||
- keyword: "no reason to live"
|
||||
severity: "high"
|
||||
- keyword: "life not worth living"
|
||||
severity: "high"
|
||||
- keyword: "burn myself"
|
||||
severity: "high"
|
||||
- keyword: "poison myself"
|
||||
severity: "high"
|
||||
- keyword: "starve myself"
|
||||
severity: "high"
|
||||
- keyword: "suffocate myself"
|
||||
severity: "high"
|
||||
- keyword: "drown myself"
|
||||
severity: "high"
|
||||
- keyword: "shoot myself"
|
||||
severity: "high"
|
||||
- keyword: "self-mutilation"
|
||||
severity: "high"
|
||||
- keyword: "self mutilation"
|
||||
severity: "high"
|
||||
- keyword: "cutting techniques"
|
||||
severity: "high"
|
||||
- keyword: "self-injury"
|
||||
severity: "high"
|
||||
- keyword: "self injury"
|
||||
severity: "high"
|
||||
- keyword: "how deep to cut"
|
||||
severity: "high"
|
||||
- keyword: "where to cut"
|
||||
severity: "high"
|
||||
- keyword: "painless suicide"
|
||||
severity: "high"
|
||||
- keyword: "quick suicide"
|
||||
severity: "high"
|
||||
- keyword: "easy suicide"
|
||||
severity: "high"
|
||||
- keyword: "suicide methods"
|
||||
severity: "high"
|
||||
|
||||
# Medium severity - may have legitimate context (news, education, prevention)
|
||||
- keyword: "suicide"
|
||||
severity: "medium"
|
||||
- keyword: "self-harm"
|
||||
severity: "medium"
|
||||
- keyword: "cutting"
|
||||
severity: "medium"
|
||||
- keyword: "anorexia"
|
||||
severity: "medium"
|
||||
- keyword: "bulimia"
|
||||
severity: "medium"
|
||||
- keyword: "suicidal"
|
||||
severity: "medium"
|
||||
- keyword: "suicidal thoughts"
|
||||
severity: "medium"
|
||||
- keyword: "self-destructive"
|
||||
severity: "medium"
|
||||
- keyword: "self destructive"
|
||||
severity: "medium"
|
||||
- keyword: "eating disorder"
|
||||
severity: "medium"
|
||||
- keyword: "purging"
|
||||
severity: "medium"
|
||||
- keyword: "binge eating"
|
||||
severity: "medium"
|
||||
- keyword: "binge and purge"
|
||||
severity: "medium"
|
||||
- keyword: "pro-ana"
|
||||
severity: "medium"
|
||||
- keyword: "pro ana"
|
||||
severity: "medium"
|
||||
- keyword: "thinspo"
|
||||
severity: "medium"
|
||||
- keyword: "thinspiration"
|
||||
severity: "medium"
|
||||
|
||||
# Exception patterns (legitimate uses that should NOT be blocked)
|
||||
exceptions:
|
||||
- "suicide prevention"
|
||||
- "suicide hotline"
|
||||
- "prevent suicide"
|
||||
- "suicide awareness"
|
||||
- "suicide crisis"
|
||||
- "suicide support"
|
||||
- "mental health"
|
||||
- "help with suicide"
|
||||
- "suicide prevention resources"
|
||||
- "talk someone out of suicide"
|
||||
- "preventing self-harm"
|
||||
- "self-harm prevention"
|
||||
- "eating disorder treatment"
|
||||
- "eating disorder recovery"
|
||||
- "eating disorder support"
|
||||
|
|
@ -0,0 +1,265 @@
|
|||
# Violence and criminal planning content detection
|
||||
category_name: "harmful_violence"
|
||||
description: "Detects content related to violence, criminal planning, and attacks"
|
||||
default_action: "BLOCK"
|
||||
|
||||
# Keywords with severity levels
|
||||
keywords:
|
||||
# High severity - explicit violence/criminal intent
|
||||
- keyword: "how to kill someone"
|
||||
severity: "high"
|
||||
- keyword: "plan attack"
|
||||
severity: "high"
|
||||
- keyword: "commit murder"
|
||||
severity: "high"
|
||||
- keyword: "school shooting"
|
||||
severity: "high"
|
||||
- keyword: "mass shooting"
|
||||
severity: "high"
|
||||
- keyword: "serial killer"
|
||||
severity: "high"
|
||||
- keyword: "hired hitman"
|
||||
severity: "high"
|
||||
- keyword: "assassinate"
|
||||
severity: "high"
|
||||
- keyword: "terrorism"
|
||||
severity: "high"
|
||||
- keyword: "terrorist attack"
|
||||
severity: "high"
|
||||
- keyword: "kidnapping plan"
|
||||
severity: "high"
|
||||
- keyword: "human trafficking"
|
||||
severity: "high"
|
||||
- keyword: "rape"
|
||||
severity: "high"
|
||||
- keyword: "sexual assault"
|
||||
severity: "high"
|
||||
- keyword: "torture method"
|
||||
severity: "high"
|
||||
- keyword: "ways to murder"
|
||||
severity: "high"
|
||||
- keyword: "plan a shooting"
|
||||
severity: "high"
|
||||
- keyword: "plan an attack"
|
||||
severity: "high"
|
||||
- keyword: "kill my"
|
||||
severity: "high"
|
||||
- keyword: "murder someone"
|
||||
severity: "high"
|
||||
- keyword: "hire a hitman"
|
||||
severity: "high"
|
||||
- keyword: "get away with murder"
|
||||
severity: "high"
|
||||
- keyword: "plan attack"
|
||||
severity: "high"
|
||||
- keyword: "planning attack"
|
||||
severity: "high"
|
||||
- keyword: "planned attack"
|
||||
severity: "high"
|
||||
- keyword: "commit crime"
|
||||
severity: "high"
|
||||
- keyword: "committing crime"
|
||||
severity: "high"
|
||||
- keyword: "organize shooting"
|
||||
severity: "high"
|
||||
- keyword: "organizing shooting"
|
||||
severity: "high"
|
||||
- keyword: "organized shooting"
|
||||
severity: "high"
|
||||
- keyword: "coordinate attack"
|
||||
severity: "high"
|
||||
- keyword: "coordinating attack"
|
||||
severity: "high"
|
||||
- keyword: "coordinated attack"
|
||||
severity: "high"
|
||||
- keyword: "gun school"
|
||||
severity: "high"
|
||||
- keyword: "shooting school"
|
||||
severity: "high"
|
||||
- keyword: "weapon attack"
|
||||
severity: "high"
|
||||
- keyword: "kill wife"
|
||||
severity: "high"
|
||||
- keyword: "kill husband"
|
||||
severity: "high"
|
||||
- keyword: "murder plan"
|
||||
severity: "high"
|
||||
- keyword: "murder planning"
|
||||
severity: "high"
|
||||
- keyword: "how to get away with murder"
|
||||
severity: "high"
|
||||
- keyword: "how to commit murder"
|
||||
severity: "high"
|
||||
- keyword: "how to kill"
|
||||
severity: "high"
|
||||
- keyword: "ways to kill"
|
||||
severity: "high"
|
||||
- keyword: "best way to kill"
|
||||
severity: "high"
|
||||
- keyword: "easiest way to kill"
|
||||
severity: "high"
|
||||
- keyword: "hire hitman"
|
||||
severity: "high"
|
||||
- keyword: "find hitman"
|
||||
severity: "high"
|
||||
- keyword: "hire assassin"
|
||||
severity: "high"
|
||||
- keyword: "find assassin"
|
||||
severity: "high"
|
||||
- keyword: "contract killer"
|
||||
severity: "high"
|
||||
- keyword: "how to assassinate"
|
||||
severity: "high"
|
||||
- keyword: "assassination plan"
|
||||
severity: "high"
|
||||
- keyword: "plan terrorist attack"
|
||||
severity: "high"
|
||||
- keyword: "organize terrorist attack"
|
||||
severity: "high"
|
||||
- keyword: "commit terrorist attack"
|
||||
severity: "high"
|
||||
- keyword: "plan mass shooting"
|
||||
severity: "high"
|
||||
- keyword: "organize mass shooting"
|
||||
severity: "high"
|
||||
- keyword: "school shooter"
|
||||
severity: "high"
|
||||
- keyword: "mass shooter"
|
||||
severity: "high"
|
||||
- keyword: "how to kidnap"
|
||||
severity: "high"
|
||||
- keyword: "kidnapping instructions"
|
||||
severity: "high"
|
||||
- keyword: "kidnap someone"
|
||||
severity: "high"
|
||||
- keyword: "abduct someone"
|
||||
severity: "high"
|
||||
- keyword: "how to rape"
|
||||
severity: "high"
|
||||
- keyword: "rape someone"
|
||||
severity: "high"
|
||||
- keyword: "how to sexually assault"
|
||||
severity: "high"
|
||||
- keyword: "sexual assault someone"
|
||||
severity: "high"
|
||||
- keyword: "torture someone"
|
||||
severity: "high"
|
||||
- keyword: "torture methods"
|
||||
severity: "high"
|
||||
- keyword: "how to torture"
|
||||
severity: "high"
|
||||
- keyword: "ways to torture"
|
||||
severity: "high"
|
||||
- keyword: "torture techniques"
|
||||
severity: "high"
|
||||
- keyword: "human trafficking operation"
|
||||
severity: "high"
|
||||
- keyword: "start human trafficking"
|
||||
severity: "high"
|
||||
- keyword: "organize human trafficking"
|
||||
severity: "high"
|
||||
- keyword: "child trafficking"
|
||||
severity: "high"
|
||||
- keyword: "sex trafficking"
|
||||
severity: "high"
|
||||
- keyword: "domestic terrorism"
|
||||
severity: "high"
|
||||
- keyword: "lone wolf attack"
|
||||
severity: "high"
|
||||
- keyword: "active shooter"
|
||||
severity: "high"
|
||||
- keyword: "mass casualty"
|
||||
severity: "high"
|
||||
- keyword: "vehicular attack"
|
||||
severity: "high"
|
||||
- keyword: "vehicle ramming"
|
||||
severity: "high"
|
||||
- keyword: "knife attack"
|
||||
severity: "high"
|
||||
- keyword: "stabbing spree"
|
||||
severity: "high"
|
||||
- keyword: "murder spree"
|
||||
severity: "high"
|
||||
- keyword: "killing spree"
|
||||
severity: "high"
|
||||
- keyword: "rampage shooting"
|
||||
severity: "high"
|
||||
- keyword: "workplace shooting"
|
||||
severity: "high"
|
||||
- keyword: "public shooting"
|
||||
severity: "high"
|
||||
- keyword: "mass murder"
|
||||
severity: "high"
|
||||
- keyword: "mass killing"
|
||||
severity: "high"
|
||||
- keyword: "massacre"
|
||||
severity: "high"
|
||||
- keyword: "genocide"
|
||||
severity: "high"
|
||||
- keyword: "ethnic cleansing"
|
||||
severity: "high"
|
||||
- keyword: "war crimes"
|
||||
severity: "high"
|
||||
- keyword: "crimes against humanity"
|
||||
severity: "high"
|
||||
- keyword: "honor killing"
|
||||
severity: "high"
|
||||
- keyword: "vigilante justice"
|
||||
severity: "high"
|
||||
- keyword: "lynch mob"
|
||||
severity: "high"
|
||||
- keyword: "mob violence"
|
||||
severity: "high"
|
||||
- keyword: "gang violence"
|
||||
severity: "high"
|
||||
- keyword: "organized crime"
|
||||
severity: "high"
|
||||
- keyword: "mafia hit"
|
||||
severity: "high"
|
||||
- keyword: "gang hit"
|
||||
severity: "high"
|
||||
- keyword: "drive by shooting"
|
||||
severity: "high"
|
||||
- keyword: "execution style"
|
||||
severity: "high"
|
||||
|
||||
# Medium severity - context-dependent
|
||||
- keyword: "murder"
|
||||
severity: "medium"
|
||||
- keyword: "kill"
|
||||
severity: "medium"
|
||||
- keyword: "assassin"
|
||||
severity: "medium"
|
||||
- keyword: "hitman"
|
||||
severity: "medium"
|
||||
- keyword: "kidnap"
|
||||
severity: "medium"
|
||||
- keyword: "attack"
|
||||
severity: "medium"
|
||||
- keyword: "violence"
|
||||
severity: "medium"
|
||||
- keyword: "weapon"
|
||||
severity: "medium"
|
||||
- keyword: "shooting"
|
||||
severity: "medium"
|
||||
- keyword: "terrorist"
|
||||
severity: "medium"
|
||||
- keyword: "crime"
|
||||
severity: "medium"
|
||||
|
||||
# Exception patterns (legitimate uses that should NOT be blocked)
|
||||
exceptions:
|
||||
- "violence prevention"
|
||||
- "crime statistics"
|
||||
- "true crime"
|
||||
- "documentary"
|
||||
- "news report"
|
||||
- "historical"
|
||||
- "prevent violence"
|
||||
- "combat violence"
|
||||
- "fight violence"
|
||||
- "violence against"
|
||||
- "victims of violence"
|
||||
- "domestic violence"
|
||||
- "reporting violence"
|
||||
- "violence awareness"
|
||||
|
|
@ -5,6 +5,7 @@ This guardrail provides regex pattern matching and keyword filtering
|
|||
to detect and block/mask sensitive content.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
|
|
@ -36,11 +37,33 @@ from litellm.types.guardrails import (
|
|||
GuardrailEventHooks,
|
||||
Mode,
|
||||
)
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.litellm_content_filter import (
|
||||
ContentFilterCategoryConfig,
|
||||
)
|
||||
from litellm.types.utils import ModelResponseStream
|
||||
|
||||
from .patterns import get_compiled_pattern
|
||||
|
||||
|
||||
# Helper data structure for category-based detection
|
||||
class CategoryConfig:
|
||||
"""Configuration for a content category."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
category_name: str,
|
||||
description: str,
|
||||
default_action: ContentFilterAction,
|
||||
keywords: List[Dict[str, str]],
|
||||
exceptions: List[str],
|
||||
):
|
||||
self.category_name = category_name
|
||||
self.description = description
|
||||
self.default_action = default_action
|
||||
self.keywords = keywords
|
||||
self.exceptions = [e.lower() for e in exceptions]
|
||||
|
||||
|
||||
class ContentFilterGuardrail(CustomGuardrail):
|
||||
"""
|
||||
Content filter guardrail that detects sensitive information using:
|
||||
|
|
@ -69,6 +92,8 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
default_on: bool = False,
|
||||
pattern_redaction_format: Optional[str] = None,
|
||||
keyword_redaction_tag: Optional[str] = None,
|
||||
categories: Optional[List[ContentFilterCategoryConfig]] = None,
|
||||
severity_threshold: str = "medium",
|
||||
**kwargs,
|
||||
):
|
||||
"""
|
||||
|
|
@ -83,6 +108,8 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
default_on: If True, runs on all requests by default
|
||||
pattern_redaction_format: Format string for pattern redaction (use {pattern_name} placeholder)
|
||||
keyword_redaction_tag: Tag to use for keyword redaction
|
||||
categories: List of category configurations with enabled/action/severity settings
|
||||
severity_threshold: Minimum severity to block ("high", "medium", "low")
|
||||
"""
|
||||
super().__init__(
|
||||
guardrail_name=guardrail_name,
|
||||
|
|
@ -101,6 +128,17 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
pattern_redaction_format or self.PATTERN_REDACTION_FORMAT
|
||||
)
|
||||
self.keyword_redaction_tag = keyword_redaction_tag or self.KEYWORD_REDACTION_STR
|
||||
self.severity_threshold = severity_threshold
|
||||
|
||||
# Store loaded categories
|
||||
self.loaded_categories: Dict[str, CategoryConfig] = {}
|
||||
self.category_keywords: Dict[str, Tuple[str, str, ContentFilterAction]] = (
|
||||
{}
|
||||
) # keyword -> (category, severity, action)
|
||||
|
||||
# Load categories if provided
|
||||
if categories:
|
||||
self._load_categories(categories)
|
||||
|
||||
# Normalize inputs: convert dicts to Pydantic models for consistent handling
|
||||
normalized_patterns: List[ContentFilterPattern] = []
|
||||
|
|
@ -144,6 +182,125 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
f"ContentFilterGuardrail initialized with {len(self.compiled_patterns)} patterns "
|
||||
f"and {len(self.blocked_words)} blocked words"
|
||||
)
|
||||
verbose_proxy_logger.debug(
|
||||
f"Loaded {len(self.loaded_categories)} categories with "
|
||||
f"{len(self.category_keywords)} keywords"
|
||||
)
|
||||
|
||||
def _load_categories(self, categories: List[ContentFilterCategoryConfig]) -> None:
|
||||
"""
|
||||
Load content categories from configuration.
|
||||
|
||||
Args:
|
||||
categories: List of category configurations with format:
|
||||
- category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
category_file: "/path/to/custom_file.yaml" # optional override
|
||||
"""
|
||||
categories_dir = os.path.join(os.path.dirname(__file__), "categories")
|
||||
|
||||
for cat_config in categories:
|
||||
category_name = cat_config.get("category")
|
||||
if not category_name or not isinstance(category_name, str):
|
||||
verbose_proxy_logger.warning(
|
||||
"Category name missing or invalid in config, skipping"
|
||||
)
|
||||
continue
|
||||
|
||||
enabled = cat_config.get("enabled", True)
|
||||
action = cat_config.get("action")
|
||||
severity_threshold = cat_config.get(
|
||||
"severity_threshold", self.severity_threshold
|
||||
)
|
||||
custom_file = cat_config.get("category_file")
|
||||
|
||||
if not enabled:
|
||||
verbose_proxy_logger.debug(
|
||||
f"Category {category_name} is disabled, skipping"
|
||||
)
|
||||
continue
|
||||
|
||||
# Load category file (custom or default)
|
||||
if custom_file:
|
||||
category_file_path = custom_file
|
||||
else:
|
||||
category_file_path = os.path.join(
|
||||
categories_dir, f"{category_name}.yaml"
|
||||
)
|
||||
|
||||
if not os.path.exists(category_file_path):
|
||||
verbose_proxy_logger.warning(
|
||||
f"Category file not found: {category_file_path}, skipping"
|
||||
)
|
||||
continue
|
||||
|
||||
try:
|
||||
category = self._load_category_file(category_file_path)
|
||||
self.loaded_categories[category_name] = category
|
||||
|
||||
# Use action from config, or default from category file
|
||||
category_action = ContentFilterAction(
|
||||
action if action else category.default_action
|
||||
)
|
||||
|
||||
# Add keywords from this category
|
||||
for keyword_data in category.keywords:
|
||||
keyword = keyword_data["keyword"].lower()
|
||||
severity = keyword_data["severity"]
|
||||
|
||||
# Check if keyword meets severity threshold
|
||||
if self._should_apply_severity(severity, severity_threshold):
|
||||
self.category_keywords[keyword] = (
|
||||
category_name,
|
||||
severity,
|
||||
category_action,
|
||||
)
|
||||
|
||||
verbose_proxy_logger.info(
|
||||
f"Loaded category {category_name}: "
|
||||
f"{len(category.keywords)} keywords"
|
||||
)
|
||||
except Exception as e:
|
||||
verbose_proxy_logger.error(
|
||||
f"Error loading category {category_name}: {e}"
|
||||
)
|
||||
|
||||
def _load_category_file(self, file_path: str) -> CategoryConfig:
|
||||
"""
|
||||
Load a category definition from a YAML file.
|
||||
|
||||
Args:
|
||||
file_path: Path to category YAML file
|
||||
|
||||
Returns:
|
||||
CategoryConfig object
|
||||
"""
|
||||
with open(file_path, "r") as f:
|
||||
data = yaml.safe_load(f)
|
||||
|
||||
return CategoryConfig(
|
||||
category_name=data.get("category_name", "unknown"),
|
||||
description=data.get("description", ""),
|
||||
default_action=ContentFilterAction(data.get("default_action", "BLOCK")),
|
||||
keywords=data.get("keywords", []),
|
||||
exceptions=data.get("exceptions", []),
|
||||
)
|
||||
|
||||
def _should_apply_severity(self, severity: str, threshold: str) -> bool:
|
||||
"""
|
||||
Check if a given severity meets the threshold.
|
||||
|
||||
Args:
|
||||
severity: The severity level of the item ("high", "medium", "low")
|
||||
threshold: The minimum severity threshold
|
||||
|
||||
Returns:
|
||||
True if severity meets or exceeds threshold
|
||||
"""
|
||||
severity_order = {"low": 0, "medium": 1, "high": 2}
|
||||
return severity_order.get(severity, 0) >= severity_order.get(threshold, 1)
|
||||
|
||||
def _add_pattern(self, pattern_config: ContentFilterPattern) -> None:
|
||||
"""
|
||||
|
|
@ -247,6 +404,64 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
return (matched_text, pattern_name, action)
|
||||
return None
|
||||
|
||||
def _check_category_keywords(
|
||||
self, text: str, exceptions: List[str]
|
||||
) -> Optional[Tuple[str, str, str, ContentFilterAction]]:
|
||||
"""
|
||||
Check text for category keywords.
|
||||
|
||||
Args:
|
||||
text: Text to check
|
||||
exceptions: List of exception phrases to ignore
|
||||
|
||||
Returns:
|
||||
Tuple of (keyword, category, severity, action) if match found, None otherwise
|
||||
"""
|
||||
text_lower = text.lower()
|
||||
|
||||
# First check if any exception applies
|
||||
for exception in exceptions:
|
||||
if exception in text_lower:
|
||||
verbose_proxy_logger.debug(
|
||||
f"Exception phrase '{exception}' found, skipping category keyword check"
|
||||
)
|
||||
return None
|
||||
|
||||
# Check category keywords
|
||||
for keyword, (category, severity, action) in self.category_keywords.items():
|
||||
# Use word boundary matching for single words to avoid false positives
|
||||
# (e.g., "men" should not match "recommend")
|
||||
# For multi-word phrases, use substring matching
|
||||
if " " in keyword:
|
||||
# Multi-word phrase - use substring matching
|
||||
keyword_found = keyword in text_lower
|
||||
else:
|
||||
# Single word - use word boundary matching to match whole words only
|
||||
keyword_pattern = r"\b" + re.escape(keyword) + r"\b"
|
||||
keyword_found = bool(re.search(keyword_pattern, text_lower))
|
||||
|
||||
if keyword_found:
|
||||
# Check if this keyword has exceptions
|
||||
category_obj = self.loaded_categories.get(category)
|
||||
if category_obj:
|
||||
# Check category-specific exceptions
|
||||
exception_found = False
|
||||
for exception in category_obj.exceptions:
|
||||
if exception in text_lower:
|
||||
verbose_proxy_logger.debug(
|
||||
f"Category exception '{exception}' found for keyword '{keyword}', skipping"
|
||||
)
|
||||
exception_found = True
|
||||
break
|
||||
if exception_found:
|
||||
continue
|
||||
|
||||
verbose_proxy_logger.debug(
|
||||
f"Category keyword '{keyword}' found in category '{category}' with severity {severity}"
|
||||
)
|
||||
return (keyword, category, severity, action)
|
||||
return None
|
||||
|
||||
def _check_blocked_words(
|
||||
self, text: str
|
||||
) -> Optional[Tuple[str, ContentFilterAction, Optional[str]]]:
|
||||
|
|
@ -337,6 +552,42 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
processed_texts = []
|
||||
|
||||
for text in texts:
|
||||
# Collect all exceptions from loaded categories
|
||||
all_exceptions = []
|
||||
for category in self.loaded_categories.values():
|
||||
all_exceptions.extend(category.exceptions)
|
||||
|
||||
# Check category keywords
|
||||
category_keyword_match = self._check_category_keywords(text, all_exceptions)
|
||||
if category_keyword_match:
|
||||
keyword, category, severity, action = category_keyword_match
|
||||
if action == ContentFilterAction.BLOCK:
|
||||
error_msg = (
|
||||
f"Content blocked: {category} category keyword '{keyword}' detected "
|
||||
f"(severity: {severity})"
|
||||
)
|
||||
verbose_proxy_logger.warning(error_msg)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail={
|
||||
"error": error_msg,
|
||||
"category": category,
|
||||
"keyword": keyword,
|
||||
"severity": severity,
|
||||
},
|
||||
)
|
||||
elif action == ContentFilterAction.MASK:
|
||||
# Replace keyword with redaction tag
|
||||
text = re.sub(
|
||||
re.escape(keyword),
|
||||
self.keyword_redaction_tag,
|
||||
text,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
verbose_proxy_logger.info(
|
||||
f"Masked category keyword '{keyword}' from {category} (severity: {severity})"
|
||||
)
|
||||
|
||||
# Check regex patterns - process ALL patterns, not just first match
|
||||
for compiled_pattern, pattern_name, action in self.compiled_patterns:
|
||||
match = compiled_pattern.search(text)
|
||||
|
|
@ -356,7 +607,9 @@ class ContentFilterGuardrail(CustomGuardrail):
|
|||
pattern_name=pattern_name.upper()
|
||||
)
|
||||
text = compiled_pattern.sub(redaction_tag, text)
|
||||
verbose_proxy_logger.info(f"Masked all {pattern_name} matches in content")
|
||||
verbose_proxy_logger.info(
|
||||
f"Masked all {pattern_name} matches in content"
|
||||
)
|
||||
|
||||
# Check blocked words - iterate through ALL blocked words
|
||||
# to ensure all matching keywords are processed, not just the first one
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""
|
||||
Prebuilt regex patterns for content filtering.
|
||||
|
||||
This module loads predefined regex patterns from patterns.json for detecting
|
||||
This module loads predefined regex patterns from patterns.json for detecting
|
||||
sensitive information like SSNs, credit cards, API keys, etc.
|
||||
"""
|
||||
|
||||
|
|
@ -25,6 +25,7 @@ _PATTERNS_DATA = _load_patterns_from_json()
|
|||
|
||||
class PrebuiltPatternName(str, Enum):
|
||||
"""Enum for prebuilt pattern names - dynamically generated from JSON"""
|
||||
|
||||
pass
|
||||
|
||||
|
||||
|
|
@ -43,13 +44,13 @@ PREBUILT_PATTERNS: Dict[str, str] = {
|
|||
def get_compiled_pattern(pattern_name: str) -> Pattern:
|
||||
"""
|
||||
Get a compiled regex pattern by name.
|
||||
|
||||
|
||||
Args:
|
||||
pattern_name: Name of the prebuilt pattern
|
||||
|
||||
|
||||
Returns:
|
||||
Compiled regex pattern
|
||||
|
||||
|
||||
Raises:
|
||||
ValueError: If pattern_name is not found in PREBUILT_PATTERNS
|
||||
"""
|
||||
|
|
@ -59,14 +60,14 @@ def get_compiled_pattern(pattern_name: str) -> Pattern:
|
|||
f"Unknown pattern name: '{pattern_name}'. "
|
||||
f"Available patterns: {available_patterns}"
|
||||
)
|
||||
|
||||
|
||||
return re.compile(PREBUILT_PATTERNS[pattern_name], re.IGNORECASE)
|
||||
|
||||
|
||||
def get_all_pattern_names() -> List[str]:
|
||||
"""
|
||||
Get a list of all available prebuilt pattern names.
|
||||
|
||||
|
||||
Returns:
|
||||
List of pattern names
|
||||
"""
|
||||
|
|
@ -99,7 +100,7 @@ PATTERN_DESCRIPTIONS: Dict[str, str] = {
|
|||
def get_pattern_metadata() -> List[Dict[str, str]]:
|
||||
"""
|
||||
Return pattern metadata for UI display.
|
||||
|
||||
|
||||
Returns:
|
||||
List of dictionaries containing pattern name, display_name, category, and description
|
||||
"""
|
||||
|
|
@ -113,3 +114,51 @@ def get_pattern_metadata() -> List[Dict[str, str]]:
|
|||
for pattern_data in _PATTERNS_DATA["patterns"]
|
||||
]
|
||||
|
||||
|
||||
def get_available_content_categories() -> List[Dict[str, str]]:
|
||||
"""
|
||||
Return available content categories for UI display.
|
||||
|
||||
Returns:
|
||||
List of dictionaries containing category name, display_name, and description
|
||||
"""
|
||||
import yaml
|
||||
|
||||
categories_dir = os.path.join(os.path.dirname(__file__), "categories")
|
||||
available_categories = []
|
||||
|
||||
if not os.path.exists(categories_dir):
|
||||
return []
|
||||
|
||||
# Scan the categories directory for YAML files
|
||||
for filename in os.listdir(categories_dir):
|
||||
if filename.endswith(".yaml") or filename.endswith(".yml"):
|
||||
category_file_path = os.path.join(categories_dir, filename)
|
||||
try:
|
||||
with open(category_file_path, "r") as f:
|
||||
category_data = yaml.safe_load(f)
|
||||
|
||||
if category_data and "category_name" in category_data:
|
||||
# Create display name from category name (convert harmful_self_harm -> Harmful Self Harm)
|
||||
display_name = (
|
||||
category_data["category_name"].replace("_", " ").title()
|
||||
)
|
||||
|
||||
available_categories.append(
|
||||
{
|
||||
"name": category_data["category_name"],
|
||||
"display_name": display_name,
|
||||
"description": category_data.get("description", ""),
|
||||
"default_action": category_data.get(
|
||||
"default_action", "BLOCK"
|
||||
),
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
# Skip files that can't be loaded
|
||||
continue
|
||||
|
||||
# Sort by name for consistent ordering
|
||||
available_categories.sort(key=lambda x: x["name"])
|
||||
|
||||
return available_categories
|
||||
|
|
|
|||
|
|
@ -1,7 +1,84 @@
|
|||
from typing import List, Literal, Optional
|
||||
|
||||
from pydantic import Field
|
||||
|
||||
from litellm.types.llms.base import BaseLiteLLMOpenAIResponseObject
|
||||
from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel
|
||||
|
||||
|
||||
class ContentFilterCategoryConfig(BaseLiteLLMOpenAIResponseObject):
|
||||
"""
|
||||
category: "harmful_self_harm"
|
||||
enabled: true
|
||||
action: "BLOCK"
|
||||
severity_threshold: "medium"
|
||||
category_file: "/path/to/custom_file.yaml" # optional override
|
||||
"""
|
||||
|
||||
category: str = Field(
|
||||
description="The category to detect",
|
||||
)
|
||||
enabled: bool = Field(
|
||||
default=True,
|
||||
description="Whether the category is enabled",
|
||||
)
|
||||
action: Literal["BLOCK", "MASK"] = Field(
|
||||
description="The action to take when the category is detected",
|
||||
)
|
||||
severity_threshold: Literal["high", "medium", "low"] = Field(
|
||||
default="medium",
|
||||
description="The severity threshold to detect the category",
|
||||
)
|
||||
category_file: Optional[str] = Field(
|
||||
default=None,
|
||||
description="Optional override. Use your own category file instead of the default one.",
|
||||
)
|
||||
|
||||
|
||||
class LitellmContentFilterGuardrailConfigModel(GuardrailConfigModel):
|
||||
"""
|
||||
Configuration model for LiteLLM Content Filter guardrail.
|
||||
|
||||
Supports:
|
||||
- Traditional keyword and pattern matching
|
||||
- Category-based detection (harmful content, bias detection)
|
||||
- Proximity-based detection (identity keywords + negative modifiers)
|
||||
"""
|
||||
|
||||
# Traditional patterns and keywords
|
||||
patterns: Optional[List[dict]] = Field(
|
||||
default=None,
|
||||
description="List of regex patterns to detect (prebuilt or custom)",
|
||||
)
|
||||
blocked_words: Optional[List[dict]] = Field(
|
||||
default=None,
|
||||
description="List of blocked keywords with actions",
|
||||
)
|
||||
blocked_words_file: Optional[str] = Field(
|
||||
default=None,
|
||||
description="Path to YAML file containing blocked words",
|
||||
)
|
||||
|
||||
# Category-based detection
|
||||
categories: Optional[List[ContentFilterCategoryConfig]] = Field(
|
||||
default=None,
|
||||
description="List of prebuilt categories to enable (harmful_*, bias_*)",
|
||||
)
|
||||
severity_threshold: str = Field(
|
||||
default="medium",
|
||||
description="Minimum severity to block (high, medium, low)",
|
||||
)
|
||||
|
||||
# Redaction customization
|
||||
pattern_redaction_format: Optional[str] = Field(
|
||||
default="[{pattern_name}_REDACTED]",
|
||||
description="Format string for pattern redaction (use {pattern_name} placeholder)",
|
||||
)
|
||||
keyword_redaction_tag: Optional[str] = Field(
|
||||
default="[KEYWORD_REDACTED]",
|
||||
description="Tag to use for keyword redaction",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def ui_friendly_name() -> str:
|
||||
return "LiteLLM Content Filter"
|
||||
return "LiteLLM Content Filter"
|
||||
|
|
|
|||
|
|
@ -58,6 +58,12 @@ interface GuardrailSettings {
|
|||
}>;
|
||||
pattern_categories: string[];
|
||||
supported_actions: string[];
|
||||
content_categories?: Array<{
|
||||
name: string;
|
||||
display_name: string;
|
||||
description: string;
|
||||
default_action: string;
|
||||
}>;
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -103,6 +109,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
// Content Filter state
|
||||
const [selectedPatterns, setSelectedPatterns] = useState<any[]>([]);
|
||||
const [blockedWords, setBlockedWords] = useState<any[]>([]);
|
||||
const [selectedContentCategories, setSelectedContentCategories] = useState<any[]>([]);
|
||||
const [toolPermissionConfig, setToolPermissionConfig] = useState<ToolPermissionConfig>({
|
||||
rules: [],
|
||||
default_action: "deny",
|
||||
|
|
@ -251,6 +258,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
setCategorySpecificThresholds({});
|
||||
setSelectedPatterns([]);
|
||||
setBlockedWords([]);
|
||||
setSelectedContentCategories([]);
|
||||
setToolPermissionConfig({
|
||||
rules: [],
|
||||
default_action: "deny",
|
||||
|
|
@ -315,7 +323,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
}
|
||||
}
|
||||
|
||||
// For Content Filter, add patterns and blocked words
|
||||
// For Content Filter, add patterns, blocked words, and categories
|
||||
if (shouldRenderContentFilterConfigSettings(values.provider)) {
|
||||
if (selectedPatterns.length > 0) {
|
||||
guardrailData.litellm_params.patterns = selectedPatterns.map((p) => ({
|
||||
|
|
@ -333,6 +341,14 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
description: w.description,
|
||||
}));
|
||||
}
|
||||
if (selectedContentCategories.length > 0) {
|
||||
guardrailData.litellm_params.categories = selectedContentCategories.map((c) => ({
|
||||
category: c.category,
|
||||
enabled: true,
|
||||
action: c.action,
|
||||
severity_threshold: c.severity_threshold || "medium",
|
||||
}));
|
||||
}
|
||||
}
|
||||
// Add config values to the guardrail_info if provided
|
||||
else if (values.config) {
|
||||
|
|
@ -581,7 +597,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
</Form.Item>
|
||||
|
||||
{/* Use the GuardrailProviderFields component to render provider-specific fields */}
|
||||
{!isToolPermissionProvider && (
|
||||
{!isToolPermissionProvider && !shouldRenderContentFilterConfigSettings(selectedProvider) && (
|
||||
<GuardrailProviderFields
|
||||
selectedProvider={selectedProvider}
|
||||
accessToken={accessToken}
|
||||
|
|
@ -608,7 +624,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
);
|
||||
};
|
||||
|
||||
const renderContentFilterConfiguration = (step: "patterns" | "keywords") => {
|
||||
const renderContentFilterConfiguration = (step: "patterns" | "keywords" | "categories") => {
|
||||
if (!guardrailSettings || !shouldRenderContentFilterConfigSettings(selectedProvider)) return null;
|
||||
|
||||
const contentFilterSettings = guardrailSettings.content_filter_settings;
|
||||
|
|
@ -634,6 +650,15 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
blockedWords.map((w) => (w.id === id ? { ...w, [field]: value } : w))
|
||||
);
|
||||
}}
|
||||
contentCategories={contentFilterSettings.content_categories || []}
|
||||
selectedContentCategories={selectedContentCategories}
|
||||
onContentCategoryAdd={(category) => setSelectedContentCategories([...selectedContentCategories, category])}
|
||||
onContentCategoryRemove={(id) => setSelectedContentCategories(selectedContentCategories.filter((c) => c.id !== id))}
|
||||
onContentCategoryUpdate={(id, field, value) => {
|
||||
setSelectedContentCategories(
|
||||
selectedContentCategories.map((c) => (c.id === id ? { ...c, [field]: value } : c))
|
||||
);
|
||||
}}
|
||||
accessToken={accessToken}
|
||||
showStep={step}
|
||||
/>
|
||||
|
|
@ -675,10 +700,15 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
return renderPiiConfiguration();
|
||||
}
|
||||
if (shouldRenderContentFilterConfigSettings(selectedProvider)) {
|
||||
return renderContentFilterConfiguration("patterns");
|
||||
return renderContentFilterConfiguration("categories");
|
||||
}
|
||||
return renderOptionalParams();
|
||||
case 2:
|
||||
if (shouldRenderContentFilterConfigSettings(selectedProvider)) {
|
||||
return renderContentFilterConfiguration("patterns");
|
||||
}
|
||||
return null;
|
||||
case 3:
|
||||
if (shouldRenderContentFilterConfigSettings(selectedProvider)) {
|
||||
return renderContentFilterConfiguration("keywords");
|
||||
}
|
||||
|
|
@ -689,7 +719,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
};
|
||||
|
||||
const renderStepButtons = () => {
|
||||
const totalSteps = shouldRenderContentFilterConfigSettings(selectedProvider) ? 3 : 2;
|
||||
const totalSteps = shouldRenderContentFilterConfigSettings(selectedProvider) ? 4 : 2;
|
||||
const isLastStep = currentStep === totalSteps - 1;
|
||||
|
||||
return (
|
||||
|
|
@ -713,7 +743,7 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
};
|
||||
|
||||
return (
|
||||
<Modal title="Add Guardrail" open={visible} onCancel={handleClose} footer={null} width={700}>
|
||||
<Modal title="Add Guardrail" open={visible} onCancel={handleClose} footer={null} width={800}>
|
||||
<Form
|
||||
form={form}
|
||||
layout="vertical"
|
||||
|
|
@ -722,19 +752,22 @@ const AddGuardrailForm: React.FC<AddGuardrailFormProps> = ({ visible, onClose, a
|
|||
default_on: false,
|
||||
}}
|
||||
>
|
||||
<Steps current={currentStep} className="mb-6">
|
||||
<Steps current={currentStep} className="mb-6" style={{ overflow: "visible" }}>
|
||||
<Step title="Basic Info" />
|
||||
<Step
|
||||
title={
|
||||
shouldRenderPIIConfigSettings(selectedProvider)
|
||||
? "PII Configuration"
|
||||
: shouldRenderContentFilterConfigSettings(selectedProvider)
|
||||
? "Pattern Detection"
|
||||
? "Default Categories"
|
||||
: "Provider Configuration"
|
||||
}
|
||||
/>
|
||||
{shouldRenderContentFilterConfigSettings(selectedProvider) && (
|
||||
<Step title="Blocked Keywords" />
|
||||
<>
|
||||
<Step title="Patterns" />
|
||||
<Step title="Keywords" />
|
||||
</>
|
||||
)}
|
||||
</Steps>
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,384 @@
|
|||
import React from "react";
|
||||
import { Card, Typography, Select, Table, Tag, Collapse } from "antd";
|
||||
import { DeleteOutlined, PlusOutlined, FileTextOutlined } from "@ant-design/icons";
|
||||
import { Button } from "@tremor/react";
|
||||
import { getCategoryYaml } from "../../networking";
|
||||
|
||||
const { Title, Text } = Typography;
|
||||
const { Option } = Select;
|
||||
const { Panel } = Collapse;
|
||||
|
||||
interface ContentCategory {
|
||||
name: string;
|
||||
display_name: string;
|
||||
description: string;
|
||||
default_action: string;
|
||||
}
|
||||
|
||||
interface SelectedCategory {
|
||||
id: string;
|
||||
category: string;
|
||||
display_name: string;
|
||||
action: "BLOCK" | "MASK";
|
||||
severity_threshold: "high" | "medium" | "low";
|
||||
}
|
||||
|
||||
interface ContentCategoryConfigurationProps {
|
||||
availableCategories: ContentCategory[];
|
||||
selectedCategories: SelectedCategory[];
|
||||
onCategoryAdd: (category: SelectedCategory) => void;
|
||||
onCategoryRemove: (id: string) => void;
|
||||
onCategoryUpdate: (id: string, field: string, value: any) => void;
|
||||
accessToken?: string | null;
|
||||
}
|
||||
|
||||
const ContentCategoryConfiguration: React.FC<ContentCategoryConfigurationProps> = ({
|
||||
availableCategories,
|
||||
selectedCategories,
|
||||
onCategoryAdd,
|
||||
onCategoryRemove,
|
||||
onCategoryUpdate,
|
||||
accessToken,
|
||||
}) => {
|
||||
const [selectedCategoryName, setSelectedCategoryName] = React.useState<string>("");
|
||||
const [categoryYaml, setCategoryYaml] = React.useState<{ [key: string]: string }>({});
|
||||
const [loadingYaml, setLoadingYaml] = React.useState<{ [key: string]: boolean }>({});
|
||||
const [expandedYamlCategories, setExpandedYamlCategories] = React.useState<string[]>([]);
|
||||
const [previewYaml, setPreviewYaml] = React.useState<string>("");
|
||||
const [loadingPreviewYaml, setLoadingPreviewYaml] = React.useState<boolean>(false);
|
||||
|
||||
const handleAddCategory = () => {
|
||||
if (!selectedCategoryName) {
|
||||
return;
|
||||
}
|
||||
|
||||
const category = availableCategories.find((c) => c.name === selectedCategoryName);
|
||||
if (!category) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Check if already added
|
||||
if (selectedCategories.some((c) => c.category === selectedCategoryName)) {
|
||||
return;
|
||||
}
|
||||
|
||||
onCategoryAdd({
|
||||
id: `category-${Date.now()}`,
|
||||
category: category.name,
|
||||
display_name: category.display_name,
|
||||
action: category.default_action as "BLOCK" | "MASK",
|
||||
severity_threshold: "medium",
|
||||
});
|
||||
|
||||
setSelectedCategoryName("");
|
||||
setPreviewYaml(""); // Clear preview when category is added
|
||||
};
|
||||
|
||||
const fetchCategoryYaml = async (categoryName: string) => {
|
||||
if (!accessToken) {
|
||||
return; // No access token
|
||||
}
|
||||
|
||||
// Check if already loaded
|
||||
if (categoryYaml[categoryName]) {
|
||||
return;
|
||||
}
|
||||
|
||||
setLoadingYaml((prev) => ({ ...prev, [categoryName]: true }));
|
||||
try {
|
||||
const data = await getCategoryYaml(accessToken, categoryName);
|
||||
setCategoryYaml((prev) => ({ ...prev, [categoryName]: data.yaml_content }));
|
||||
} catch (error) {
|
||||
console.error(`Failed to fetch YAML for category ${categoryName}:`, error);
|
||||
} finally {
|
||||
setLoadingYaml((prev) => ({ ...prev, [categoryName]: false }));
|
||||
}
|
||||
};
|
||||
|
||||
// Fetch preview YAML when a category is selected in dropdown
|
||||
React.useEffect(() => {
|
||||
if (selectedCategoryName && accessToken) {
|
||||
// Check if we already have this YAML cached
|
||||
const cachedYaml = categoryYaml[selectedCategoryName];
|
||||
if (cachedYaml) {
|
||||
setPreviewYaml(cachedYaml);
|
||||
return;
|
||||
}
|
||||
|
||||
// Fetch the YAML for preview
|
||||
setLoadingPreviewYaml(true);
|
||||
console.log(`Fetching YAML for category: ${selectedCategoryName}`, { accessToken: accessToken ? "present" : "missing" });
|
||||
getCategoryYaml(accessToken, selectedCategoryName)
|
||||
.then((data) => {
|
||||
console.log(`Successfully fetched YAML for ${selectedCategoryName}:`, data);
|
||||
setPreviewYaml(data.yaml_content);
|
||||
// Also cache it for later use
|
||||
setCategoryYaml((prev) => ({ ...prev, [selectedCategoryName]: data.yaml_content }));
|
||||
})
|
||||
.catch((error) => {
|
||||
console.error(`Failed to fetch preview YAML for category ${selectedCategoryName}:`, error);
|
||||
setPreviewYaml("");
|
||||
})
|
||||
.finally(() => {
|
||||
setLoadingPreviewYaml(false);
|
||||
});
|
||||
} else {
|
||||
setPreviewYaml("");
|
||||
setLoadingPreviewYaml(false);
|
||||
}
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [selectedCategoryName, accessToken]);
|
||||
|
||||
const columns = [
|
||||
{
|
||||
title: "Category",
|
||||
dataIndex: "display_name",
|
||||
key: "display_name",
|
||||
render: (text: string, record: SelectedCategory) => {
|
||||
const category = availableCategories.find((c) => c.name === record.category);
|
||||
return (
|
||||
<div>
|
||||
<div style={{ fontWeight: 500 }}>{text}</div>
|
||||
{category?.description && (
|
||||
<div style={{ fontSize: "12px", color: "#888", marginTop: "4px" }}>
|
||||
{category.description}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
title: "Action",
|
||||
dataIndex: "action",
|
||||
key: "action",
|
||||
width: 150,
|
||||
render: (action: string, record: SelectedCategory) => (
|
||||
<Select
|
||||
value={action}
|
||||
onChange={(value) => onCategoryUpdate(record.id, "action", value)}
|
||||
style={{ width: "100%" }}
|
||||
>
|
||||
<Option value="BLOCK">
|
||||
<Tag color="red">BLOCK</Tag>
|
||||
</Option>
|
||||
<Option value="MASK">
|
||||
<Tag color="orange">MASK</Tag>
|
||||
</Option>
|
||||
</Select>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: "Severity Threshold",
|
||||
dataIndex: "severity_threshold",
|
||||
key: "severity_threshold",
|
||||
width: 180,
|
||||
render: (threshold: string, record: SelectedCategory) => (
|
||||
<Select
|
||||
value={threshold}
|
||||
onChange={(value) => onCategoryUpdate(record.id, "severity_threshold", value)}
|
||||
style={{ width: "100%" }}
|
||||
>
|
||||
<Option value="low">Low</Option>
|
||||
<Option value="medium">Medium</Option>
|
||||
<Option value="high">High</Option>
|
||||
</Select>
|
||||
),
|
||||
},
|
||||
{
|
||||
title: "",
|
||||
key: "actions",
|
||||
width: 80,
|
||||
render: (_: any, record: SelectedCategory) => (
|
||||
<Button
|
||||
icon={DeleteOutlined}
|
||||
onClick={() => onCategoryRemove(record.id)}
|
||||
variant="secondary"
|
||||
size="xs"
|
||||
>
|
||||
Remove
|
||||
</Button>
|
||||
),
|
||||
},
|
||||
];
|
||||
|
||||
const unselectedCategories = availableCategories.filter(
|
||||
(cat) => !selectedCategories.some((sel) => sel.category === cat.name)
|
||||
);
|
||||
|
||||
return (
|
||||
<Card
|
||||
title={
|
||||
<div style={{ display: "flex", justifyContent: "space-between", alignItems: "center" }}>
|
||||
<Title level={5} style={{ margin: 0 }}>
|
||||
Content Categories
|
||||
</Title>
|
||||
<Text type="secondary" style={{ fontSize: 14, fontWeight: 400 }}>
|
||||
Detect harmful content, bias, and inappropriate advice using semantic analysis
|
||||
</Text>
|
||||
</div>
|
||||
}
|
||||
size="small"
|
||||
>
|
||||
<div style={{ marginBottom: 16, display: "flex", gap: 8 }}>
|
||||
<Select
|
||||
placeholder="Select a content category"
|
||||
value={selectedCategoryName || undefined}
|
||||
onChange={setSelectedCategoryName}
|
||||
style={{ flex: 1 }}
|
||||
showSearch
|
||||
optionLabelProp="label"
|
||||
filterOption={(input, option) =>
|
||||
(option?.label?.toString().toLowerCase() ?? "").includes(input.toLowerCase())
|
||||
}
|
||||
>
|
||||
{unselectedCategories.map((cat) => (
|
||||
<Option key={cat.name} value={cat.name} label={cat.display_name}>
|
||||
<div>
|
||||
<div style={{ fontWeight: 500 }}>{cat.display_name}</div>
|
||||
<div style={{ fontSize: "12px", color: "#666", marginTop: "2px" }}>
|
||||
{cat.description}
|
||||
</div>
|
||||
</div>
|
||||
</Option>
|
||||
))}
|
||||
</Select>
|
||||
<Button
|
||||
onClick={handleAddCategory}
|
||||
disabled={!selectedCategoryName}
|
||||
icon={PlusOutlined}
|
||||
>
|
||||
Add
|
||||
</Button>
|
||||
</div>
|
||||
|
||||
{/* Preview YAML box - shown when category is selected but not yet added */}
|
||||
{selectedCategoryName && (
|
||||
<div
|
||||
style={{
|
||||
marginBottom: 16,
|
||||
padding: "12px",
|
||||
background: "#f9f9f9",
|
||||
border: "1px solid #e0e0e0",
|
||||
borderRadius: "4px",
|
||||
}}
|
||||
>
|
||||
<div style={{ marginBottom: 8, fontWeight: 500, fontSize: "14px" }}>
|
||||
Preview: {availableCategories.find((c) => c.name === selectedCategoryName)?.display_name}
|
||||
</div>
|
||||
{loadingPreviewYaml ? (
|
||||
<div style={{ padding: "16px", textAlign: "center", color: "#888" }}>
|
||||
Loading YAML...
|
||||
</div>
|
||||
) : previewYaml ? (
|
||||
<pre
|
||||
style={{
|
||||
background: "#fff",
|
||||
padding: "12px",
|
||||
borderRadius: "4px",
|
||||
overflow: "auto",
|
||||
maxHeight: "300px",
|
||||
fontSize: "12px",
|
||||
lineHeight: "1.5",
|
||||
margin: 0,
|
||||
border: "1px solid #e0e0e0",
|
||||
}}
|
||||
>
|
||||
<code>{previewYaml}</code>
|
||||
</pre>
|
||||
) : (
|
||||
<div style={{ padding: "8px", textAlign: "center", color: "#888", fontSize: "12px" }}>
|
||||
Unable to load YAML content
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{selectedCategories.length > 0 ? (
|
||||
<>
|
||||
<Table
|
||||
dataSource={selectedCategories}
|
||||
columns={columns}
|
||||
pagination={false}
|
||||
size="small"
|
||||
rowKey="id"
|
||||
/>
|
||||
<div style={{ marginTop: 16 }}>
|
||||
<Collapse
|
||||
activeKey={expandedYamlCategories}
|
||||
onChange={(keys) => {
|
||||
const keyArray = Array.isArray(keys) ? keys : keys ? [keys] : [];
|
||||
const newExpanded = new Set(keyArray as string[]);
|
||||
const oldExpanded = new Set(expandedYamlCategories);
|
||||
|
||||
// Find newly expanded categories and fetch their YAML
|
||||
keyArray.forEach((key) => {
|
||||
const categoryName = key as string;
|
||||
if (!oldExpanded.has(categoryName) && !categoryYaml[categoryName]) {
|
||||
fetchCategoryYaml(categoryName);
|
||||
}
|
||||
});
|
||||
|
||||
setExpandedYamlCategories(keyArray as string[]);
|
||||
}}
|
||||
ghost
|
||||
>
|
||||
{selectedCategories.map((category) => (
|
||||
<Panel
|
||||
header={
|
||||
<div style={{ display: "flex", alignItems: "center", gap: 8 }}>
|
||||
<FileTextOutlined />
|
||||
<span>View YAML for {category.display_name}</span>
|
||||
</div>
|
||||
}
|
||||
key={category.category}
|
||||
>
|
||||
{loadingYaml[category.category] ? (
|
||||
<div style={{ padding: "16px", textAlign: "center", color: "#888" }}>
|
||||
Loading YAML...
|
||||
</div>
|
||||
) : categoryYaml[category.category] ? (
|
||||
<pre
|
||||
style={{
|
||||
background: "#f5f5f5",
|
||||
padding: "16px",
|
||||
borderRadius: "4px",
|
||||
overflow: "auto",
|
||||
maxHeight: "400px",
|
||||
fontSize: "12px",
|
||||
lineHeight: "1.5",
|
||||
margin: 0,
|
||||
}}
|
||||
>
|
||||
<code>{categoryYaml[category.category]}</code>
|
||||
</pre>
|
||||
) : (
|
||||
<div style={{ padding: "16px", textAlign: "center", color: "#888" }}>
|
||||
YAML will load when expanded
|
||||
</div>
|
||||
)}
|
||||
</Panel>
|
||||
))}
|
||||
</Collapse>
|
||||
</div>
|
||||
</>
|
||||
) : (
|
||||
<div
|
||||
style={{
|
||||
textAlign: "center",
|
||||
padding: "24px",
|
||||
color: "#888",
|
||||
border: "1px dashed #d9d9d9",
|
||||
borderRadius: "4px",
|
||||
}}
|
||||
>
|
||||
No content categories selected. Add categories to detect harmful content, bias, or
|
||||
inappropriate advice.
|
||||
</div>
|
||||
)}
|
||||
</Card>
|
||||
);
|
||||
};
|
||||
|
||||
export default ContentCategoryConfiguration;
|
||||
|
||||
|
|
@ -9,6 +9,7 @@ import CustomPatternModal from "./CustomPatternModal";
|
|||
import KeywordModal from "./KeywordModal";
|
||||
import PatternTable from "./PatternTable";
|
||||
import KeywordTable from "./KeywordTable";
|
||||
import ContentCategoryConfiguration from "./ContentCategoryConfiguration";
|
||||
|
||||
const { Title, Text } = Typography;
|
||||
|
||||
|
|
@ -35,6 +36,21 @@ interface BlockedWord {
|
|||
description?: string;
|
||||
}
|
||||
|
||||
interface ContentCategory {
|
||||
name: string;
|
||||
display_name: string;
|
||||
description: string;
|
||||
default_action: string;
|
||||
}
|
||||
|
||||
interface SelectedContentCategory {
|
||||
id: string;
|
||||
category: string;
|
||||
display_name: string;
|
||||
action: "BLOCK" | "MASK";
|
||||
severity_threshold: "high" | "medium" | "low";
|
||||
}
|
||||
|
||||
interface ContentFilterConfigurationProps {
|
||||
prebuiltPatterns: PrebuiltPattern[];
|
||||
categories: string[];
|
||||
|
|
@ -48,7 +64,12 @@ interface ContentFilterConfigurationProps {
|
|||
onBlockedWordUpdate: (id: string, field: string, value: any) => void;
|
||||
onFileUpload?: (content: string) => void;
|
||||
accessToken: string | null;
|
||||
showStep?: "patterns" | "keywords";
|
||||
showStep?: "patterns" | "keywords" | "categories";
|
||||
contentCategories?: ContentCategory[];
|
||||
selectedContentCategories?: SelectedContentCategory[];
|
||||
onContentCategoryAdd?: (category: SelectedContentCategory) => void;
|
||||
onContentCategoryRemove?: (id: string) => void;
|
||||
onContentCategoryUpdate?: (id: string, field: string, value: any) => void;
|
||||
}
|
||||
|
||||
const ContentFilterConfiguration: React.FC<ContentFilterConfigurationProps> = ({
|
||||
|
|
@ -65,6 +86,11 @@ const ContentFilterConfiguration: React.FC<ContentFilterConfigurationProps> = ({
|
|||
onFileUpload,
|
||||
accessToken,
|
||||
showStep,
|
||||
contentCategories = [],
|
||||
selectedContentCategories = [],
|
||||
onContentCategoryAdd,
|
||||
onContentCategoryRemove,
|
||||
onContentCategoryUpdate,
|
||||
}) => {
|
||||
const [patternModalVisible, setPatternModalVisible] = useState(false);
|
||||
const [keywordModalVisible, setKeywordModalVisible] = useState(false);
|
||||
|
|
@ -167,13 +193,14 @@ const ContentFilterConfiguration: React.FC<ContentFilterConfigurationProps> = ({
|
|||
|
||||
const showPatterns = !showStep || showStep === "patterns";
|
||||
const showKeywords = !showStep || showStep === "keywords";
|
||||
const showCategories = !showStep || showStep === "categories";
|
||||
|
||||
return (
|
||||
<div className="space-y-6">
|
||||
{!showStep && (
|
||||
<div>
|
||||
<Text type="secondary">
|
||||
Configure patterns and keywords to detect and filter sensitive information in requests and responses.
|
||||
Configure patterns, keywords, and content categories to detect and filter sensitive information in requests and responses.
|
||||
</Text>
|
||||
</div>
|
||||
)}
|
||||
|
|
@ -244,6 +271,17 @@ const ContentFilterConfiguration: React.FC<ContentFilterConfigurationProps> = ({
|
|||
</Card>
|
||||
)}
|
||||
|
||||
{showCategories && contentCategories.length > 0 && onContentCategoryAdd && onContentCategoryRemove && onContentCategoryUpdate && (
|
||||
<ContentCategoryConfiguration
|
||||
availableCategories={contentCategories}
|
||||
selectedCategories={selectedContentCategories}
|
||||
onCategoryAdd={onContentCategoryAdd}
|
||||
onCategoryRemove={onContentCategoryRemove}
|
||||
onCategoryUpdate={onContentCategoryUpdate}
|
||||
accessToken={accessToken}
|
||||
/>
|
||||
)}
|
||||
|
||||
<PatternModal
|
||||
visible={patternModalVisible}
|
||||
prebuiltPatterns={prebuiltPatterns}
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import {
|
|||
guardrail_provider_map,
|
||||
populateGuardrailProviders,
|
||||
populateGuardrailProviderMap,
|
||||
shouldRenderContentFilterConfigSettings,
|
||||
} from "./guardrail_info_helpers";
|
||||
import { getGuardrailProviderSpecificParams } from "../networking";
|
||||
import NumericalInput from "../shared/numerical_input";
|
||||
|
|
@ -107,6 +108,20 @@ const GuardrailProviderFields: React.FC<GuardrailProviderFieldsProps> = ({
|
|||
}
|
||||
|
||||
console.log("Value:", value);
|
||||
|
||||
// Fields to skip for content filter provider (handled in dedicated steps)
|
||||
const contentFilterFieldsToSkip = new Set([
|
||||
"patterns",
|
||||
"blocked_words",
|
||||
"blocked_words_file",
|
||||
"categories",
|
||||
"severity_threshold",
|
||||
"pattern_redaction_format",
|
||||
"keyword_redaction_tag",
|
||||
]);
|
||||
|
||||
const isContentFilterProvider = shouldRenderContentFilterConfigSettings(selectedProvider);
|
||||
|
||||
// Convert object to array of entries and render fields
|
||||
const renderFields = (fields: { [key: string]: ProviderParam }, parentKey = "", parentValue?: any) => {
|
||||
return Object.entries(fields).map(([fieldKey, field]) => {
|
||||
|
|
@ -123,6 +138,11 @@ const GuardrailProviderFields: React.FC<GuardrailProviderFieldsProps> = ({
|
|||
return null;
|
||||
}
|
||||
|
||||
// Skip content filter specific fields when it's a content filter provider (handled in dedicated steps)
|
||||
if (isContentFilterProvider && contentFilterFieldsToSkip.has(fieldKey)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Handle other nested fields (like azure/text_moderations optional_params)
|
||||
if (field.type === "nested" && field.fields) {
|
||||
return (
|
||||
|
|
|
|||
|
|
@ -6836,6 +6836,40 @@ export const getGuardrailProviderSpecificParams = async (accessToken: string) =>
|
|||
}
|
||||
};
|
||||
|
||||
export const getCategoryYaml = async (accessToken: string, categoryName: string) => {
|
||||
try {
|
||||
// URL encode the category name to handle special characters
|
||||
const encodedCategoryName = encodeURIComponent(categoryName);
|
||||
const url = proxyBaseUrl
|
||||
? `${proxyBaseUrl}/guardrails/ui/category_yaml/${encodedCategoryName}`
|
||||
: `/guardrails/ui/category_yaml/${encodedCategoryName}`;
|
||||
|
||||
console.log(`Fetching category YAML from: ${url}`);
|
||||
|
||||
const response = await fetch(url, {
|
||||
method: "GET",
|
||||
headers: {
|
||||
[globalLitellmHeaderName]: `Bearer ${accessToken}`,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorData = await response.text();
|
||||
console.error(`Failed to get category YAML. Status: ${response.status}, Error:`, errorData);
|
||||
handleError(errorData);
|
||||
throw new Error(`Failed to get category YAML: ${response.status} ${errorData}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
console.log("Category YAML response:", data);
|
||||
return data;
|
||||
} catch (error) {
|
||||
console.error("Failed to get category YAML:", error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
export const getAgentsList = async (accessToken: string) => {
|
||||
try {
|
||||
const url = proxyBaseUrl ? `${proxyBaseUrl}/v1/agents` : `/v1/agents`;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue