From 26cd2c4473d87378b9ba86733f76edb9d9074702 Mon Sep 17 00:00:00 2001 From: Krish Dholakia Date: Thu, 18 Dec 2025 16:28:58 +0530 Subject: [PATCH] Guardrails - add built in guardrails for harmful content, bias, etc. (#18029) * feat(litellm_content_filter.py): add support for content filtering categories make it easy for proxy admin to prevent messages about violence, self harm or illegal weapons going through litellm * feat: initial commit adding bias detection allows admin to block inappropriate content about sexual orientation, etc. * refactor: simplify content_filter.py use a more exhaustive set of keywords, instead of guessing at potential phrases user can use * feat(content_filter.py): add new denied topics for in-built content filter guardrails allow user to automatically block content relating to certain categories from being sent to the LLML * refactor(content-filter): document new params to litellm content filter * feat(ui/): litellm content filter - select content categories on ui * docs: update documentation * docs(litellm_content_filter.md): document new content filters --- .../guardrails/litellm_content_filter.md | 490 +++++++++++++++++- docs/my-website/sidebars.js | 2 +- .../index.html} | 0 .../index.html} | 0 .../{budgets.html => budgets/index.html} | 0 .../{caching.html => caching/index.html} | 0 .../{old-usage.html => old-usage/index.html} | 0 .../{prompts.html => prompts/index.html} | 0 .../index.html} | 0 .../proxy/_experimental/out/guardrails.html | 1 - .../out/{login.html => login/index.html} | 0 .../out/{logs.html => logs/index.html} | 0 .../{callback.html => callback/index.html} | 0 .../{model-hub.html => model-hub/index.html} | 0 .../index.html} | 0 .../index.html} | 0 .../proxy/_experimental/out/onboarding.html | 1 - .../index.html} | 0 .../index.html} | 0 .../index.html} | 0 .../index.html} | 0 .../index.html} | 0 .../{ui-theme.html => ui-theme/index.html} | 0 .../out/{teams.html => teams/index.html} | 0 .../{test-key.html => test-key/index.html} | 0 .../index.html} | 0 .../index.html} | 0 .../out/{usage.html => usage/index.html} | 0 .../out/{users.html => users/index.html} | 0 .../index.html} | 0 litellm/proxy/_new_secret_config.yaml | 55 ++ .../proxy/guardrails/guardrail_endpoints.py | 47 ++ .../litellm_content_filter/__init__.py | 17 +- .../categories/bias_gender.yaml | 53 ++ .../categories/bias_racial.yaml | 148 ++++++ .../categories/bias_religious.yaml | 118 +++++ .../categories/bias_sexual_orientation.yaml | 251 +++++++++ .../categories/denied_financial_advice.yaml | 139 +++++ .../categories/denied_legal_advice.yaml | 137 +++++ .../categories/denied_medical_advice.yaml | 133 +++++ .../categories/harmful_illegal_weapons.yaml | 299 +++++++++++ .../categories/harmful_self_harm.yaml | 184 +++++++ .../categories/harmful_violence.yaml | 265 ++++++++++ .../litellm_content_filter/content_filter.py | 255 ++++++++- .../litellm_content_filter/patterns.py | 63 ++- .../guardrail_hooks/litellm_content_filter.py | 79 ++- .../guardrails/add_guardrail_form.tsx | 51 +- .../ContentCategoryConfiguration.tsx | 384 ++++++++++++++ .../ContentFilterConfiguration.tsx | 42 +- .../guardrails/guardrail_provider_fields.tsx | 20 + .../src/components/networking.tsx | 34 ++ 51 files changed, 3228 insertions(+), 40 deletions(-) rename litellm/proxy/_experimental/out/{api-reference.html => api-reference/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{api-playground.html => api-playground/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{budgets.html => budgets/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{caching.html => caching/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{old-usage.html => old-usage/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{prompts.html => prompts/index.html} (100%) rename litellm/proxy/_experimental/out/experimental/{tag-management.html => tag-management/index.html} (100%) delete mode 100644 litellm/proxy/_experimental/out/guardrails.html rename litellm/proxy/_experimental/out/{login.html => login/index.html} (100%) rename litellm/proxy/_experimental/out/{logs.html => logs/index.html} (100%) rename litellm/proxy/_experimental/out/mcp/oauth/{callback.html => callback/index.html} (100%) rename litellm/proxy/_experimental/out/{model-hub.html => model-hub/index.html} (100%) rename litellm/proxy/_experimental/out/{model_hub_table.html => model_hub_table/index.html} (100%) rename litellm/proxy/_experimental/out/{models-and-endpoints.html => models-and-endpoints/index.html} (100%) delete mode 100644 litellm/proxy/_experimental/out/onboarding.html rename litellm/proxy/_experimental/out/{organizations.html => organizations/index.html} (100%) rename litellm/proxy/_experimental/out/{playground.html => playground/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{admin-settings.html => admin-settings/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{logging-and-alerts.html => logging-and-alerts/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{router-settings.html => router-settings/index.html} (100%) rename litellm/proxy/_experimental/out/settings/{ui-theme.html => ui-theme/index.html} (100%) rename litellm/proxy/_experimental/out/{teams.html => teams/index.html} (100%) rename litellm/proxy/_experimental/out/{test-key.html => test-key/index.html} (100%) rename litellm/proxy/_experimental/out/tools/{mcp-servers.html => mcp-servers/index.html} (100%) rename litellm/proxy/_experimental/out/tools/{vector-stores.html => vector-stores/index.html} (100%) rename litellm/proxy/_experimental/out/{usage.html => usage/index.html} (100%) rename litellm/proxy/_experimental/out/{users.html => users/index.html} (100%) rename litellm/proxy/_experimental/out/{virtual-keys.html => virtual-keys/index.html} (100%) create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_gender.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_racial.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_religious.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_sexual_orientation.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_financial_advice.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_legal_advice.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_medical_advice.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_illegal_weapons.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_self_harm.yaml create mode 100644 litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_violence.yaml create mode 100644 ui/litellm-dashboard/src/components/guardrails/content_filter/ContentCategoryConfiguration.tsx diff --git a/docs/my-website/docs/proxy/guardrails/litellm_content_filter.md b/docs/my-website/docs/proxy/guardrails/litellm_content_filter.md index 29183c693a4..20bed9a0488 100644 --- a/docs/my-website/docs/proxy/guardrails/litellm_content_filter.md +++ b/docs/my-website/docs/proxy/guardrails/litellm_content_filter.md @@ -3,10 +3,12 @@ import TabItem from '@theme/TabItem'; import Image from '@theme/IdealImage'; -# LiteLLM Content Filter +# LiteLLM Content Filter (Built-in Guardrails) **Built-in guardrail** for detecting and filtering sensitive information using regex patterns and keyword matching. No external dependencies required. +**When to use?** Good for cases which do not require an ML model to detect sensitive information. + ## Overview | Property | Details | @@ -56,6 +58,44 @@ Test examples: ### Step 1: Define Guardrails in config.yaml + + + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: "harmful-content-filter" + litellm_params: + guardrail: litellm_content_filter + mode: "pre_call" + + # Enable harmful content categories + categories: + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + - category: "harmful_violence" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + - category: "harmful_illegal_weapons" + enabled: true + action: "BLOCK" + severity_threshold: "medium" +``` + + + + + ```yaml showLineNumbers title="config.yaml" model_list: - model_name: gpt-3.5-turbo @@ -86,6 +126,48 @@ guardrails: description: "Sensitive internal information" ``` + + + + +```yaml showLineNumbers title="config.yaml" +model_list: + - model_name: gpt-3.5-turbo + litellm_params: + model: openai/gpt-3.5-turbo + api_key: os.environ/OPENAI_API_KEY + +guardrails: + - guardrail_name: "comprehensive-filter" + litellm_params: + guardrail: litellm_content_filter + mode: "pre_call" + + # Harmful content categories + categories: + - category: "harmful_violence" + enabled: true + action: "BLOCK" + severity_threshold: "high" + + # PII patterns + patterns: + - pattern_type: "prebuilt" + pattern_name: "us_ssn" + action: "BLOCK" + - pattern_type: "prebuilt" + pattern_name: "email" + action: "MASK" + + # Custom keywords + blocked_words: + - keyword: "confidential" + action: "BLOCK" +``` + + + + ### Step 2: Start LiteLLM Gateway ```shell @@ -363,9 +445,171 @@ Output: "Email ***EMAIL***, SSN ***US_SSN***, ***REDACTED*** data" - Pattern names are automatically uppercased (e.g., `email` → `EMAIL`) - `keyword_redaction_tag` is a fixed string (no placeholders) +## Content Categories + +Prebuilt categories use **keyword matching** to detect harmful content, bias, and inappropriate advice. Keywords are matched with word boundaries (single words) or as substrings (multi-word phrases), case-insensitive. + +### Available Categories + +| Category | Description | +|----------|-------------| +| **Harmful Content** | | +| `harmful_self_harm` | Self-harm, suicide, eating disorders | +| `harmful_violence` | Violence, criminal planning, attacks | +| `harmful_illegal_weapons` | Illegal weapons, explosives, dangerous materials | +| **Bias Detection** | | +| `bias_gender` | Gender-based discrimination, stereotypes | +| `bias_sexual_orientation` | LGBTQ+ discrimination, homophobia, transphobia | +| `bias_racial` | Racial/ethnic discrimination, stereotypes | +| `bias_religious` | Religious discrimination, stereotypes | +| **Denied Advice** | | +| `denied_financial_advice` | Personalized financial advice, investment recommendations | +| `denied_medical_advice` | Medical advice, diagnosis, treatment recommendations | +| `denied_legal_advice` | Legal advice, representation, legal strategy | + +:::info Bias Detection Considerations + +Bias detection is **complex and context-dependent**. Rule-based systems catch explicit discriminatory language but may generate false positives on legitimate discussions. Start with **high severity thresholds** and test thoroughly. For mission-critical bias detection, consider combining with AI-based guardrails (e.g., HiddenLayer, Lakera). + +::: + +### Configuration + +```yaml showLineNumbers title="config.yaml" +guardrails: + - guardrail_name: "content-filter" + litellm_params: + guardrail: litellm_content_filter + mode: "pre_call" + + categories: + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" # Blocks medium+ severity + + - category: "bias_gender" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit discrimination + + - category: "denied_financial_advice" + enabled: true + action: "BLOCK" + severity_threshold: "medium" +``` + +**Severity Thresholds:** +- `"high"` - Only blocks high severity items +- `"medium"` - Blocks medium and high severity (default) +- `"low"` - Blocks all severity levels + +### Custom Category Files + +Override default categories with custom keyword lists: + +```yaml showLineNumbers title="config.yaml" +categories: + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + category_file: "/path/to/custom.yaml" +``` + +```yaml showLineNumbers title="custom.yaml" +category_name: "harmful_self_harm" +description: "Custom self-harm detection" +default_action: "BLOCK" + +keywords: + - keyword: "suicide" + severity: "high" + - keyword: "harm myself" + severity: "high" + +exceptions: + - "suicide prevention" + - "mental health" +``` + ## Use Cases -### 1. PII Protection +### 1. Harmful Content Detection + +Block or detect requests containing harmful, illegal, or dangerous content: + +```yaml +categories: + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + - category: "harmful_violence" + enabled: true + action: "BLOCK" + severity_threshold: "high" + - category: "harmful_illegal_weapons" + enabled: true + action: "BLOCK" + severity_threshold: "medium" +``` + +### 2. Bias and Discrimination Detection + +Detect and block biased, discriminatory, or hateful content across multiple dimensions: + +```yaml +categories: + # Gender-based discrimination + - category: "bias_gender" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + # LGBTQ+ discrimination + - category: "bias_sexual_orientation" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + # Racial/ethnic discrimination + - category: "bias_racial" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit to reduce false positives + + # Religious discrimination + - category: "bias_religious" + enabled: true + action: "BLOCK" + severity_threshold: "medium" +``` + +**Sensitivity Tuning:** + +For bias detection, severity thresholds are critical to balance safety and legitimate discourse: + +```yaml +# Conservative (low false positives, may miss subtle bias) +categories: + - category: "bias_racial" + severity_threshold: "high" # Only blocks explicit discriminatory language + +# Balanced (recommended) +categories: + - category: "bias_gender" + severity_threshold: "medium" # Blocks stereotypes and explicit discrimination + +# Strict (high safety, may have more false positives) +categories: + - category: "bias_sexual_orientation" + severity_threshold: "low" # Blocks all potentially problematic content +``` + + + +### 3. PII Protection Block or mask personally identifiable information before sending to LLMs: ```yaml @@ -409,7 +653,54 @@ For large lists of sensitive terms, use a file: blocked_words_file: "/path/to/sensitive_terms.yaml" ``` -### 4. Compliance +### 4. Safe AI for Consumer Applications + +Combining harmful content and bias detection for consumer-facing AI: + +```yaml +guardrails: + - guardrail_name: "safe-consumer-ai" + litellm_params: + guardrail: litellm_content_filter + mode: "pre_call" + + categories: + # Harmful content - strict + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + - category: "harmful_violence" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + # Bias detection - balanced + - category: "bias_gender" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Avoid blocking legitimate gender discussions + + - category: "bias_sexual_orientation" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + + - category: "bias_racial" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Education and news may discuss race +``` + +**Perfect for:** +- Chatbots and virtual assistants +- Educational AI tools +- Customer service AI +- Content generation platforms +- Public-facing AI applications + +### 5. Compliance Ensure regulatory compliance by filtering sensitive data types: ```yaml @@ -422,8 +713,184 @@ patterns: action: "BLOCK" ``` +## Best Practices for Bias Detection + +### Choosing the Right Severity Threshold + +Bias detection requires careful tuning to avoid blocking legitimate content: + +**High Threshold (Recommended for most use cases)** +- Blocks only explicit discriminatory language +- Lower false positives +- Allows nuanced discussions about identity, diversity, and social issues +- Good for: Public-facing applications, education, research + +**Medium Threshold** +- Blocks stereotypes and generalizations +- Balanced approach +- May catch some edge cases in legitimate discourse +- Good for: Consumer applications, internal tools, moderated environments + +**Low Threshold** +- Strictest filtering +- Blocks even borderline language +- Higher false positives but maximum safety +- Good for: Youth-focused applications, highly controlled environments + +### Testing Your Bias Filters + +Always test with realistic use cases: + +```yaml +# Test legitimate discussions (should NOT be blocked) +- "Our company has a gender diversity initiative" +- "Research shows racial disparities in healthcare" +- "We support LGBTQ+ rights and equality" +- "Religious freedom is a fundamental right" + +# Test discriminatory content (SHOULD be blocked) +- "Women are too emotional to lead" +- "All [group] are [negative stereotype]" +- "Being gay is unnatural" +- "[Religious group] are all extremists" +``` + +### Monitoring and Iteration + +1. **Log blocked requests** to review false positives +2. **Add exceptions** for legitimate terms in your domain +3. **Adjust severity thresholds** based on your audience +4. **Use custom category files** for domain-specific bias patterns + +### Cultural and Linguistic Considerations + +The prebuilt categories focus on English and common patterns. For other languages or cultural contexts: + +1. Create custom category files with region-specific terms +2. Consult with native speakers and cultural experts +3. Include local slurs and stereotypes +4. Adjust severity based on regional norms + ## Troubleshooting +### False Positives with Bias Detection + +**Issue:** Legitimate discussions about diversity, identity, or social issues are being blocked + +**Solutions:** + +1. **Raise severity threshold:** +```yaml +categories: + - category: "bias_racial" + severity_threshold: "high" # Only explicit discrimination +``` + +2. **Add domain-specific exceptions:** +```yaml +# my_custom_bias_gender.yaml +exceptions: + - "gender pay gap" + - "gender diversity" + - "women in tech" + - "gender equality" + - "dei initiative" + - "inclusion program" +``` + +3. **Review what was blocked:** +Check error details to understand what triggered the block: +```json +{ + "error": "Content blocked: bias_gender category keyword 'women' detected (severity: medium)", + "category": "bias_gender", + "keyword": "women", + "severity": "medium" +} +``` + +If this is a false positive, add "women in leadership" or other legitimate phrases to exceptions in your custom category file. + +### False Positives with Categories + +**Issue:** Legitimate content is being blocked by category filters + +**Solution 1:** Adjust severity threshold to only block high-severity items: +```yaml +categories: + - category: "harmful_violence" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only block explicit harmful content +``` + +**Solution 2:** Add exceptions to your custom category file: +```yaml +# my_custom_violence.yaml +exceptions: + - "crime statistics" + - "documentary" + - "news report" + - "historical context" +``` + +**Solution 3:** Use a custom category file with your own curated keyword list: +```yaml +categories: + - category: "harmful_violence" + enabled: true + action: "BLOCK" + category_file: "/path/to/my_violence_keywords.yaml" +``` + +### Category Not Loading + +**Issue:** Category is not being applied + +**Checklist:** +1. Verify category is enabled: `enabled: true` +2. Check category name matches file: `harmful_self_harm.yaml` +3. Check file exists in `litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/` +4. Review logs for loading errors: `litellm --config config.yaml --detailed_debug` + +### Keyword Not Matching + +**Issue:** Expected keyword is not being detected + +**Solutions:** + +1. **For single words:** Ensure the keyword appears as a whole word. The system uses word boundary matching, so "men" won't match "recommend". + +2. **For multi-word phrases:** Use the exact phrase as it should appear. Multi-word keywords are matched as substrings (case-insensitive), so "harm myself" will match "I want to harm myself" or "harming myself". + +3. **Check exceptions:** If your keyword is in the exceptions list, it won't be detected. Review the category file's exceptions section. + +4. **Verify severity threshold:** Lower severity keywords won't match if your threshold is set too high. For example, if a keyword has `severity: "low"` but your `severity_threshold: "high"`, it won't match. + +### Too Many False Negatives + +**Issue:** Harmful content is not being caught + +**Solution 1:** Lower severity threshold: +```yaml +severity_threshold: "low" # Catch more but may increase false positives +``` + +**Solution 2:** Add custom keywords for your specific use case: +```yaml +categories: + - category: "harmful_violence" + enabled: true + category_file: "/path/to/enhanced_violence.yaml" + +# In enhanced_violence.yaml, add domain-specific keywords +keywords: + - keyword: "your specific harmful phrase" + severity: "high" + - keyword: "another harmful term" + severity: "medium" +``` + ### Pattern Not Matching **Issue:** Regex pattern isn't detecting expected content @@ -440,16 +907,23 @@ print(re.search(pattern, test_text)) # Should match **Issue:** Text contains multiple sensitive patterns -**Solution:** First matching pattern/keyword is processed. Order patterns by priority: +**Solution:** Guardrail checks in this order: categories (keywords), regex patterns, then blocked words. Order by priority: ```yaml +# Categories checked first (high priority) +# Category keywords are matched first +categories: + - category: "harmful_self_harm" + severity_threshold: "high" + +# Then regex patterns patterns: - # Most critical first - pattern_type: "prebuilt" pattern_name: "us_ssn" action: "BLOCK" - # Less critical - - pattern_type: "prebuilt" - pattern_name: "email" + +# Then simple blocked keywords +blocked_words: + - keyword: "confidential" action: "MASK" ``` diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 0044bf65471..3d4c6580a86 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -52,6 +52,7 @@ const sidebars = { ] }, "proxy/guardrails/test_playground", + "proxy/guardrails/litellm_content_filter", ...[ "proxy/guardrails/aim_security", "proxy/guardrails/onyx_security", @@ -63,7 +64,6 @@ const sidebars = { "proxy/guardrails/grayswan", "proxy/guardrails/hiddenlayer", "proxy/guardrails/lasso_security", - "proxy/guardrails/litellm_content_filter", "proxy/guardrails/guardrails_ai", "proxy/guardrails/lakera_ai", "proxy/guardrails/model_armor", diff --git a/litellm/proxy/_experimental/out/api-reference.html b/litellm/proxy/_experimental/out/api-reference/index.html similarity index 100% rename from litellm/proxy/_experimental/out/api-reference.html rename to litellm/proxy/_experimental/out/api-reference/index.html diff --git a/litellm/proxy/_experimental/out/experimental/api-playground.html b/litellm/proxy/_experimental/out/experimental/api-playground/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/api-playground.html rename to litellm/proxy/_experimental/out/experimental/api-playground/index.html diff --git a/litellm/proxy/_experimental/out/experimental/budgets.html b/litellm/proxy/_experimental/out/experimental/budgets/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/budgets.html rename to litellm/proxy/_experimental/out/experimental/budgets/index.html diff --git a/litellm/proxy/_experimental/out/experimental/caching.html b/litellm/proxy/_experimental/out/experimental/caching/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/caching.html rename to litellm/proxy/_experimental/out/experimental/caching/index.html diff --git a/litellm/proxy/_experimental/out/experimental/old-usage.html b/litellm/proxy/_experimental/out/experimental/old-usage/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/old-usage.html rename to litellm/proxy/_experimental/out/experimental/old-usage/index.html diff --git a/litellm/proxy/_experimental/out/experimental/prompts.html b/litellm/proxy/_experimental/out/experimental/prompts/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/prompts.html rename to litellm/proxy/_experimental/out/experimental/prompts/index.html diff --git a/litellm/proxy/_experimental/out/experimental/tag-management.html b/litellm/proxy/_experimental/out/experimental/tag-management/index.html similarity index 100% rename from litellm/proxy/_experimental/out/experimental/tag-management.html rename to litellm/proxy/_experimental/out/experimental/tag-management/index.html diff --git a/litellm/proxy/_experimental/out/guardrails.html b/litellm/proxy/_experimental/out/guardrails.html deleted file mode 100644 index 0d14de33739..00000000000 --- a/litellm/proxy/_experimental/out/guardrails.html +++ /dev/null @@ -1 +0,0 @@ -LiteLLM Dashboard \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/login.html b/litellm/proxy/_experimental/out/login/index.html similarity index 100% rename from litellm/proxy/_experimental/out/login.html rename to litellm/proxy/_experimental/out/login/index.html diff --git a/litellm/proxy/_experimental/out/logs.html b/litellm/proxy/_experimental/out/logs/index.html similarity index 100% rename from litellm/proxy/_experimental/out/logs.html rename to litellm/proxy/_experimental/out/logs/index.html diff --git a/litellm/proxy/_experimental/out/mcp/oauth/callback.html b/litellm/proxy/_experimental/out/mcp/oauth/callback/index.html similarity index 100% rename from litellm/proxy/_experimental/out/mcp/oauth/callback.html rename to litellm/proxy/_experimental/out/mcp/oauth/callback/index.html diff --git a/litellm/proxy/_experimental/out/model-hub.html b/litellm/proxy/_experimental/out/model-hub/index.html similarity index 100% rename from litellm/proxy/_experimental/out/model-hub.html rename to litellm/proxy/_experimental/out/model-hub/index.html diff --git a/litellm/proxy/_experimental/out/model_hub_table.html b/litellm/proxy/_experimental/out/model_hub_table/index.html similarity index 100% rename from litellm/proxy/_experimental/out/model_hub_table.html rename to litellm/proxy/_experimental/out/model_hub_table/index.html diff --git a/litellm/proxy/_experimental/out/models-and-endpoints.html b/litellm/proxy/_experimental/out/models-and-endpoints/index.html similarity index 100% rename from litellm/proxy/_experimental/out/models-and-endpoints.html rename to litellm/proxy/_experimental/out/models-and-endpoints/index.html diff --git a/litellm/proxy/_experimental/out/onboarding.html b/litellm/proxy/_experimental/out/onboarding.html deleted file mode 100644 index e47fae11884..00000000000 --- a/litellm/proxy/_experimental/out/onboarding.html +++ /dev/null @@ -1 +0,0 @@ -LiteLLM Dashboard \ No newline at end of file diff --git a/litellm/proxy/_experimental/out/organizations.html b/litellm/proxy/_experimental/out/organizations/index.html similarity index 100% rename from litellm/proxy/_experimental/out/organizations.html rename to litellm/proxy/_experimental/out/organizations/index.html diff --git a/litellm/proxy/_experimental/out/playground.html b/litellm/proxy/_experimental/out/playground/index.html similarity index 100% rename from litellm/proxy/_experimental/out/playground.html rename to litellm/proxy/_experimental/out/playground/index.html diff --git a/litellm/proxy/_experimental/out/settings/admin-settings.html b/litellm/proxy/_experimental/out/settings/admin-settings/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/admin-settings.html rename to litellm/proxy/_experimental/out/settings/admin-settings/index.html diff --git a/litellm/proxy/_experimental/out/settings/logging-and-alerts.html b/litellm/proxy/_experimental/out/settings/logging-and-alerts/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/logging-and-alerts.html rename to litellm/proxy/_experimental/out/settings/logging-and-alerts/index.html diff --git a/litellm/proxy/_experimental/out/settings/router-settings.html b/litellm/proxy/_experimental/out/settings/router-settings/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/router-settings.html rename to litellm/proxy/_experimental/out/settings/router-settings/index.html diff --git a/litellm/proxy/_experimental/out/settings/ui-theme.html b/litellm/proxy/_experimental/out/settings/ui-theme/index.html similarity index 100% rename from litellm/proxy/_experimental/out/settings/ui-theme.html rename to litellm/proxy/_experimental/out/settings/ui-theme/index.html diff --git a/litellm/proxy/_experimental/out/teams.html b/litellm/proxy/_experimental/out/teams/index.html similarity index 100% rename from litellm/proxy/_experimental/out/teams.html rename to litellm/proxy/_experimental/out/teams/index.html diff --git a/litellm/proxy/_experimental/out/test-key.html b/litellm/proxy/_experimental/out/test-key/index.html similarity index 100% rename from litellm/proxy/_experimental/out/test-key.html rename to litellm/proxy/_experimental/out/test-key/index.html diff --git a/litellm/proxy/_experimental/out/tools/mcp-servers.html b/litellm/proxy/_experimental/out/tools/mcp-servers/index.html similarity index 100% rename from litellm/proxy/_experimental/out/tools/mcp-servers.html rename to litellm/proxy/_experimental/out/tools/mcp-servers/index.html diff --git a/litellm/proxy/_experimental/out/tools/vector-stores.html b/litellm/proxy/_experimental/out/tools/vector-stores/index.html similarity index 100% rename from litellm/proxy/_experimental/out/tools/vector-stores.html rename to litellm/proxy/_experimental/out/tools/vector-stores/index.html diff --git a/litellm/proxy/_experimental/out/usage.html b/litellm/proxy/_experimental/out/usage/index.html similarity index 100% rename from litellm/proxy/_experimental/out/usage.html rename to litellm/proxy/_experimental/out/usage/index.html diff --git a/litellm/proxy/_experimental/out/users.html b/litellm/proxy/_experimental/out/users/index.html similarity index 100% rename from litellm/proxy/_experimental/out/users.html rename to litellm/proxy/_experimental/out/users/index.html diff --git a/litellm/proxy/_experimental/out/virtual-keys.html b/litellm/proxy/_experimental/out/virtual-keys/index.html similarity index 100% rename from litellm/proxy/_experimental/out/virtual-keys.html rename to litellm/proxy/_experimental/out/virtual-keys/index.html diff --git a/litellm/proxy/_new_secret_config.yaml b/litellm/proxy/_new_secret_config.yaml index 322021e4e05..e8074c82512 100644 --- a/litellm/proxy/_new_secret_config.yaml +++ b/litellm/proxy/_new_secret_config.yaml @@ -21,6 +21,61 @@ model_list: # api_base: http://localhost:8080 # default_on: true +guardrails: + - guardrail_name: "harmful-content-filter" + litellm_params: + guardrail: litellm_content_filter + mode: "pre_call" + default_on: true + # Model configuration + guardrail_model: + - model: "gpt-4o" # From model_list + supported_multimodal_content: # Supported content types = images, documents, audio, video, text + - images + - documents + + categories: + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" # Block medium+ + + - category: "harmful_violence" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit + + - category: "harmful_illegal_weapons" + enabled: true + action: "BLOCK" + severity_threshold: "low" # Strictest + + - category: "bias_gender" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit to reduce false positives + + - category: "bias_sexual_orientation" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit to reduce false positives + + - category: "denied_medical_advice" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit to reduce false positives + + - category: "denied_legal_advice" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit to reduce false positives + + - category: "denied_financial_advice" + enabled: true + action: "BLOCK" + severity_threshold: "high" # Only explicit to reduce false positives + + prompts: - prompt_id: "simple_prompt" litellm_params: diff --git a/litellm/proxy/guardrails/guardrail_endpoints.py b/litellm/proxy/guardrails/guardrail_endpoints.py index bb78383ce44..3ce819439cb 100644 --- a/litellm/proxy/guardrails/guardrail_endpoints.py +++ b/litellm/proxy/guardrails/guardrail_endpoints.py @@ -699,6 +699,7 @@ async def get_guardrail_ui_settings(): """ from litellm.proxy.guardrails.guardrail_hooks.litellm_content_filter.patterns import ( PATTERN_CATEGORIES, + get_available_content_categories, get_pattern_metadata, ) @@ -721,10 +722,56 @@ async def get_guardrail_ui_settings(): "prebuilt_patterns": get_pattern_metadata(), "pattern_categories": list(PATTERN_CATEGORIES.keys()), "supported_actions": ["BLOCK", "MASK"], + "content_categories": get_available_content_categories(), }, ) +@router.get( + "/guardrails/ui/category_yaml/{category_name}", + tags=["Guardrails"], + dependencies=[Depends(user_api_key_auth)], +) +async def get_category_yaml(category_name: str): + """ + Get the YAML content for a specific content filter category. + + Args: + category_name: The name of the category (e.g., "bias_gender", "harmful_self_harm") + + Returns: + The raw YAML content of the category file + """ + import os + + # Get the categories directory path + categories_dir = os.path.join( + os.path.dirname(__file__), + "guardrail_hooks", + "litellm_content_filter", + "categories", + ) + + # Construct the file path + category_file_path = os.path.join(categories_dir, f"{category_name}.yaml") + + if not os.path.exists(category_file_path): + raise HTTPException( + status_code=404, detail=f"Category file not found: {category_name}" + ) + + try: + # Read and return the raw YAML content + with open(category_file_path, "r") as f: + yaml_content = f.read() + + return {"category_name": category_name, "yaml_content": yaml_content} + except Exception as e: + raise HTTPException( + status_code=500, detail=f"Error reading category file: {str(e)}" + ) + + @router.post( "/guardrails/validate_blocked_words_file", tags=["Guardrails"], diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py index 89bb53ef72b..d7f06cf6a53 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/__init__.py @@ -13,18 +13,18 @@ if TYPE_CHECKING: def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"): """ Initialize the Content Filter Guardrail. - + Args: litellm_params: Guardrail configuration parameters guardrail: Guardrail metadata - + Returns: Initialized ContentFilterGuardrail instance """ guardrail_name = guardrail.get("guardrail_name") if not guardrail_name: raise ValueError("Content Filter: guardrail_name is required") - + content_filter_guardrail = ContentFilterGuardrail( guardrail_name=guardrail_name, patterns=litellm_params.patterns, @@ -32,12 +32,12 @@ def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail" blocked_words_file=litellm_params.blocked_words_file, event_hook=litellm_params.mode, # type: ignore default_on=litellm_params.default_on or False, + categories=getattr(litellm_params, "categories", None), + severity_threshold=getattr(litellm_params, "severity_threshold", "medium"), ) - - litellm.logging_callback_manager.add_litellm_callback( - content_filter_guardrail - ) - + + litellm.logging_callback_manager.add_litellm_callback(content_filter_guardrail) + return content_filter_guardrail @@ -49,4 +49,3 @@ guardrail_initializer_registry = { guardrail_class_registry = { SupportedGuardrailIntegrations.LITELLM_CONTENT_FILTER.value: ContentFilterGuardrail, } - diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_gender.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_gender.yaml new file mode 100644 index 00000000000..fbc164733b8 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_gender.yaml @@ -0,0 +1,53 @@ +# Gender-based bias and discrimination detection +category_name: "bias_gender" +description: "Detects gender-based discriminatory language, stereotypes, and biased content" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - gender identity terms + - keyword: "women" + severity: "high" + - keyword: "woman" + severity: "high" + - keyword: "men" + severity: "high" + - keyword: "man" + severity: "high" + - keyword: "female" + severity: "high" + - keyword: "females" + severity: "high" + - keyword: "male" + severity: "high" + - keyword: "males" + severity: "high" + - keyword: "girl" + severity: "high" + - keyword: "girls" + severity: "high" + - keyword: "boy" + severity: "high" + - keyword: "boys" + severity: "high" + +# Exceptions - legitimate discussions about gender +exceptions: + - "gender equality" + - "gender diversity" + - "gender studies" + - "gender gap" + - "gender discrimination" + - "combat gender" + - "address gender" + - "research shows" + - "study found" + - "gender identity" + - "gender expression" + - "transgender" + - "gender neutral" + - "women's rights" + - "women's health" + - "men's health" + - "gender bias" + - "gender equity" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_racial.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_racial.yaml new file mode 100644 index 00000000000..86d9182e83d --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_racial.yaml @@ -0,0 +1,148 @@ +# Racial and ethnic bias detection +category_name: "bias_racial" +description: "Detects racial and ethnic discrimination, stereotypes, and biased content" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - racial/ethnic identity terms + - keyword: "black" + severity: "high" + - keyword: "blacks" + severity: "high" + - keyword: "white" + severity: "high" + - keyword: "whites" + severity: "high" + - keyword: "asian" + severity: "high" + - keyword: "asians" + severity: "high" + - keyword: "hispanic" + severity: "high" + - keyword: "hispanics" + severity: "high" + - keyword: "latino" + severity: "high" + - keyword: "latina" + severity: "high" + - keyword: "latinos" + severity: "high" + - keyword: "mexican" + severity: "high" + - keyword: "mexicans" + severity: "high" + - keyword: "chinese" + severity: "high" + - keyword: "indian" + severity: "high" + - keyword: "indians" + severity: "high" + - keyword: "arab" + severity: "high" + - keyword: "arabs" + severity: "high" + - keyword: "african" + severity: "high" + - keyword: "africans" + severity: "high" + - keyword: "immigrant" + severity: "high" + - keyword: "immigrants" + severity: "high" + - keyword: "refugee" + severity: "high" + - keyword: "refugees" + severity: "high" + - keyword: "minority" + severity: "high" + - keyword: "minorities" + severity: "high" + + # High severity - discriminatory terms + - keyword: "inferior race" + severity: "high" + - keyword: "superior race" + severity: "high" + - keyword: "racial purity" + severity: "high" + - keyword: "master race" + severity: "high" + - keyword: "white supremacy" + severity: "high" + - keyword: "white genocide" + severity: "high" + - keyword: "great replacement" + severity: "high" + - keyword: "race traitor" + severity: "high" + - keyword: "race mixing" + severity: "high" + - keyword: "model minority" + severity: "high" + - keyword: "ghetto culture" + severity: "high" + - keyword: "thug culture" + severity: "high" + - keyword: "diversity hire" + severity: "high" + - keyword: "black crime" + severity: "high" + - keyword: "immigrant crime" + severity: "high" + - keyword: "minority lazy" + severity: "high" + - keyword: "stealing jobs" + severity: "high" + - keyword: "go back" + severity: "high" + - keyword: "you people" + severity: "medium" + - keyword: "those people" + severity: "medium" + - keyword: "all blacks" + severity: "high" + - keyword: "all whites" + severity: "high" + - keyword: "all asians" + severity: "high" + - keyword: "all hispanics" + severity: "high" + - keyword: "all latinos" + severity: "high" + - keyword: "all mexicans" + severity: "high" + - keyword: "all immigrants" + severity: "high" + +# Exceptions - legitimate discussions about race, diversity, anti-racism +exceptions: + - "racial equality" + - "racial justice" + - "racial discrimination" + - "anti-racism" + - "combat racism" + - "racial bias" + - "systemic racism" + - "structural racism" + - "racial equity" + - "diversity and inclusion" + - "black lives matter" + - "civil rights" + - "fight racism" + - "address racism" + - "racial disparities" + - "racism is" + - "racist" + - "study shows" + - "research indicates" + - "historical" + - "black history" + - "black culture" + - "asian culture" + - "hispanic culture" + - "latino culture" + - "immigrant rights" + - "immigrant support" + - "protect immigrants" + - "minority rights" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_religious.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_religious.yaml new file mode 100644 index 00000000000..fe1da6ae464 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_religious.yaml @@ -0,0 +1,118 @@ +# Religious bias and discrimination detection +category_name: "bias_religious" +description: "Detects religious discrimination, stereotypes, and biased content" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - religion terms that could be used discriminatorily + - keyword: "muslim" + severity: "high" + - keyword: "muslims" + severity: "high" + - keyword: "islam" + severity: "high" + - keyword: "islamic" + severity: "high" + - keyword: "jew" + severity: "high" + - keyword: "jews" + severity: "high" + - keyword: "jewish" + severity: "high" + - keyword: "christian" + severity: "high" + - keyword: "christians" + severity: "high" + - keyword: "christianity" + severity: "high" + - keyword: "atheist" + severity: "high" + - keyword: "atheists" + severity: "high" + - keyword: "hindu" + severity: "high" + - keyword: "hindus" + severity: "high" + - keyword: "hinduism" + severity: "high" + - keyword: "buddhist" + severity: "high" + - keyword: "buddhists" + severity: "high" + - keyword: "buddhism" + severity: "high" + - keyword: "sikh" + severity: "high" + - keyword: "sikhs" + severity: "high" + - keyword: "catholic" + severity: "high" + - keyword: "catholics" + severity: "high" + - keyword: "protestant" + severity: "high" + - keyword: "protestants" + severity: "high" + - keyword: "mormon" + severity: "high" + - keyword: "mormons" + severity: "high" + + + # Medium severity - context-dependent + - keyword: "religious extremist" + severity: "medium" + - keyword: "religious fanatic" + severity: "medium" + - keyword: "infidel" + severity: "medium" + - keyword: "infidels" + severity: "medium" + - keyword: "jihad" + severity: "medium" + - keyword: "crusade" + severity: "medium" + +# Exceptions - legitimate religious discussion, education, interfaith work +exceptions: + - "religious freedom" + - "religious tolerance" + - "religious discrimination" + - "religious persecution" + - "interfaith" + - "religious studies" + - "religious history" + - "combat islamophobia" + - "fight antisemitism" + - "religious diversity" + - "freedom of religion" + - "religious equality" + - "anti-muslim bias" + - "anti-semitism" + - "antisemitism" + - "religious bigotry" + - "theological" + - "scripture" + - "comparative religion" + - "protect muslims" + - "protect jews" + - "protect christians" + - "muslim rights" + - "jewish rights" + - "christian rights" + - "muslim community" + - "jewish community" + - "christian community" + - "muslim culture" + - "jewish culture" + - "christian culture" + - "islam teaches" + - "judaism teaches" + - "christianity teaches" + - "muslim faith" + - "jewish faith" + - "christian faith" + - "muslim americans" + - "jewish americans" + - "christian americans" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_sexual_orientation.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_sexual_orientation.yaml new file mode 100644 index 00000000000..60721e761db --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/bias_sexual_orientation.yaml @@ -0,0 +1,251 @@ +# Sexual orientation bias and discrimination detection +category_name: "bias_sexual_orientation" +description: "Detects discriminatory language targeting LGBTQ+ individuals and communities" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # HIGH SEVERITY - Explicit slurs (always block regardless of context) + - keyword: "faggot" + severity: "high" + - keyword: "fag" + severity: "high" + - keyword: "dyke" + severity: "high" + - keyword: "tranny" + severity: "high" + - keyword: "shemale" + severity: "high" + - keyword: "sodomite" + severity: "high" + - keyword: "homo" + severity: "high" + - keyword: "f*ggot" + severity: "high" + - keyword: "f@ggot" + severity: "high" + - keyword: "gay" + severity: "high" + + # HIGH SEVERITY - Always discriminatory phrases + - keyword: "gay agenda" + severity: "high" + - keyword: "homosexual agenda" + severity: "high" + - keyword: "conversion therapy" + severity: "high" + - keyword: "reparative therapy" + severity: "high" + - keyword: "groomer" + severity: "high" + - keyword: "lifestyle choice" + severity: "high" + - keyword: "gay lifestyle" + severity: "high" + - keyword: "homosexual lifestyle" + severity: "high" + - keyword: "choose to be gay" + severity: "high" + - keyword: "gay is a choice" + severity: "high" + - keyword: "homosexuality is a choice" + severity: "high" + - keyword: "gay is unnatural" + severity: "high" + - keyword: "gay is wrong" + severity: "high" + - keyword: "gay is a sin" + severity: "high" + - keyword: "gay is sinful" + severity: "high" + - keyword: "gay is a disease" + severity: "high" + - keyword: "gay is an illness" + severity: "high" + - keyword: "gay is a disorder" + severity: "high" + - keyword: "gay is abnormal" + severity: "high" + - keyword: "gay is deviant" + severity: "high" + - keyword: "gay is perverted" + severity: "high" + - keyword: "gay is a perversion" + severity: "high" + - keyword: "gay is degenerate" + severity: "high" + - keyword: "gay is an abomination" + severity: "high" + - keyword: "gay is disgusting" + severity: "high" + - keyword: "gay is immoral" + severity: "high" + - keyword: "gay is evil" + severity: "high" + - keyword: "gay is sick" + severity: "high" + - keyword: "gay is twisted" + severity: "high" + - keyword: "lesbian is unnatural" + severity: "high" + - keyword: "lesbian is wrong" + severity: "high" + - keyword: "lesbian is a sin" + severity: "high" + - keyword: "lesbian is abnormal" + severity: "high" + - keyword: "bisexual is confused" + severity: "high" + - keyword: "bisexual is fake" + severity: "high" + - keyword: "transgender is unnatural" + severity: "high" + - keyword: "transgender is wrong" + severity: "high" + - keyword: "transgender is a sin" + severity: "high" + - keyword: "transgender is a disease" + severity: "high" + - keyword: "transgender is a disorder" + severity: "high" + - keyword: "transgender is abnormal" + severity: "high" + - keyword: "transgender is fake" + severity: "high" + - keyword: "transgender is confused" + severity: "high" + - keyword: "transgender is a delusion" + severity: "high" + - keyword: "transgender is mental illness" + severity: "high" + - keyword: "trans is unnatural" + severity: "high" + - keyword: "trans is wrong" + severity: "high" + - keyword: "trans is fake" + severity: "high" + - keyword: "trans is confused" + severity: "high" + - keyword: "trans is a delusion" + severity: "high" + - keyword: "trans is mental" + severity: "high" + - keyword: "queer is unnatural" + severity: "high" + - keyword: "queer is wrong" + severity: "high" + - keyword: "homosexual is unnatural" + severity: "high" + - keyword: "homosexual is wrong" + severity: "high" + - keyword: "homosexual is a sin" + severity: "high" + - keyword: "homosexual is a disease" + severity: "high" + - keyword: "homosexual is abnormal" + severity: "high" + - keyword: "lgbtq is unnatural" + severity: "high" + - keyword: "lgbtq is wrong" + severity: "high" + - keyword: "lgbtq is a sin" + severity: "high" + - keyword: "cure gay" + severity: "high" + - keyword: "cure homosexuality" + severity: "high" + - keyword: "cure transgender" + severity: "high" + - keyword: "fix gay" + severity: "high" + - keyword: "change gay" + severity: "high" + - keyword: "convert gay" + severity: "high" + - keyword: "stop being gay" + severity: "high" + - keyword: "prevent gay" + severity: "high" + - keyword: "eliminate gay" + severity: "high" + - keyword: "eradicate gay" + severity: "high" + - keyword: "ban gay" + severity: "high" + - keyword: "punish gay" + severity: "high" + - keyword: "cure lesbian" + severity: "high" + - keyword: "fix lesbian" + severity: "high" + - keyword: "change lesbian" + severity: "high" + - keyword: "cure trans" + severity: "high" + - keyword: "fix trans" + severity: "high" + - keyword: "change trans" + severity: "high" + - keyword: "cure lgbtq" + severity: "high" + - keyword: "fix lgbtq" + severity: "high" + - keyword: "all gays are" + severity: "high" + - keyword: "all lesbians are" + severity: "high" + - keyword: "all transgender are" + severity: "high" + - keyword: "all trans are" + severity: "high" + - keyword: "all lgbtq are" + severity: "high" + - keyword: "gays are pedophiles" + severity: "high" + - keyword: "gays are groomers" + severity: "high" + - keyword: "gays are perverts" + severity: "high" + - keyword: "gays are sinners" + severity: "high" + - keyword: "lesbians are perverts" + severity: "high" + - keyword: "transgender are predators" + severity: "high" + - keyword: "trans are predators" + severity: "high" + - keyword: "trans are groomers" + severity: "high" + - keyword: "trans women are men" + severity: "high" + - keyword: "trans men are women" + severity: "high" + - keyword: "not real women" + severity: "medium" + - keyword: "not real men" + severity: "medium" + +# Exceptions - legitimate discussions +exceptions: + - "lgbtq rights" + - "lgbtq equality" + - "lgbtq community" + - "lgbtq support" + - "lgbtq advocacy" + - "sexual orientation" + - "coming out" + - "pride" + - "marriage equality" + - "anti-lgbtq" + - "discrimination against" + - "protect lgbtq" + - "lgbtq history" + - "gender identity" + - "conversion therapy ban" + - "harmful practice" + - "fight discrimination" + - "lgbtq acceptance" + - "lgbtq inclusion" + - "support lgbtq" + - "lgbtq youth" + - "lgbtq healthcare" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_financial_advice.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_financial_advice.yaml new file mode 100644 index 00000000000..14f7b394e62 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_financial_advice.yaml @@ -0,0 +1,139 @@ +# Financial advice and investment guidance detection +category_name: "denied_financial_advice" +description: "Detects requests for personalized financial advice, investment recommendations, or financial planning that should be provided by licensed financial advisors" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - core financial terms + - keyword: "invest" + severity: "high" + - keyword: "investing" + severity: "high" + - keyword: "investment" + severity: "high" + - keyword: "investments" + severity: "high" + - keyword: "stock" + severity: "high" + - keyword: "stocks" + severity: "high" + - keyword: "portfolio" + severity: "high" + - keyword: "crypto" + severity: "high" + - keyword: "cryptocurrency" + severity: "high" + - keyword: "bitcoin" + severity: "high" + - keyword: "ethereum" + severity: "high" + - keyword: "trading" + severity: "high" + - keyword: "trade" + severity: "high" + - keyword: "trader" + severity: "high" + - keyword: "retirement" + severity: "high" + - keyword: "401k" + severity: "high" + - keyword: "ira" + severity: "high" + - keyword: "roth" + severity: "high" + - keyword: "mortgage" + severity: "high" + - keyword: "refinance" + severity: "high" + - keyword: "loan" + severity: "high" + - keyword: "loans" + severity: "high" + - keyword: "debt" + severity: "high" + - keyword: "tax" + severity: "high" + - keyword: "taxes" + severity: "high" + - keyword: "etf" + severity: "high" + - keyword: "bond" + severity: "high" + - keyword: "bonds" + severity: "high" + - keyword: "mutual" + severity: "high" + - keyword: "forex" + severity: "high" + - keyword: "futures" + severity: "high" + - keyword: "diversify" + severity: "high" + - keyword: "diversification" + severity: "high" + +# Exceptions - legitimate financial discussions +exceptions: + - "consult a financial advisor" + - "consult your financial advisor" + - "speak with financial advisor" + - "hire financial advisor" + - "seek financial advice" + - "financial professional" + - "licensed financial advisor" + - "certified financial planner" + - "financial consultant" + - "investment professional" + - "tax professional" + - "certified public accountant" + - "speak to a professional" + - "talk to a professional" + - "cpa" + - "tax preparer" + - "financial education" + - "financial literacy" + - "personal finance education" + - "investment education" + - "general financial information" + - "general information" + - "educational purposes" + - "for educational purposes" + - "not financial advice" + - "not investment advice" + - "this is not financial advice" + - "this is not investment advice" + - "not a substitute for" + - "financial disclaimer" + - "investment disclaimer" + - "financial research" + - "market research" + - "economic research" + - "financial analysis" + - "market analysis" + - "financial news" + - "market news" + - "economic news" + - "financial history" + - "investment history" + - "market trends" + - "economic trends" + - "financial concepts" + - "investment concepts" + - "financial terminology" + - "investment terminology" + - "stock market basics" + - "investment basics" + - "finance 101" + - "budgeting basics" + - "saving tips" + - "general tips" + - "debt reduction strategies" + - "credit score information" + - "how does" + - "what is" + - "what are" + - "explain" + - "definition of" + - "means" + diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_legal_advice.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_legal_advice.yaml new file mode 100644 index 00000000000..fe47c570033 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_legal_advice.yaml @@ -0,0 +1,137 @@ +# Legal advice and representation detection +category_name: "denied_legal_advice" +description: "Detects requests for legal advice, representation, or legal strategy that should be provided by licensed attorneys" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - core legal terms + - keyword: "lawyer" + severity: "high" + - keyword: "attorney" + severity: "high" + - keyword: "lawsuit" + severity: "high" + - keyword: "sue" + severity: "high" + - keyword: "suing" + severity: "high" + - keyword: "court" + severity: "high" + - keyword: "trial" + severity: "high" + - keyword: "case" + severity: "high" + - keyword: "contract" + severity: "high" + - keyword: "litigation" + severity: "high" + - keyword: "plead" + severity: "high" + - keyword: "guilty" + severity: "high" + - keyword: "divorce" + severity: "high" + - keyword: "custody" + severity: "high" + - keyword: "immigration" + severity: "high" + - keyword: "visa" + severity: "high" + - keyword: "asylum" + severity: "high" + - keyword: "deportation" + severity: "high" + - keyword: "criminal" + severity: "high" + - keyword: "charges" + severity: "high" + - keyword: "arrest" + severity: "high" + - keyword: "warrant" + severity: "high" + - keyword: "sentence" + severity: "high" + - keyword: "prosecution" + severity: "high" + - keyword: "bankruptcy" + severity: "high" + - keyword: "patent" + severity: "high" + - keyword: "trademark" + severity: "high" + - keyword: "copyright" + severity: "high" + - keyword: "settlement" + severity: "high" + - keyword: "defendant" + severity: "high" + - keyword: "plaintiff" + severity: "high" + - keyword: "testimony" + severity: "high" + +# Exceptions - legitimate legal discussions +exceptions: + - "consult a lawyer" + - "consult an attorney" + - "consult your lawyer" + - "consult your attorney" + - "hire a lawyer" + - "hire an attorney" + - "find a lawyer" + - "find an attorney" + - "seek legal counsel" + - "seek legal advice" + - "get legal advice" + - "legal professional" + - "qualified attorney" + - "licensed lawyer" + - "licensed attorney" + - "legal representation" + - "retain counsel" + - "contact a lawyer" + - "contact an attorney" + - "speak with attorney" + - "speak with lawyer" + - "talk to a lawyer" + - "talk to an attorney" + - "legal consultation" + - "attorney consultation" + - "legal education" + - "legal studies" + - "law school" + - "legal research" + - "legal terminology" + - "legal terms" + - "legal system" + - "court system" + - "legal process" + - "legal procedure" + - "general legal information" + - "general information" + - "educational purposes" + - "for educational purposes" + - "not legal advice" + - "this is not legal advice" + - "not a substitute for" + - "legal disclaimer" + - "legal history" + - "legal precedent" + - "case law" + - "supreme court" + - "constitutional law" + - "legal rights awareness" + - "know your rights" + - "civil rights" + - "human rights" + - "legal framework" + - "how does" + - "what is" + - "what are" + - "explain" + - "definition of" + - "means" + - "criminal justice system" + - "immigration system" + diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_medical_advice.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_medical_advice.yaml new file mode 100644 index 00000000000..d74631b9341 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/denied_medical_advice.yaml @@ -0,0 +1,133 @@ +# Medical advice and diagnosis detection +category_name: "denied_medical_advice" +description: "Detects requests for medical advice, diagnosis, or treatment recommendations that should be provided by licensed healthcare professionals" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - core medical terms + - keyword: "diagnose" + severity: "high" + - keyword: "diagnosis" + severity: "high" + - keyword: "doctor" + severity: "high" + - keyword: "physician" + severity: "high" + - keyword: "medication" + severity: "high" + - keyword: "medicine" + severity: "high" + - keyword: "prescription" + severity: "high" + - keyword: "prescribe" + severity: "high" + - keyword: "drug" + severity: "high" + - keyword: "drugs" + severity: "high" + - keyword: "treatment" + severity: "high" + - keyword: "treat" + severity: "high" + - keyword: "cure" + severity: "high" + - keyword: "surgery" + severity: "high" + - keyword: "symptoms" + severity: "high" + - keyword: "symptom" + severity: "high" + - keyword: "disease" + severity: "high" + - keyword: "illness" + severity: "high" + - keyword: "condition" + severity: "high" + - keyword: "cancer" + severity: "high" + - keyword: "diabetes" + severity: "high" + - keyword: "depression" + severity: "high" + - keyword: "anxiety" + severity: "high" + - keyword: "adhd" + severity: "high" + - keyword: "bipolar" + severity: "high" + - keyword: "psychiatric" + severity: "high" + - keyword: "vaccine" + severity: "high" + - keyword: "vaccination" + severity: "high" + - keyword: "dosage" + severity: "high" + - keyword: "dose" + severity: "high" + - keyword: "injury" + severity: "high" + - keyword: "treatment" + severity: "high" + - keyword: "injection" + severity: "high" + +# Exceptions - legitimate medical discussions +exceptions: + - "medical history" + - "medical research" + - "medical studies" + - "according to research" + - "research shows" + - "studies show" + - "consult a doctor" + - "consult your doctor" + - "see a doctor" + - "see your doctor" + - "talk to your doctor" + - "speak with your doctor" + - "ask your doctor" + - "visit your doctor" + - "contact your doctor" + - "call your doctor" + - "medical professional" + - "healthcare provider" + - "healthcare professional" + - "seek medical attention" + - "seek medical advice" + - "emergency medical" + - "call emergency" + - "call 911" + - "go to hospital" + - "visit hospital" + - "go to the emergency" + - "medical emergency" + - "general information" + - "general health information" + - "educational purposes" + - "for educational purposes" + - "not medical advice" + - "this is not medical advice" + - "not a substitute for" + - "medical documentation" + - "medical terminology" + - "medical terms" + - "anatomy and physiology" + - "medical science" + - "public health" + - "health education" + - "wellness information" + - "preventive care" + - "healthy lifestyle" + - "medical disclaimer" + - "how does" + - "what is" + - "what are" + - "explain" + - "definition of" + - "means" + - "cancer research" + - "diabetes research" + - "mental health awareness" + diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_illegal_weapons.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_illegal_weapons.yaml new file mode 100644 index 00000000000..3a5a28daf0c --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_illegal_weapons.yaml @@ -0,0 +1,299 @@ +# Illegal weapons and explosives content detection +category_name: "harmful_illegal_weapons" +description: "Detects content related to illegal weapons, explosives, and dangerous materials" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - explicit illegal weapons/explosives instructions + - keyword: "make a bomb" + severity: "high" + - keyword: "build a bomb" + severity: "high" + - keyword: "create explosive" + severity: "high" + - keyword: "homemade bomb" + severity: "high" + - keyword: "pipe bomb" + severity: "high" + - keyword: "pressure cooker bomb" + severity: "high" + - keyword: "fertilizer bomb" + severity: "high" + - keyword: "car bomb" + severity: "high" + - keyword: "suicide bomb" + severity: "high" + - keyword: "explosive device" + severity: "high" + - keyword: "bomb instructions" + severity: "high" + - keyword: "bomb recipe" + severity: "high" + - keyword: "bomb tutorial" + severity: "high" + - keyword: "homemade explosive" + severity: "high" + - keyword: "illegal firearm" + severity: "high" + - keyword: "unregistered gun" + severity: "high" + - keyword: "make poison" + severity: "high" + - keyword: "create poison" + severity: "high" + - keyword: "homemade poison" + severity: "high" + - keyword: "lethal poison" + severity: "high" + - keyword: "chemical weapon" + severity: "high" + - keyword: "biological weapon" + severity: "high" + - keyword: "ricin" + severity: "high" + - keyword: "cyanide poison" + severity: "high" + - keyword: "how to make a bomb" + severity: "high" + - keyword: "how to build a bomb" + severity: "high" + - keyword: "how to create explosive" + severity: "high" + - keyword: "bomb making instructions" + severity: "high" + - keyword: "bomb making tutorial" + severity: "high" + - keyword: "homemade explosive recipe" + severity: "high" + - keyword: "how to make poison" + severity: "high" + - keyword: "how to create poison" + severity: "high" + - keyword: "undetectable poison" + severity: "high" + - keyword: "untraceable weapon" + severity: "high" + - keyword: "3d print gun" + severity: "high" + - keyword: "3d printed gun" + severity: "high" + - keyword: "3d printing gun" + severity: "high" + - keyword: "make bomb" + severity: "high" + - keyword: "making bomb" + severity: "high" + - keyword: "build explosive" + severity: "high" + - keyword: "building explosive" + severity: "high" + - keyword: "create weapon" + severity: "high" + - keyword: "creating weapon" + severity: "high" + - keyword: "manufacture explosive" + severity: "high" + - keyword: "manufacturing explosive" + severity: "high" + - keyword: "manufactured explosive" + severity: "high" + - keyword: "construct bomb" + severity: "high" + - keyword: "constructing bomb" + severity: "high" + - keyword: "constructed bomb" + severity: "high" + - keyword: "fertilizer explosive" + severity: "high" + - keyword: "pressure cooker explosive" + severity: "high" + - keyword: "pipe explosive" + severity: "high" + - keyword: "homemade weapon" + severity: "high" + - keyword: "ingredients bomb" + severity: "high" + - keyword: "bomb ingredients" + severity: "high" + - keyword: "instructions explosive" + severity: "high" + - keyword: "explosive instructions" + severity: "high" + - keyword: "improvised explosive device" + severity: "high" + - keyword: "ied instructions" + severity: "high" + - keyword: "make ied" + severity: "high" + - keyword: "build ied" + severity: "high" + - keyword: "nail bomb" + severity: "high" + - keyword: "shrapnel bomb" + severity: "high" + - keyword: "dirty bomb" + severity: "high" + - keyword: "radiological weapon" + severity: "high" + - keyword: "nuclear weapon" + severity: "high" + - keyword: "molotov cocktail" + severity: "high" + - keyword: "incendiary device" + severity: "high" + - keyword: "fire bomb" + severity: "high" + - keyword: "napalm recipe" + severity: "high" + - keyword: "thermite recipe" + severity: "high" + - keyword: "ammonium nitrate bomb" + severity: "high" + - keyword: "anfo explosive" + severity: "high" + - keyword: "c4 explosive" + severity: "high" + - keyword: "semtex explosive" + severity: "high" + - keyword: "tnt recipe" + severity: "high" + - keyword: "dynamite recipe" + severity: "high" + - keyword: "plastic explosive" + severity: "high" + - keyword: "det cord" + severity: "high" + - keyword: "detonator instructions" + severity: "high" + - keyword: "remote detonator" + severity: "high" + - keyword: "timer bomb" + severity: "high" + - keyword: "time bomb" + severity: "high" + - keyword: "letter bomb" + severity: "high" + - keyword: "package bomb" + severity: "high" + - keyword: "mail bomb" + severity: "high" + - keyword: "anthrax weapon" + severity: "high" + - keyword: "bioweapon" + severity: "high" + - keyword: "weaponized anthrax" + severity: "high" + - keyword: "weaponized smallpox" + severity: "high" + - keyword: "nerve agent" + severity: "high" + - keyword: "sarin gas" + severity: "high" + - keyword: "vx gas" + severity: "high" + - keyword: "mustard gas" + severity: "high" + - keyword: "chlorine gas weapon" + severity: "high" + - keyword: "poison gas" + severity: "high" + - keyword: "ricin poison" + severity: "high" + - keyword: "ricin extraction" + severity: "high" + - keyword: "ricin recipe" + severity: "high" + - keyword: "botulinum toxin weapon" + severity: "high" + - keyword: "arsenic poison" + severity: "high" + - keyword: "thallium poison" + severity: "high" + - keyword: "polonium poison" + severity: "high" + - keyword: "undetectable poison recipe" + severity: "high" + - keyword: "untraceable poison" + severity: "high" + - keyword: "silent weapon" + severity: "high" + - keyword: "suppressor gun" + severity: "high" + - keyword: "silencer gun" + severity: "high" + - keyword: "homemade silencer" + severity: "high" + - keyword: "homemade suppressor" + severity: "high" + - keyword: "ghost gun" + severity: "high" + - keyword: "unregistered weapon" + severity: "high" + - keyword: "untraceable gun" + severity: "high" + - keyword: "illegal gun" + severity: "high" + - keyword: "black market gun" + severity: "high" + - keyword: "black market weapons" + severity: "high" + - keyword: "buy illegal gun" + severity: "high" + - keyword: "buy illegal weapon" + severity: "high" + - keyword: "automatic weapon conversion" + severity: "high" + - keyword: "full auto conversion" + severity: "high" + - keyword: "machine gun conversion" + severity: "high" + - keyword: "bump stock" + severity: "medium" + - keyword: "binary trigger" + severity: "medium" + - keyword: "explosive ammunition" + severity: "high" + - keyword: "armor piercing rounds" + severity: "high" + - keyword: "cop killer bullets" + severity: "high" + - keyword: "illegal ammunition" + severity: "high" + + # Medium severity - context-dependent + - keyword: "bomb" + severity: "medium" + - keyword: "explosive" + severity: "medium" + - keyword: "grenade" + severity: "medium" + - keyword: "landmine" + severity: "medium" + - keyword: "ammunition" + severity: "medium" + - keyword: "firearm" + severity: "medium" + - keyword: "gun" + severity: "medium" + - keyword: "rifle" + severity: "medium" + - keyword: "poison" + severity: "medium" + +# Exception patterns (legitimate uses that should NOT be blocked) +exceptions: + - "bomb disposal" + - "bomb squad" + - "explosive ordnance disposal" + - "eod" + - "historical" + - "educational" + - "safety training" + - "weapons history" + - "weapons education" + - "explosive safety" + - "bomb detection" + - "bomb threat" + - "report bomb" + - "reporting bomb" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_self_harm.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_self_harm.yaml new file mode 100644 index 00000000000..c3b5e781e8a --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_self_harm.yaml @@ -0,0 +1,184 @@ +# Self-harm and suicide-related content detection +category_name: "harmful_self_harm" +description: "Detects content related to self-harm, suicide, and eating disorders" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - explicit self-harm intent + - keyword: "kill myself" + severity: "high" + - keyword: "end my life" + severity: "high" + - keyword: "commit suicide" + severity: "high" + - keyword: "want to die" + severity: "high" + - keyword: "how to die" + severity: "high" + - keyword: "suicide method" + severity: "high" + - keyword: "painless death" + severity: "high" + - keyword: "overdose on" + severity: "high" + - keyword: "hang myself" + severity: "high" + - keyword: "cut myself" + severity: "high" + - keyword: "self harm" + severity: "high" + - keyword: "hurt myself" + severity: "high" + - keyword: "how to commit suicide" + severity: "high" + - keyword: "ways to kill myself" + severity: "high" + - keyword: "easiest way to die" + severity: "high" + - keyword: "suicide instructions" + severity: "high" + - keyword: "how to end my life" + severity: "high" + - keyword: "plan suicide" + severity: "high" + - keyword: "planning suicide" + severity: "high" + - keyword: "planned suicide" + severity: "high" + - keyword: "attempt suicide" + severity: "high" + - keyword: "attempting suicide" + severity: "high" + - keyword: "attempted suicide" + severity: "high" + - keyword: "self harm method" + severity: "high" + - keyword: "self harm methods" + severity: "high" + - keyword: "suicide note" + severity: "high" + - keyword: "end life" + severity: "high" + - keyword: "ending my life" + severity: "high" + - keyword: "take my life" + severity: "high" + - keyword: "taking my life" + severity: "high" + - keyword: "slash my wrists" + severity: "high" + - keyword: "slit my wrists" + severity: "high" + - keyword: "jump off bridge" + severity: "high" + - keyword: "jump off building" + severity: "high" + - keyword: "fatal overdose" + severity: "high" + - keyword: "lethal dose" + severity: "high" + - keyword: "suicide plan" + severity: "high" + - keyword: "suicide pact" + severity: "high" + - keyword: "want to end it" + severity: "high" + - keyword: "can't go on" + severity: "high" + - keyword: "better off dead" + severity: "high" + - keyword: "wish i was dead" + severity: "high" + - keyword: "no reason to live" + severity: "high" + - keyword: "life not worth living" + severity: "high" + - keyword: "burn myself" + severity: "high" + - keyword: "poison myself" + severity: "high" + - keyword: "starve myself" + severity: "high" + - keyword: "suffocate myself" + severity: "high" + - keyword: "drown myself" + severity: "high" + - keyword: "shoot myself" + severity: "high" + - keyword: "self-mutilation" + severity: "high" + - keyword: "self mutilation" + severity: "high" + - keyword: "cutting techniques" + severity: "high" + - keyword: "self-injury" + severity: "high" + - keyword: "self injury" + severity: "high" + - keyword: "how deep to cut" + severity: "high" + - keyword: "where to cut" + severity: "high" + - keyword: "painless suicide" + severity: "high" + - keyword: "quick suicide" + severity: "high" + - keyword: "easy suicide" + severity: "high" + - keyword: "suicide methods" + severity: "high" + + # Medium severity - may have legitimate context (news, education, prevention) + - keyword: "suicide" + severity: "medium" + - keyword: "self-harm" + severity: "medium" + - keyword: "cutting" + severity: "medium" + - keyword: "anorexia" + severity: "medium" + - keyword: "bulimia" + severity: "medium" + - keyword: "suicidal" + severity: "medium" + - keyword: "suicidal thoughts" + severity: "medium" + - keyword: "self-destructive" + severity: "medium" + - keyword: "self destructive" + severity: "medium" + - keyword: "eating disorder" + severity: "medium" + - keyword: "purging" + severity: "medium" + - keyword: "binge eating" + severity: "medium" + - keyword: "binge and purge" + severity: "medium" + - keyword: "pro-ana" + severity: "medium" + - keyword: "pro ana" + severity: "medium" + - keyword: "thinspo" + severity: "medium" + - keyword: "thinspiration" + severity: "medium" + +# Exception patterns (legitimate uses that should NOT be blocked) +exceptions: + - "suicide prevention" + - "suicide hotline" + - "prevent suicide" + - "suicide awareness" + - "suicide crisis" + - "suicide support" + - "mental health" + - "help with suicide" + - "suicide prevention resources" + - "talk someone out of suicide" + - "preventing self-harm" + - "self-harm prevention" + - "eating disorder treatment" + - "eating disorder recovery" + - "eating disorder support" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_violence.yaml b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_violence.yaml new file mode 100644 index 00000000000..52d6d5d2d66 --- /dev/null +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/categories/harmful_violence.yaml @@ -0,0 +1,265 @@ +# Violence and criminal planning content detection +category_name: "harmful_violence" +description: "Detects content related to violence, criminal planning, and attacks" +default_action: "BLOCK" + +# Keywords with severity levels +keywords: + # High severity - explicit violence/criminal intent + - keyword: "how to kill someone" + severity: "high" + - keyword: "plan attack" + severity: "high" + - keyword: "commit murder" + severity: "high" + - keyword: "school shooting" + severity: "high" + - keyword: "mass shooting" + severity: "high" + - keyword: "serial killer" + severity: "high" + - keyword: "hired hitman" + severity: "high" + - keyword: "assassinate" + severity: "high" + - keyword: "terrorism" + severity: "high" + - keyword: "terrorist attack" + severity: "high" + - keyword: "kidnapping plan" + severity: "high" + - keyword: "human trafficking" + severity: "high" + - keyword: "rape" + severity: "high" + - keyword: "sexual assault" + severity: "high" + - keyword: "torture method" + severity: "high" + - keyword: "ways to murder" + severity: "high" + - keyword: "plan a shooting" + severity: "high" + - keyword: "plan an attack" + severity: "high" + - keyword: "kill my" + severity: "high" + - keyword: "murder someone" + severity: "high" + - keyword: "hire a hitman" + severity: "high" + - keyword: "get away with murder" + severity: "high" + - keyword: "plan attack" + severity: "high" + - keyword: "planning attack" + severity: "high" + - keyword: "planned attack" + severity: "high" + - keyword: "commit crime" + severity: "high" + - keyword: "committing crime" + severity: "high" + - keyword: "organize shooting" + severity: "high" + - keyword: "organizing shooting" + severity: "high" + - keyword: "organized shooting" + severity: "high" + - keyword: "coordinate attack" + severity: "high" + - keyword: "coordinating attack" + severity: "high" + - keyword: "coordinated attack" + severity: "high" + - keyword: "gun school" + severity: "high" + - keyword: "shooting school" + severity: "high" + - keyword: "weapon attack" + severity: "high" + - keyword: "kill wife" + severity: "high" + - keyword: "kill husband" + severity: "high" + - keyword: "murder plan" + severity: "high" + - keyword: "murder planning" + severity: "high" + - keyword: "how to get away with murder" + severity: "high" + - keyword: "how to commit murder" + severity: "high" + - keyword: "how to kill" + severity: "high" + - keyword: "ways to kill" + severity: "high" + - keyword: "best way to kill" + severity: "high" + - keyword: "easiest way to kill" + severity: "high" + - keyword: "hire hitman" + severity: "high" + - keyword: "find hitman" + severity: "high" + - keyword: "hire assassin" + severity: "high" + - keyword: "find assassin" + severity: "high" + - keyword: "contract killer" + severity: "high" + - keyword: "how to assassinate" + severity: "high" + - keyword: "assassination plan" + severity: "high" + - keyword: "plan terrorist attack" + severity: "high" + - keyword: "organize terrorist attack" + severity: "high" + - keyword: "commit terrorist attack" + severity: "high" + - keyword: "plan mass shooting" + severity: "high" + - keyword: "organize mass shooting" + severity: "high" + - keyword: "school shooter" + severity: "high" + - keyword: "mass shooter" + severity: "high" + - keyword: "how to kidnap" + severity: "high" + - keyword: "kidnapping instructions" + severity: "high" + - keyword: "kidnap someone" + severity: "high" + - keyword: "abduct someone" + severity: "high" + - keyword: "how to rape" + severity: "high" + - keyword: "rape someone" + severity: "high" + - keyword: "how to sexually assault" + severity: "high" + - keyword: "sexual assault someone" + severity: "high" + - keyword: "torture someone" + severity: "high" + - keyword: "torture methods" + severity: "high" + - keyword: "how to torture" + severity: "high" + - keyword: "ways to torture" + severity: "high" + - keyword: "torture techniques" + severity: "high" + - keyword: "human trafficking operation" + severity: "high" + - keyword: "start human trafficking" + severity: "high" + - keyword: "organize human trafficking" + severity: "high" + - keyword: "child trafficking" + severity: "high" + - keyword: "sex trafficking" + severity: "high" + - keyword: "domestic terrorism" + severity: "high" + - keyword: "lone wolf attack" + severity: "high" + - keyword: "active shooter" + severity: "high" + - keyword: "mass casualty" + severity: "high" + - keyword: "vehicular attack" + severity: "high" + - keyword: "vehicle ramming" + severity: "high" + - keyword: "knife attack" + severity: "high" + - keyword: "stabbing spree" + severity: "high" + - keyword: "murder spree" + severity: "high" + - keyword: "killing spree" + severity: "high" + - keyword: "rampage shooting" + severity: "high" + - keyword: "workplace shooting" + severity: "high" + - keyword: "public shooting" + severity: "high" + - keyword: "mass murder" + severity: "high" + - keyword: "mass killing" + severity: "high" + - keyword: "massacre" + severity: "high" + - keyword: "genocide" + severity: "high" + - keyword: "ethnic cleansing" + severity: "high" + - keyword: "war crimes" + severity: "high" + - keyword: "crimes against humanity" + severity: "high" + - keyword: "honor killing" + severity: "high" + - keyword: "vigilante justice" + severity: "high" + - keyword: "lynch mob" + severity: "high" + - keyword: "mob violence" + severity: "high" + - keyword: "gang violence" + severity: "high" + - keyword: "organized crime" + severity: "high" + - keyword: "mafia hit" + severity: "high" + - keyword: "gang hit" + severity: "high" + - keyword: "drive by shooting" + severity: "high" + - keyword: "execution style" + severity: "high" + + # Medium severity - context-dependent + - keyword: "murder" + severity: "medium" + - keyword: "kill" + severity: "medium" + - keyword: "assassin" + severity: "medium" + - keyword: "hitman" + severity: "medium" + - keyword: "kidnap" + severity: "medium" + - keyword: "attack" + severity: "medium" + - keyword: "violence" + severity: "medium" + - keyword: "weapon" + severity: "medium" + - keyword: "shooting" + severity: "medium" + - keyword: "terrorist" + severity: "medium" + - keyword: "crime" + severity: "medium" + +# Exception patterns (legitimate uses that should NOT be blocked) +exceptions: + - "violence prevention" + - "crime statistics" + - "true crime" + - "documentary" + - "news report" + - "historical" + - "prevent violence" + - "combat violence" + - "fight violence" + - "violence against" + - "victims of violence" + - "domestic violence" + - "reporting violence" + - "violence awareness" diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py index 4058d734a5b..344383af2bb 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py @@ -5,6 +5,7 @@ This guardrail provides regex pattern matching and keyword filtering to detect and block/mask sensitive content. """ +import os import re from typing import ( TYPE_CHECKING, @@ -36,11 +37,33 @@ from litellm.types.guardrails import ( GuardrailEventHooks, Mode, ) +from litellm.types.proxy.guardrails.guardrail_hooks.litellm_content_filter import ( + ContentFilterCategoryConfig, +) from litellm.types.utils import ModelResponseStream from .patterns import get_compiled_pattern +# Helper data structure for category-based detection +class CategoryConfig: + """Configuration for a content category.""" + + def __init__( + self, + category_name: str, + description: str, + default_action: ContentFilterAction, + keywords: List[Dict[str, str]], + exceptions: List[str], + ): + self.category_name = category_name + self.description = description + self.default_action = default_action + self.keywords = keywords + self.exceptions = [e.lower() for e in exceptions] + + class ContentFilterGuardrail(CustomGuardrail): """ Content filter guardrail that detects sensitive information using: @@ -69,6 +92,8 @@ class ContentFilterGuardrail(CustomGuardrail): default_on: bool = False, pattern_redaction_format: Optional[str] = None, keyword_redaction_tag: Optional[str] = None, + categories: Optional[List[ContentFilterCategoryConfig]] = None, + severity_threshold: str = "medium", **kwargs, ): """ @@ -83,6 +108,8 @@ class ContentFilterGuardrail(CustomGuardrail): default_on: If True, runs on all requests by default pattern_redaction_format: Format string for pattern redaction (use {pattern_name} placeholder) keyword_redaction_tag: Tag to use for keyword redaction + categories: List of category configurations with enabled/action/severity settings + severity_threshold: Minimum severity to block ("high", "medium", "low") """ super().__init__( guardrail_name=guardrail_name, @@ -101,6 +128,17 @@ class ContentFilterGuardrail(CustomGuardrail): pattern_redaction_format or self.PATTERN_REDACTION_FORMAT ) self.keyword_redaction_tag = keyword_redaction_tag or self.KEYWORD_REDACTION_STR + self.severity_threshold = severity_threshold + + # Store loaded categories + self.loaded_categories: Dict[str, CategoryConfig] = {} + self.category_keywords: Dict[str, Tuple[str, str, ContentFilterAction]] = ( + {} + ) # keyword -> (category, severity, action) + + # Load categories if provided + if categories: + self._load_categories(categories) # Normalize inputs: convert dicts to Pydantic models for consistent handling normalized_patterns: List[ContentFilterPattern] = [] @@ -144,6 +182,125 @@ class ContentFilterGuardrail(CustomGuardrail): f"ContentFilterGuardrail initialized with {len(self.compiled_patterns)} patterns " f"and {len(self.blocked_words)} blocked words" ) + verbose_proxy_logger.debug( + f"Loaded {len(self.loaded_categories)} categories with " + f"{len(self.category_keywords)} keywords" + ) + + def _load_categories(self, categories: List[ContentFilterCategoryConfig]) -> None: + """ + Load content categories from configuration. + + Args: + categories: List of category configurations with format: + - category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + category_file: "/path/to/custom_file.yaml" # optional override + """ + categories_dir = os.path.join(os.path.dirname(__file__), "categories") + + for cat_config in categories: + category_name = cat_config.get("category") + if not category_name or not isinstance(category_name, str): + verbose_proxy_logger.warning( + "Category name missing or invalid in config, skipping" + ) + continue + + enabled = cat_config.get("enabled", True) + action = cat_config.get("action") + severity_threshold = cat_config.get( + "severity_threshold", self.severity_threshold + ) + custom_file = cat_config.get("category_file") + + if not enabled: + verbose_proxy_logger.debug( + f"Category {category_name} is disabled, skipping" + ) + continue + + # Load category file (custom or default) + if custom_file: + category_file_path = custom_file + else: + category_file_path = os.path.join( + categories_dir, f"{category_name}.yaml" + ) + + if not os.path.exists(category_file_path): + verbose_proxy_logger.warning( + f"Category file not found: {category_file_path}, skipping" + ) + continue + + try: + category = self._load_category_file(category_file_path) + self.loaded_categories[category_name] = category + + # Use action from config, or default from category file + category_action = ContentFilterAction( + action if action else category.default_action + ) + + # Add keywords from this category + for keyword_data in category.keywords: + keyword = keyword_data["keyword"].lower() + severity = keyword_data["severity"] + + # Check if keyword meets severity threshold + if self._should_apply_severity(severity, severity_threshold): + self.category_keywords[keyword] = ( + category_name, + severity, + category_action, + ) + + verbose_proxy_logger.info( + f"Loaded category {category_name}: " + f"{len(category.keywords)} keywords" + ) + except Exception as e: + verbose_proxy_logger.error( + f"Error loading category {category_name}: {e}" + ) + + def _load_category_file(self, file_path: str) -> CategoryConfig: + """ + Load a category definition from a YAML file. + + Args: + file_path: Path to category YAML file + + Returns: + CategoryConfig object + """ + with open(file_path, "r") as f: + data = yaml.safe_load(f) + + return CategoryConfig( + category_name=data.get("category_name", "unknown"), + description=data.get("description", ""), + default_action=ContentFilterAction(data.get("default_action", "BLOCK")), + keywords=data.get("keywords", []), + exceptions=data.get("exceptions", []), + ) + + def _should_apply_severity(self, severity: str, threshold: str) -> bool: + """ + Check if a given severity meets the threshold. + + Args: + severity: The severity level of the item ("high", "medium", "low") + threshold: The minimum severity threshold + + Returns: + True if severity meets or exceeds threshold + """ + severity_order = {"low": 0, "medium": 1, "high": 2} + return severity_order.get(severity, 0) >= severity_order.get(threshold, 1) def _add_pattern(self, pattern_config: ContentFilterPattern) -> None: """ @@ -247,6 +404,64 @@ class ContentFilterGuardrail(CustomGuardrail): return (matched_text, pattern_name, action) return None + def _check_category_keywords( + self, text: str, exceptions: List[str] + ) -> Optional[Tuple[str, str, str, ContentFilterAction]]: + """ + Check text for category keywords. + + Args: + text: Text to check + exceptions: List of exception phrases to ignore + + Returns: + Tuple of (keyword, category, severity, action) if match found, None otherwise + """ + text_lower = text.lower() + + # First check if any exception applies + for exception in exceptions: + if exception in text_lower: + verbose_proxy_logger.debug( + f"Exception phrase '{exception}' found, skipping category keyword check" + ) + return None + + # Check category keywords + for keyword, (category, severity, action) in self.category_keywords.items(): + # Use word boundary matching for single words to avoid false positives + # (e.g., "men" should not match "recommend") + # For multi-word phrases, use substring matching + if " " in keyword: + # Multi-word phrase - use substring matching + keyword_found = keyword in text_lower + else: + # Single word - use word boundary matching to match whole words only + keyword_pattern = r"\b" + re.escape(keyword) + r"\b" + keyword_found = bool(re.search(keyword_pattern, text_lower)) + + if keyword_found: + # Check if this keyword has exceptions + category_obj = self.loaded_categories.get(category) + if category_obj: + # Check category-specific exceptions + exception_found = False + for exception in category_obj.exceptions: + if exception in text_lower: + verbose_proxy_logger.debug( + f"Category exception '{exception}' found for keyword '{keyword}', skipping" + ) + exception_found = True + break + if exception_found: + continue + + verbose_proxy_logger.debug( + f"Category keyword '{keyword}' found in category '{category}' with severity {severity}" + ) + return (keyword, category, severity, action) + return None + def _check_blocked_words( self, text: str ) -> Optional[Tuple[str, ContentFilterAction, Optional[str]]]: @@ -337,6 +552,42 @@ class ContentFilterGuardrail(CustomGuardrail): processed_texts = [] for text in texts: + # Collect all exceptions from loaded categories + all_exceptions = [] + for category in self.loaded_categories.values(): + all_exceptions.extend(category.exceptions) + + # Check category keywords + category_keyword_match = self._check_category_keywords(text, all_exceptions) + if category_keyword_match: + keyword, category, severity, action = category_keyword_match + if action == ContentFilterAction.BLOCK: + error_msg = ( + f"Content blocked: {category} category keyword '{keyword}' detected " + f"(severity: {severity})" + ) + verbose_proxy_logger.warning(error_msg) + raise HTTPException( + status_code=400, + detail={ + "error": error_msg, + "category": category, + "keyword": keyword, + "severity": severity, + }, + ) + elif action == ContentFilterAction.MASK: + # Replace keyword with redaction tag + text = re.sub( + re.escape(keyword), + self.keyword_redaction_tag, + text, + flags=re.IGNORECASE, + ) + verbose_proxy_logger.info( + f"Masked category keyword '{keyword}' from {category} (severity: {severity})" + ) + # Check regex patterns - process ALL patterns, not just first match for compiled_pattern, pattern_name, action in self.compiled_patterns: match = compiled_pattern.search(text) @@ -356,7 +607,9 @@ class ContentFilterGuardrail(CustomGuardrail): pattern_name=pattern_name.upper() ) text = compiled_pattern.sub(redaction_tag, text) - verbose_proxy_logger.info(f"Masked all {pattern_name} matches in content") + verbose_proxy_logger.info( + f"Masked all {pattern_name} matches in content" + ) # Check blocked words - iterate through ALL blocked words # to ensure all matching keywords are processed, not just the first one diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/patterns.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/patterns.py index b4649d73e34..776cf5bd8d2 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/patterns.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/patterns.py @@ -1,7 +1,7 @@ """ Prebuilt regex patterns for content filtering. -This module loads predefined regex patterns from patterns.json for detecting +This module loads predefined regex patterns from patterns.json for detecting sensitive information like SSNs, credit cards, API keys, etc. """ @@ -25,6 +25,7 @@ _PATTERNS_DATA = _load_patterns_from_json() class PrebuiltPatternName(str, Enum): """Enum for prebuilt pattern names - dynamically generated from JSON""" + pass @@ -43,13 +44,13 @@ PREBUILT_PATTERNS: Dict[str, str] = { def get_compiled_pattern(pattern_name: str) -> Pattern: """ Get a compiled regex pattern by name. - + Args: pattern_name: Name of the prebuilt pattern - + Returns: Compiled regex pattern - + Raises: ValueError: If pattern_name is not found in PREBUILT_PATTERNS """ @@ -59,14 +60,14 @@ def get_compiled_pattern(pattern_name: str) -> Pattern: f"Unknown pattern name: '{pattern_name}'. " f"Available patterns: {available_patterns}" ) - + return re.compile(PREBUILT_PATTERNS[pattern_name], re.IGNORECASE) def get_all_pattern_names() -> List[str]: """ Get a list of all available prebuilt pattern names. - + Returns: List of pattern names """ @@ -99,7 +100,7 @@ PATTERN_DESCRIPTIONS: Dict[str, str] = { def get_pattern_metadata() -> List[Dict[str, str]]: """ Return pattern metadata for UI display. - + Returns: List of dictionaries containing pattern name, display_name, category, and description """ @@ -113,3 +114,51 @@ def get_pattern_metadata() -> List[Dict[str, str]]: for pattern_data in _PATTERNS_DATA["patterns"] ] + +def get_available_content_categories() -> List[Dict[str, str]]: + """ + Return available content categories for UI display. + + Returns: + List of dictionaries containing category name, display_name, and description + """ + import yaml + + categories_dir = os.path.join(os.path.dirname(__file__), "categories") + available_categories = [] + + if not os.path.exists(categories_dir): + return [] + + # Scan the categories directory for YAML files + for filename in os.listdir(categories_dir): + if filename.endswith(".yaml") or filename.endswith(".yml"): + category_file_path = os.path.join(categories_dir, filename) + try: + with open(category_file_path, "r") as f: + category_data = yaml.safe_load(f) + + if category_data and "category_name" in category_data: + # Create display name from category name (convert harmful_self_harm -> Harmful Self Harm) + display_name = ( + category_data["category_name"].replace("_", " ").title() + ) + + available_categories.append( + { + "name": category_data["category_name"], + "display_name": display_name, + "description": category_data.get("description", ""), + "default_action": category_data.get( + "default_action", "BLOCK" + ), + } + ) + except Exception: + # Skip files that can't be loaded + continue + + # Sort by name for consistent ordering + available_categories.sort(key=lambda x: x["name"]) + + return available_categories diff --git a/litellm/types/proxy/guardrails/guardrail_hooks/litellm_content_filter.py b/litellm/types/proxy/guardrails/guardrail_hooks/litellm_content_filter.py index 4ccab3718ed..b5e36334ede 100644 --- a/litellm/types/proxy/guardrails/guardrail_hooks/litellm_content_filter.py +++ b/litellm/types/proxy/guardrails/guardrail_hooks/litellm_content_filter.py @@ -1,7 +1,84 @@ +from typing import List, Literal, Optional + +from pydantic import Field + +from litellm.types.llms.base import BaseLiteLLMOpenAIResponseObject from litellm.types.proxy.guardrails.guardrail_hooks.base import GuardrailConfigModel +class ContentFilterCategoryConfig(BaseLiteLLMOpenAIResponseObject): + """ + category: "harmful_self_harm" + enabled: true + action: "BLOCK" + severity_threshold: "medium" + category_file: "/path/to/custom_file.yaml" # optional override + """ + + category: str = Field( + description="The category to detect", + ) + enabled: bool = Field( + default=True, + description="Whether the category is enabled", + ) + action: Literal["BLOCK", "MASK"] = Field( + description="The action to take when the category is detected", + ) + severity_threshold: Literal["high", "medium", "low"] = Field( + default="medium", + description="The severity threshold to detect the category", + ) + category_file: Optional[str] = Field( + default=None, + description="Optional override. Use your own category file instead of the default one.", + ) + + class LitellmContentFilterGuardrailConfigModel(GuardrailConfigModel): + """ + Configuration model for LiteLLM Content Filter guardrail. + + Supports: + - Traditional keyword and pattern matching + - Category-based detection (harmful content, bias detection) + - Proximity-based detection (identity keywords + negative modifiers) + """ + + # Traditional patterns and keywords + patterns: Optional[List[dict]] = Field( + default=None, + description="List of regex patterns to detect (prebuilt or custom)", + ) + blocked_words: Optional[List[dict]] = Field( + default=None, + description="List of blocked keywords with actions", + ) + blocked_words_file: Optional[str] = Field( + default=None, + description="Path to YAML file containing blocked words", + ) + + # Category-based detection + categories: Optional[List[ContentFilterCategoryConfig]] = Field( + default=None, + description="List of prebuilt categories to enable (harmful_*, bias_*)", + ) + severity_threshold: str = Field( + default="medium", + description="Minimum severity to block (high, medium, low)", + ) + + # Redaction customization + pattern_redaction_format: Optional[str] = Field( + default="[{pattern_name}_REDACTED]", + description="Format string for pattern redaction (use {pattern_name} placeholder)", + ) + keyword_redaction_tag: Optional[str] = Field( + default="[KEYWORD_REDACTED]", + description="Tag to use for keyword redaction", + ) + @staticmethod def ui_friendly_name() -> str: - return "LiteLLM Content Filter" \ No newline at end of file + return "LiteLLM Content Filter" diff --git a/ui/litellm-dashboard/src/components/guardrails/add_guardrail_form.tsx b/ui/litellm-dashboard/src/components/guardrails/add_guardrail_form.tsx index 20ca36f6d16..f4bfa304811 100644 --- a/ui/litellm-dashboard/src/components/guardrails/add_guardrail_form.tsx +++ b/ui/litellm-dashboard/src/components/guardrails/add_guardrail_form.tsx @@ -58,6 +58,12 @@ interface GuardrailSettings { }>; pattern_categories: string[]; supported_actions: string[]; + content_categories?: Array<{ + name: string; + display_name: string; + description: string; + default_action: string; + }>; }; } @@ -103,6 +109,7 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a // Content Filter state const [selectedPatterns, setSelectedPatterns] = useState([]); const [blockedWords, setBlockedWords] = useState([]); + const [selectedContentCategories, setSelectedContentCategories] = useState([]); const [toolPermissionConfig, setToolPermissionConfig] = useState({ rules: [], default_action: "deny", @@ -251,6 +258,7 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a setCategorySpecificThresholds({}); setSelectedPatterns([]); setBlockedWords([]); + setSelectedContentCategories([]); setToolPermissionConfig({ rules: [], default_action: "deny", @@ -315,7 +323,7 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a } } - // For Content Filter, add patterns and blocked words + // For Content Filter, add patterns, blocked words, and categories if (shouldRenderContentFilterConfigSettings(values.provider)) { if (selectedPatterns.length > 0) { guardrailData.litellm_params.patterns = selectedPatterns.map((p) => ({ @@ -333,6 +341,14 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a description: w.description, })); } + if (selectedContentCategories.length > 0) { + guardrailData.litellm_params.categories = selectedContentCategories.map((c) => ({ + category: c.category, + enabled: true, + action: c.action, + severity_threshold: c.severity_threshold || "medium", + })); + } } // Add config values to the guardrail_info if provided else if (values.config) { @@ -581,7 +597,7 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a {/* Use the GuardrailProviderFields component to render provider-specific fields */} - {!isToolPermissionProvider && ( + {!isToolPermissionProvider && !shouldRenderContentFilterConfigSettings(selectedProvider) && ( = ({ visible, onClose, a ); }; - const renderContentFilterConfiguration = (step: "patterns" | "keywords") => { + const renderContentFilterConfiguration = (step: "patterns" | "keywords" | "categories") => { if (!guardrailSettings || !shouldRenderContentFilterConfigSettings(selectedProvider)) return null; const contentFilterSettings = guardrailSettings.content_filter_settings; @@ -634,6 +650,15 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a blockedWords.map((w) => (w.id === id ? { ...w, [field]: value } : w)) ); }} + contentCategories={contentFilterSettings.content_categories || []} + selectedContentCategories={selectedContentCategories} + onContentCategoryAdd={(category) => setSelectedContentCategories([...selectedContentCategories, category])} + onContentCategoryRemove={(id) => setSelectedContentCategories(selectedContentCategories.filter((c) => c.id !== id))} + onContentCategoryUpdate={(id, field, value) => { + setSelectedContentCategories( + selectedContentCategories.map((c) => (c.id === id ? { ...c, [field]: value } : c)) + ); + }} accessToken={accessToken} showStep={step} /> @@ -675,10 +700,15 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a return renderPiiConfiguration(); } if (shouldRenderContentFilterConfigSettings(selectedProvider)) { - return renderContentFilterConfiguration("patterns"); + return renderContentFilterConfiguration("categories"); } return renderOptionalParams(); case 2: + if (shouldRenderContentFilterConfigSettings(selectedProvider)) { + return renderContentFilterConfiguration("patterns"); + } + return null; + case 3: if (shouldRenderContentFilterConfigSettings(selectedProvider)) { return renderContentFilterConfiguration("keywords"); } @@ -689,7 +719,7 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a }; const renderStepButtons = () => { - const totalSteps = shouldRenderContentFilterConfigSettings(selectedProvider) ? 3 : 2; + const totalSteps = shouldRenderContentFilterConfigSettings(selectedProvider) ? 4 : 2; const isLastStep = currentStep === totalSteps - 1; return ( @@ -713,7 +743,7 @@ const AddGuardrailForm: React.FC = ({ visible, onClose, a }; return ( - +
= ({ visible, onClose, a default_on: false, }} > - + {shouldRenderContentFilterConfigSettings(selectedProvider) && ( - + <> + + + )} diff --git a/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentCategoryConfiguration.tsx b/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentCategoryConfiguration.tsx new file mode 100644 index 00000000000..7c8652a7b18 --- /dev/null +++ b/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentCategoryConfiguration.tsx @@ -0,0 +1,384 @@ +import React from "react"; +import { Card, Typography, Select, Table, Tag, Collapse } from "antd"; +import { DeleteOutlined, PlusOutlined, FileTextOutlined } from "@ant-design/icons"; +import { Button } from "@tremor/react"; +import { getCategoryYaml } from "../../networking"; + +const { Title, Text } = Typography; +const { Option } = Select; +const { Panel } = Collapse; + +interface ContentCategory { + name: string; + display_name: string; + description: string; + default_action: string; +} + +interface SelectedCategory { + id: string; + category: string; + display_name: string; + action: "BLOCK" | "MASK"; + severity_threshold: "high" | "medium" | "low"; +} + +interface ContentCategoryConfigurationProps { + availableCategories: ContentCategory[]; + selectedCategories: SelectedCategory[]; + onCategoryAdd: (category: SelectedCategory) => void; + onCategoryRemove: (id: string) => void; + onCategoryUpdate: (id: string, field: string, value: any) => void; + accessToken?: string | null; +} + +const ContentCategoryConfiguration: React.FC = ({ + availableCategories, + selectedCategories, + onCategoryAdd, + onCategoryRemove, + onCategoryUpdate, + accessToken, +}) => { + const [selectedCategoryName, setSelectedCategoryName] = React.useState(""); + const [categoryYaml, setCategoryYaml] = React.useState<{ [key: string]: string }>({}); + const [loadingYaml, setLoadingYaml] = React.useState<{ [key: string]: boolean }>({}); + const [expandedYamlCategories, setExpandedYamlCategories] = React.useState([]); + const [previewYaml, setPreviewYaml] = React.useState(""); + const [loadingPreviewYaml, setLoadingPreviewYaml] = React.useState(false); + + const handleAddCategory = () => { + if (!selectedCategoryName) { + return; + } + + const category = availableCategories.find((c) => c.name === selectedCategoryName); + if (!category) { + return; + } + + // Check if already added + if (selectedCategories.some((c) => c.category === selectedCategoryName)) { + return; + } + + onCategoryAdd({ + id: `category-${Date.now()}`, + category: category.name, + display_name: category.display_name, + action: category.default_action as "BLOCK" | "MASK", + severity_threshold: "medium", + }); + + setSelectedCategoryName(""); + setPreviewYaml(""); // Clear preview when category is added + }; + + const fetchCategoryYaml = async (categoryName: string) => { + if (!accessToken) { + return; // No access token + } + + // Check if already loaded + if (categoryYaml[categoryName]) { + return; + } + + setLoadingYaml((prev) => ({ ...prev, [categoryName]: true })); + try { + const data = await getCategoryYaml(accessToken, categoryName); + setCategoryYaml((prev) => ({ ...prev, [categoryName]: data.yaml_content })); + } catch (error) { + console.error(`Failed to fetch YAML for category ${categoryName}:`, error); + } finally { + setLoadingYaml((prev) => ({ ...prev, [categoryName]: false })); + } + }; + + // Fetch preview YAML when a category is selected in dropdown + React.useEffect(() => { + if (selectedCategoryName && accessToken) { + // Check if we already have this YAML cached + const cachedYaml = categoryYaml[selectedCategoryName]; + if (cachedYaml) { + setPreviewYaml(cachedYaml); + return; + } + + // Fetch the YAML for preview + setLoadingPreviewYaml(true); + console.log(`Fetching YAML for category: ${selectedCategoryName}`, { accessToken: accessToken ? "present" : "missing" }); + getCategoryYaml(accessToken, selectedCategoryName) + .then((data) => { + console.log(`Successfully fetched YAML for ${selectedCategoryName}:`, data); + setPreviewYaml(data.yaml_content); + // Also cache it for later use + setCategoryYaml((prev) => ({ ...prev, [selectedCategoryName]: data.yaml_content })); + }) + .catch((error) => { + console.error(`Failed to fetch preview YAML for category ${selectedCategoryName}:`, error); + setPreviewYaml(""); + }) + .finally(() => { + setLoadingPreviewYaml(false); + }); + } else { + setPreviewYaml(""); + setLoadingPreviewYaml(false); + } + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [selectedCategoryName, accessToken]); + + const columns = [ + { + title: "Category", + dataIndex: "display_name", + key: "display_name", + render: (text: string, record: SelectedCategory) => { + const category = availableCategories.find((c) => c.name === record.category); + return ( +
+
{text}
+ {category?.description && ( +
+ {category.description} +
+ )} +
+ ); + }, + }, + { + title: "Action", + dataIndex: "action", + key: "action", + width: 150, + render: (action: string, record: SelectedCategory) => ( + + ), + }, + { + title: "Severity Threshold", + dataIndex: "severity_threshold", + key: "severity_threshold", + width: 180, + render: (threshold: string, record: SelectedCategory) => ( + + ), + }, + { + title: "", + key: "actions", + width: 80, + render: (_: any, record: SelectedCategory) => ( + + ), + }, + ]; + + const unselectedCategories = availableCategories.filter( + (cat) => !selectedCategories.some((sel) => sel.category === cat.name) + ); + + return ( + + + Content Categories + + + Detect harmful content, bias, and inappropriate advice using semantic analysis + + + } + size="small" + > +
+ + +
+ + {/* Preview YAML box - shown when category is selected but not yet added */} + {selectedCategoryName && ( +
+
+ Preview: {availableCategories.find((c) => c.name === selectedCategoryName)?.display_name} +
+ {loadingPreviewYaml ? ( +
+ Loading YAML... +
+ ) : previewYaml ? ( +
+              {previewYaml}
+            
+ ) : ( +
+ Unable to load YAML content +
+ )} +
+ )} + + {selectedCategories.length > 0 ? ( + <> + +
+ { + const keyArray = Array.isArray(keys) ? keys : keys ? [keys] : []; + const newExpanded = new Set(keyArray as string[]); + const oldExpanded = new Set(expandedYamlCategories); + + // Find newly expanded categories and fetch their YAML + keyArray.forEach((key) => { + const categoryName = key as string; + if (!oldExpanded.has(categoryName) && !categoryYaml[categoryName]) { + fetchCategoryYaml(categoryName); + } + }); + + setExpandedYamlCategories(keyArray as string[]); + }} + ghost + > + {selectedCategories.map((category) => ( + + + View YAML for {category.display_name} +
+ } + key={category.category} + > + {loadingYaml[category.category] ? ( +
+ Loading YAML... +
+ ) : categoryYaml[category.category] ? ( +
+                      {categoryYaml[category.category]}
+                    
+ ) : ( +
+ YAML will load when expanded +
+ )} + + ))} + + + + ) : ( +
+ No content categories selected. Add categories to detect harmful content, bias, or + inappropriate advice. +
+ )} + + ); +}; + +export default ContentCategoryConfiguration; + diff --git a/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentFilterConfiguration.tsx b/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentFilterConfiguration.tsx index 168fefdfe64..bae95aac6ce 100644 --- a/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentFilterConfiguration.tsx +++ b/ui/litellm-dashboard/src/components/guardrails/content_filter/ContentFilterConfiguration.tsx @@ -9,6 +9,7 @@ import CustomPatternModal from "./CustomPatternModal"; import KeywordModal from "./KeywordModal"; import PatternTable from "./PatternTable"; import KeywordTable from "./KeywordTable"; +import ContentCategoryConfiguration from "./ContentCategoryConfiguration"; const { Title, Text } = Typography; @@ -35,6 +36,21 @@ interface BlockedWord { description?: string; } +interface ContentCategory { + name: string; + display_name: string; + description: string; + default_action: string; +} + +interface SelectedContentCategory { + id: string; + category: string; + display_name: string; + action: "BLOCK" | "MASK"; + severity_threshold: "high" | "medium" | "low"; +} + interface ContentFilterConfigurationProps { prebuiltPatterns: PrebuiltPattern[]; categories: string[]; @@ -48,7 +64,12 @@ interface ContentFilterConfigurationProps { onBlockedWordUpdate: (id: string, field: string, value: any) => void; onFileUpload?: (content: string) => void; accessToken: string | null; - showStep?: "patterns" | "keywords"; + showStep?: "patterns" | "keywords" | "categories"; + contentCategories?: ContentCategory[]; + selectedContentCategories?: SelectedContentCategory[]; + onContentCategoryAdd?: (category: SelectedContentCategory) => void; + onContentCategoryRemove?: (id: string) => void; + onContentCategoryUpdate?: (id: string, field: string, value: any) => void; } const ContentFilterConfiguration: React.FC = ({ @@ -65,6 +86,11 @@ const ContentFilterConfiguration: React.FC = ({ onFileUpload, accessToken, showStep, + contentCategories = [], + selectedContentCategories = [], + onContentCategoryAdd, + onContentCategoryRemove, + onContentCategoryUpdate, }) => { const [patternModalVisible, setPatternModalVisible] = useState(false); const [keywordModalVisible, setKeywordModalVisible] = useState(false); @@ -167,13 +193,14 @@ const ContentFilterConfiguration: React.FC = ({ const showPatterns = !showStep || showStep === "patterns"; const showKeywords = !showStep || showStep === "keywords"; + const showCategories = !showStep || showStep === "categories"; return (
{!showStep && (
- Configure patterns and keywords to detect and filter sensitive information in requests and responses. + Configure patterns, keywords, and content categories to detect and filter sensitive information in requests and responses.
)} @@ -244,6 +271,17 @@ const ContentFilterConfiguration: React.FC = ({ )} + {showCategories && contentCategories.length > 0 && onContentCategoryAdd && onContentCategoryRemove && onContentCategoryUpdate && ( + + )} + = ({ } console.log("Value:", value); + + // Fields to skip for content filter provider (handled in dedicated steps) + const contentFilterFieldsToSkip = new Set([ + "patterns", + "blocked_words", + "blocked_words_file", + "categories", + "severity_threshold", + "pattern_redaction_format", + "keyword_redaction_tag", + ]); + + const isContentFilterProvider = shouldRenderContentFilterConfigSettings(selectedProvider); + // Convert object to array of entries and render fields const renderFields = (fields: { [key: string]: ProviderParam }, parentKey = "", parentValue?: any) => { return Object.entries(fields).map(([fieldKey, field]) => { @@ -123,6 +138,11 @@ const GuardrailProviderFields: React.FC = ({ return null; } + // Skip content filter specific fields when it's a content filter provider (handled in dedicated steps) + if (isContentFilterProvider && contentFilterFieldsToSkip.has(fieldKey)) { + return null; + } + // Handle other nested fields (like azure/text_moderations optional_params) if (field.type === "nested" && field.fields) { return ( diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index ddabc6f5212..f0464c61f88 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -6836,6 +6836,40 @@ export const getGuardrailProviderSpecificParams = async (accessToken: string) => } }; +export const getCategoryYaml = async (accessToken: string, categoryName: string) => { + try { + // URL encode the category name to handle special characters + const encodedCategoryName = encodeURIComponent(categoryName); + const url = proxyBaseUrl + ? `${proxyBaseUrl}/guardrails/ui/category_yaml/${encodedCategoryName}` + : `/guardrails/ui/category_yaml/${encodedCategoryName}`; + + console.log(`Fetching category YAML from: ${url}`); + + const response = await fetch(url, { + method: "GET", + headers: { + [globalLitellmHeaderName]: `Bearer ${accessToken}`, + "Content-Type": "application/json", + }, + }); + + if (!response.ok) { + const errorData = await response.text(); + console.error(`Failed to get category YAML. Status: ${response.status}, Error:`, errorData); + handleError(errorData); + throw new Error(`Failed to get category YAML: ${response.status} ${errorData}`); + } + + const data = await response.json(); + console.log("Category YAML response:", data); + return data; + } catch (error) { + console.error("Failed to get category YAML:", error); + throw error; + } +}; + export const getAgentsList = async (accessToken: string) => { try { const url = proxyBaseUrl ? `${proxyBaseUrl}/v1/agents` : `/v1/agents`;