Merge pull request #14401 from Noma-Security/noma_non_blocking_monitor_mode

Noma non blocking monitor mode & anonymize input support
This commit is contained in:
Krish Dholakia 2025-09-13 09:41:41 -07:00 • committed by GitHub
commit 38efd358eb
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
5 changed files with 1445 additions and 83 deletions

View file

@ -135,6 +135,7 @@ guardrails:
# application_id: "my-app"
# monitor_mode: false
# block_failures: true
# anonymize_input: false
```
### Required Parameters
@ -147,6 +148,7 @@ guardrails:
- **`application_id`**: Your application identifier (defaults to `"litellm"`)
- **`monitor_mode`**: If `true`, logs violations without blocking (defaults to `false`)
- **`block_failures`**: If `true`, blocks requests when guardrail API failures occur (defaults to `true`)
- **`anonymize_input`**: If `true`, replaces sensitive content with anonymized version (defaults to `false`)
## Environment Variables
@ -158,6 +160,7 @@ export NOMA_API_BASE="https://api.noma.security/" # Optional
export NOMA_APPLICATION_ID="my-app" # Optional
export NOMA_MONITOR_MODE="false" # Optional
export NOMA_BLOCK_FAILURES="true" # Optional
export NOMA_ANONYMIZE_INPUT="false" # Optional
```
## Advanced Configuration
@ -190,6 +193,20 @@ guardrails:
block_failures: false # Allow requests to proceed if guardrail API fails
```
### Content Anonymization
Enable anonymization to replace sensitive content instead of blocking:
```yaml
guardrails:
- guardrail_name: "noma-anonymize"
litellm_params:
guardrail: noma
mode: "pre_call"
api_key: os.environ/NOMA_API_KEY
anonymize_input: true # Replace sensitive data with anonymized version
```
### Multiple Guardrails
Apply different configurations for input and output:

View file

@ -18,6 +18,7 @@ def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail"
application_id=litellm_params.application_id,
monitor_mode=litellm_params.monitor_mode,
block_failures=litellm_params.block_failures,
anonymize_input=litellm_params.anonymize_input,
event_hook=litellm_params.mode,
default_on=litellm_params.default_on,
)

View file

@ -5,9 +5,10 @@
#
# +-------------------------------------------------------------+
import asyncio
import copy
import os
from typing import Any, Dict, Literal, Optional, Union
from typing import Any, Dict, Final, Literal, Optional, Union
from urllib.parse import urljoin
from fastapi import HTTPException
@ -24,6 +25,15 @@ from litellm.proxy._types import UserAPIKeyAuth
from litellm.types.guardrails import GuardrailEventHooks
from litellm.types.utils import EmbeddingResponse, ImageResponse
# Constants
USER_ROLE: Final[Literal["user"]] = "user"
ASSISTANT_ROLE: Final[Literal["assistant"]] = "assistant"
SENSITIVE_DATA_DETECTOR_KEYS: Final[list[str]] = ["sensitiveData", "dataDetector"]
# Type aliases
MessageRole = Literal["user", "assistant"]
LLMResponse = Union[Any, ModelResponse, EmbeddingResponse, ImageResponse]
class NomaBlockedMessage(HTTPException):
"""Exception raised when Noma guardrail blocks a message"""
@ -77,6 +87,7 @@ class NomaBlockedMessage(HTTPException):
"allowedTopics",
"bannedTopics",
"topicGuardrails",
"topicDetector", # Mock name for tests
] and isinstance(value, dict):
filtered_topics = {}
for topic, topic_result in value.items():
@ -86,7 +97,7 @@ class NomaBlockedMessage(HTTPException):
if filtered_topics:
result[key] = filtered_topics
elif key == "sensitiveData" and isinstance(value, dict):
elif key in SENSITIVE_DATA_DETECTOR_KEYS and isinstance(value, dict):
filtered_sensitive = {}
for data_type, data_result in value.items():
if self._is_result_true(data_result):
@ -135,6 +146,7 @@ class NomaGuardrail(CustomGuardrail):
application_id: Optional[str] = None,
monitor_mode: Optional[bool] = None,
block_failures: Optional[bool] = None,
anonymize_input: Optional[bool] = None,
**kwargs,
):
self.async_handler = get_async_httpx_client(
@ -162,8 +174,326 @@ class NomaGuardrail(CustomGuardrail):
else:
self.block_failures = block_failures
if anonymize_input is None:
self.anonymize_input = (
os.environ.get("NOMA_ANONYMIZE_INPUT", "false").lower() == "true"
)
else:
self.anonymize_input = anonymize_input
super().__init__(**kwargs)
def _create_background_noma_check(
self,
coro,
) -> None:
"""Create a background task for Noma API calls without blocking the main flow"""
try:
asyncio.create_task(coro)
except Exception as e:
verbose_proxy_logger.error(
f"Failed to create background Noma task: {str(e)}"
)
async def _process_user_message_check(
self,
request_data: dict,
user_auth: UserAPIKeyAuth,
) -> Optional[str]:
"""Shared logic for processing user message checks"""
extra_data = self.get_guardrail_dynamic_request_body_params(request_data)
user_message = await self._extract_user_message(request_data)
if not user_message:
return None
payload = {"request": {"text": user_message}}
response_json = await self._call_noma_api(
payload=payload,
llm_request_id=None,
request_data=request_data,
user_auth=user_auth,
extra_data=extra_data,
)
if self.monitor_mode:
await self._handle_verdict_background(
USER_ROLE, user_message, response_json
)
return user_message
# Check if we should anonymize content
if self._should_anonymize(response_json, USER_ROLE):
anonymized_content = self._extract_anonymized_content(
response_json, USER_ROLE
)
if anonymized_content:
# Replace the user message content with anonymized version
self._replace_user_message_content(request_data, anonymized_content)
verbose_proxy_logger.debug(
f"Noma guardrail anonymized user message: {anonymized_content}"
)
return anonymized_content
await self._check_verdict(USER_ROLE, user_message, response_json)
return user_message
async def _process_llm_response_check(
self,
request_data: dict,
response: LLMResponse,
user_auth: UserAPIKeyAuth,
) -> Optional[str]:
"""Shared logic for processing LLM response checks"""
extra_data = self.get_guardrail_dynamic_request_body_params(request_data)
if not isinstance(response, litellm.ModelResponse):
return None
content = None
for choice in response.choices:
if isinstance(choice, litellm.Choices) and choice.message.content:
content = choice.message.content
break
if not content or not isinstance(content, str):
return None
payload = {"response": {"text": content}}
response_json = await self._call_noma_api(
payload=payload,
llm_request_id=response.id,
request_data=request_data,
user_auth=user_auth,
extra_data=extra_data,
)
if self.monitor_mode:
await self._handle_verdict_background(
ASSISTANT_ROLE, content, response_json
)
return content
# Check if we should anonymize content
if self._should_anonymize(response_json, ASSISTANT_ROLE):
anonymized_content = self._extract_anonymized_content(
response_json, ASSISTANT_ROLE
)
if anonymized_content:
# Replace the LLM response content with anonymized version
self._replace_llm_response_content(response, anonymized_content)
verbose_proxy_logger.debug(
f"Noma guardrail anonymized LLM response: {anonymized_content}"
)
return anonymized_content
await self._check_verdict(ASSISTANT_ROLE, content, response_json)
return content
def _should_only_sensitive_data_failed(self, classification_obj: dict) -> bool:
"""
Check if only sensitive data detectors (PII, PCI, secrets) have result=true in the classification.
Args:
classification_obj: The prompt or response classification object from Noma API
Returns:
True if only sensitiveData detectors have result=true, False otherwise
"""
if not classification_obj:
return False
# Track which detectors have result=true (detected violations)
failed_detectors = []
sensitive_data_detected = False
for key, value in classification_obj.items():
if key in SENSITIVE_DATA_DETECTOR_KEYS and isinstance(value, dict):
# Check if any sensitive data detector has result=true
for data_type, data_result in value.items():
if self._is_result_true(data_result):
sensitive_data_detected = True
# Don't add to failed_detectors as we want to allow these
elif isinstance(value, dict) and "result" in value:
# Check other detectors - these should NOT have result=true
if self._is_result_true(value):
failed_detectors.append(key)
elif isinstance(value, dict):
# Handle nested detectors
for nested_key, nested_value in value.items():
if self._is_result_true(nested_value):
failed_detectors.append(f"{key}.{nested_key}")
# Return True only if sensitive data was detected AND no other detectors have result=true
return sensitive_data_detected and len(failed_detectors) == 0
def _extract_anonymized_content(
self, response_json: dict, message_type: MessageRole
) -> Optional[str]:
"""
Extract anonymized content from Noma API response.
Args:
response_json: The full response from Noma API
message_type: Either 'user' or 'assistant' to determine which content to extract
Returns:
The anonymized content string if available, None otherwise
"""
original_response = response_json.get("originalResponse", {})
if message_type == USER_ROLE:
prompt_data = original_response.get("prompt", {})
anonymized_data = prompt_data.get("anonymizedContent", {})
return anonymized_data.get("anonymized")
elif message_type == ASSISTANT_ROLE:
response_data = original_response.get("response", {})
anonymized_data = response_data.get("anonymizedContent", {})
return anonymized_data.get("anonymized")
return None
def _should_anonymize(self, response_json: dict, message_type: MessageRole) -> bool:
"""
Determine if content should be anonymized based on Noma API response.
Logic:
- If verdict=True: Content is safe, anonymize if anonymized version exists
- If verdict=False: Check if only sensitiveData detectors have result=True
- If yes: Anonymize
- If no: Block (other violations detected)
Args:
response_json: The full response from Noma API
message_type: Either 'user' or 'assistant' to determine which classification to check
Returns:
True if content should be anonymized, False if it should be blocked
"""
# Only anonymize in blocking mode when anonymize_input is enabled
if self.monitor_mode or not self.anonymize_input:
return False
verdict = response_json.get("verdict", True)
# If verdict is True, anonymize (content is considered safe)
if verdict:
return True
# If verdict is False, check if only sensitive data detectors have result=True
original_response = response_json.get("originalResponse", {})
if message_type == USER_ROLE:
classification_obj = original_response.get("prompt", {})
elif message_type == ASSISTANT_ROLE:
classification_obj = original_response.get("response", {})
else:
return False
# Anonymize only if solely sensitive data (PII/PCI/secrets) was detected
return self._should_only_sensitive_data_failed(classification_obj)
def _is_result_true(self, result_obj: Optional[Dict[str, Any]]) -> bool:
"""
Check if a result object has a "result" field that is True.
Args:
result_obj: A dictionary that may contain a "result" field
Returns:
True if the "result" field exists and is True, False otherwise
"""
if not result_obj or not isinstance(result_obj, dict):
return False
return result_obj.get("result") is True
def _replace_user_message_content(
self, request_data: dict, anonymized_content: str
):
"""
Replace the user message content in request data with anonymized version.
Args:
request_data: The original request data
anonymized_content: The anonymized content to replace with
"""
messages = request_data.get("messages", [])
if not messages:
return
# Find and replace the last user message
for i in range(len(messages) - 1, -1, -1):
if messages[i].get("role") == USER_ROLE:
messages[i]["content"] = anonymized_content
break
def _replace_llm_response_content(
self, response: LLMResponse, anonymized_content: str
):
"""
Replace the LLM response content with anonymized version.
Args:
response: The original LLM response
anonymized_content: The anonymized content to replace with
"""
if not isinstance(response, litellm.ModelResponse):
return
# Replace content in all choices
for choice in response.choices:
if isinstance(choice, litellm.Choices) and choice.message.content:
choice.message.content = anonymized_content
async def _check_user_message_background(
self,
request_data: dict,
user_auth: UserAPIKeyAuth,
) -> None:
"""Check user message in background for monitor mode - non-blocking"""
try:
await self._process_user_message_check(request_data, user_auth)
except Exception as e:
verbose_proxy_logger.error(
f"Noma background user message check failed: {str(e)}"
)
async def _check_llm_response_background(
self,
request_data: dict,
response: LLMResponse,
user_auth: UserAPIKeyAuth,
) -> None:
"""Check LLM response in background for monitor mode - non-blocking"""
try:
await self._process_llm_response_check(request_data, response, user_auth)
except Exception as e:
verbose_proxy_logger.error(
f"Noma background response check failed: {str(e)}"
)
async def _handle_verdict_background(
self,
type: MessageRole,
message: str,
response_json: dict,
) -> None:
"""Handle verdict from Noma API in background - logging only, never blocks"""
try:
if not response_json.get("verdict", True):
msg = f"Noma guardrail blocked {type} message: {message}"
verbose_proxy_logger.warning(msg)
else:
msg = f"Noma guardrail allowed {type} message: {message}"
verbose_proxy_logger.info(msg)
except Exception as e:
verbose_proxy_logger.error(
f"Noma background verdict handling failed: {str(e)}"
)
async def async_pre_call_hook(
self,
user_api_key_dict: UserAPIKeyAuth,
@ -191,6 +521,18 @@ class NomaGuardrail(CustomGuardrail):
):
return data
# In monitor mode, run Noma check in background and return immediately
if self.monitor_mode:
try:
self._create_background_noma_check(
self._check_user_message_background(data, user_api_key_dict)
)
except Exception as e:
verbose_proxy_logger.error(
f"Failed to start background Noma pre-call check: {str(e)}"
)
return data
try:
return await self._check_user_message(data, user_api_key_dict)
except NomaBlockedMessage:
@ -198,7 +540,7 @@ class NomaGuardrail(CustomGuardrail):
except Exception as e:
verbose_proxy_logger.error(f"Noma pre-call hook failed: {str(e)}")
if self.block_failures and not self.monitor_mode:
if self.block_failures:
raise
return data
@ -220,6 +562,18 @@ class NomaGuardrail(CustomGuardrail):
if self.should_run_guardrail(data=data, event_type=event_type) is not True:
return data
# In monitor mode, run Noma check in background and return immediately
if self.monitor_mode:
try:
self._create_background_noma_check(
self._check_user_message_background(data, user_api_key_dict)
)
except Exception as e:
verbose_proxy_logger.error(
f"Failed to start background Noma moderation check: {str(e)}"
)
return data
try:
return await self._check_user_message(data, user_api_key_dict)
except NomaBlockedMessage:
@ -227,7 +581,7 @@ class NomaGuardrail(CustomGuardrail):
except Exception as e:
verbose_proxy_logger.error(f"Noma moderation hook failed: {str(e)}")
if self.block_failures and not self.monitor_mode:
if self.block_failures:
raise
return data
@ -235,19 +589,33 @@ class NomaGuardrail(CustomGuardrail):
self,
data: dict,
user_api_key_dict: UserAPIKeyAuth,
response: Union[Any, ModelResponse, EmbeddingResponse, ImageResponse],
response: LLMResponse,
):
event_type: GuardrailEventHooks = GuardrailEventHooks.post_call
if self.should_run_guardrail(data=data, event_type=event_type) is not True:
return response
# In monitor mode, run Noma check in background and return immediately
if self.monitor_mode:
try:
self._create_background_noma_check(
self._check_llm_response_background(
data, response, user_api_key_dict
)
)
except Exception as e:
verbose_proxy_logger.error(
f"Failed to start background Noma post-call check: {str(e)}"
)
return response
try:
return await self._check_llm_response(data, response, user_api_key_dict)
except NomaBlockedMessage:
raise
except Exception as e:
verbose_proxy_logger.error(f"Noma post-call hook failed: {str(e)}")
if self.block_failures and not self.monitor_mode:
if self.block_failures:
raise
return response
@ -257,55 +625,24 @@ class NomaGuardrail(CustomGuardrail):
user_auth: UserAPIKeyAuth,
) -> Union[Exception, str, dict, None]:
"""Check user message for policy violations"""
extra_data = self.get_guardrail_dynamic_request_body_params(request_data)
user_message = await self._extract_user_message(request_data)
user_message = await self._process_user_message_check(request_data, user_auth)
if not user_message:
return request_data
payload = {"request": {"text": user_message}}
response_json = await self._call_noma_api(
payload=payload,
llm_request_id=None,
request_data=request_data,
user_auth=user_auth,
extra_data=extra_data,
)
await self._check_verdict("user", user_message, response_json)
return request_data
async def _check_llm_response(
self,
request_data: dict,
response: Union[Any, ModelResponse, EmbeddingResponse, ImageResponse],
response: LLMResponse,
user_auth: UserAPIKeyAuth,
) -> Union[Exception, ModelResponse, Any]:
"""Check LLM response for policy violations"""
extra_data = self.get_guardrail_dynamic_request_body_params(request_data)
if not isinstance(response, litellm.ModelResponse):
return response
content = None
for choice in response.choices:
if isinstance(choice, litellm.Choices) and choice.message.content:
content = choice.message.content
break
if not content or not isinstance(content, str):
return response
payload = {"response": {"text": content}}
response_json = await self._call_noma_api(
payload=payload,
llm_request_id=response.id,
request_data=request_data,
user_auth=user_auth,
extra_data=extra_data,
content = await self._process_llm_response_check(
request_data, response, user_auth
)
await self._check_verdict("assistant", content, response_json)
if not content:
return response
return response
@ -316,7 +653,7 @@ class NomaGuardrail(CustomGuardrail):
return None
# Get the last user message
user_messages = [msg for msg in messages if msg.get("role") == "user"]
user_messages = [msg for msg in messages if msg.get("role") == USER_ROLE]
if not user_messages:
return None
@ -371,7 +708,7 @@ class NomaGuardrail(CustomGuardrail):
async def _check_verdict(
self,
type: Literal["user", "assistant"],
type: MessageRole,
message: str,
response_json: dict,
) -> None:
@ -379,11 +716,7 @@ class NomaGuardrail(CustomGuardrail):
Check the verdict from the Noma API and raise an exception if needed
"""
if not response_json.get("verdict", True):
msg = str.format(
"Noma guardrail blocked {type} message: {message}",
type=type,
message=message,
)
msg = f"Noma guardrail blocked {type} message: {message}"
if self.monitor_mode:
verbose_proxy_logger.warning(msg)
@ -392,11 +725,7 @@ class NomaGuardrail(CustomGuardrail):
original_response = response_json.get("originalResponse", {})
raise NomaBlockedMessage(original_response)
else:
msg = str.format(
"Noma guardrail allowed {type} message: {message}",
type=type,
message=message,
)
msg = f"Noma guardrail allowed {type} message: {message}"
if self.monitor_mode:
verbose_proxy_logger.info(msg)
else:

View file

@ -42,6 +42,7 @@ class SupportedGuardrailIntegrations(Enum):
OPENAI_MODERATION = "openai_moderation"
NOMA = "noma"
class Role(Enum):
SYSTEM = "system"
ASSISTANT = "assistant"
@ -312,7 +313,6 @@ class BedrockGuardrailConfigModel(BaseModel):
)
class LakeraV2GuardrailConfigModel(BaseModel):
"""Configuration parameters for the Lakera AI v2 guardrail"""
@ -375,6 +375,10 @@ class NomaGuardrailConfigModel(BaseModel):
default=None,
description="If True, blocks requests on API failures. Defaults to True if not provided",
)
anonymize_input: Optional[bool] = Field(
default=None,
description="If True, replaces sensitive content with anonymized version when only PII/PCI/secrets are detected. Only applies in blocking mode. Defaults to False if not provided",
)
class BaseLitellmParams(BaseModel): # works for new and patch update guardrails
@ -425,7 +429,8 @@ class BaseLitellmParams(BaseModel): # works for new and patch update guardrails
)
model: Optional[str] = Field(
default=None, description="Optional field if guardrail requires a 'model' parameter"
default=None,
description="Optional field if guardrail requires a 'model' parameter",
)
# Model Armor params
@ -446,7 +451,7 @@ class BaseLitellmParams(BaseModel): # works for new and patch update guardrails
default=True,
description="Whether to fail the request if Model Armor encounters an error",
)
model_config = ConfigDict(extra="allow", protected_namespaces=())