diff --git a/cookbook/guardrails/ismalicious/README.md b/cookbook/guardrails/ismalicious/README.md index 46bcecc9e8b..b1f4b52af53 100644 --- a/cookbook/guardrails/ismalicious/README.md +++ b/cookbook/guardrails/ismalicious/README.md @@ -4,7 +4,7 @@ Inspect normalized MCP argument strings before tool execution and text/structure ## Configuration -Obtain an API key and secret from [your account](https://ismalicious.com/app/account). In your secret manager, set `ISMALICIOUS_ENCODED_API_KEY` to the Base64 encoding of `apiKey:apiSecret`, with no newline. Do not log that value or put it directly into YAML +Obtain an API key and secret from [your account](https://ismalicious.com/app/account). In your secret manager, set `ISMALICIOUS_ENCODED_API_KEY` to the Base64 encoding of `apiKey:apiSecret`, with no newline. Pass the secret explicitly through LiteLLM's `api_key` configuration as shown below; this provider does not read an ambient credential fallback. Do not log that value or put it directly into YAML Merge [`config.yaml`](config.yaml) into the proxy configuration containing your MCP servers. It sets both MCP modes and `default_on: true`. MCP subcalls do not necessarily inherit a parent chat request's guardrail selection, so relying only on a `guardrails` field on the parent request does not establish MCP enforcement @@ -32,7 +32,7 @@ This provider uses LiteLLM's native MCP guardrail translation. It supports scann The normalized list is not the complete MCP envelope: transport metadata, content-block metadata, annotations, binary fields and fields not exposed by LiteLLM's string extraction are not inspected. Scanning this list must not be described as scanning every field in the original response -The result can have existed in process memory or logging structures before the post-call inspection. Disable message/content logging and prompt storage as shown in the sample, do not install callbacks that expose raw tool output, and review your tracing configuration. Tool side effects cannot be undone. The failure message omits the raw text, URLs, credentials and upstream exception details. Normal MCP error responses may retain HTTP200 while indicating `isError`; clients must inspect the JSON-RPC/tool result rather than treating HTTP200 as an allow decision +The result can have existed in process memory or logging structures before the post-call inspection. Disable message/content logging and prompt storage as shown in the sample, do not enable detailed/debug logging or install callbacks that expose raw tool output, and review your tracing configuration. The native logging decorator records a decision summary on success and the fixed refusal on failure, but native debug logging can include allowed content. Tool side effects cannot be undone. The failure message omits the raw text, URLs, credentials and upstream exception details. Normal MCP error responses may retain HTTP200 while indicating `isError`; clients must inspect the JSON-RPC/tool result rather than treating HTTP200 as an allow decision ## Validation diff --git a/litellm/proxy/guardrails/guardrail_hooks/ismalicious/ismalicious.py b/litellm/proxy/guardrails/guardrail_hooks/ismalicious/ismalicious.py index be19ff13598..4ac5592b0a2 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/ismalicious/ismalicious.py +++ b/litellm/proxy/guardrails/guardrail_hooks/ismalicious/ismalicious.py @@ -2,15 +2,15 @@ import base64 import binascii import json from dataclasses import dataclass -from typing import TYPE_CHECKING, Final, Literal, NoReturn, Self +from typing import TYPE_CHECKING, Final, Literal, NoReturn from urllib.parse import urlsplit import httpx from pydantic import BaseModel, ConfigDict, Field, ValidationError, model_validator +from typing_extensions import Self from litellm.exceptions import GuardrailRaisedException -from litellm.integrations.custom_guardrail import CustomGuardrail -from litellm.secret_managers.main import get_secret_str +from litellm.integrations.custom_guardrail import CustomGuardrail, log_guardrail_information from litellm.types.guardrails import GuardrailEventHooks from litellm.types.utils import GenericGuardrailAPIInputs @@ -123,7 +123,7 @@ class IsMaliciousGuardrail(CustomGuardrail): default_on: bool = False, transport: httpx.AsyncBaseTransport | None = None, ) -> None: - resolved_key: Final = api_key or get_secret_str("ISMALICIOUS_ENCODED_API_KEY") + resolved_key: Final = api_key if not resolved_key: raise ValueError("IsMalicious requires a Base64 API key and secret pair") try: @@ -185,6 +185,7 @@ class IsMaliciousGuardrail(CustomGuardrail): if failure is not None: self._raise_failure(failure) + @log_guardrail_information async def apply_guardrail( self, inputs: GenericGuardrailAPIInputs, diff --git a/tests/unit/proxy/guardrails/guardrail_hooks/test_ismalicious.py b/tests/unit/proxy/guardrails/guardrail_hooks/test_ismalicious.py index 916136489d9..65c9051b4a8 100644 --- a/tests/unit/proxy/guardrails/guardrail_hooks/test_ismalicious.py +++ b/tests/unit/proxy/guardrails/guardrail_hooks/test_ismalicious.py @@ -188,6 +188,30 @@ def test_credentials_never_redirect_to_a_custom_endpoint(): ) +def test_missing_explicit_credentials_fail_at_startup(): + with pytest.raises(ValueError, match="requires a Base64"): + IsMaliciousGuardrail(event_hook=GuardrailEventHooks.pre_mcp_call) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("verdict", ["allow", "block"]) +async def test_native_decision_logging_excludes_content_and_credentials(verdict): + def handler(request): + return service_response(request, verdict) + + data = {"request_id": "test"} + inputs = {"texts": ["Private untrusted result"]} + if verdict == "block": + with pytest.raises(GuardrailRaisedException): + await guardrail(handler).apply_guardrail(inputs=inputs, request_data=data, input_type="response") + else: + await guardrail(handler).apply_guardrail(inputs=inputs, request_data=data, input_type="response") + recorded = json.dumps(data) + assert "standard_logging_guardrail_information" in recorded + assert "Private untrusted result" not in recorded + assert URL not in recorded and KEY not in recorded + + def test_mcp_subcalls_without_guardrail_metadata_use_explicit_default_on(): gate = guardrail() assert gate.should_run_guardrail(data={}, event_type=GuardrailEventHooks.pre_mcp_call)