From 3726ce2cfc4606bf55769024c0afe21331e05ef5 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:05:47 -0700 Subject: [PATCH] refactor(guardrails): fix agent 365 to the production endpoint and log the opt-in fail_open at error level (#43189) * feat(guardrails): fail open by default when Agent 365 cannot evaluate and count it in Prometheus Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * style(tests): ruff format the Prometheus fail-open registry test Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(guardrails): add Agent 365 authority host override, fail-open integration test and per-guardrail YAML default Add `authority_host` to the Agent 365 config (also read from AGENT365_AUTHORITY_HOST, then AZURE_AUTHORITY_HOST) so sovereign clouds and the integration test can point the OBO exchange at a different Entra host. Add tests/integration/mcp/test_mcp_agent_365_guardrail.py, a real proxy test with Postgres, Redis, a scripted MCP upstream and local Entra and Agent 365 doubles covering the default fail-open, explicit fail-closed and fail-open, Defender Skipped, policy denial, persisted status and Prometheus counter. Use PrometheusLogger.get_instance for the fail-open metric lookup instead of a hand-rolled callback scan. Clarify the config description: gateway credential failures fail open, caller token failures block. Extract the dashboard YAML preview into teamGuardrailConfigYaml.ts so the effective per-guardrail default is unit tested and the "default" hint only shows when nothing was set explicitly. Regenerate the lazy OpenAPI snapshot and schema.d.ts for the new field. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(guardrails): default a scheme-less Agent 365 authority host to https and treat a null fallback as unset in the YAML preview Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): audit cells for the Agent 365 fail-open default across entry points, Entra faults, throttling and two workers Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * test(integration): prove both Agent 365 workers serve and that a killed worker is replaced Each fresh connection reports its worker pid from /debug/memory/summary and its MCP catalog on the same connection, so the two-worker readiness wait covers both workers by identity. The kill test now kills a pid the proxy reported as a worker and waits for a replacement pid, instead of the first psutil child Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(guardrails): drop the prometheus fail-open counter from the agent 365 guardrail Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(guardrails): keep agent 365 fail closed by default and make fail_open an explicit opt-in Restores the shared unreachable_fallback default and the sibling guardrail initializers, drops the Admin UI YAML preview that only existed for the per-guardrail default, and reworks the unit and integration tests so the default blocks with HTTP 503 while unreachable_fallback: fail_open lets availability failures through as Unscanned Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(guardrails): append authority_host after the existing Agent365Guardrail parameters Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * feat(guardrails): agent 365 fails open by default and hides the production overrides from the UI form Agent 365 sits in the runtime path of every MCP tool call, so an Entra or Agent 365 outage now lets the call through unscanned (logged at error level, recorded as Unscanned with guardrail_failed_to_respond) instead of blocking it. unreachable_fallback: fail_closed stays as the opt-in strict mode. Policy blocks, throttling, 4xx rejections and a rejected caller token still block The shared unreachable_fallback field becomes nullable so each guardrail owns its default; every sibling still resolves None to fail_closed and typesafe keeps failing open api_base, resource_app_id and agent_id have production defaults and leave the dashboard form (ui_hidden); they stay available in config.yaml and env. The authority_host override and its env keys are gone, the OBO exchange always uses login.microsoftonline.com. The integration suite keeps only the cells that need no Entra double, the evaluation paths live in unit tests with an injected handler Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * chore(ui): regenerate openapi snapshot and schema.d.ts for the nullable unreachable_fallback Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(guardrails): fix agent 365 to the production endpoint and keep fail_closed as the default Remove api_base, resource_app_id and agent_id from the Agent 365 config model, their AGENT365_* env fallbacks and the _is_ui_hidden helper: the evaluation URL and the Agent Tools app id are fixed production constants and the agent identity is always the caller's key alias. Revert the fail_open default; unreachable_fallback: fail_open stays an explicit opt-in. Restore the shared unreachable_fallback field, the sibling guardrail initializers and typesafe to main. Move the Entra dependent cells from the subprocess integration suite to unit tests with an injected HTTP handler. Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * fix(guardrails): warn when agent 365 yaml still carries the removed override keys Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> * refactor(guardrails): inject the http handler into the agent 365 initializer instead of assigning it after construction Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --------- Co-authored-by: yucheng Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/proxy/_lazy_openapi_snapshot.json | 24 --- .../guardrail_hooks/agent_365/__init__.py | 37 ++-- .../guardrail_hooks/agent_365/agent_365.py | 19 +- .../guardrails/guardrail_hooks/agent_365.py | 31 +--- .../mcp/test_mcp_agent_365_guardrail.py | 163 ++++++++++++++++++ .../guardrail_hooks/test_agent_365.py | 124 +++++++++++-- ui/litellm-dashboard/src/lib/http/schema.d.ts | 10 -- 7 files changed, 309 insertions(+), 99 deletions(-) create mode 100644 tests/integration/mcp/test_mcp_agent_365_guardrail.py diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index ded4db6d2aa..db57aa4f046 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -11368,18 +11368,6 @@ "description": "Custom advisory message template used when on_flagged='inject_system_message'. Must contain a {reason} placeholder. Defaults to a generic advisory message if unset.", "title": "Advisory System Message" }, - "agent_id": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "description": "Agent identity reported to Agent 365 with every tool evaluation. When unset, the caller's key alias is used.", - "title": "Agent Id" - }, "akto_account_id": { "anyOf": [ { @@ -12951,18 +12939,6 @@ "description": "The message the bot speaks aloud when a /v1/realtime guardrail fires. Falls back to violation_message_template if not set.", "title": "Realtime Violation Message" }, - "resource_app_id": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "description": "Application id of the Agent 365 resource the OBO token is minted for. Defaults to the production resource ea9ffc3e-8a23-4a7d-836d-234d7c7565c1; the Test and PreProd environments use a different id. Falls back to the AGENT365_RESOURCE_APP_ID environment variable.", - "title": "Resource App Id" - }, "rules": { "anyOf": [ { diff --git a/litellm/proxy/guardrails/guardrail_hooks/agent_365/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/agent_365/__init__.py index 9aacdec0602..2a8c6479ae6 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/agent_365/__init__.py +++ b/litellm/proxy/guardrails/guardrail_hooks/agent_365/__init__.py @@ -1,18 +1,21 @@ from typing import TYPE_CHECKING, Final +from litellm._logging import verbose_proxy_logger from litellm.types.guardrails import SupportedGuardrailIntegrations -from litellm.types.proxy.guardrails.guardrail_hooks.agent_365 import ( - AGENT_365_PROD_API_BASE, - AGENT_365_PROD_RESOURCE_APP_ID, -) from .agent_365 import Agent365Guardrail if TYPE_CHECKING: + from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from litellm.types.guardrails import Guardrail, LitellmParams -def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail") -> Agent365Guardrail: +def initialize_guardrail( + litellm_params: "LitellmParams", + guardrail: "Guardrail", + *, + async_handler: "AsyncHTTPHandler | None" = None, +) -> Agent365Guardrail: import litellm from litellm.secret_managers.main import get_secret_str @@ -21,8 +24,6 @@ def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail" client_secret: Final = ( litellm_params.client_secret or litellm_params.api_key or get_secret_str("AGENT365_CLIENT_SECRET") ) - api_base: Final = litellm_params.api_base or get_secret_str("AGENT365_API_BASE") - resource_app_id: Final = litellm_params.resource_app_id or get_secret_str("AGENT365_RESOURCE_APP_ID") if not tenant_id: raise ValueError("Microsoft Agent 365: tenant_id is required") @@ -37,16 +38,32 @@ def initialize_guardrail(litellm_params: "LitellmParams", guardrail: "Guardrail" if not guardrail_name: raise ValueError("Microsoft Agent 365: guardrail_name is required") + extras: Final = litellm_params.model_extra or {} + ignored_overrides: Final = tuple( + key + for key, value in ( + ("api_base", litellm_params.api_base), + ("resource_app_id", extras.get("resource_app_id")), + ("agent_id", extras.get("agent_id")), + ) + if value is not None + ) + if ignored_overrides: + verbose_proxy_logger.warning( + "Microsoft Agent 365 (%s): ignoring %s; evaluations always go to the production Agent 365 endpoint " + "and the agent identity is the caller's key alias", + guardrail_name, + ", ".join(ignored_overrides), + ) + agent_365_guardrail: Final = Agent365Guardrail( guardrail_name=guardrail_name, tenant_id=tenant_id, client_id=client_id, client_secret=client_secret, - api_base=api_base or AGENT_365_PROD_API_BASE, - resource_app_id=resource_app_id or AGENT_365_PROD_RESOURCE_APP_ID, - agent_id=litellm_params.agent_id, request_timeout=litellm_params.timeout if litellm_params.timeout is not None else 10.0, unreachable_fallback=litellm_params.unreachable_fallback, + async_handler=async_handler, event_hook=litellm_params.mode, default_on=litellm_params.default_on, ) diff --git a/litellm/proxy/guardrails/guardrail_hooks/agent_365/agent_365.py b/litellm/proxy/guardrails/guardrail_hooks/agent_365/agent_365.py index 975d321104d..52f3eeb4ce7 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/agent_365/agent_365.py +++ b/litellm/proxy/guardrails/guardrail_hooks/agent_365/agent_365.py @@ -49,7 +49,8 @@ if TYPE_CHECKING: from litellm.types.utils import GuardrailStatus TOKEN_ENDPOINT_TEMPLATE: Final = "https://login.microsoftonline.com/{tenant_id}/oauth2/v2.0/token" -EVALUATE_PATH: Final = "/agents/tool-evaluation/evaluate" +EVALUATE_URL: Final = f"{AGENT_365_PROD_API_BASE}/agents/tool-evaluation/evaluate" +OBO_SCOPE: Final = f"{AGENT_365_PROD_RESOURCE_APP_ID}/{AGENT_365_SCOPE_NAME}" MCP_SESSION_ID_HEADER: Final = "mcp-session-id" DEFENDER_STATUS_EVALUATED: Final = "Evaluated" _GATEWAY_OWNED_TOKEN_ERRORS: Final = frozenset( @@ -153,9 +154,6 @@ class Agent365Guardrail(CustomGuardrail): tenant_id: str, client_id: str, client_secret: str, - api_base: str = AGENT_365_PROD_API_BASE, - resource_app_id: str = AGENT_365_PROD_RESOURCE_APP_ID, - agent_id: str | None = None, request_timeout: float = 10.0, unreachable_fallback: Literal["fail_closed", "fail_open"] = "fail_closed", async_handler: AsyncHTTPHandler | None = None, @@ -171,9 +169,6 @@ class Agent365Guardrail(CustomGuardrail): self.tenant_id = tenant_id self.client_id = client_id self.client_secret = client_secret - self.api_base = api_base.rstrip("/") - self.resource_app_id = resource_app_id - self.agent_id = agent_id self.request_timeout = request_timeout self.unreachable_fallback: Literal["fail_closed", "fail_open"] = ( "fail_open" if unreachable_fallback == "fail_open" else "fail_closed" @@ -230,7 +225,7 @@ class Agent365Guardrail(CustomGuardrail): tool_name=tool_name, reason=( f"Entra rejected the gateway's own Agent 365 credentials ({exc.error_code}); " - "check the guardrail's client_id, client_secret and resource_app_id" + "check the guardrail's client_id and client_secret" ), ) self._handle_caller_fault( @@ -262,7 +257,7 @@ class Agent365Guardrail(CustomGuardrail): start: Final = time.perf_counter() try: response: Final = await self._post_allowing_error_status( - url=f"{self.api_base}{EVALUATE_PATH}", + url=EVALUATE_URL, json=self._build_evaluate_payload(data=data, user_api_key_dict=user_api_key_dict), headers={"Authorization": f"Bearer {obo_token}"}, # mutable-ok: httpx header dict ) @@ -396,7 +391,7 @@ class Agent365Guardrail(CustomGuardrail): tool_name: Final = str(data.get("mcp_tool_name") or "") arguments: Final = data.get("mcp_arguments") server_name: Final = str(data.get("mcp_server_name") or "litellm") - agent_id: Final = self.agent_id or user_api_key_dict.key_alias + agent_id: Final = user_api_key_dict.key_alias payload: Final[dict[str, object]] = { # mutable-ok: JSON body with optional fields added below "tool": {"name": tool_name}, "serverName": server_name, @@ -454,7 +449,7 @@ class Agent365Guardrail(CustomGuardrail): "client_id": self.client_id, "client_secret": self.client_secret, "assertion": assertion, - "scope": f"{self.resource_app_id}/{AGENT_365_SCOPE_NAME}", + "scope": OBO_SCOPE, "requested_token_use": "on_behalf_of", }, headers={"Content-Type": "application/x-www-form-urlencoded"}, # mutable-ok: httpx header dict @@ -575,7 +570,7 @@ class Agent365Guardrail(CustomGuardrail): latency_ms: float | None = None, ) -> dict: # mutable-ok: returns the request data dict per hook contract if self.unreachable_fallback == "fail_open": - verbose_proxy_logger.warning( + verbose_proxy_logger.error( "Agent 365 guardrail (%s): %s; unreachable_fallback='fail_open', allowing tool call '%s' unscanned", self.guardrail_name, reason, diff --git a/litellm/types/proxy/guardrails/guardrail_hooks/agent_365.py b/litellm/types/proxy/guardrails/guardrail_hooks/agent_365.py index dd3d7fe5f74..70ac10ae082 100644 --- a/litellm/types/proxy/guardrails/guardrail_hooks/agent_365.py +++ b/litellm/types/proxy/guardrails/guardrail_hooks/agent_365.py @@ -1,4 +1,4 @@ -from typing import Final +from typing import Final, Literal from pydantic import Field @@ -34,30 +34,13 @@ class Agent365GuardrailConfigModel(GuardrailConfigModel): ), ) - api_base: str | None = Field( - default=None, + unreachable_fallback: Literal["fail_closed", "fail_open"] = Field( + default="fail_closed", description=( - "Base URL of the Microsoft Agent 365 tool-evaluation endpoint. " - f"Defaults to the production endpoint {AGENT_365_PROD_API_BASE}. " - "Falls back to the AGENT365_API_BASE environment variable." - ), - ) - - resource_app_id: str | None = Field( - default=None, - description=( - "Application id of the Agent 365 resource the OBO token is minted for. " - f"Defaults to the production resource {AGENT_365_PROD_RESOURCE_APP_ID}; " - "the Test and PreProd environments use a different id. " - "Falls back to the AGENT365_RESOURCE_APP_ID environment variable." - ), - ) - - agent_id: str | None = Field( - default=None, - description=( - "Agent identity reported to Agent 365 with every tool evaluation. " - "When unset, the caller's key alias is used." + "Behavior when Agent 365 or Entra is unreachable, times out, returns 5xx, skips the evaluation, or " + "rejects the gateway's own client credentials. 'fail_closed' (default) blocks the tool call with HTTP 503. " + "'fail_open' allows it, logs an error and records it as Unscanned in the logs and OpenTelemetry. " + "Policy blocks, 4xx rejections, throttling and a rejected caller token always block." ), ) diff --git a/tests/integration/mcp/test_mcp_agent_365_guardrail.py b/tests/integration/mcp/test_mcp_agent_365_guardrail.py new file mode 100644 index 00000000000..5842d3ce4a7 --- /dev/null +++ b/tests/integration/mcp/test_mcp_agent_365_guardrail.py @@ -0,0 +1,163 @@ +"""Agent 365 guardrail paths that end before the OBO exchange, so neither Entra nor Agent 365 is ever contacted.""" + +import json +import uuid +from collections.abc import Iterator +from contextlib import contextmanager +from dataclasses import dataclass +from hashlib import sha256 +from pathlib import Path +from typing import Final + +import httpx +import pytest +import yaml +from integration._support.client import Gateway, eventually +from integration._support.database import read_rows +from integration._support.mcp import ( + ENTRY_POINTS, + EntryPoint, + McpCaller, + McpPeer, + echo_tool, + register_mcp, + scripted_peer, + tool_calls, +) +from integration._support.process import owned_proxy_process +from integration._support.wire import Reply, Request, wire_server + +TENANT: Final = "00000000-0000-4000-8000-0000000a3650" +REJECTED: Final = "Agent 365 guardrail rejected the tool call" +GUARDRAIL_ROWS: Final = ( + "SELECT metadata->'guardrail_information' AS gi FROM \"LiteLLM_SpendLogs\" " + 'WHERE api_key = %s AND call_type = %s ORDER BY "startTime"' +) +FALLBACKS: Final = (None, "fail_open", "fail_closed") + + +def _generic_guardrail_outage(request: Request) -> Reply: + assert request.target == "/beta/litellm_basic_guardrail_api", request.target + return Reply(status=503, body=json.dumps({"error": "synthetic sibling guardrail outage"}).encode()) + + +def _config(tmp_path: Path, name: str, fallback: str | None, sibling_url: str | None) -> Path: + config: dict = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["guardrails"] = [ + { + "guardrail_name": name, + "litellm_params": { + "guardrail": "agent_365", + "mode": "pre_mcp_call", + "default_on": True, + "tenant_id": TENANT, + "client_id": "synthetic-client-id", + "client_secret": "synthetic-client-secret", + **({"unreachable_fallback": fallback} if fallback else {}), + }, + }, + *( + [ + { + "guardrail_name": f"{name}-sibling", + "litellm_params": { + "guardrail": "generic_guardrail_api", + "mode": "pre_call", + "default_on": True, + "api_base": sibling_url, + }, + } + ] + if sibling_url + else [] + ), + ] + path: Final = tmp_path / "agent_365.yaml" + tmp_path.mkdir(parents=True, exist_ok=True) + path.write_text(yaml.safe_dump(config)) + return path + + +@dataclass(frozen=True, slots=True) +class Rig: + candidate: Gateway + key: str + alias: str + peer: McpPeer + server_id: str + + def caller(self, entry: EntryPoint = "mcp", bearer: str | None = None) -> McpCaller: + headers: Final = {"Authorization": f"Bearer {bearer}"} if bearer else {} + return McpCaller(self.candidate, self.key, entry, self.alias, headers=headers) + + def upstream_tool_names(self) -> tuple[str, ...]: + return tuple(str(call["body"]["params"]["name"]) for call in tool_calls(self.peer.drain())) + + def guardrail_statuses(self, call_type: str, at_least: int) -> list[str]: + rows: Final = eventually( + lambda: read_rows(GUARDRAIL_ROWS, (sha256(self.key.encode()).hexdigest(), call_type)), + lambda seen: len(seen) >= at_least, + seconds=70, + ) + return [row["gi"][0]["guardrail_status"] if row["gi"] else "none" for row in rows] + + +@contextmanager +def _rig(gateway: Gateway, tmp_path: Path, fallback: str | None, *, sibling: bool = False) -> Iterator[Rig]: + alias: Final = "a365" + uuid.uuid4().hex[:8] + with ( + wire_server(_generic_guardrail_outage) as sibling_outage, + scripted_peer(echo_tool("add")) as peer, + owned_proxy_process( + gateway, + tmp_path, + {"PROXY_CONFIG_RELOAD_INTERVAL_SECONDS": "2"}, + config=_config(tmp_path, alias, fallback, sibling_outage.url if sibling else None), + ) as owned, + owned.gateway.scenario() as scenario, + ): + identity: Final = register_mcp(scenario, peer, alias) + key: Final = scenario.key(object_permission={"mcp_servers": [identity]}) + peer.drain() + yield Rig(owned.gateway, key, alias, peer, identity) + + +def _chat(rig: Rig, model: str, marker: str) -> httpx.Response: + return rig.candidate.request( + "POST", "/v1/chat/completions", {"model": model, "messages": [{"role": "user", "content": marker}]}, key=rig.key + ) + + +@pytest.mark.parametrize("fallback", FALLBACKS) +def test_a_missing_or_malformed_caller_bearer_blocks_on_every_entry_point_whatever_the_fallback( + gateway: Gateway, tmp_path: Path, fallback: str | None +) -> None: + with _rig(gateway, tmp_path, fallback) as rig: + assert f"{rig.alias}-add" in rig.caller().list_tools().tools, "the catalog needs only the virtual key" + for entry in ENTRY_POINTS: + missing: Final = rig.caller(entry).call(f"{rig.alias}-add", {"entry": entry}, server_id=rig.server_id) + assert missing.error is not None and REJECTED in missing.raw, f"{entry} without a bearer: {missing.raw}" + malformed: Final = rig.caller(entry, "not-a-jws").call( + f"{rig.alias}-add", {"entry": entry}, server_id=rig.server_id + ) + assert malformed.error is not None and REJECTED in malformed.raw, f"{entry} opaque bearer: {malformed.raw}" + assert rig.upstream_tool_names() == () + expected: Final = 2 * len(ENTRY_POINTS) + assert rig.guardrail_statuses("call_mcp_tool", expected) == ["guardrail_intervened"] * expected + + +def test_chat_completions_are_unaffected_by_the_mcp_guardrail(gateway: Gateway, tmp_path: Path) -> None: + with _rig(gateway, tmp_path, fallback=None) as rig, rig.candidate.scenario() as scenario: + model: Final = scenario.model() + chat: Final = _chat(rig, model, "unaffected-" + uuid.uuid4().hex) + assert chat.status_code == 200, chat.text + assert rig.guardrail_statuses("acompletion", 1) == ["none"] + + +def test_a_sibling_guardrail_keeps_its_own_fail_closed_default_next_to_agent_365( + gateway: Gateway, tmp_path: Path +) -> None: + with _rig(gateway, tmp_path, fallback=None, sibling=True) as rig, rig.candidate.scenario() as scenario: + model: Final = scenario.model() + chat: Final = _chat(rig, model, "sibling-" + uuid.uuid4().hex) + assert chat.status_code == 500 and "Generic Guardrail API failed" in chat.text, chat.text diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_agent_365.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_agent_365.py index f9b7561b9d3..23131321938 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_agent_365.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/test_agent_365.py @@ -1,3 +1,4 @@ +import logging import time import uuid from types import SimpleNamespace @@ -121,16 +122,12 @@ def _make_guardrail( handler: FakeHandler, *, unreachable_fallback: str = "fail_closed", - agent_id: str | None = None, - api_base: str = AGENT_365_PROD_API_BASE, ) -> Agent365Guardrail: return Agent365Guardrail( guardrail_name="agent-365-guard", tenant_id="tenant-abc", client_id="client-xyz", client_secret="secret-123", - api_base=api_base, - agent_id=agent_id, unreachable_fallback=unreachable_fallback, async_handler=handler, event_hook="pre_mcp_call", @@ -138,6 +135,18 @@ def _make_guardrail( ) +def _default_fallback_guardrail(handler: FakeHandler) -> Agent365Guardrail: + return Agent365Guardrail( + guardrail_name="agent-365-guard", + tenant_id="tenant-abc", + client_id="client-xyz", + client_secret="secret-123", + async_handler=handler, + event_hook="pre_mcp_call", + default_on=True, + ) + + def _mcp_data(**overrides: Any) -> dict: data: Final[dict] = { "mcp_tool_name": "send_email", @@ -206,20 +215,59 @@ class TestInitializeGuardrail: assert redact_string(str(exc_info.value)) == str(exc_info.value) def test_env_var_fallbacks(self, monkeypatch): - monkeypatch.delenv("AGENT365_RESOURCE_APP_ID", raising=False) monkeypatch.setenv("AGENT365_TENANT_ID", "env-tenant") monkeypatch.setenv("AGENT365_CLIENT_ID", "env-client") monkeypatch.setenv("AGENT365_CLIENT_SECRET", "env-secret") - monkeypatch.setenv("AGENT365_API_BASE", "https://env.example.test") params: Final = LitellmParams(guardrail="agent_365", mode="pre_mcp_call") guardrail: Final = initialize_guardrail(params, {"guardrail_name": "a365-env"}) assert guardrail.tenant_id == "env-tenant" assert guardrail.client_id == "env-client" assert guardrail.client_secret == "env-secret" - assert guardrail.api_base == "https://env.example.test" - assert guardrail.resource_app_id == AGENT_365_PROD_RESOURCE_APP_ID assert guardrail.unreachable_fallback == "fail_closed" + def test_fail_open_is_opt_in_through_litellm_params(self): + params: Final = LitellmParams( + guardrail="agent_365", + mode="pre_mcp_call", + tenant_id="t", + client_id="c", + client_secret="s", + unreachable_fallback="fail_open", + ) + guardrail: Final = initialize_guardrail(params, {"guardrail_name": "a365-open"}) + assert guardrail.unreachable_fallback == "fail_open" + + def test_ui_form_offers_only_the_credentials_and_the_fallback(self): + from litellm.proxy.guardrails.guardrail_endpoints import _get_fields_from_model + + fields: Final = _get_fields_from_model(Agent365GuardrailConfigModel) + assert set(fields) == {"tenant_id", "client_id", "client_secret", "unreachable_fallback"} + assert fields["unreachable_fallback"]["default_value"] == "fail_closed" + + @pytest.mark.asyncio + async def test_stale_yaml_overrides_are_ignored_and_logged(self, caplog): + params: Final = LitellmParams( + guardrail="agent_365", + mode="pre_mcp_call", + tenant_id="tenant-abc", + client_id="client-xyz", + client_secret="secret-123", + default_on=True, + api_base="https://agent365.example.test", + resource_app_id="00000000-0000-0000-0000-000000000000", + agent_id="yaml-agent", + ) + handler: Final = FakeHandler([_token_response(), _allow_response()]) + with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"): + guardrail: Final = initialize_guardrail(params, {"guardrail_name": "a365-stale"}, async_handler=handler) + assert "ignoring api_base, resource_app_id, agent_id" in caplog.text + await _run(guardrail, _mcp_data()) + token_call, evaluate_call = handler.calls + assert token_call.url == TOKEN_URL + assert token_call.data["scope"] == f"{AGENT_365_PROD_RESOURCE_APP_ID}/ThreatProtection.Evaluate.All" + assert evaluate_call.url == EVALUATE_URL + assert evaluate_call.json["agentId"] == "my-agent-key" + def test_explicit_params_win(self, monkeypatch): monkeypatch.setenv("AGENT365_TENANT_ID", "env-tenant") params: Final = LitellmParams( @@ -228,14 +276,12 @@ class TestInitializeGuardrail: tenant_id="param-tenant", client_id="client-xyz", client_secret="param-secret", - agent_id="agent-007", unreachable_fallback="fail_open", timeout=5, ) guardrail: Final = initialize_guardrail(params, {"guardrail_name": "a365-params"}) assert guardrail.tenant_id == "param-tenant" assert guardrail.client_secret == "param-secret" - assert guardrail.agent_id == "agent-007" assert guardrail.unreachable_fallback == "fail_open" assert guardrail.request_timeout == 5.0 @@ -289,7 +335,7 @@ class TestAllowFlow: @pytest.mark.asyncio async def test_evaluate_payload(self): handler: Final = FakeHandler([_token_response(), _allow_response()]) - guardrail: Final = _make_guardrail(handler, agent_id="agent-007") + guardrail: Final = _make_guardrail(handler) await _run(guardrail, _mcp_data()) evaluate_call: Final = handler.calls[1] assert evaluate_call.url == EVALUATE_URL @@ -298,14 +344,7 @@ class TestAllowFlow: assert evaluate_call.json["serverName"] == "outlook_mcp" assert evaluate_call.json["arguments"] == {"to": "user@example.com", "body": "hello"} assert evaluate_call.json["conversationId"] == "sess-123" - assert evaluate_call.json["agentId"] == "agent-007" - - @pytest.mark.asyncio - async def test_agent_id_falls_back_to_key_alias(self): - handler: Final = FakeHandler([_token_response(), _allow_response()]) - guardrail: Final = _make_guardrail(handler) - await _run(guardrail, _mcp_data()) - assert handler.calls[1].json["agentId"] == "my-agent-key" + assert evaluate_call.json["agentId"] == "my-agent-key" @pytest.mark.asyncio async def test_non_mcp_call_type_skipped(self): @@ -471,6 +510,53 @@ class TestDefenderNotEvaluated: assert "rejected" in exc_info.value.detail["error"] +AVAILABILITY_FAILURES: Final = ( + pytest.param([_token_response(), httpx.ReadTimeout("timed out")], id="evaluate-timeout"), + pytest.param([_token_response(), _response(502, text="bad gateway")], id="evaluate-5xx"), + pytest.param([_token_response(), _not_evaluated_response("Skipped")], id="evaluate-skipped"), + pytest.param([_response(503, text="entra down")], id="entra-5xx"), +) + + +class TestFailOpenOptIn: + @pytest.mark.asyncio + @pytest.mark.parametrize("responses", AVAILABILITY_FAILURES) + async def test_constructor_default_blocks_each_availability_failure_with_503(self, responses): + guardrail: Final = _default_fallback_guardrail(FakeHandler(responses)) + assert guardrail.unreachable_fallback == "fail_closed" + with pytest.raises(HTTPException) as exc_info: + await _run(guardrail, _mcp_data()) + assert exc_info.value.status_code == 503 + assert "fail_closed" in exc_info.value.detail["message"] + + @pytest.mark.asyncio + @pytest.mark.parametrize("responses", AVAILABILITY_FAILURES) + async def test_opted_in_fail_open_lets_each_availability_failure_through_as_failed_to_respond(self, responses): + guardrail: Final = _make_guardrail(FakeHandler(responses), unreachable_fallback="fail_open") + data: Final = _mcp_data() + assert await _run(guardrail, data) is data + info: Final = _guardrail_info(data) + assert info["guardrail_status"] == "guardrail_failed_to_respond" + assert info["guardrail_response"]["verdict"] == "Unscanned" + + @pytest.mark.asyncio + async def test_opted_in_fail_open_logs_the_unscanned_call_at_error_level(self, caplog): + handler: Final = FakeHandler([_token_response(), httpx.ReadTimeout("timed out")]) + guardrail: Final = _make_guardrail(handler, unreachable_fallback="fail_open") + with caplog.at_level(logging.ERROR, logger="LiteLLM Proxy"): + await _run(guardrail, _mcp_data()) + fail_open_logs: Final = [r for r in caplog.records if "unreachable_fallback='fail_open'" in r.getMessage()] + assert [r.levelno for r in fail_open_logs] == [logging.ERROR], caplog.text + + @pytest.mark.asyncio + async def test_opted_in_fail_open_still_blocks_a_policy_block(self): + handler: Final = FakeHandler([_token_response(), _block_response()]) + guardrail: Final = _make_guardrail(handler, unreachable_fallback="fail_open") + with pytest.raises(HTTPException) as exc_info: + await _run(guardrail, _mcp_data()) + assert exc_info.value.status_code == 400 + + class TestUnreachableFallback: @pytest.mark.asyncio async def test_evaluate_litellm_timeout_fail_closed(self): diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 2b8ed9aa58d..5924abfb0c6 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -34030,11 +34030,6 @@ export interface components { * @description Custom advisory message template used when on_flagged='inject_system_message'. Must contain a {reason} placeholder. Defaults to a generic advisory message if unset. */ advisory_system_message?: string | null; - /** - * Agent Id - * @description Agent identity reported to Agent 365 with every tool evaluation. When unset, the caller's key alias is used. - */ - agent_id?: string | null; /** * Akto Account Id * @description Akto account ID for multi-tenant deployments. Env: AKTO_ACCOUNT_ID. Default: '1000000'. @@ -34684,11 +34679,6 @@ export interface components { * @description The message the bot speaks aloud when a /v1/realtime guardrail fires. Falls back to violation_message_template if not set. */ realtime_violation_message?: string | null; - /** - * Resource App Id - * @description Application id of the Agent 365 resource the OBO token is minted for. Defaults to the production resource ea9ffc3e-8a23-4a7d-836d-234d7c7565c1; the Test and PreProd environments use a different id. Falls back to the AGENT365_RESOURCE_APP_ID environment variable. - */ - resource_app_id?: string | null; /** * Rules * @description Ordered allow/deny rules. Patterns use regex for tool names/types and optional regex constraints on tool arguments.