From e394867fc915c54a0535e4c9fef6cc339101aa39 Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Thu, 19 Feb 2026 14:32:42 -0800 Subject: [PATCH] feat: implement guardrails usage dashboard backend Add backend infrastructure for guardrails performance monitoring dashboard: Database Schema: - Add LiteLLM_DailyGuardrailMetrics table for daily aggregated metrics - Track total_requests, success/intervened/failed/not_run counts per guardrail - Store aggregated latency metrics in milliseconds - Unique constraint on [guardrail_name, provider, mode, date, api_key] Data Collection & Aggregation: - Add DailyGuardrailMetricsTransaction type for queue transactions - Implement guardrail metrics extraction from spend log metadata - Add batch upsert logic with retry handling (60s commit interval) - Process each guardrail separately with status-based counting API Endpoints: - GET /guardrail/metrics - List all guardrails with aggregated metrics - GET /guardrail/{name}/metrics - Detail view with daily time-series - GET /guardrail/{name}/logs - Request logs with status filtering Type Definitions: - Add Pydantic models for API request/response validation - GuardrailSummary, GuardrailDetailMetrics, GuardrailLogsResponse Key Features: - Fail rate = (intervened_count / total_requests) * 100 - Avg latency measures guardrail execution overhead only - Reuses LiteLLM_SpendLogs for per-request drill-down - Follows existing daily spend tracking patterns Testing Required: - Run: poetry run prisma migrate dev --name add_guardrail_metrics - Frontend implementation pending (Phase 5) - See IMPLEMENTATION_STATUS.md for details Co-Authored-By: Claude Opus 4.6 --- IMPLEMENTATION_STATUS.md | 196 +++++++++++ litellm/proxy/_types.py | 15 + litellm/proxy/db/db_spend_update_writer.py | 264 ++++++++++++++- .../guardrail_metrics_endpoints.py | 317 ++++++++++++++++++ litellm/proxy/proxy_server.py | 18 +- litellm/proxy/schema.prisma | 35 +- .../management_endpoints/guardrail_metrics.py | 75 +++++ 7 files changed, 901 insertions(+), 19 deletions(-) create mode 100644 IMPLEMENTATION_STATUS.md create mode 100644 litellm/proxy/management_endpoints/guardrail_metrics_endpoints.py create mode 100644 litellm/types/proxy/management_endpoints/guardrail_metrics.py diff --git a/IMPLEMENTATION_STATUS.md b/IMPLEMENTATION_STATUS.md new file mode 100644 index 00000000000..62e7618ad81 --- /dev/null +++ b/IMPLEMENTATION_STATUS.md @@ -0,0 +1,196 @@ +# Guardrails Usage Dashboard - Implementation Status + +## Completed: Backend Implementation (Phases 1-4) + +### ✅ Phase 1: Database Schema +- **File**: `litellm/proxy/schema.prisma` +- Added `LiteLLM_DailyGuardrailMetrics` table with: + - Unique constraint on `[guardrail_name, guardrail_provider, guardrail_mode, date, api_key]` + - Indexes on `date`, `guardrail_name`, `guardrail_provider`, `api_key` + - Fields for tracking total_requests, success/intervened/failed/not_run counts + - Aggregated latency metrics in milliseconds +- **Action Required**: Run `poetry run prisma migrate dev --name add_guardrail_metrics` to create the table + +### ✅ Phase 2: Data Collection & Aggregation +- **File**: `litellm/proxy/_types.py` + - Added `DailyGuardrailMetricsTransaction` TypedDict + +- **File**: `litellm/proxy/db/db_spend_update_writer.py` + - Added `daily_guardrail_metrics_update_queue` to `__init__` + - Implemented `add_spend_log_transaction_to_daily_guardrail_transaction()`: + - Extracts guardrail_information from metadata + - Creates separate transaction per guardrail + - Calculates status counts and latency in milliseconds + - Implemented `update_daily_guardrail_metrics()` static method: + - Batch upserts to database with retry logic + - Increments counters on conflict + - Added guardrail transaction call to `update_database()` flow + - Added commit logic to `_commit_spend_updates_to_db_without_redis_buffer()` + +### ✅ Phase 3: Type Definitions +- **File**: `litellm/types/proxy/management_endpoints/guardrail_metrics.py` + - Created Pydantic models for: + - `GuardrailMetrics` - Aggregated metrics + - `GuardrailSummary` - Table view + - `GuardrailMetricsResponse` - List endpoint response + - `GuardrailDailyMetrics` - Time-series data + - `GuardrailDetailMetrics` - Detail view with daily metrics + - `GuardrailLogEntry` - Individual request log + - `GuardrailLogsResponse` - Logs endpoint response + +### ✅ Phase 4: API Endpoints +- **File**: `litellm/proxy/management_endpoints/guardrail_metrics_endpoints.py` + - Implemented 3 endpoints: + 1. `GET /guardrail/metrics` - List guardrails with aggregated metrics + - Query params: start_date, end_date, guardrail_name, provider, page, page_size + - Returns sorted by fail_rate descending + 2. `GET /guardrail/{guardrail_name}/metrics` - Detail view metrics + - Returns overview stats + daily time-series + 3. `GET /guardrail/{guardrail_name}/logs` - Request logs + - Query params: start_date, end_date, status_filter, page, page_size + - Filters LiteLLM_SpendLogs by guardrail_information + +- **File**: `litellm/proxy/proxy_server.py` + - Added import for `guardrail_metrics_router` + - Registered router with `app.include_router(guardrail_metrics_router)` + +## 🔲 Pending: Frontend Implementation (Phase 5) + +### TypeScript Types +- **File to create**: `ui/litellm-dashboard/src/components/GuardrailsPage/types.ts` +- Defines interfaces matching backend Pydantic models + +### Networking Functions +- **File to update**: `ui/litellm-dashboard/src/components/networking.tsx` +- Add API call functions: + - `guardrailMetricsCall()` + - `guardrailDetailMetricsCall()` + - `guardrailLogsCall()` + +### Components +1. **Table View**: `ui/litellm-dashboard/src/components/GuardrailsPage/GuardrailsTableView.tsx` + - Displays list of guardrails with metrics + - Clickable rows navigate to detail view + +2. **Detail View**: `ui/litellm-dashboard/src/components/GuardrailsPage/GuardrailDetailView.tsx` + - Metric cards (Requests, Fail Rate, Latency, Blocked) + - Tabs for Overview and Logs + - Area chart for fail rate trend + +3. **Logs Tab**: `ui/litellm-dashboard/src/components/GuardrailsPage/GuardrailLogsTab.tsx` + - Filterable table (All, Blocked, Passed) + - Expandable rows for guardrail response details + +### Pages +1. **Main Page**: `ui/litellm-dashboard/src/app/(dashboard)/guardrails/page.tsx` + - Date range picker + - Renders GuardrailsTableView + +2. **Detail Page**: `ui/litellm-dashboard/src/app/(dashboard)/guardrails/[name]/page.tsx` + - Back button to overview + - Date range picker + - Renders GuardrailDetailView + +## Testing & Verification + +### Backend Testing Steps +1. Run Prisma migration: + ```bash + cd /Users/krrishdholakia/Documents/litellm-guardrails-dashboard + poetry run prisma migrate dev --name add_guardrail_metrics --schema=litellm/proxy/schema.prisma + ``` + +2. Start the proxy server with guardrails configured + +3. Send requests that trigger guardrails (both pass and fail cases) + +4. Wait 60s for batch commit to database + +5. Test API endpoints: + ```bash + # List guardrails + curl -X GET "http://localhost:4000/guardrail/metrics?start_date=2026-02-01&end_date=2026-02-19" \ + -H "Authorization: Bearer " + + # Get guardrail details + curl -X GET "http://localhost:4000/guardrail/my-guardrail/metrics?start_date=2026-02-01&end_date=2026-02-19" \ + -H "Authorization: Bearer " + + # Get logs + curl -X GET "http://localhost:4000/guardrail/my-guardrail/logs?start_date=2026-02-01&end_date=2026-02-19" \ + -H "Authorization: Bearer " + ``` + +6. Verify data in database: + ```sql + SELECT * FROM "LiteLLM_DailyGuardrailMetrics" LIMIT 10; + ``` + +### Frontend Testing Steps (Once Implemented) +1. Navigate to `/guardrails` page +2. Verify table loads with metrics +3. Click guardrail row → navigate to detail page +4. Verify Overview tab shows metrics and chart +5. Switch to Logs tab → verify logs display +6. Test status filter (All, Blocked, Passed) +7. Test date range filtering +8. Test pagination + +## Key Design Decisions + +### Fail Rate Calculation +- **Formula**: `(intervened_count / total_requests) * 100` +- Only counts `guardrail_intervened` as failures (policy violations) +- Excludes `guardrail_failed_to_respond` (infrastructure errors) + +### Average Latency +- **Formula**: `total_latency_ms / total_requests` +- Measures **guardrail execution overhead** (not total request latency) +- Captured from `StandardLoggingGuardrailInformation.duration` field +- Stored in milliseconds for better precision + +### Per-Request Logs +- No new table needed +- Existing `LiteLLM_SpendLogs.metadata.guardrail_information` used +- Query optimized with date filtering and over-fetching strategy + +### Aggregation Strategy +- Daily aggregation reduces full table scans on large datasets +- Batch commits every 60s reduce database load +- Indexed queries for fast filtering +- Pagination limits memory usage + +## Files Modified + +### Backend +1. `litellm/proxy/schema.prisma` - Added table +2. `litellm/proxy/_types.py` - Added transaction type +3. `litellm/proxy/db/db_spend_update_writer.py` - Data collection & commit +4. `litellm/types/proxy/management_endpoints/guardrail_metrics.py` - New file +5. `litellm/proxy/management_endpoints/guardrail_metrics_endpoints.py` - New file +6. `litellm/proxy/proxy_server.py` - Router registration + +### Frontend (Pending - 7 files) +1. `ui/litellm-dashboard/src/components/GuardrailsPage/types.ts` +2. `ui/litellm-dashboard/src/components/networking.tsx` +3. `ui/litellm-dashboard/src/components/GuardrailsPage/GuardrailsTableView.tsx` +4. `ui/litellm-dashboard/src/components/GuardrailsPage/GuardrailDetailView.tsx` +5. `ui/litellm-dashboard/src/components/GuardrailsPage/GuardrailLogsTab.tsx` +6. `ui/litellm-dashboard/src/app/(dashboard)/guardrails/page.tsx` +7. `ui/litellm-dashboard/src/app/(dashboard)/guardrails/[name]/page.tsx` + +## Next Steps + +1. **Run Prisma Migration** to create the database table +2. **Test Backend** with curl requests after sending guardrail traffic +3. **Implement Frontend** following Phase 5 specifications +4. **Run `make test-unit`** to ensure no regressions +5. **Test Integration** with high request volume (1000+ requests) +6. **Create Pull Request** with proper tests and documentation + +## Notes + +- Migration needs to be run in an environment with proper Prisma setup +- Some Pyright diagnostics appeared but are pre-existing in codebase +- Frontend implementation follows existing LiteLLM dashboard patterns (Tremor UI components) +- All backend code follows existing patterns from daily spend tracking diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 0e4fab9c79d..e3caaea9f09 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -4079,6 +4079,21 @@ class DailyAgentSpendTransaction(BaseDailySpendTransaction): agent_id: str +class DailyGuardrailMetricsTransaction(TypedDict): + """Transaction for daily guardrail metrics aggregation.""" + guardrail_name: str + guardrail_provider: str + guardrail_mode: str + date: str # YYYY-MM-DD + api_key: str + total_requests: int + success_count: int + intervened_count: int + failed_count: int + not_run_count: int + total_latency_ms: float + + class DBSpendUpdateTransactions(TypedDict): """ Internal Data Structure for buffering spend updates in Redis or in memory before committing them to the database diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py index 03628fda47f..08b32d99650 100644 --- a/litellm/proxy/db/db_spend_update_writer.py +++ b/litellm/proxy/db/db_spend_update_writer.py @@ -13,7 +13,17 @@ import random import time import traceback from datetime import datetime, timedelta, timezone -from typing import TYPE_CHECKING, Any, Dict, List, Literal, Optional, Union, cast, overload +from typing import ( + TYPE_CHECKING, + Any, + Dict, + List, + Literal, + Optional, + Union, + cast, + overload, +) import litellm from litellm._logging import verbose_proxy_logger @@ -23,12 +33,13 @@ from litellm.litellm_core_utils.safe_json_loads import safe_json_loads from litellm.proxy._types import ( DB_CONNECTION_ERROR_TYPES, BaseDailySpendTransaction, - DailyTagSpendTransaction, - DailyOrganizationSpendTransaction, - DailyTeamSpendTransaction, - DailyEndUserSpendTransaction, - DailyUserSpendTransaction, DailyAgentSpendTransaction, + DailyEndUserSpendTransaction, + DailyGuardrailMetricsTransaction, + DailyOrganizationSpendTransaction, + DailyTagSpendTransaction, + DailyTeamSpendTransaction, + DailyUserSpendTransaction, DBSpendUpdateTransactions, Litellm_EntityType, LiteLLM_UserTable, @@ -73,6 +84,7 @@ class DBSpendUpdateWriter: self.daily_agent_spend_update_queue = DailySpendUpdateQueue() self.daily_org_spend_update_queue = DailySpendUpdateQueue() self.daily_tag_spend_update_queue = DailySpendUpdateQueue() + self.daily_guardrail_metrics_update_queue = DailySpendUpdateQueue() async def update_database( # LiteLLM management object fields @@ -221,6 +233,12 @@ class DBSpendUpdateWriter: prisma_client=prisma_client, ) ) + asyncio.create_task( + self.add_spend_log_transaction_to_daily_guardrail_transaction( + payload=copy.deepcopy(payload), + prisma_client=prisma_client, + ) + ) verbose_proxy_logger.debug("Runs spend update on all tables") except Exception: @@ -699,6 +717,20 @@ class DBSpendUpdateWriter: daily_spend_transactions=daily_agent_spend_update_transactions, ) + ################## Daily Guardrail Metrics Update Transactions ################## + # Aggregate all in memory daily guardrail metrics transactions and commit to db + daily_guardrail_metrics_transactions = cast( + Dict[str, DailyGuardrailMetricsTransaction], + await self.daily_guardrail_metrics_update_queue.flush_and_get_aggregated_daily_spend_update_transactions(), + ) + + await DBSpendUpdateWriter.update_daily_guardrail_metrics( + n_retry_times=n_retry_times, + prisma_client=prisma_client, + proxy_logging_obj=proxy_logging_obj, + daily_metrics_transactions=daily_guardrail_metrics_transactions, + ) + async def _commit_spend_updates_to_db( # noqa: PLR0915 self, prisma_client: PrismaClient, @@ -1475,6 +1507,134 @@ class DBSpendUpdateWriter: unique_constraint_name="tag_date_api_key_model_custom_llm_provider_mcp_namespaced_tool_name_endpoint", ) + @staticmethod + async def update_daily_guardrail_metrics( + n_retry_times: int, + prisma_client: PrismaClient, + proxy_logging_obj: ProxyLogging, + daily_metrics_transactions: Dict[str, DailyGuardrailMetricsTransaction], + ): + """ + Batch upsert daily guardrail metrics to database. + + Uses same pattern as daily spend tables with upsert on conflict. + """ + if not daily_metrics_transactions: + return + + from litellm.proxy.utils import _raise_failed_update_spend_exception + + verbose_proxy_logger.debug( + f"Daily Guardrail Metrics transactions: {len(daily_metrics_transactions)}" + ) + BATCH_SIZE = 100 + start_time = time.time() + + try: + for i in range(n_retry_times + 1): + try: + # Sort transactions to minimize deadlocks + transactions_to_process = dict( + sorted( + daily_metrics_transactions.items(), + key=lambda x: ( + x[1].get("date") or "", + x[1].get("guardrail_name") or "", + x[1].get("api_key") or "", + ), + )[:BATCH_SIZE] + ) + + if len(transactions_to_process) == 0: + verbose_proxy_logger.debug( + "No new transactions to process for daily guardrail metrics update" + ) + break + + try: + async with prisma_client.db.batch_() as batcher: + for _, transaction in transactions_to_process.items(): + where_clause = { + "guardrail_daily_unique": { + "guardrail_name": transaction["guardrail_name"], + "guardrail_provider": transaction.get("guardrail_provider", "unknown"), + "guardrail_mode": transaction.get("guardrail_mode", "unknown"), + "date": transaction["date"], + "api_key": transaction["api_key"], + } + } + + common_data = { + "guardrail_name": transaction["guardrail_name"], + "guardrail_provider": transaction.get("guardrail_provider"), + "guardrail_mode": transaction.get("guardrail_mode"), + "date": transaction["date"], + "api_key": transaction["api_key"], + "total_requests": transaction["total_requests"], + "success_count": transaction["success_count"], + "intervened_count": transaction["intervened_count"], + "failed_count": transaction["failed_count"], + "not_run_count": transaction["not_run_count"], + "total_latency_ms": transaction["total_latency_ms"], + } + + update_data = { + "total_requests": {"increment": transaction["total_requests"]}, + "success_count": {"increment": transaction["success_count"]}, + "intervened_count": {"increment": transaction["intervened_count"]}, + "failed_count": {"increment": transaction["failed_count"]}, + "not_run_count": {"increment": transaction["not_run_count"]}, + "total_latency_ms": {"increment": transaction["total_latency_ms"]}, + } + + batcher.litellm_dailyguardrailmetrics.upsert( + where=where_clause, + data={ + "create": common_data, + "update": update_data, + }, + ) + except Exception as batch_error: + verbose_proxy_logger.exception( + f"Daily guardrail metrics batch upsert failed. " + f"Batch size: {len(transactions_to_process)}, " + f"Error: {str(batch_error)}" + ) + raise + + verbose_proxy_logger.debug( + f"Processed {len(transactions_to_process)} daily guardrail metrics transactions in {time.time() - start_time:.2f}s" + ) + + # Remove processed transactions + for key in transactions_to_process.keys(): + daily_metrics_transactions.pop(key, None) + + break + + except DB_CONNECTION_ERROR_TYPES as e: + if i >= n_retry_times: + _raise_failed_update_spend_exception( + e=e, + start_time=start_time, + proxy_logging_obj=proxy_logging_obj, + ) + verbose_proxy_logger.debug( + f"Error updating daily guardrail metrics, retrying. {i + 1}/{n_retry_times + 1}" + ) + await asyncio.sleep(0.5) + + except Exception as e: + verbose_proxy_logger.error(f"Error updating daily guardrail metrics: {e}") + if n_retry_times > 0: + await asyncio.sleep(0.5) + await DBSpendUpdateWriter.update_daily_guardrail_metrics( + n_retry_times - 1, + prisma_client, + proxy_logging_obj, + daily_metrics_transactions, + ) + async def _common_add_spend_log_transaction_to_daily_transaction( self, payload: Union[dict, SpendLogsPayload], @@ -1797,3 +1957,95 @@ class DBSpendUpdateWriter: await self.daily_tag_spend_update_queue.add_update( update={daily_transaction_key: daily_transaction} ) + + async def add_spend_log_transaction_to_daily_guardrail_transaction( + self, + payload: Union[dict, SpendLogsPayload], + prisma_client: Optional[PrismaClient] = None, + ) -> None: + """ + Extract guardrail information from payload and queue for daily aggregation. + + Creates separate transaction for each guardrail in the request. + """ + if prisma_client is None: + return + + try: + # Parse metadata + metadata_str = payload.get("metadata") + if isinstance(metadata_str, str): + _metadata = json.loads(metadata_str) + else: + _metadata = metadata_str or {} + + guardrail_information = _metadata.get("guardrail_information") + + if not guardrail_information or not isinstance(guardrail_information, list): + return + + # Get date from startTime + start_time = payload.get("startTime") + if isinstance(start_time, datetime): + date = start_time.date().isoformat() + elif isinstance(start_time, str): + date = start_time.split("T")[0] + else: + return + + api_key = payload.get("api_key", "") + + # Process each guardrail separately + for guardrail in guardrail_information: + guardrail_name = guardrail.get("guardrail_name") + if not guardrail_name: + continue + + guardrail_provider = guardrail.get("guardrail_provider") or "unknown" + guardrail_mode = self._serialize_guardrail_mode(guardrail.get("guardrail_mode")) + guardrail_status = guardrail.get("guardrail_status", "not_run") + + # Calculate status counts + success_count = 1 if guardrail_status == "success" else 0 + intervened_count = 1 if guardrail_status == "guardrail_intervened" else 0 + failed_count = 1 if guardrail_status == "guardrail_failed_to_respond" else 0 + not_run_count = 1 if guardrail_status == "not_run" else 0 + + # Get guardrail execution latency in milliseconds + # duration = time from guardrail start_time to end_time (guardrail overhead only) + duration = guardrail.get("duration") or 0 + duration_ms = float(duration) * 1000 # convert seconds to ms + + # Create unique transaction key + transaction_key = f"{guardrail_name}_{guardrail_provider}_{guardrail_mode}_{date}_{api_key}" + + daily_transaction = DailyGuardrailMetricsTransaction( + guardrail_name=guardrail_name, + guardrail_provider=guardrail_provider, + guardrail_mode=guardrail_mode, + date=date, + api_key=api_key, + total_requests=1, + success_count=success_count, + intervened_count=intervened_count, + failed_count=failed_count, + not_run_count=not_run_count, + total_latency_ms=duration_ms, + ) + + await self.daily_guardrail_metrics_update_queue.add_update( + update={transaction_key: daily_transaction} # type: ignore + ) + + except Exception as e: + verbose_proxy_logger.error(f"Error adding guardrail transaction: {e}") + + @staticmethod + def _serialize_guardrail_mode(mode) -> str: + """Convert guardrail_mode to string for storage.""" + if isinstance(mode, str): + return mode + elif isinstance(mode, list): + return json.dumps(mode) + else: + return str(mode) if mode else "unknown" diff --git a/litellm/proxy/management_endpoints/guardrail_metrics_endpoints.py b/litellm/proxy/management_endpoints/guardrail_metrics_endpoints.py new file mode 100644 index 00000000000..cc435e77c6a --- /dev/null +++ b/litellm/proxy/management_endpoints/guardrail_metrics_endpoints.py @@ -0,0 +1,317 @@ +import json +from typing import Any, Dict, Optional + +from fastapi import APIRouter, Depends, HTTPException, Query + +from litellm.proxy._types import CommonProxyErrors +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.types.proxy.management_endpoints.guardrail_metrics import ( + GuardrailDailyMetrics, + GuardrailDetailMetrics, + GuardrailLogEntry, + GuardrailLogsResponse, + GuardrailMetricsResponse, + GuardrailSummary, +) + +router = APIRouter() + + +@router.get( + "/guardrail/metrics", + tags=["guardrails"], + dependencies=[Depends(user_api_key_auth)], + response_model=GuardrailMetricsResponse, +) +async def get_guardrail_metrics( + start_date: str = Query(..., description="Start date YYYY-MM-DD"), + end_date: str = Query(..., description="End date YYYY-MM-DD"), + guardrail_name: Optional[str] = Query(None), + provider: Optional[str] = Query(None), + page: int = Query(1, ge=1), + page_size: int = Query(50, ge=1, le=500), +): + """ + Get aggregated guardrail metrics for dashboard table view. + + Returns list of guardrails with total requests, fail rate, avg latency. + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException( + status_code=500, + detail={"error": CommonProxyErrors.db_not_connected_error.value}, + ) + + # Build query filters + where_conditions: Dict[str, Any] = { + "date": { + "gte": start_date, + "lte": end_date, + } + } + + if guardrail_name: + where_conditions["guardrail_name"] = guardrail_name + + if provider: + where_conditions["guardrail_provider"] = provider + + # Query daily metrics + daily_metrics = await prisma_client.db.litellm_dailyguardrailmetrics.find_many( + where=where_conditions, + order=[{"date": "desc"}], + ) + + # Aggregate by guardrail_name across all dates + guardrail_aggregates = {} + for record in daily_metrics: + name = record.guardrail_name + if name not in guardrail_aggregates: + guardrail_aggregates[name] = { + "provider": record.guardrail_provider or "unknown", + "total_requests": 0, + "intervened_count": 0, + "total_latency_ms": 0.0, + } + + agg = guardrail_aggregates[name] + agg["total_requests"] += int(record.total_requests) + agg["intervened_count"] += int(record.intervened_count) + agg["total_latency_ms"] += float(record.total_latency_ms) + + # Calculate fail rate and avg latency + results = [] + for name, agg in guardrail_aggregates.items(): + fail_rate = ( + (agg["intervened_count"] / agg["total_requests"] * 100) + if agg["total_requests"] > 0 + else 0.0 + ) + avg_latency = ( + agg["total_latency_ms"] / agg["total_requests"] + if agg["total_requests"] > 0 + else 0.0 + ) + + results.append( + GuardrailSummary( + guardrail_name=name, + provider=agg["provider"], + total_requests=agg["total_requests"], + fail_rate=round(fail_rate, 2), + avg_latency_ms=round(avg_latency, 2), + ) + ) + + # Sort by fail rate descending + results.sort(key=lambda x: x.fail_rate, reverse=True) + + # Pagination + start_idx = (page - 1) * page_size + end_idx = start_idx + page_size + paginated_results = results[start_idx:end_idx] + + total_count = len(results) + + return GuardrailMetricsResponse( + results=paginated_results, + metadata={ + "page": page, + "total_pages": (total_count + page_size - 1) // page_size, + "has_more": end_idx < total_count, + "total_count": total_count, + }, + ) + + +@router.get( + "/guardrail/{guardrail_name}/metrics", + tags=["guardrails"], + dependencies=[Depends(user_api_key_auth)], + response_model=GuardrailDetailMetrics, +) +async def get_guardrail_detail_metrics( + guardrail_name: str, + start_date: str = Query(..., description="Start date YYYY-MM-DD"), + end_date: str = Query(..., description="End date YYYY-MM-DD"), +): + """ + Get detailed metrics for a specific guardrail (for overview tab). + + Returns aggregated metrics plus daily time-series data. + """ + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException( + status_code=500, + detail={"error": CommonProxyErrors.db_not_connected_error.value}, + ) + + # Query daily metrics for this guardrail + daily_records = await prisma_client.db.litellm_dailyguardrailmetrics.find_many( + where={ + "guardrail_name": guardrail_name, + "date": { + "gte": start_date, + "lte": end_date, + }, + }, + order=[{"date": "asc"}], + ) + + if not daily_records: + return GuardrailDetailMetrics( + requests_evaluated=0, + fail_rate=0.0, + avg_latency_ms=0.0, + blocked_count=0, + daily_metrics=[], + ) + + # Calculate totals + total_requests = sum(int(r.total_requests) for r in daily_records) + total_intervened = sum(int(r.intervened_count) for r in daily_records) + total_latency_ms = sum(float(r.total_latency_ms) for r in daily_records) + + fail_rate = (total_intervened / total_requests * 100) if total_requests > 0 else 0.0 + avg_latency_ms = total_latency_ms / total_requests if total_requests > 0 else 0.0 + + # Build daily time-series + daily_metrics = [] + for record in daily_records: + requests = int(record.total_requests) + intervened = int(record.intervened_count) + latency = float(record.total_latency_ms) + + daily_fail_rate = (intervened / requests * 100) if requests > 0 else 0.0 + daily_avg_latency = latency / requests if requests > 0 else 0.0 + + daily_metrics.append( + GuardrailDailyMetrics( + date=record.date, + total_requests=requests, + intervened_count=intervened, + success_count=int(record.success_count), + fail_rate=round(daily_fail_rate, 2), + avg_latency_ms=round(daily_avg_latency, 2), + ) + ) + + return GuardrailDetailMetrics( + requests_evaluated=total_requests, + fail_rate=round(fail_rate, 2), + avg_latency_ms=round(avg_latency_ms, 2), + blocked_count=total_intervened, + daily_metrics=daily_metrics, + ) + + +@router.get( + "/guardrail/{guardrail_name}/logs", + tags=["guardrails"], + dependencies=[Depends(user_api_key_auth)], + response_model=GuardrailLogsResponse, +) +async def get_guardrail_logs( + guardrail_name: str, + start_date: str = Query(..., description="Start date YYYY-MM-DD"), + end_date: str = Query(..., description="End date YYYY-MM-DD"), + status_filter: Optional[str] = Query(None, description="'blocked' or 'passed'"), + page: int = Query(1, ge=1), + page_size: int = Query(50, ge=1, le=100), +): + """ + Get individual request logs for a guardrail (for logs tab). + + Queries LiteLLM_SpendLogs and filters by guardrail_information in metadata. + """ + from datetime import datetime + + from litellm.proxy.proxy_server import prisma_client + + if prisma_client is None: + raise HTTPException( + status_code=500, + detail={"error": CommonProxyErrors.db_not_connected_error.value}, + ) + + # Convert dates to datetime for filtering + start_datetime = datetime.fromisoformat(start_date).isoformat() + end_datetime = datetime.fromisoformat(end_date + "T23:59:59").isoformat() + + # Query spend logs with metadata containing guardrail_information + # Note: This is a simplified approach - may need optimization for large datasets + spend_logs = await prisma_client.db.litellm_spendlogs.find_many( + where={ + "startTime": { + "gte": start_datetime, + "lte": end_datetime, + }, + }, + order=[{"startTime": "desc"}], + skip=(page - 1) * page_size, + take=page_size * 3, # Over-fetch to account for filtering + ) + + # Parse and filter logs + filtered_logs = [] + for log in spend_logs: + try: + metadata = json.loads(log.metadata) if isinstance(log.metadata, str) else log.metadata + guardrail_info = metadata.get("guardrail_information", []) + + # Find matching guardrail in list + for g in guardrail_info: + if g.get("guardrail_name") == guardrail_name: + status = g.get("guardrail_status", "") + + # Map status to blocked/passed + if status == "guardrail_intervened": + log_status = "blocked" + elif status == "success": + log_status = "passed" + else: + continue # Skip other statuses + + # Apply status filter + if status_filter and log_status != status_filter: + continue + + # Extract request content + request_content = None + messages = metadata.get("messages", []) + if messages and isinstance(messages, list) and len(messages) > 0: + last_msg = messages[-1] + if isinstance(last_msg, dict): + request_content = last_msg.get("content", "") + + filtered_logs.append( + GuardrailLogEntry( + request_id=log.request_id, + timestamp=log.startTime.isoformat() if log.startTime else "", + model=log.model or "unknown", + status=log_status, + guardrail_response=g.get("guardrail_response"), + request_content=request_content, + latency_ms=round((g.get("duration") or 0) * 1000, 2), + ) + ) + + if len(filtered_logs) >= page_size: + break + + if len(filtered_logs) >= page_size: + break + + except Exception as e: + continue + + return GuardrailLogsResponse( + logs=filtered_logs[:page_size], + total_count=len(filtered_logs), # Approximate + page=page, + page_size=page_size, + ) diff --git a/litellm/proxy/proxy_server.py b/litellm/proxy/proxy_server.py index 0e702abfc6c..529c31532be 100644 --- a/litellm/proxy/proxy_server.py +++ b/litellm/proxy/proxy_server.py @@ -357,12 +357,13 @@ from litellm.proxy.management_endpoints.customer_endpoints import ( from litellm.proxy.management_endpoints.fallback_management_endpoints import ( router as fallback_management_router, ) +from litellm.proxy.management_endpoints.guardrail_metrics_endpoints import ( + router as guardrail_metrics_router, +) from litellm.proxy.management_endpoints.internal_user_endpoints import ( router as internal_user_router, ) -from litellm.proxy.management_endpoints.internal_user_endpoints import ( - user_update, -) +from litellm.proxy.management_endpoints.internal_user_endpoints import user_update from litellm.proxy.management_endpoints.key_management_endpoints import ( delete_verification_tokens, duration_in_seconds, @@ -388,10 +389,10 @@ from litellm.proxy.management_endpoints.model_management_endpoints import ( from litellm.proxy.management_endpoints.organization_endpoints import ( router as organization_router, ) +from litellm.proxy.management_endpoints.policy_endpoints import router as policy_router from litellm.proxy.management_endpoints.project_endpoints import ( router as project_router, ) -from litellm.proxy.management_endpoints.policy_endpoints import router as policy_router from litellm.proxy.management_endpoints.router_settings_endpoints import ( router as router_settings_router, ) @@ -421,9 +422,7 @@ from litellm.proxy.openai_evals_endpoints.endpoints import router as evals_route from litellm.proxy.openai_files_endpoints.files_endpoints import ( router as openai_files_router, ) -from litellm.proxy.openai_files_endpoints.files_endpoints import ( - set_files_config, -) +from litellm.proxy.openai_files_endpoints.files_endpoints import set_files_config from litellm.proxy.pass_through_endpoints.llm_passthrough_endpoints import ( passthrough_endpoint_router, ) @@ -522,9 +521,7 @@ from litellm.types.proxy.management_endpoints.ui_sso import ( LiteLLM_UpperboundKeyGenerateParams, ) from litellm.types.realtime import RealtimeQueryParams -from litellm.types.router import ( - DeploymentTypedDict, -) +from litellm.types.router import DeploymentTypedDict from litellm.types.router import ModelInfo as RouterModelInfo from litellm.types.router import ( RouterGeneralSettings, @@ -12489,6 +12486,7 @@ app.include_router(cloudzero_router) app.include_router(caching_router) app.include_router(analytics_router) app.include_router(guardrails_router) +app.include_router(guardrail_metrics_router) app.include_router(policy_router) app.include_router(policy_crud_router) app.include_router(policy_resolve_router) diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 6eaeabe8916..27eba1ecf84 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -747,11 +747,11 @@ model LiteLLM_DailyTeamSpend { model LiteLLM_DailyTagSpend { id String @id @default(uuid()) request_id String? - tag String? + tag String? date String - api_key String + api_key String model String? - model_group String? + model_group String? custom_llm_provider String? mcp_namespaced_tool_name String? endpoint String? @@ -775,6 +775,35 @@ model LiteLLM_DailyTagSpend { @@index([endpoint]) } +// Track daily guardrail metrics per guardrail +model LiteLLM_DailyGuardrailMetrics { + id String @id @default(uuid()) + guardrail_name String + guardrail_provider String? + guardrail_mode String? // pre_call, post_call, during_call + date String // YYYY-MM-DD format + api_key String // for per-key breakdowns + + // Aggregated counts + total_requests BigInt @default(0) + success_count BigInt @default(0) // guardrail_status = "success" + intervened_count BigInt @default(0) // guardrail_status = "guardrail_intervened" + failed_count BigInt @default(0) // guardrail_status = "guardrail_failed_to_respond" + not_run_count BigInt @default(0) // guardrail_status = "not_run" + + // Aggregated latency (guardrail execution overhead only) + total_latency_ms Float @default(0.0) // sum of guardrail durations in milliseconds + + created_at DateTime @default(now()) + updated_at DateTime @updatedAt + + // Unique constraint: one row per guardrail per day per api_key + @@unique([guardrail_name, guardrail_provider, guardrail_mode, date, api_key], name: "guardrail_daily_unique") + @@index([date]) + @@index([guardrail_name]) + @@index([guardrail_provider]) + @@index([api_key]) +} // Track the status of cron jobs running. Only allow one pod to run the job at a time model LiteLLM_CronJob { diff --git a/litellm/types/proxy/management_endpoints/guardrail_metrics.py b/litellm/types/proxy/management_endpoints/guardrail_metrics.py new file mode 100644 index 00000000000..6bb9612716b --- /dev/null +++ b/litellm/types/proxy/management_endpoints/guardrail_metrics.py @@ -0,0 +1,75 @@ +from datetime import date +from typing import Any, Dict, List, Optional + +from pydantic import BaseModel + + +class GuardrailMetrics(BaseModel): + """Aggregated metrics for a guardrail.""" + + total_requests: int = 0 + success_count: int = 0 + intervened_count: int = 0 + failed_count: int = 0 + not_run_count: int = 0 + fail_rate: float = 0.0 # percentage + avg_latency_ms: float = 0.0 + + +class GuardrailSummary(BaseModel): + """Summary view for guardrails table.""" + + guardrail_name: str + provider: str + total_requests: int + fail_rate: float # percentage + avg_latency_ms: float + + +class GuardrailMetricsResponse(BaseModel): + """Response for /guardrail/metrics endpoint.""" + + results: List[GuardrailSummary] + metadata: Dict[str, Any] + + +class GuardrailDailyMetrics(BaseModel): + """Daily time-series data for a guardrail.""" + + date: str + total_requests: int + intervened_count: int + success_count: int + fail_rate: float + avg_latency_ms: float + + +class GuardrailDetailMetrics(BaseModel): + """Detailed metrics for guardrail overview page.""" + + requests_evaluated: int + fail_rate: float + avg_latency_ms: float + blocked_count: int # intervened in selected period + daily_metrics: List[GuardrailDailyMetrics] + + +class GuardrailLogEntry(BaseModel): + """Individual request log entry.""" + + request_id: str + timestamp: str + model: str + status: str # "blocked" or "passed" + guardrail_response: Optional[dict] = None + request_content: Optional[str] = None + latency_ms: float = 0.0 + + +class GuardrailLogsResponse(BaseModel): + """Response for logs tab.""" + + logs: List[GuardrailLogEntry] + total_count: int + page: int + page_size: int