diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index 096e23e673d..e1dc849413c 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -1,7 +1,10 @@ import asyncio import copy +import importlib.metadata +import json import logging import os +import sys import time import traceback from datetime import datetime, timedelta @@ -9,10 +12,12 @@ from typing import Any, Dict, Iterable, Literal, Optional, Union, cast import fastapi from fastapi import APIRouter, Depends, HTTPException, Request, Response, status +from pydantic import BaseModel, Field import litellm from litellm._logging import verbose_logger, verbose_proxy_logger from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS +from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker from litellm.litellm_core_utils.custom_logger_registry import CustomLoggerRegistry from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler from litellm.proxy._types import ( @@ -131,6 +136,104 @@ services = Union[ str, ] +_DIAGNOSE_REDACTED_VALUE = "[REDACTED]" +_diagnose_sensitive_masker = SensitiveDataMasker() + + +class DiagnoseRequest(BaseModel): + model: Optional[str] = Field( + default=None, + description=( + "Optional proxy model name to use for generating the diagnostic report. " + "When omitted, LiteLLM uses the first configured proxy deployment." + ), + ) + issue_description: Optional[str] = Field( + default=None, + description="Short description of the issue the admin is trying to reproduce.", + ) + reproduction_steps: Optional[str] = Field( + default=None, + description="Known reproduction steps or the behavior the admin has already tried.", + ) + + +def _redact_diagnose_payload(payload: Any, sensitive_parent: bool = False) -> Any: + if isinstance(payload, dict): + redacted: dict = {} + for key, value in payload.items(): + key_is_sensitive = _diagnose_sensitive_masker.is_sensitive_key(str(key)) + if key_is_sensitive: + redacted[key] = _DIAGNOSE_REDACTED_VALUE + else: + redacted[key] = _redact_diagnose_payload(value, key_is_sensitive) + return redacted + if isinstance(payload, list): + return [_redact_diagnose_payload(item, sensitive_parent) for item in payload] + if sensitive_parent: + return _DIAGNOSE_REDACTED_VALUE + if isinstance(payload, (str, int, float, bool)) or payload is None: + return payload + return str(payload) + + +def _get_litellm_package_version() -> str: + try: + return importlib.metadata.version("litellm") + except importlib.metadata.PackageNotFoundError: + return "unknown" + + +def _get_diagnose_model_list(llm_router: Optional[Any]) -> list: + if llm_router is None: + return [] + model_list = llm_router.get_model_list(model_name=None) + return model_list or [] + + +def _select_diagnose_model( + requested_model: Optional[str], model_list: list +) -> Optional[str]: + if requested_model: + return requested_model + for deployment in model_list: + if isinstance(deployment, dict) and deployment.get("model_name"): + return deployment["model_name"] + return None + + +def _build_diagnose_prompt( + *, + request: DiagnoseRequest, + selected_model: str, + diagnostic_context: dict, +) -> str: + serialized_context = json.dumps( + diagnostic_context, indent=2, sort_keys=True, default=str + ) + return ( + "You are helping the LiteLLM team reproduce a proxy issue. " + "Act like a concise support engineer doing a mini diagnostic grill: " + "identify the exact LiteLLM version, summarize the configured proxy models, " + "highlight relevant YAML/config settings, list missing details to ask the " + "admin, and produce clean Markdown reproduction steps.\n\n" + f"Model selected for this diagnostic LLM call: {selected_model}\n" + f"Issue description from admin: {request.issue_description or 'Not provided'}\n" + f"Known reproduction steps from admin: {request.reproduction_steps or 'Not provided'}\n\n" + "Use only this redacted diagnostic context; never invent secrets:\n" + f"```json\n{serialized_context}\n```\n\n" + "Return Markdown with these headings: Summary, Environment, Config and Models, " + "Reproduction Steps, Questions for Admin, Suspected Areas." + ) + + +def _extract_diagnose_response_text(response: Any) -> str: + try: + content = response.choices[0].message.content + except (AttributeError, IndexError, TypeError): + return "" + return content or "" + @router.get( "/test", @@ -1570,6 +1673,84 @@ async def health_readiness_details(response: Response): return await _get_health_readiness_details(response=response) +@router.post( + "/diagnose", + tags=["health"], + dependencies=[Depends(user_api_key_auth)], +) +async def diagnose_endpoint( + diagnose_request: DiagnoseRequest, + user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), +): + """ + Generate a redacted Markdown diagnostic report for LiteLLM support. + + This endpoint is restricted to proxy admins and proxy admin view-only users + because it summarizes proxy configuration and configured deployments. + """ + if not _is_proxy_admin(user_api_key_dict): + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail="Only proxy admins can call /diagnose.", + ) + + from litellm.proxy.proxy_server import llm_router, proxy_config + + configured_models = _get_diagnose_model_list(llm_router=llm_router) + selected_model = _select_diagnose_model( + requested_model=diagnose_request.model, + model_list=configured_models, + ) + redacted_config = _redact_diagnose_payload(proxy_config.get_config_state()) + redacted_models = _redact_diagnose_payload(configured_models) + diagnostic_context = { + "litellm_version": _get_litellm_package_version(), + "python_version": sys.version, + "configured_models": redacted_models, + "config": redacted_config, + "admin_user": { + "user_id": user_api_key_dict.user_id, + "user_role": ( + user_api_key_dict.user_role.value + if hasattr(user_api_key_dict.user_role, "value") + else user_api_key_dict.user_role + ), + }, + } + + if selected_model is None or llm_router is None: + return { + "used_llm": False, + "selected_model": selected_model, + "diagnostic_report": ( + "# LiteLLM Diagnostic Report\n\n" + "No proxy model is configured for the diagnostic LLM call. " + "Call `/diagnose` again with a configured `model`, or add a " + "model to the proxy config, so LiteLLM can generate the full " + "Markdown reproduction report.\n\n" + f"```json\n{json.dumps(diagnostic_context, indent=2, sort_keys=True, default=str)}\n```" + ), + "diagnostic_context": diagnostic_context, + } + + prompt = _build_diagnose_prompt( + request=diagnose_request, + selected_model=selected_model, + diagnostic_context=diagnostic_context, + ) + response = await llm_router.acompletion( + model=selected_model, + messages=[{"role": "user", "content": prompt}], + temperature=0.0, + ) + return { + "used_llm": True, + "selected_model": selected_model, + "diagnostic_report": _extract_diagnose_response_text(response), + "diagnostic_context": diagnostic_context, + } + + @router.get( "/health/backlog", tags=["health"], diff --git a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py index 2edcb00c967..d094a139a12 100644 --- a/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/test_litellm/proxy/health_endpoints/test_health_endpoints.py @@ -643,6 +643,135 @@ def test_health_readiness_details_returns_diagnostic_fields(monkeypatch): assert "cache" in response_data +def test_diagnose_requires_admin_view(): + """ + /diagnose returns support-ready diagnostics, so it must reject non-admins. + """ + app = FastAPI() + app.include_router(_health_endpoints_module.router) + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.INTERNAL_USER + ) + client = TestClient(app) + + response = client.post("/diagnose", json={"model": "gpt-4o"}) + + assert response.status_code == 403, response.text + + +def test_diagnose_generates_redacted_llm_report(monkeypatch): + """ + /diagnose should prompt a configured LLM with redacted proxy state and + return the generated Markdown reproduction report. + """ + captured_call = {} + + class FakeRouter: + def get_model_list(self, model_name=None): + deployments = [ + { + "model_name": "gpt-4o", + "litellm_params": { + "model": "openai/gpt-4o", + "api_key": "sk-secret-value-that-must-not-leak", + "api_base": "https://api.openai.example", + }, + "model_info": {"id": "deployment-1"}, + } + ] + if model_name is None: + return deployments + return [d for d in deployments if d["model_name"] == model_name] + + async def acompletion(self, **kwargs): + captured_call.update(kwargs) + return SimpleNamespace( + choices=[ + SimpleNamespace( + message=SimpleNamespace( + content="# Reproduction report\n\nUse this report." + ) + ) + ] + ) + + app = FastAPI() + app.include_router(_health_endpoints_module.router) + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN + ) + client = TestClient(app) + + monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", FakeRouter()) + monkeypatch.setattr( + "litellm.proxy.proxy_server.proxy_config", + SimpleNamespace( + get_config_state=lambda: { + "model_list": [ + { + "model_name": "gpt-4o", + "litellm_params": { + "model": "openai/gpt-4o", + "api_key": "sk-secret-value-that-must-not-leak", + }, + } + ], + "general_settings": {"master_key": "sk-master-secret"}, + } + ), + ) + + response = client.post( + "/diagnose", + json={"issue_description": "Requests fail after enabling the proxy."}, + ) + + assert response.status_code == 200, response.text + response_data = response.json() + assert response_data["selected_model"] == "gpt-4o" + assert response_data["used_llm"] is True + assert response_data["diagnostic_report"].startswith("# Reproduction report") + assert captured_call["model"] == "gpt-4o" + + prompt_text = captured_call["messages"][0]["content"] + assert "Requests fail after enabling the proxy." in prompt_text + assert "sk-secret-value-that-must-not-leak" not in prompt_text + assert "sk-master-secret" not in prompt_text + assert "REDACTED" in prompt_text + + +def test_diagnose_prompts_for_model_when_no_llm_is_configured(monkeypatch): + """ + /diagnose should return redacted context plus next-step instructions when + the proxy has no configured LLM to generate the full report. + """ + app = FastAPI() + app.include_router(_health_endpoints_module.router) + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth( + user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY + ) + client = TestClient(app) + + monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", None) + monkeypatch.setattr( + "litellm.proxy.proxy_server.proxy_config", + SimpleNamespace( + get_config_state=lambda: { + "general_settings": {"master_key": "sk-master-secret"} + } + ), + ) + + response = client.post("/diagnose", json={}) + + assert response.status_code == 200, response.text + response_data = response.json() + assert response_data["used_llm"] is False + assert response_data["selected_model"] is None + assert "configured `model`" in response_data["diagnostic_report"] + assert "sk-master-secret" not in response_data["diagnostic_report"] + + def test_health_readiness_allows_explicit_legacy_public_details(monkeypatch): """ Operators can explicitly preserve the legacy public readiness payload. diff --git a/tests/test_litellm/proxy/test_sensitive_route_auth.py b/tests/test_litellm/proxy/test_sensitive_route_auth.py index 19998e52779..c315113534d 100644 --- a/tests/test_litellm/proxy/test_sensitive_route_auth.py +++ b/tests/test_litellm/proxy/test_sensitive_route_auth.py @@ -2,6 +2,7 @@ from fastapi.routing import APIRoute from litellm.proxy.auth.user_api_key_auth import user_api_key_auth from litellm.proxy.common_utils.debug_utils import router as debug_router +from litellm.proxy.health_endpoints._health_endpoints import router as health_router from litellm.proxy.spend_tracking.spend_management_endpoints import ( router as spend_router, ) @@ -32,3 +33,9 @@ def test_provider_budgets_requires_auth_dependency(): assert user_api_key_auth in _get_route_dependency_calls( spend_router, "/provider/budgets", "GET" ) + + +def test_diagnose_requires_auth_dependency(): + assert user_api_key_auth in _get_route_dependency_calls( + health_router, "/diagnose", "POST" + )