Add admin diagnose endpoint

Co-authored-by: ishaan-berri <ishaan-berri@users.noreply.github.com>
This commit is contained in:
oss-agent-shin 2026-05-06 01:44:20 +00:00
parent e75c7a312a
commit 53f8a4cac5
No known key found for this signature in database
3 changed files with 317 additions and 0 deletions

View file

@ -1,7 +1,10 @@
import asyncio
import copy
import importlib.metadata
import json
import logging
import os
import sys
import time
import traceback
from datetime import datetime, timedelta
@ -9,10 +12,12 @@ from typing import Any, Dict, Iterable, Literal, Optional, Union, cast
import fastapi
from fastapi import APIRouter, Depends, HTTPException, Request, Response, status
from pydantic import BaseModel, Field
import litellm
from litellm._logging import verbose_logger, verbose_proxy_logger
from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS
from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker
from litellm.litellm_core_utils.custom_logger_registry import CustomLoggerRegistry
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
from litellm.proxy._types import (
@ -131,6 +136,104 @@ services = Union[
str,
]
_DIAGNOSE_REDACTED_VALUE = "[REDACTED]"
_diagnose_sensitive_masker = SensitiveDataMasker()
class DiagnoseRequest(BaseModel):
model: Optional[str] = Field(
default=None,
description=(
"Optional proxy model name to use for generating the diagnostic report. "
"When omitted, LiteLLM uses the first configured proxy deployment."
),
)
issue_description: Optional[str] = Field(
default=None,
description="Short description of the issue the admin is trying to reproduce.",
)
reproduction_steps: Optional[str] = Field(
default=None,
description="Known reproduction steps or the behavior the admin has already tried.",
)
def _redact_diagnose_payload(payload: Any, sensitive_parent: bool = False) -> Any:
if isinstance(payload, dict):
redacted: dict = {}
for key, value in payload.items():
key_is_sensitive = _diagnose_sensitive_masker.is_sensitive_key(str(key))
if key_is_sensitive:
redacted[key] = _DIAGNOSE_REDACTED_VALUE
else:
redacted[key] = _redact_diagnose_payload(value, key_is_sensitive)
return redacted
if isinstance(payload, list):
return [_redact_diagnose_payload(item, sensitive_parent) for item in payload]
if sensitive_parent:
return _DIAGNOSE_REDACTED_VALUE
if isinstance(payload, (str, int, float, bool)) or payload is None:
return payload
return str(payload)
def _get_litellm_package_version() -> str:
try:
return importlib.metadata.version("litellm")
except importlib.metadata.PackageNotFoundError:
return "unknown"
def _get_diagnose_model_list(llm_router: Optional[Any]) -> list:
if llm_router is None:
return []
model_list = llm_router.get_model_list(model_name=None)
return model_list or []
def _select_diagnose_model(
requested_model: Optional[str], model_list: list
) -> Optional[str]:
if requested_model:
return requested_model
for deployment in model_list:
if isinstance(deployment, dict) and deployment.get("model_name"):
return deployment["model_name"]
return None
def _build_diagnose_prompt(
*,
request: DiagnoseRequest,
selected_model: str,
diagnostic_context: dict,
) -> str:
serialized_context = json.dumps(
diagnostic_context, indent=2, sort_keys=True, default=str
)
return (
"You are helping the LiteLLM team reproduce a proxy issue. "
"Act like a concise support engineer doing a mini diagnostic grill: "
"identify the exact LiteLLM version, summarize the configured proxy models, "
"highlight relevant YAML/config settings, list missing details to ask the "
"admin, and produce clean Markdown reproduction steps.\n\n"
f"Model selected for this diagnostic LLM call: {selected_model}\n"
f"Issue description from admin: {request.issue_description or 'Not provided'}\n"
f"Known reproduction steps from admin: {request.reproduction_steps or 'Not provided'}\n\n"
"Use only this redacted diagnostic context; never invent secrets:\n"
f"```json\n{serialized_context}\n```\n\n"
"Return Markdown with these headings: Summary, Environment, Config and Models, "
"Reproduction Steps, Questions for Admin, Suspected Areas."
)
def _extract_diagnose_response_text(response: Any) -> str:
try:
content = response.choices[0].message.content
except (AttributeError, IndexError, TypeError):
return ""
return content or ""
@router.get(
"/test",
@ -1570,6 +1673,84 @@ async def health_readiness_details(response: Response):
return await _get_health_readiness_details(response=response)
@router.post(
"/diagnose",
tags=["health"],
dependencies=[Depends(user_api_key_auth)],
)
async def diagnose_endpoint(
diagnose_request: DiagnoseRequest,
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
):
"""
Generate a redacted Markdown diagnostic report for LiteLLM support.
This endpoint is restricted to proxy admins and proxy admin view-only users
because it summarizes proxy configuration and configured deployments.
"""
if not _is_proxy_admin(user_api_key_dict):
raise HTTPException(
status_code=status.HTTP_403_FORBIDDEN,
detail="Only proxy admins can call /diagnose.",
)
from litellm.proxy.proxy_server import llm_router, proxy_config
configured_models = _get_diagnose_model_list(llm_router=llm_router)
selected_model = _select_diagnose_model(
requested_model=diagnose_request.model,
model_list=configured_models,
)
redacted_config = _redact_diagnose_payload(proxy_config.get_config_state())
redacted_models = _redact_diagnose_payload(configured_models)
diagnostic_context = {
"litellm_version": _get_litellm_package_version(),
"python_version": sys.version,
"configured_models": redacted_models,
"config": redacted_config,
"admin_user": {
"user_id": user_api_key_dict.user_id,
"user_role": (
user_api_key_dict.user_role.value
if hasattr(user_api_key_dict.user_role, "value")
else user_api_key_dict.user_role
),
},
}
if selected_model is None or llm_router is None:
return {
"used_llm": False,
"selected_model": selected_model,
"diagnostic_report": (
"# LiteLLM Diagnostic Report\n\n"
"No proxy model is configured for the diagnostic LLM call. "
"Call `/diagnose` again with a configured `model`, or add a "
"model to the proxy config, so LiteLLM can generate the full "
"Markdown reproduction report.\n\n"
f"```json\n{json.dumps(diagnostic_context, indent=2, sort_keys=True, default=str)}\n```"
),
"diagnostic_context": diagnostic_context,
}
prompt = _build_diagnose_prompt(
request=diagnose_request,
selected_model=selected_model,
diagnostic_context=diagnostic_context,
)
response = await llm_router.acompletion(
model=selected_model,
messages=[{"role": "user", "content": prompt}],
temperature=0.0,
)
return {
"used_llm": True,
"selected_model": selected_model,
"diagnostic_report": _extract_diagnose_response_text(response),
"diagnostic_context": diagnostic_context,
}
@router.get(
"/health/backlog",
tags=["health"],

View file

@ -643,6 +643,135 @@ def test_health_readiness_details_returns_diagnostic_fields(monkeypatch):
assert "cache" in response_data
def test_diagnose_requires_admin_view():
"""
/diagnose returns support-ready diagnostics, so it must reject non-admins.
"""
app = FastAPI()
app.include_router(_health_endpoints_module.router)
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
user_role=LitellmUserRoles.INTERNAL_USER
)
client = TestClient(app)
response = client.post("/diagnose", json={"model": "gpt-4o"})
assert response.status_code == 403, response.text
def test_diagnose_generates_redacted_llm_report(monkeypatch):
"""
/diagnose should prompt a configured LLM with redacted proxy state and
return the generated Markdown reproduction report.
"""
captured_call = {}
class FakeRouter:
def get_model_list(self, model_name=None):
deployments = [
{
"model_name": "gpt-4o",
"litellm_params": {
"model": "openai/gpt-4o",
"api_key": "sk-secret-value-that-must-not-leak",
"api_base": "https://api.openai.example",
},
"model_info": {"id": "deployment-1"},
}
]
if model_name is None:
return deployments
return [d for d in deployments if d["model_name"] == model_name]
async def acompletion(self, **kwargs):
captured_call.update(kwargs)
return SimpleNamespace(
choices=[
SimpleNamespace(
message=SimpleNamespace(
content="# Reproduction report\n\nUse this report."
)
)
]
)
app = FastAPI()
app.include_router(_health_endpoints_module.router)
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN
)
client = TestClient(app)
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", FakeRouter())
monkeypatch.setattr(
"litellm.proxy.proxy_server.proxy_config",
SimpleNamespace(
get_config_state=lambda: {
"model_list": [
{
"model_name": "gpt-4o",
"litellm_params": {
"model": "openai/gpt-4o",
"api_key": "sk-secret-value-that-must-not-leak",
},
}
],
"general_settings": {"master_key": "sk-master-secret"},
}
),
)
response = client.post(
"/diagnose",
json={"issue_description": "Requests fail after enabling the proxy."},
)
assert response.status_code == 200, response.text
response_data = response.json()
assert response_data["selected_model"] == "gpt-4o"
assert response_data["used_llm"] is True
assert response_data["diagnostic_report"].startswith("# Reproduction report")
assert captured_call["model"] == "gpt-4o"
prompt_text = captured_call["messages"][0]["content"]
assert "Requests fail after enabling the proxy." in prompt_text
assert "sk-secret-value-that-must-not-leak" not in prompt_text
assert "sk-master-secret" not in prompt_text
assert "REDACTED" in prompt_text
def test_diagnose_prompts_for_model_when_no_llm_is_configured(monkeypatch):
"""
/diagnose should return redacted context plus next-step instructions when
the proxy has no configured LLM to generate the full report.
"""
app = FastAPI()
app.include_router(_health_endpoints_module.router)
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY
)
client = TestClient(app)
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", None)
monkeypatch.setattr(
"litellm.proxy.proxy_server.proxy_config",
SimpleNamespace(
get_config_state=lambda: {
"general_settings": {"master_key": "sk-master-secret"}
}
),
)
response = client.post("/diagnose", json={})
assert response.status_code == 200, response.text
response_data = response.json()
assert response_data["used_llm"] is False
assert response_data["selected_model"] is None
assert "configured `model`" in response_data["diagnostic_report"]
assert "sk-master-secret" not in response_data["diagnostic_report"]
def test_health_readiness_allows_explicit_legacy_public_details(monkeypatch):
"""
Operators can explicitly preserve the legacy public readiness payload.

View file

@ -2,6 +2,7 @@ from fastapi.routing import APIRoute
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.common_utils.debug_utils import router as debug_router
from litellm.proxy.health_endpoints._health_endpoints import router as health_router
from litellm.proxy.spend_tracking.spend_management_endpoints import (
router as spend_router,
)
@ -32,3 +33,9 @@ def test_provider_budgets_requires_auth_dependency():
assert user_api_key_auth in _get_route_dependency_calls(
spend_router, "/provider/budgets", "GET"
)
def test_diagnose_requires_auth_dependency():
assert user_api_key_auth in _get_route_dependency_calls(
health_router, "/diagnose", "POST"
)