mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
Add admin diagnose endpoint
Co-authored-by: ishaan-berri <ishaan-berri@users.noreply.github.com>
This commit is contained in:
parent
e75c7a312a
commit
53f8a4cac5
3 changed files with 317 additions and 0 deletions
|
|
@ -1,7 +1,10 @@
|
|||
import asyncio
|
||||
import copy
|
||||
import importlib.metadata
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import traceback
|
||||
from datetime import datetime, timedelta
|
||||
|
|
@ -9,10 +12,12 @@ from typing import Any, Dict, Iterable, Literal, Optional, Union, cast
|
|||
|
||||
import fastapi
|
||||
from fastapi import APIRouter, Depends, HTTPException, Request, Response, status
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
import litellm
|
||||
from litellm._logging import verbose_logger, verbose_proxy_logger
|
||||
from litellm.constants import HEALTH_CHECK_TIMEOUT_SECONDS
|
||||
from litellm.litellm_core_utils.sensitive_data_masker import SensitiveDataMasker
|
||||
from litellm.litellm_core_utils.custom_logger_registry import CustomLoggerRegistry
|
||||
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler
|
||||
from litellm.proxy._types import (
|
||||
|
|
@ -131,6 +136,104 @@ services = Union[
|
|||
str,
|
||||
]
|
||||
|
||||
_DIAGNOSE_REDACTED_VALUE = "[REDACTED]"
|
||||
_diagnose_sensitive_masker = SensitiveDataMasker()
|
||||
|
||||
|
||||
class DiagnoseRequest(BaseModel):
|
||||
model: Optional[str] = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Optional proxy model name to use for generating the diagnostic report. "
|
||||
"When omitted, LiteLLM uses the first configured proxy deployment."
|
||||
),
|
||||
)
|
||||
issue_description: Optional[str] = Field(
|
||||
default=None,
|
||||
description="Short description of the issue the admin is trying to reproduce.",
|
||||
)
|
||||
reproduction_steps: Optional[str] = Field(
|
||||
default=None,
|
||||
description="Known reproduction steps or the behavior the admin has already tried.",
|
||||
)
|
||||
|
||||
|
||||
def _redact_diagnose_payload(payload: Any, sensitive_parent: bool = False) -> Any:
|
||||
if isinstance(payload, dict):
|
||||
redacted: dict = {}
|
||||
for key, value in payload.items():
|
||||
key_is_sensitive = _diagnose_sensitive_masker.is_sensitive_key(str(key))
|
||||
if key_is_sensitive:
|
||||
redacted[key] = _DIAGNOSE_REDACTED_VALUE
|
||||
else:
|
||||
redacted[key] = _redact_diagnose_payload(value, key_is_sensitive)
|
||||
return redacted
|
||||
if isinstance(payload, list):
|
||||
return [_redact_diagnose_payload(item, sensitive_parent) for item in payload]
|
||||
if sensitive_parent:
|
||||
return _DIAGNOSE_REDACTED_VALUE
|
||||
if isinstance(payload, (str, int, float, bool)) or payload is None:
|
||||
return payload
|
||||
return str(payload)
|
||||
|
||||
|
||||
def _get_litellm_package_version() -> str:
|
||||
try:
|
||||
return importlib.metadata.version("litellm")
|
||||
except importlib.metadata.PackageNotFoundError:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _get_diagnose_model_list(llm_router: Optional[Any]) -> list:
|
||||
if llm_router is None:
|
||||
return []
|
||||
model_list = llm_router.get_model_list(model_name=None)
|
||||
return model_list or []
|
||||
|
||||
|
||||
def _select_diagnose_model(
|
||||
requested_model: Optional[str], model_list: list
|
||||
) -> Optional[str]:
|
||||
if requested_model:
|
||||
return requested_model
|
||||
for deployment in model_list:
|
||||
if isinstance(deployment, dict) and deployment.get("model_name"):
|
||||
return deployment["model_name"]
|
||||
return None
|
||||
|
||||
|
||||
def _build_diagnose_prompt(
|
||||
*,
|
||||
request: DiagnoseRequest,
|
||||
selected_model: str,
|
||||
diagnostic_context: dict,
|
||||
) -> str:
|
||||
serialized_context = json.dumps(
|
||||
diagnostic_context, indent=2, sort_keys=True, default=str
|
||||
)
|
||||
return (
|
||||
"You are helping the LiteLLM team reproduce a proxy issue. "
|
||||
"Act like a concise support engineer doing a mini diagnostic grill: "
|
||||
"identify the exact LiteLLM version, summarize the configured proxy models, "
|
||||
"highlight relevant YAML/config settings, list missing details to ask the "
|
||||
"admin, and produce clean Markdown reproduction steps.\n\n"
|
||||
f"Model selected for this diagnostic LLM call: {selected_model}\n"
|
||||
f"Issue description from admin: {request.issue_description or 'Not provided'}\n"
|
||||
f"Known reproduction steps from admin: {request.reproduction_steps or 'Not provided'}\n\n"
|
||||
"Use only this redacted diagnostic context; never invent secrets:\n"
|
||||
f"```json\n{serialized_context}\n```\n\n"
|
||||
"Return Markdown with these headings: Summary, Environment, Config and Models, "
|
||||
"Reproduction Steps, Questions for Admin, Suspected Areas."
|
||||
)
|
||||
|
||||
|
||||
def _extract_diagnose_response_text(response: Any) -> str:
|
||||
try:
|
||||
content = response.choices[0].message.content
|
||||
except (AttributeError, IndexError, TypeError):
|
||||
return ""
|
||||
return content or ""
|
||||
|
||||
|
||||
@router.get(
|
||||
"/test",
|
||||
|
|
@ -1570,6 +1673,84 @@ async def health_readiness_details(response: Response):
|
|||
return await _get_health_readiness_details(response=response)
|
||||
|
||||
|
||||
@router.post(
|
||||
"/diagnose",
|
||||
tags=["health"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
)
|
||||
async def diagnose_endpoint(
|
||||
diagnose_request: DiagnoseRequest,
|
||||
user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
|
||||
):
|
||||
"""
|
||||
Generate a redacted Markdown diagnostic report for LiteLLM support.
|
||||
|
||||
This endpoint is restricted to proxy admins and proxy admin view-only users
|
||||
because it summarizes proxy configuration and configured deployments.
|
||||
"""
|
||||
if not _is_proxy_admin(user_api_key_dict):
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_403_FORBIDDEN,
|
||||
detail="Only proxy admins can call /diagnose.",
|
||||
)
|
||||
|
||||
from litellm.proxy.proxy_server import llm_router, proxy_config
|
||||
|
||||
configured_models = _get_diagnose_model_list(llm_router=llm_router)
|
||||
selected_model = _select_diagnose_model(
|
||||
requested_model=diagnose_request.model,
|
||||
model_list=configured_models,
|
||||
)
|
||||
redacted_config = _redact_diagnose_payload(proxy_config.get_config_state())
|
||||
redacted_models = _redact_diagnose_payload(configured_models)
|
||||
diagnostic_context = {
|
||||
"litellm_version": _get_litellm_package_version(),
|
||||
"python_version": sys.version,
|
||||
"configured_models": redacted_models,
|
||||
"config": redacted_config,
|
||||
"admin_user": {
|
||||
"user_id": user_api_key_dict.user_id,
|
||||
"user_role": (
|
||||
user_api_key_dict.user_role.value
|
||||
if hasattr(user_api_key_dict.user_role, "value")
|
||||
else user_api_key_dict.user_role
|
||||
),
|
||||
},
|
||||
}
|
||||
|
||||
if selected_model is None or llm_router is None:
|
||||
return {
|
||||
"used_llm": False,
|
||||
"selected_model": selected_model,
|
||||
"diagnostic_report": (
|
||||
"# LiteLLM Diagnostic Report\n\n"
|
||||
"No proxy model is configured for the diagnostic LLM call. "
|
||||
"Call `/diagnose` again with a configured `model`, or add a "
|
||||
"model to the proxy config, so LiteLLM can generate the full "
|
||||
"Markdown reproduction report.\n\n"
|
||||
f"```json\n{json.dumps(diagnostic_context, indent=2, sort_keys=True, default=str)}\n```"
|
||||
),
|
||||
"diagnostic_context": diagnostic_context,
|
||||
}
|
||||
|
||||
prompt = _build_diagnose_prompt(
|
||||
request=diagnose_request,
|
||||
selected_model=selected_model,
|
||||
diagnostic_context=diagnostic_context,
|
||||
)
|
||||
response = await llm_router.acompletion(
|
||||
model=selected_model,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
temperature=0.0,
|
||||
)
|
||||
return {
|
||||
"used_llm": True,
|
||||
"selected_model": selected_model,
|
||||
"diagnostic_report": _extract_diagnose_response_text(response),
|
||||
"diagnostic_context": diagnostic_context,
|
||||
}
|
||||
|
||||
|
||||
@router.get(
|
||||
"/health/backlog",
|
||||
tags=["health"],
|
||||
|
|
|
|||
|
|
@ -643,6 +643,135 @@ def test_health_readiness_details_returns_diagnostic_fields(monkeypatch):
|
|||
assert "cache" in response_data
|
||||
|
||||
|
||||
def test_diagnose_requires_admin_view():
|
||||
"""
|
||||
/diagnose returns support-ready diagnostics, so it must reject non-admins.
|
||||
"""
|
||||
app = FastAPI()
|
||||
app.include_router(_health_endpoints_module.router)
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.INTERNAL_USER
|
||||
)
|
||||
client = TestClient(app)
|
||||
|
||||
response = client.post("/diagnose", json={"model": "gpt-4o"})
|
||||
|
||||
assert response.status_code == 403, response.text
|
||||
|
||||
|
||||
def test_diagnose_generates_redacted_llm_report(monkeypatch):
|
||||
"""
|
||||
/diagnose should prompt a configured LLM with redacted proxy state and
|
||||
return the generated Markdown reproduction report.
|
||||
"""
|
||||
captured_call = {}
|
||||
|
||||
class FakeRouter:
|
||||
def get_model_list(self, model_name=None):
|
||||
deployments = [
|
||||
{
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o",
|
||||
"api_key": "sk-secret-value-that-must-not-leak",
|
||||
"api_base": "https://api.openai.example",
|
||||
},
|
||||
"model_info": {"id": "deployment-1"},
|
||||
}
|
||||
]
|
||||
if model_name is None:
|
||||
return deployments
|
||||
return [d for d in deployments if d["model_name"] == model_name]
|
||||
|
||||
async def acompletion(self, **kwargs):
|
||||
captured_call.update(kwargs)
|
||||
return SimpleNamespace(
|
||||
choices=[
|
||||
SimpleNamespace(
|
||||
message=SimpleNamespace(
|
||||
content="# Reproduction report\n\nUse this report."
|
||||
)
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
app = FastAPI()
|
||||
app.include_router(_health_endpoints_module.router)
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN
|
||||
)
|
||||
client = TestClient(app)
|
||||
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", FakeRouter())
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_config",
|
||||
SimpleNamespace(
|
||||
get_config_state=lambda: {
|
||||
"model_list": [
|
||||
{
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {
|
||||
"model": "openai/gpt-4o",
|
||||
"api_key": "sk-secret-value-that-must-not-leak",
|
||||
},
|
||||
}
|
||||
],
|
||||
"general_settings": {"master_key": "sk-master-secret"},
|
||||
}
|
||||
),
|
||||
)
|
||||
|
||||
response = client.post(
|
||||
"/diagnose",
|
||||
json={"issue_description": "Requests fail after enabling the proxy."},
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
response_data = response.json()
|
||||
assert response_data["selected_model"] == "gpt-4o"
|
||||
assert response_data["used_llm"] is True
|
||||
assert response_data["diagnostic_report"].startswith("# Reproduction report")
|
||||
assert captured_call["model"] == "gpt-4o"
|
||||
|
||||
prompt_text = captured_call["messages"][0]["content"]
|
||||
assert "Requests fail after enabling the proxy." in prompt_text
|
||||
assert "sk-secret-value-that-must-not-leak" not in prompt_text
|
||||
assert "sk-master-secret" not in prompt_text
|
||||
assert "REDACTED" in prompt_text
|
||||
|
||||
|
||||
def test_diagnose_prompts_for_model_when_no_llm_is_configured(monkeypatch):
|
||||
"""
|
||||
/diagnose should return redacted context plus next-step instructions when
|
||||
the proxy has no configured LLM to generate the full report.
|
||||
"""
|
||||
app = FastAPI()
|
||||
app.include_router(_health_endpoints_module.router)
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
|
||||
user_role=LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY
|
||||
)
|
||||
client = TestClient(app)
|
||||
|
||||
monkeypatch.setattr("litellm.proxy.proxy_server.llm_router", None)
|
||||
monkeypatch.setattr(
|
||||
"litellm.proxy.proxy_server.proxy_config",
|
||||
SimpleNamespace(
|
||||
get_config_state=lambda: {
|
||||
"general_settings": {"master_key": "sk-master-secret"}
|
||||
}
|
||||
),
|
||||
)
|
||||
|
||||
response = client.post("/diagnose", json={})
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
response_data = response.json()
|
||||
assert response_data["used_llm"] is False
|
||||
assert response_data["selected_model"] is None
|
||||
assert "configured `model`" in response_data["diagnostic_report"]
|
||||
assert "sk-master-secret" not in response_data["diagnostic_report"]
|
||||
|
||||
|
||||
def test_health_readiness_allows_explicit_legacy_public_details(monkeypatch):
|
||||
"""
|
||||
Operators can explicitly preserve the legacy public readiness payload.
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ from fastapi.routing import APIRoute
|
|||
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.common_utils.debug_utils import router as debug_router
|
||||
from litellm.proxy.health_endpoints._health_endpoints import router as health_router
|
||||
from litellm.proxy.spend_tracking.spend_management_endpoints import (
|
||||
router as spend_router,
|
||||
)
|
||||
|
|
@ -32,3 +33,9 @@ def test_provider_budgets_requires_auth_dependency():
|
|||
assert user_api_key_auth in _get_route_dependency_calls(
|
||||
spend_router, "/provider/budgets", "GET"
|
||||
)
|
||||
|
||||
|
||||
def test_diagnose_requires_auth_dependency():
|
||||
assert user_api_key_auth in _get_route_dependency_calls(
|
||||
health_router, "/diagnose", "POST"
|
||||
)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue