mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-30 01:52:18 +00:00
feat: add model leaderboard page (#43649)
* feat(proxy): add model leaderboard analytics * feat: add model insights task and range constants * feat: record task type from task tags in model usage rollup * feat: serve 365 days of model insights by UTC date * test: cover task tag resolution in model usage rollup * test: update model insights range limit test to 365 days * chore: regenerate dashboard api types for model insights * feat: add model insights aggregation helpers * test: cover model insights aggregation helpers * feat: redesign model leaderboard with stacked bars, treemap and ranking * test: update model leaderboard view test * feat: mark model leaderboard as beta in sidebar * chore: sync schema.prisma copies from root * fix: only treat task: prefixed tags as model insight tasks * feat: add metric type for model insights ranking * fix: rank model insights by selected metric and scope detail queries to ranked deployments * test: plain tags are not model insight tasks * test: cover metric ranking, deployment scoping and rollup round trip * fix: build model insights weeks and halves from the requested date range * test: cover empty weeks and range-based change comparison * fix: refetch by metric, show load errors and ignore stale responses * test: cover metric refetch and error state * feat: define model insight tasks in a JSON file * feat: return task labels and categories from model insights * feat: load model insight tasks from JSON * refactor: validate rollup task tags against the JSON task list * feat: serve the task list with model insights * refactor: drop hardcoded task list from constants * build: ship model insight tasks JSON in the wheel * test: cover model insight task JSON * refactor: take task labels and categories from the API * test: pass task info to task tile builder * refactor: color treemap by API-provided category * test: include tasks in model leaderboard fixture * fix: make daily model usage migration idempotent * feat: bound the model insights task query size * fix: compute task breakdown independent of the chart metric * test: task breakdown is stable across chart metrics * chore: regenerate lazy openapi snapshot for model insights * chore: regenerate dashboard api types for model insights * fix: keep previous ranking dimmed while a new metric loads * test: cover stale metric state in model leaderboard * refactor: drop task row cap constant * fix: return the full task breakdown instead of a truncated one * test: task query is not truncated * feat: add task summary types for model insights * feat: summarise tasks server-side on a separate model insights endpoint * test: cover the model insights tasks endpoint * chore: regenerate lazy openapi snapshot for model insights tasks * chore: regenerate dashboard api types for model insights tasks * refactor: drop client-side task aggregation * test: remove client-side task aggregation tests * feat: load task breakdown separately from the chart metric * test: task breakdown is not refetched on chart metric change --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
This commit is contained in:
parent
39d14bd855
commit
118ce3cc91
28 changed files with 2225 additions and 0 deletions
|
|
@ -0,0 +1,19 @@
|
|||
CREATE TABLE IF NOT EXISTS "LiteLLM_DailyModelUsage" (
|
||||
"date" TEXT NOT NULL,
|
||||
"model_group" TEXT NOT NULL,
|
||||
"model" TEXT NOT NULL,
|
||||
"custom_llm_provider" TEXT NOT NULL,
|
||||
"task_type" TEXT NOT NULL,
|
||||
"spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
|
||||
"prompt_tokens" BIGINT NOT NULL DEFAULT 0,
|
||||
"completion_tokens" BIGINT NOT NULL DEFAULT 0,
|
||||
"request_count" BIGINT NOT NULL DEFAULT 0,
|
||||
"successful_requests" BIGINT NOT NULL DEFAULT 0,
|
||||
"failed_requests" BIGINT NOT NULL DEFAULT 0,
|
||||
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
"updated_at" TIMESTAMP(3) NOT NULL,
|
||||
CONSTRAINT "LiteLLM_DailyModelUsage_pkey" PRIMARY KEY ("date", "model_group", "model", "custom_llm_provider", "task_type")
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS "LiteLLM_DailyModelUsage_date_idx" ON "LiteLLM_DailyModelUsage"("date");
|
||||
CREATE INDEX IF NOT EXISTS "LiteLLM_DailyModelUsage_model_group_idx" ON "LiteLLM_DailyModelUsage"("model_group");
|
||||
|
|
@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend {
|
|||
@@id([date, tool_name])
|
||||
}
|
||||
|
||||
model LiteLLM_DailyModelUsage {
|
||||
date String
|
||||
model_group String
|
||||
model String
|
||||
custom_llm_provider String
|
||||
task_type String
|
||||
spend Float @default(0.0)
|
||||
prompt_tokens BigInt @default(0)
|
||||
completion_tokens BigInt @default(0)
|
||||
request_count BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
failed_requests BigInt @default(0)
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
@@id([date, model_group, model, custom_llm_provider, task_type])
|
||||
@@index([date])
|
||||
@@index([model_group])
|
||||
}
|
||||
|
||||
// Gateway request counts recorded at the ASGI edge by
|
||||
// BillableRequestMetricsMiddleware. This is the source of truth for SGR
|
||||
// (successful gateway requests): it counts what the proxy actually answered,
|
||||
|
|
|
|||
|
|
@ -1777,6 +1777,10 @@ SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS: Final = float(
|
|||
os.getenv("SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS", "5")
|
||||
)
|
||||
TOOL_SPEND_TOP_TOOLS: Final = 100
|
||||
MODEL_INSIGHTS_TOP_MODELS: Final = 10
|
||||
MODEL_INSIGHTS_MAX_RANGE_DAYS: Final = 365
|
||||
MODEL_INSIGHTS_DEFAULT_TASK: Final = "uncategorized"
|
||||
MODEL_INSIGHTS_TASK_TAG_PREFIX: Final = "task:"
|
||||
SPEND_LOG_PARTITION_INTERVAL: Final = os.getenv("SPEND_LOG_PARTITION_INTERVAL", "day")
|
||||
SPEND_LOG_PARTITION_PRECREATE_AHEAD: Final = int(os.getenv("SPEND_LOG_PARTITION_PRECREATE_AHEAD", 7))
|
||||
SPEND_LOG_WRITE_BATCH_MAX_BYTES: Final = max(1, int(os.getenv("SPEND_LOG_WRITE_BATCH_MAX_BYTES", 2_000_000)))
|
||||
|
|
|
|||
|
|
@ -128,6 +128,11 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = (
|
|||
module_path="litellm.proxy.management_endpoints.tool_management_endpoints",
|
||||
path_prefixes=("/v1/tool", "/tool"),
|
||||
),
|
||||
LazyFeature(
|
||||
name="model_insights",
|
||||
module_path="litellm.proxy.management_endpoints.model_insights_endpoints",
|
||||
path_prefixes=("/model-insights",),
|
||||
),
|
||||
LazyFeature(
|
||||
name="search_tools",
|
||||
module_path="litellm.proxy.search_endpoints.search_tool_management",
|
||||
|
|
|
|||
|
|
@ -40700,6 +40700,463 @@
|
|||
}
|
||||
}
|
||||
},
|
||||
"model_insights": {
|
||||
"components": {
|
||||
"schemas": {
|
||||
"HTTPValidationError": {
|
||||
"properties": {
|
||||
"detail": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/ValidationError"
|
||||
},
|
||||
"title": "Detail",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"title": "HTTPValidationError",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelInsightDailyMetric": {
|
||||
"properties": {
|
||||
"completion_tokens": {
|
||||
"title": "Completion Tokens",
|
||||
"type": "integer"
|
||||
},
|
||||
"date": {
|
||||
"title": "Date",
|
||||
"type": "string"
|
||||
},
|
||||
"failed_requests": {
|
||||
"title": "Failed Requests",
|
||||
"type": "integer"
|
||||
},
|
||||
"model": {
|
||||
"title": "Model",
|
||||
"type": "string"
|
||||
},
|
||||
"model_group": {
|
||||
"title": "Model Group",
|
||||
"type": "string"
|
||||
},
|
||||
"prompt_tokens": {
|
||||
"title": "Prompt Tokens",
|
||||
"type": "integer"
|
||||
},
|
||||
"provider": {
|
||||
"title": "Provider",
|
||||
"type": "string"
|
||||
},
|
||||
"requests": {
|
||||
"title": "Requests",
|
||||
"type": "integer"
|
||||
},
|
||||
"spend": {
|
||||
"title": "Spend",
|
||||
"type": "number"
|
||||
},
|
||||
"successful_requests": {
|
||||
"title": "Successful Requests",
|
||||
"type": "integer"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"model_group",
|
||||
"model",
|
||||
"provider",
|
||||
"spend",
|
||||
"prompt_tokens",
|
||||
"completion_tokens",
|
||||
"requests",
|
||||
"successful_requests",
|
||||
"failed_requests",
|
||||
"date"
|
||||
],
|
||||
"title": "ModelInsightDailyMetric",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelInsightMetric": {
|
||||
"properties": {
|
||||
"completion_tokens": {
|
||||
"title": "Completion Tokens",
|
||||
"type": "integer"
|
||||
},
|
||||
"failed_requests": {
|
||||
"title": "Failed Requests",
|
||||
"type": "integer"
|
||||
},
|
||||
"model": {
|
||||
"title": "Model",
|
||||
"type": "string"
|
||||
},
|
||||
"model_group": {
|
||||
"title": "Model Group",
|
||||
"type": "string"
|
||||
},
|
||||
"prompt_tokens": {
|
||||
"title": "Prompt Tokens",
|
||||
"type": "integer"
|
||||
},
|
||||
"provider": {
|
||||
"title": "Provider",
|
||||
"type": "string"
|
||||
},
|
||||
"requests": {
|
||||
"title": "Requests",
|
||||
"type": "integer"
|
||||
},
|
||||
"spend": {
|
||||
"title": "Spend",
|
||||
"type": "number"
|
||||
},
|
||||
"successful_requests": {
|
||||
"title": "Successful Requests",
|
||||
"type": "integer"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"model_group",
|
||||
"model",
|
||||
"provider",
|
||||
"spend",
|
||||
"prompt_tokens",
|
||||
"completion_tokens",
|
||||
"requests",
|
||||
"successful_requests",
|
||||
"failed_requests"
|
||||
],
|
||||
"title": "ModelInsightMetric",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelInsightTaskSummary": {
|
||||
"properties": {
|
||||
"category": {
|
||||
"title": "Category",
|
||||
"type": "string"
|
||||
},
|
||||
"label": {
|
||||
"title": "Label",
|
||||
"type": "string"
|
||||
},
|
||||
"leader": {
|
||||
"title": "Leader",
|
||||
"type": "string"
|
||||
},
|
||||
"provider": {
|
||||
"title": "Provider",
|
||||
"type": "string"
|
||||
},
|
||||
"share": {
|
||||
"title": "Share",
|
||||
"type": "number"
|
||||
},
|
||||
"task_type": {
|
||||
"title": "Task Type",
|
||||
"type": "string"
|
||||
},
|
||||
"value": {
|
||||
"title": "Value",
|
||||
"type": "number"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"task_type",
|
||||
"label",
|
||||
"category",
|
||||
"value",
|
||||
"share",
|
||||
"leader",
|
||||
"provider"
|
||||
],
|
||||
"title": "ModelInsightTaskSummary",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelInsightTasksResponse": {
|
||||
"properties": {
|
||||
"end_date": {
|
||||
"title": "End Date",
|
||||
"type": "string"
|
||||
},
|
||||
"start_date": {
|
||||
"title": "Start Date",
|
||||
"type": "string"
|
||||
},
|
||||
"tasks": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/ModelInsightTaskSummary"
|
||||
},
|
||||
"title": "Tasks",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"start_date",
|
||||
"end_date",
|
||||
"tasks"
|
||||
],
|
||||
"title": "ModelInsightTasksResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"ModelInsightsResponse": {
|
||||
"properties": {
|
||||
"daily": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/ModelInsightDailyMetric"
|
||||
},
|
||||
"title": "Daily",
|
||||
"type": "array"
|
||||
},
|
||||
"end_date": {
|
||||
"title": "End Date",
|
||||
"type": "string"
|
||||
},
|
||||
"start_date": {
|
||||
"title": "Start Date",
|
||||
"type": "string"
|
||||
},
|
||||
"top_models": {
|
||||
"items": {
|
||||
"$ref": "#/components/schemas/ModelInsightMetric"
|
||||
},
|
||||
"title": "Top Models",
|
||||
"type": "array"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"start_date",
|
||||
"end_date",
|
||||
"daily",
|
||||
"top_models"
|
||||
],
|
||||
"title": "ModelInsightsResponse",
|
||||
"type": "object"
|
||||
},
|
||||
"ValidationError": {
|
||||
"properties": {
|
||||
"ctx": {
|
||||
"title": "Context",
|
||||
"type": "object"
|
||||
},
|
||||
"input": {
|
||||
"title": "Input"
|
||||
},
|
||||
"loc": {
|
||||
"items": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "integer"
|
||||
}
|
||||
]
|
||||
},
|
||||
"title": "Location",
|
||||
"type": "array"
|
||||
},
|
||||
"msg": {
|
||||
"title": "Message",
|
||||
"type": "string"
|
||||
},
|
||||
"type": {
|
||||
"title": "Error Type",
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"loc",
|
||||
"msg",
|
||||
"type"
|
||||
],
|
||||
"title": "ValidationError",
|
||||
"type": "object"
|
||||
}
|
||||
}
|
||||
},
|
||||
"paths": {
|
||||
"/model-insights": {
|
||||
"get": {
|
||||
"operationId": "get_model_insights_model_insights_get",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "YYYY-MM-DD, defaults to 365 days ago",
|
||||
"in": "query",
|
||||
"name": "start_date",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "YYYY-MM-DD, defaults to 365 days ago",
|
||||
"title": "Start Date"
|
||||
}
|
||||
},
|
||||
{
|
||||
"description": "YYYY-MM-DD, defaults to today",
|
||||
"in": "query",
|
||||
"name": "end_date",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "YYYY-MM-DD, defaults to today",
|
||||
"title": "End Date"
|
||||
}
|
||||
},
|
||||
{
|
||||
"description": "Metric the top models are ranked by",
|
||||
"in": "query",
|
||||
"name": "metric",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"default": "tokens",
|
||||
"description": "Metric the top models are ranked by",
|
||||
"enum": [
|
||||
"requests",
|
||||
"spend",
|
||||
"tokens"
|
||||
],
|
||||
"title": "Metric",
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelInsightsResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Successful Response"
|
||||
},
|
||||
"422": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/HTTPValidationError"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Validation Error"
|
||||
}
|
||||
},
|
||||
"security": [
|
||||
{
|
||||
"APIKeyHeader": []
|
||||
}
|
||||
],
|
||||
"summary": "Get Model Insights",
|
||||
"tags": [
|
||||
"model_insights"
|
||||
]
|
||||
}
|
||||
},
|
||||
"/model-insights/tasks": {
|
||||
"get": {
|
||||
"operationId": "get_model_insight_tasks_model_insights_tasks_get",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "YYYY-MM-DD, defaults to 365 days ago",
|
||||
"in": "query",
|
||||
"name": "start_date",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "YYYY-MM-DD, defaults to 365 days ago",
|
||||
"title": "Start Date"
|
||||
}
|
||||
},
|
||||
{
|
||||
"description": "YYYY-MM-DD, defaults to today",
|
||||
"in": "query",
|
||||
"name": "end_date",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"description": "YYYY-MM-DD, defaults to today",
|
||||
"title": "End Date"
|
||||
}
|
||||
},
|
||||
{
|
||||
"description": "Metric task shares are computed from",
|
||||
"in": "query",
|
||||
"name": "metric",
|
||||
"required": false,
|
||||
"schema": {
|
||||
"default": "spend",
|
||||
"description": "Metric task shares are computed from",
|
||||
"enum": [
|
||||
"requests",
|
||||
"spend",
|
||||
"tokens"
|
||||
],
|
||||
"title": "Metric",
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/ModelInsightTasksResponse"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Successful Response"
|
||||
},
|
||||
"422": {
|
||||
"content": {
|
||||
"application/json": {
|
||||
"schema": {
|
||||
"$ref": "#/components/schemas/HTTPValidationError"
|
||||
}
|
||||
}
|
||||
},
|
||||
"description": "Validation Error"
|
||||
}
|
||||
},
|
||||
"security": [
|
||||
{
|
||||
"APIKeyHeader": []
|
||||
}
|
||||
],
|
||||
"summary": "Get Model Insight Tasks",
|
||||
"tags": [
|
||||
"model_insights"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"policies": {
|
||||
"components": {
|
||||
"schemas": {
|
||||
|
|
|
|||
|
|
@ -1148,6 +1148,16 @@ class DBSpendUpdateWriter:
|
|||
traceback.format_exc(),
|
||||
)
|
||||
|
||||
try:
|
||||
from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage
|
||||
|
||||
await increment_daily_model_usage(prisma_client=prisma_client, payload=payload_copy)
|
||||
except Exception:
|
||||
verbose_proxy_logger.debug(
|
||||
"_batch_database_updates: increment_daily_model_usage failed: %s",
|
||||
traceback.format_exc(),
|
||||
)
|
||||
|
||||
async def _update_key_db(
|
||||
self,
|
||||
response_cost: float | None,
|
||||
|
|
|
|||
14
litellm/proxy/db/model_insights_tasks.py
Normal file
14
litellm/proxy/db/model_insights_tasks.py
Normal file
|
|
@ -0,0 +1,14 @@
|
|||
import json
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
from litellm.types.model_insights import ModelInsightTask
|
||||
|
||||
_TASKS_FILE: Final = Path(__file__).resolve().parent.parent / "model_insights_tasks.json"
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def load_model_insight_tasks() -> dict[str, ModelInsightTask]:
|
||||
raw: Final = json.loads(_TASKS_FILE.read_text())
|
||||
return {name: ModelInsightTask(task_type=name, **entry) for name, entry in raw.items()}
|
||||
86
litellm/proxy/db/model_usage_rollup.py
Normal file
86
litellm/proxy/db/model_usage_rollup.py
Normal file
|
|
@ -0,0 +1,86 @@
|
|||
from datetime import datetime
|
||||
from typing import Final
|
||||
|
||||
from pydantic import TypeAdapter, ValidationError
|
||||
|
||||
from litellm.constants import (
|
||||
INTERNAL_CALL_ORIGIN_METADATA_KEY,
|
||||
MODEL_INSIGHTS_DEFAULT_TASK,
|
||||
MODEL_INSIGHTS_TASK_TAG_PREFIX,
|
||||
)
|
||||
from litellm.proxy._types import SpendLogsPayload
|
||||
from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
from litellm.repositories.table_repositories import DailyModelUsageRepository
|
||||
|
||||
_METADATA: Final = TypeAdapter(dict[str, object])
|
||||
_TAGS: Final = TypeAdapter(list[object])
|
||||
|
||||
|
||||
def model_usage_task_type(request_tags: str) -> str:
|
||||
try:
|
||||
tags: Final = _TAGS.validate_json(request_tags)
|
||||
except ValidationError:
|
||||
return MODEL_INSIGHTS_DEFAULT_TASK
|
||||
for tag in tags:
|
||||
if isinstance(tag, str) and tag.startswith(MODEL_INSIGHTS_TASK_TAG_PREFIX):
|
||||
task = tag.removeprefix(MODEL_INSIGHTS_TASK_TAG_PREFIX)
|
||||
if task in load_model_insight_tasks():
|
||||
return task
|
||||
return MODEL_INSIGHTS_DEFAULT_TASK
|
||||
|
||||
|
||||
def _is_internal_call(metadata: str) -> bool:
|
||||
try:
|
||||
decoded: Final = _METADATA.validate_json(metadata)
|
||||
except ValidationError:
|
||||
return False
|
||||
return bool(decoded.get(INTERNAL_CALL_ORIGIN_METADATA_KEY))
|
||||
|
||||
|
||||
def _date_from_start_time(start_time: datetime | str) -> str | None:
|
||||
if isinstance(start_time, datetime):
|
||||
return start_time.date().isoformat()
|
||||
return start_time[:10] if len(start_time) >= 10 else None
|
||||
|
||||
|
||||
async def increment_daily_model_usage(prisma_client: PrismaClient, payload: SpendLogsPayload) -> None:
|
||||
date: Final = _date_from_start_time(payload["startTime"])
|
||||
if date is None or _is_internal_call(payload["metadata"]):
|
||||
return
|
||||
|
||||
model: Final = payload["model"] or "unknown"
|
||||
model_group: Final = payload["model_group"] or model
|
||||
provider: Final = payload["custom_llm_provider"] or "unknown"
|
||||
task_type: Final = model_usage_task_type(payload["request_tags"])
|
||||
successful: Final = 1 if payload["status"] == "success" else 0
|
||||
failed: Final = 1 - successful
|
||||
key: Final = {
|
||||
"date": date,
|
||||
"model_group": model_group,
|
||||
"model": model,
|
||||
"custom_llm_provider": provider,
|
||||
"task_type": task_type,
|
||||
}
|
||||
await DailyModelUsageRepository(prisma_client).table.upsert(
|
||||
where={"date_model_group_model_custom_llm_provider_task_type": key},
|
||||
data={
|
||||
"create": {
|
||||
**key,
|
||||
"spend": payload["spend"],
|
||||
"prompt_tokens": payload["prompt_tokens"],
|
||||
"completion_tokens": payload["completion_tokens"],
|
||||
"request_count": 1,
|
||||
"successful_requests": successful,
|
||||
"failed_requests": failed,
|
||||
},
|
||||
"update": {
|
||||
"spend": {"increment": payload["spend"]},
|
||||
"prompt_tokens": {"increment": payload["prompt_tokens"]},
|
||||
"completion_tokens": {"increment": payload["completion_tokens"]},
|
||||
"request_count": {"increment": 1},
|
||||
"successful_requests": {"increment": successful},
|
||||
"failed_requests": {"increment": failed},
|
||||
},
|
||||
},
|
||||
)
|
||||
220
litellm/proxy/management_endpoints/model_insights_endpoints.py
Normal file
220
litellm/proxy/management_endpoints/model_insights_endpoints.py
Normal file
|
|
@ -0,0 +1,220 @@
|
|||
from collections.abc import Mapping
|
||||
from datetime import date, datetime, timedelta, timezone
|
||||
from typing import Annotated, Final
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException, Query
|
||||
from pydantic import BaseModel, Field, TypeAdapter
|
||||
|
||||
from litellm.constants import MODEL_INSIGHTS_DEFAULT_TASK, MODEL_INSIGHTS_MAX_RANGE_DAYS, MODEL_INSIGHTS_TOP_MODELS
|
||||
from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks
|
||||
from litellm.repositories.table_repositories import DailyModelUsageRepository
|
||||
from litellm.types.model_insights import (
|
||||
ModelInsightDailyMetric,
|
||||
ModelInsightMetric,
|
||||
ModelInsightsMetric,
|
||||
ModelInsightsResponse,
|
||||
ModelInsightTask,
|
||||
ModelInsightTasksResponse,
|
||||
ModelInsightTaskSummary,
|
||||
)
|
||||
|
||||
router: Final = APIRouter()
|
||||
|
||||
|
||||
class _Sums(BaseModel):
|
||||
spend: float = 0.0
|
||||
prompt_tokens: int = 0
|
||||
completion_tokens: int = 0
|
||||
request_count: int = 0
|
||||
successful_requests: int = 0
|
||||
failed_requests: int = 0
|
||||
|
||||
|
||||
class _GroupedModel(BaseModel):
|
||||
model_group: str
|
||||
model: str
|
||||
custom_llm_provider: str
|
||||
sums: _Sums = Field(alias="_sum")
|
||||
|
||||
|
||||
class _GroupedDaily(_GroupedModel):
|
||||
date: str
|
||||
|
||||
|
||||
class _GroupedTask(_GroupedModel):
|
||||
task_type: str
|
||||
|
||||
|
||||
_MODEL_ROWS: Final = TypeAdapter(list[_GroupedModel])
|
||||
_DAILY_ROWS: Final = TypeAdapter(list[_GroupedDaily])
|
||||
_TASK_ROWS: Final = TypeAdapter(list[_GroupedTask])
|
||||
_UNCATEGORIZED_TASK: Final = ModelInsightTask(
|
||||
task_type=MODEL_INSIGHTS_DEFAULT_TASK, label="Uncategorized", category="General"
|
||||
)
|
||||
_SUM_FIELDS: Final = {
|
||||
"spend": True,
|
||||
"prompt_tokens": True,
|
||||
"completion_tokens": True,
|
||||
"request_count": True,
|
||||
"successful_requests": True,
|
||||
"failed_requests": True,
|
||||
}
|
||||
|
||||
|
||||
def _parse_date(value: str | None, fallback: date) -> date:
|
||||
if value is None:
|
||||
return fallback
|
||||
try:
|
||||
return date.fromisoformat(value)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail="Dates must use YYYY-MM-DD") from exc
|
||||
|
||||
|
||||
def _metric(row: _GroupedModel) -> ModelInsightMetric:
|
||||
return ModelInsightMetric(
|
||||
model_group=row.model_group,
|
||||
model=row.model,
|
||||
provider=row.custom_llm_provider,
|
||||
spend=row.sums.spend,
|
||||
prompt_tokens=row.sums.prompt_tokens,
|
||||
completion_tokens=row.sums.completion_tokens,
|
||||
requests=row.sums.request_count,
|
||||
successful_requests=row.sums.successful_requests,
|
||||
failed_requests=row.sums.failed_requests,
|
||||
)
|
||||
|
||||
|
||||
def _rank_value(row: _GroupedModel, metric: ModelInsightsMetric) -> float:
|
||||
if metric == "requests":
|
||||
return row.sums.request_count
|
||||
if metric == "spend":
|
||||
return row.sums.spend
|
||||
return row.sums.prompt_tokens + row.sums.completion_tokens
|
||||
|
||||
|
||||
def _top_model_rows(rows: list[_GroupedModel], metric: ModelInsightsMetric) -> list[_GroupedModel]:
|
||||
return sorted(rows, key=lambda row: _rank_value(row, metric), reverse=True)[:MODEL_INSIGHTS_TOP_MODELS]
|
||||
|
||||
|
||||
def _deployment_filter(rows: list[_GroupedModel]) -> list[dict[str, str]]:
|
||||
return [
|
||||
{"model_group": row.model_group, "model": row.model, "custom_llm_provider": row.custom_llm_provider}
|
||||
for row in rows
|
||||
]
|
||||
|
||||
|
||||
def _daily_metric(row: _GroupedDaily) -> ModelInsightDailyMetric:
|
||||
return ModelInsightDailyMetric(date=row.date, **_metric(row).model_dump())
|
||||
|
||||
|
||||
def _summarize_tasks(rows: list[_GroupedTask], metric: ModelInsightsMetric) -> list[ModelInsightTaskSummary]:
|
||||
catalog: Final = load_model_insight_tasks()
|
||||
totals: Final[dict[str, float]] = {}
|
||||
leaders: Final[dict[str, _GroupedTask]] = {}
|
||||
for row in rows:
|
||||
value = _rank_value(row, metric)
|
||||
totals[row.task_type] = totals.get(row.task_type, 0.0) + value
|
||||
leader = leaders.get(row.task_type)
|
||||
if leader is None or value > _rank_value(leader, metric):
|
||||
leaders[row.task_type] = row
|
||||
grand: Final = sum(totals.values())
|
||||
return [
|
||||
ModelInsightTaskSummary(
|
||||
**(catalog.get(task) or _UNCATEGORIZED_TASK).model_copy(update={"task_type": task}).model_dump(),
|
||||
value=value,
|
||||
share=value / grand * 100 if grand else 0.0,
|
||||
leader=leaders[task].model_group,
|
||||
provider=leaders[task].custom_llm_provider,
|
||||
)
|
||||
for task, value in sorted(totals.items(), key=lambda item: item[1], reverse=True)
|
||||
]
|
||||
|
||||
|
||||
def _resolve_window(
|
||||
user_api_key_dict: UserAPIKeyAuth, start_date: str | None, end_date: str | None
|
||||
) -> tuple[date, date, Mapping[str, object], DailyModelUsageRepository]:
|
||||
from litellm.proxy.proxy_server import prisma_client
|
||||
|
||||
if user_api_key_dict.user_role not in (LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY):
|
||||
raise HTTPException(status_code=403, detail="Only proxy admins can view deployment-wide model insights")
|
||||
if prisma_client is None:
|
||||
raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value)
|
||||
|
||||
end_day: Final = _parse_date(end_date, datetime.now(timezone.utc).date())
|
||||
start_day: Final = _parse_date(start_date, end_day - timedelta(days=MODEL_INSIGHTS_MAX_RANGE_DAYS - 1))
|
||||
if start_day > end_day or (end_day - start_day).days >= MODEL_INSIGHTS_MAX_RANGE_DAYS:
|
||||
raise HTTPException(
|
||||
status_code=400, detail=f"Date range must be between 1 and {MODEL_INSIGHTS_MAX_RANGE_DAYS} days"
|
||||
)
|
||||
date_window: Final[Mapping[str, object]] = {"date": {"gte": start_day.isoformat(), "lte": end_day.isoformat()}}
|
||||
return start_day, end_day, date_window, DailyModelUsageRepository(prisma_client)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/model-insights",
|
||||
tags=["model insights"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
response_model=ModelInsightsResponse,
|
||||
)
|
||||
async def get_model_insights(
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
start_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to 365 days ago")] = None,
|
||||
end_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to today")] = None,
|
||||
metric: Annotated[ModelInsightsMetric, Query(description="Metric the top models are ranked by")] = "tokens",
|
||||
) -> ModelInsightsResponse:
|
||||
start_day, end_day, date_window, repository = _resolve_window(user_api_key_dict, start_date, end_date)
|
||||
table: Final = repository.table
|
||||
grouped_model_rows: Final = _MODEL_ROWS.validate_python(
|
||||
await table.group_by(
|
||||
by=["model_group", "model", "custom_llm_provider"],
|
||||
sum=_SUM_FIELDS,
|
||||
where=date_window,
|
||||
)
|
||||
)
|
||||
model_rows: Final = _top_model_rows(grouped_model_rows, metric)
|
||||
selected_window: Final = {**date_window, "OR": _deployment_filter(model_rows)}
|
||||
daily_rows: Final = _DAILY_ROWS.validate_python(
|
||||
await table.group_by(
|
||||
by=["date", "model_group", "model", "custom_llm_provider"],
|
||||
sum=_SUM_FIELDS,
|
||||
where=selected_window,
|
||||
order={"date": "asc"},
|
||||
)
|
||||
if model_rows
|
||||
else []
|
||||
)
|
||||
return ModelInsightsResponse(
|
||||
start_date=start_day.isoformat(),
|
||||
end_date=end_day.isoformat(),
|
||||
top_models=[_metric(row) for row in model_rows],
|
||||
daily=[_daily_metric(row) for row in daily_rows],
|
||||
)
|
||||
|
||||
|
||||
@router.get(
|
||||
"/model-insights/tasks",
|
||||
tags=["model insights"],
|
||||
dependencies=[Depends(user_api_key_auth)],
|
||||
response_model=ModelInsightTasksResponse,
|
||||
)
|
||||
async def get_model_insight_tasks(
|
||||
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
|
||||
start_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to 365 days ago")] = None,
|
||||
end_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to today")] = None,
|
||||
metric: Annotated[ModelInsightsMetric, Query(description="Metric task shares are computed from")] = "spend",
|
||||
) -> ModelInsightTasksResponse:
|
||||
start_day, end_day, date_window, repository = _resolve_window(user_api_key_dict, start_date, end_date)
|
||||
task_rows: Final = _TASK_ROWS.validate_python(
|
||||
await repository.table.group_by(
|
||||
by=["task_type", "model_group", "model", "custom_llm_provider"],
|
||||
sum=_SUM_FIELDS,
|
||||
where=date_window,
|
||||
)
|
||||
)
|
||||
return ModelInsightTasksResponse(
|
||||
start_date=start_day.isoformat(),
|
||||
end_date=end_day.isoformat(),
|
||||
tasks=_summarize_tasks(task_rows, metric),
|
||||
)
|
||||
22
litellm/proxy/model_insights_tasks.json
Normal file
22
litellm/proxy/model_insights_tasks.json
Normal file
|
|
@ -0,0 +1,22 @@
|
|||
{
|
||||
"classification": {"label": "Classification", "category": "General"},
|
||||
"content_writing": {"label": "Content Writing", "category": "General"},
|
||||
"roleplay_fiction": {"label": "Roleplay & Fiction", "category": "General"},
|
||||
"conversation": {"label": "Conversation", "category": "General"},
|
||||
"research_reports": {"label": "Research & Reports", "category": "General"},
|
||||
"qa_knowledge": {"label": "Q&A & Knowledge", "category": "General"},
|
||||
"customer_support": {"label": "Customer Support", "category": "General"},
|
||||
"summarization": {"label": "Summarization", "category": "General"},
|
||||
"translation": {"label": "Translation", "category": "General"},
|
||||
"workflow_execution": {"label": "Workflow Execution", "category": "Agent"},
|
||||
"multi_step_planning": {"label": "Multi-step Planning", "category": "Agent"},
|
||||
"tool_dispatch": {"label": "Tool Dispatch", "category": "Agent"},
|
||||
"code_generation": {"label": "Code Generation", "category": "Code"},
|
||||
"debugging": {"label": "Debugging", "category": "Code"},
|
||||
"code_review": {"label": "Code Review", "category": "Code"},
|
||||
"frontend_ui": {"label": "Frontend & UI", "category": "Code"},
|
||||
"file_io": {"label": "File I/O", "category": "Code"},
|
||||
"shell_execution": {"label": "Shell Execution", "category": "Code"},
|
||||
"data_extraction": {"label": "Data Extraction", "category": "Data"},
|
||||
"data_transformation": {"label": "Data Transformation", "category": "Data"}
|
||||
}
|
||||
|
|
@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend {
|
|||
@@id([date, tool_name])
|
||||
}
|
||||
|
||||
model LiteLLM_DailyModelUsage {
|
||||
date String
|
||||
model_group String
|
||||
model String
|
||||
custom_llm_provider String
|
||||
task_type String
|
||||
spend Float @default(0.0)
|
||||
prompt_tokens BigInt @default(0)
|
||||
completion_tokens BigInt @default(0)
|
||||
request_count BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
failed_requests BigInt @default(0)
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
@@id([date, model_group, model, custom_llm_provider, task_type])
|
||||
@@index([date])
|
||||
@@index([model_group])
|
||||
}
|
||||
|
||||
// Gateway request counts recorded at the ASGI edge by
|
||||
// BillableRequestMetricsMiddleware. This is the source of truth for SGR
|
||||
// (successful gateway requests): it counts what the proxy actually answered,
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ from litellm.repositories.table_repositories import (
|
|||
ConfigOverridesRepository,
|
||||
DailyGuardrailMetricsRepository,
|
||||
DailyGuardrailUsageUnitsRepository,
|
||||
DailyModelUsageRepository,
|
||||
DailyPolicyMetricsRepository,
|
||||
DailyTagSpendRepository,
|
||||
DailyToolSpendRepository,
|
||||
|
|
@ -105,6 +106,7 @@ __all__ = [
|
|||
"CredentialsRepository",
|
||||
"DailyGuardrailMetricsRepository",
|
||||
"DailyGuardrailUsageUnitsRepository",
|
||||
"DailyModelUsageRepository",
|
||||
"DailyPolicyMetricsRepository",
|
||||
"DailyTagSpendRepository",
|
||||
"DailyToolSpendRepository",
|
||||
|
|
|
|||
|
|
@ -212,6 +212,10 @@ class DailyToolSpendRepository(PrismaTableRepository["prisma_models.LiteLLM_Dail
|
|||
table_name = "litellm_dailytoolspend"
|
||||
|
||||
|
||||
class DailyModelUsageRepository(PrismaTableRepository["prisma_models.LiteLLM_DailyModelUsage"]):
|
||||
table_name = "litellm_dailymodelusage"
|
||||
|
||||
|
||||
class SpendLogGuardrailIndexRepository(PrismaTableRepository["prisma_models.LiteLLM_SpendLogGuardrailIndex"]):
|
||||
table_name = "litellm_spendlogguardrailindex"
|
||||
|
||||
|
|
|
|||
47
litellm/types/model_insights.py
Normal file
47
litellm/types/model_insights.py
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
ModelInsightsMetric = Literal["requests", "spend", "tokens"]
|
||||
|
||||
|
||||
class ModelInsightMetric(BaseModel):
|
||||
model_group: str
|
||||
model: str
|
||||
provider: str
|
||||
spend: float
|
||||
prompt_tokens: int
|
||||
completion_tokens: int
|
||||
requests: int
|
||||
successful_requests: int
|
||||
failed_requests: int
|
||||
|
||||
|
||||
class ModelInsightDailyMetric(ModelInsightMetric):
|
||||
date: str
|
||||
|
||||
|
||||
class ModelInsightTask(BaseModel):
|
||||
task_type: str
|
||||
label: str
|
||||
category: str
|
||||
|
||||
|
||||
class ModelInsightTaskSummary(ModelInsightTask):
|
||||
value: float
|
||||
share: float
|
||||
leader: str
|
||||
provider: str
|
||||
|
||||
|
||||
class ModelInsightsResponse(BaseModel):
|
||||
start_date: str
|
||||
end_date: str
|
||||
daily: list[ModelInsightDailyMetric]
|
||||
top_models: list[ModelInsightMetric]
|
||||
|
||||
|
||||
class ModelInsightTasksResponse(BaseModel):
|
||||
start_date: str
|
||||
end_date: str
|
||||
tasks: list[ModelInsightTaskSummary]
|
||||
|
|
@ -317,6 +317,7 @@ include = [
|
|||
"litellm/proxy/_experimental/out/**",
|
||||
"litellm/router_strategy/complexity_router/artifacts/*.json",
|
||||
"litellm/router_strategy/complexity_router/fuse_presets.json",
|
||||
"litellm/proxy/model_insights_tasks.json",
|
||||
"litellm/proxy/client/cli/commands/codex_base_instructions.md",
|
||||
]
|
||||
exclude = [
|
||||
|
|
|
|||
|
|
@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend {
|
|||
@@id([date, tool_name])
|
||||
}
|
||||
|
||||
model LiteLLM_DailyModelUsage {
|
||||
date String
|
||||
model_group String
|
||||
model String
|
||||
custom_llm_provider String
|
||||
task_type String
|
||||
spend Float @default(0.0)
|
||||
prompt_tokens BigInt @default(0)
|
||||
completion_tokens BigInt @default(0)
|
||||
request_count BigInt @default(0)
|
||||
successful_requests BigInt @default(0)
|
||||
failed_requests BigInt @default(0)
|
||||
created_at DateTime @default(now())
|
||||
updated_at DateTime @updatedAt
|
||||
|
||||
@@id([date, model_group, model, custom_llm_provider, task_type])
|
||||
@@index([date])
|
||||
@@index([model_group])
|
||||
}
|
||||
|
||||
// Gateway request counts recorded at the ASGI edge by
|
||||
// BillableRequestMetricsMiddleware. This is the source of truth for SGR
|
||||
// (successful gateway requests): it counts what the proxy actually answered,
|
||||
|
|
|
|||
19
tests/test_litellm/proxy/db/test_model_insights_tasks.py
Normal file
19
tests/test_litellm/proxy/db/test_model_insights_tasks.py
Normal file
|
|
@ -0,0 +1,19 @@
|
|||
from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks
|
||||
from litellm.proxy.db.model_usage_rollup import model_usage_task_type
|
||||
|
||||
|
||||
def test_every_task_has_a_label_and_a_category() -> None:
|
||||
tasks = load_model_insight_tasks()
|
||||
|
||||
assert tasks
|
||||
for name, task in tasks.items():
|
||||
assert task.task_type == name
|
||||
assert task.label
|
||||
assert task.category in {"General", "Agent", "Code", "Data"}
|
||||
|
||||
|
||||
def test_tasks_in_the_json_file_are_the_ones_the_rollup_accepts() -> None:
|
||||
for name in load_model_insight_tasks():
|
||||
assert model_usage_task_type(f'["task:{name}"]') == name
|
||||
|
||||
assert model_usage_task_type('["task:not_in_the_file"]') == "uncategorized"
|
||||
89
tests/test_litellm/proxy/db/test_model_usage_rollup.py
Normal file
89
tests/test_litellm/proxy/db/test_model_usage_rollup.py
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
from datetime import datetime, timezone
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage, model_usage_task_type
|
||||
|
||||
|
||||
def test_model_usage_task_type_reads_task_tag_or_defaults() -> None:
|
||||
assert model_usage_task_type('["team-a", "task:classification"]') == "classification"
|
||||
assert model_usage_task_type('["task:made-up"]') == "uncategorized"
|
||||
assert model_usage_task_type('["debugging"]') == "uncategorized"
|
||||
assert model_usage_task_type("[]") == "uncategorized"
|
||||
assert model_usage_task_type("not json") == "uncategorized"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_increment_daily_model_usage_uses_atomic_prisma_upsert() -> None:
|
||||
table = MagicMock()
|
||||
table.upsert = AsyncMock()
|
||||
prisma_client = MagicMock()
|
||||
prisma_client.db.litellm_dailymodelusage = table
|
||||
payload = {
|
||||
"request_id": "request-1",
|
||||
"call_type": "acompletion",
|
||||
"api_key": "key",
|
||||
"spend": 0.25,
|
||||
"total_tokens": 30,
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 20,
|
||||
"startTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
|
||||
"endTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
|
||||
"completionStartTime": None,
|
||||
"model": "openai/gpt-5.4-mini",
|
||||
"model_id": None,
|
||||
"model_group": "fast-chat",
|
||||
"mcp_namespaced_tool_name": None,
|
||||
"agent_id": None,
|
||||
"api_base": "",
|
||||
"user": "user",
|
||||
"metadata": "{}",
|
||||
"cache_hit": "False",
|
||||
"cache_key": "",
|
||||
"request_tags": "[]",
|
||||
"team_id": None,
|
||||
"organization_id": None,
|
||||
"end_user": None,
|
||||
"requester_ip_address": None,
|
||||
"custom_llm_provider": "openai",
|
||||
"messages": None,
|
||||
"response": None,
|
||||
"proxy_server_request": None,
|
||||
"session_id": None,
|
||||
"request_duration_ms": 20,
|
||||
"status": "success",
|
||||
"litellm_call_id": None,
|
||||
}
|
||||
|
||||
await increment_daily_model_usage(prisma_client, payload)
|
||||
|
||||
call = table.upsert.await_args.kwargs
|
||||
assert call["data"]["create"]["request_count"] == 1
|
||||
assert call["data"]["update"]["completion_tokens"] == {"increment": 20}
|
||||
assert call["data"]["create"]["task_type"] == "uncategorized"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_increment_daily_model_usage_records_task_from_request_tags() -> None:
|
||||
table = MagicMock()
|
||||
table.upsert = AsyncMock()
|
||||
prisma_client = MagicMock()
|
||||
prisma_client.db.litellm_dailymodelusage = table
|
||||
payload = {
|
||||
"call_type": "acompletion",
|
||||
"spend": 0.1,
|
||||
"prompt_tokens": 1,
|
||||
"completion_tokens": 2,
|
||||
"startTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
|
||||
"model": "gpt-5",
|
||||
"model_group": "gpt-5",
|
||||
"metadata": "{}",
|
||||
"request_tags": '["task:debugging"]',
|
||||
"custom_llm_provider": "openai",
|
||||
"status": "success",
|
||||
}
|
||||
|
||||
await increment_daily_model_usage(prisma_client, payload)
|
||||
|
||||
assert table.upsert.await_args.kwargs["data"]["create"]["task_type"] == "debugging"
|
||||
|
|
@ -0,0 +1,233 @@
|
|||
from datetime import datetime, timezone
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from fastapi import FastAPI
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage
|
||||
from litellm.proxy.management_endpoints.model_insights_endpoints import router
|
||||
|
||||
|
||||
def _override_auth() -> UserAPIKeyAuth:
|
||||
return UserAPIKeyAuth(api_key="sk-test", user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
|
||||
|
||||
|
||||
def _grouped_row(*, prompt_tokens: str = "100", completion_tokens: str = "200", **dimensions: str) -> dict[str, object]:
|
||||
return {
|
||||
**dimensions,
|
||||
"_sum": {
|
||||
"spend": 1.25,
|
||||
"prompt_tokens": prompt_tokens,
|
||||
"completion_tokens": completion_tokens,
|
||||
"request_count": "3",
|
||||
"successful_requests": "3",
|
||||
"failed_requests": "0",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def test_model_insights_reads_only_bounded_rollup() -> None:
|
||||
model = _grouped_row(model_group="fast-chat", model="openai/gpt-5.4-mini", custom_llm_provider="openai")
|
||||
prompt_heavy_model = _grouped_row(
|
||||
prompt_tokens="500",
|
||||
completion_tokens="10",
|
||||
model_group="long-context",
|
||||
model="anthropic/claude-sonnet-4-5",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
daily = _grouped_row(
|
||||
date="2026-09-28",
|
||||
model_group="fast-chat",
|
||||
model="openai/gpt-5.4-mini",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
table = MagicMock()
|
||||
table.group_by = AsyncMock(side_effect=[[model, prompt_heavy_model], [daily]])
|
||||
prisma = MagicMock()
|
||||
prisma.db.litellm_dailymodelusage = table
|
||||
prisma.db.query_raw = AsyncMock()
|
||||
prisma.db.litellm_spendlogs.find_many = AsyncMock()
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
app.dependency_overrides[user_api_key_auth] = _override_auth
|
||||
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
|
||||
response = TestClient(app).get("/model-insights?start_date=2026-09-09&end_date=2026-09-28")
|
||||
|
||||
assert response.status_code == 200
|
||||
assert response.json()["top_models"][0]["model_group"] == "long-context"
|
||||
assert "by_task" not in response.json()
|
||||
assert table.group_by.await_count == 2
|
||||
prisma.db.query_raw.assert_not_awaited()
|
||||
prisma.db.litellm_spendlogs.find_many.assert_not_awaited()
|
||||
|
||||
|
||||
def test_model_insights_rejects_ranges_over_365_days() -> None:
|
||||
prisma = MagicMock()
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
app.dependency_overrides[user_api_key_auth] = _override_auth
|
||||
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
|
||||
response = TestClient(app).get("/model-insights?start_date=2025-09-01&end_date=2026-09-28")
|
||||
|
||||
assert response.status_code == 400
|
||||
|
||||
|
||||
def _call(table: MagicMock, query: str, path: str = "/model-insights") -> object:
|
||||
prisma = MagicMock()
|
||||
prisma.db.litellm_dailymodelusage = table
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
app.dependency_overrides[user_api_key_auth] = _override_auth
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
|
||||
return TestClient(app).get(f"{path}?start_date=2026-09-01&end_date=2026-09-28&{query}")
|
||||
|
||||
|
||||
def test_model_insights_ranks_top_models_by_selected_metric() -> None:
|
||||
token_heavy = _grouped_row(
|
||||
prompt_tokens="9000", completion_tokens="9000", model_group="big", model="m1", custom_llm_provider="openai"
|
||||
)
|
||||
request_heavy = _grouped_row(
|
||||
prompt_tokens="1", completion_tokens="1", model_group="busy", model="m2", custom_llm_provider="openai"
|
||||
)
|
||||
request_heavy["_sum"]["request_count"] = "500"
|
||||
table = MagicMock()
|
||||
table.group_by = AsyncMock(side_effect=[[token_heavy, request_heavy], []])
|
||||
|
||||
by_requests = _call(table, "metric=requests").json()
|
||||
by_tokens = _call(
|
||||
MagicMock(group_by=AsyncMock(side_effect=[[token_heavy, request_heavy], []])), "metric=tokens"
|
||||
).json()
|
||||
|
||||
assert by_requests["top_models"][0]["model_group"] == "busy"
|
||||
assert by_tokens["top_models"][0]["model_group"] == "big"
|
||||
|
||||
|
||||
def test_model_insights_scopes_daily_to_ranked_deployments() -> None:
|
||||
ranked = _grouped_row(model_group="shared", model="m1", custom_llm_provider="openai")
|
||||
table = MagicMock()
|
||||
table.group_by = AsyncMock(side_effect=[[ranked], []])
|
||||
|
||||
_call(table, "metric=tokens")
|
||||
|
||||
daily_where = table.group_by.await_args_list[1].kwargs["where"]
|
||||
assert daily_where["OR"] == [{"model_group": "shared", "model": "m1", "custom_llm_provider": "openai"}]
|
||||
assert "model_group" not in daily_where
|
||||
|
||||
|
||||
def _task_rows() -> list[dict[str, object]]:
|
||||
def row(task: str, group: str, requests: str, spend: float) -> dict[str, object]:
|
||||
base = _grouped_row(task_type=task, model_group=group, model=group, custom_llm_provider="openai")
|
||||
base["_sum"].update({"request_count": requests, "spend": spend}) # type: ignore[union-attr]
|
||||
return base
|
||||
|
||||
return [
|
||||
row("debugging", "big", "1", 9.0),
|
||||
row("debugging", "busy", "50", 1.0),
|
||||
row("classification", "busy", "10", 1.0),
|
||||
]
|
||||
|
||||
|
||||
def test_model_insight_tasks_are_summarised_on_the_server() -> None:
|
||||
table = MagicMock(group_by=AsyncMock(return_value=_task_rows()))
|
||||
|
||||
body = _call(table, "metric=spend", path="/model-insights/tasks").json()
|
||||
|
||||
assert [(t["task_type"], t["label"], t["category"], t["leader"]) for t in body["tasks"]] == [
|
||||
("debugging", "Debugging", "Code", "big"),
|
||||
("classification", "Classification", "General", "busy"),
|
||||
]
|
||||
assert [round(t["share"], 1) for t in body["tasks"]] == [90.9, 9.1]
|
||||
assert "OR" not in table.group_by.await_args.kwargs["where"]
|
||||
assert "take" not in table.group_by.await_args.kwargs
|
||||
|
||||
|
||||
def test_model_insight_tasks_leader_follows_the_selected_metric() -> None:
|
||||
by_spend = _call(MagicMock(group_by=AsyncMock(return_value=_task_rows())), "metric=spend", "/model-insights/tasks")
|
||||
by_requests = _call(
|
||||
MagicMock(group_by=AsyncMock(return_value=_task_rows())), "metric=requests", "/model-insights/tasks"
|
||||
)
|
||||
|
||||
assert by_spend.json()["tasks"][0]["leader"] == "big"
|
||||
assert by_requests.json()["tasks"][0]["leader"] == "busy"
|
||||
|
||||
|
||||
def test_model_insight_tasks_unknown_task_shows_as_uncategorized() -> None:
|
||||
row = _grouped_row(task_type="uncategorized", model_group="a", model="a", custom_llm_provider="openai")
|
||||
body = _call(MagicMock(group_by=AsyncMock(return_value=[row])), "metric=spend", "/model-insights/tasks").json()
|
||||
|
||||
assert [(t["label"], t["category"]) for t in body["tasks"]] == [("Uncategorized", "General")]
|
||||
|
||||
|
||||
def test_model_insight_tasks_require_an_admin() -> None:
|
||||
app = FastAPI()
|
||||
app.include_router(router)
|
||||
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
|
||||
api_key="sk-test", user_id="u", user_role=LitellmUserRoles.INTERNAL_USER
|
||||
)
|
||||
with patch("litellm.proxy.proxy_server.prisma_client", MagicMock()):
|
||||
assert TestClient(app).get("/model-insights/tasks").status_code == 403
|
||||
|
||||
|
||||
def test_model_insights_rejects_unknown_metric() -> None:
|
||||
assert _call(MagicMock(group_by=AsyncMock()), "metric=bogus").status_code == 422
|
||||
|
||||
|
||||
class _InMemoryUsageTable:
|
||||
def __init__(self) -> None:
|
||||
self.rows: dict[tuple[str, ...], dict[str, float]] = {}
|
||||
|
||||
async def upsert(self, where: dict, data: dict) -> None:
|
||||
key_fields = where["date_model_group_model_custom_llm_provider_task_type"]
|
||||
key = tuple(key_fields.values())
|
||||
if key not in self.rows:
|
||||
self.rows[key] = {**key_fields, **{k: v for k, v in data["create"].items() if k not in key_fields}}
|
||||
return
|
||||
for field, change in data["update"].items():
|
||||
self.rows[key][field] += change["increment"]
|
||||
|
||||
async def group_by(self, by: list[str], sum: dict, where: dict, **_: object) -> list[dict]:
|
||||
grouped: dict[tuple, dict] = {}
|
||||
for row in self.rows.values():
|
||||
if not where["date"]["gte"] <= row["date"] <= where["date"]["lte"]:
|
||||
continue
|
||||
if where.get("OR") and not any(all(row[k] == v for k, v in option.items()) for option in where["OR"]):
|
||||
continue
|
||||
bucket = grouped.setdefault(tuple(row[k] for k in by), {**{k: row[k] for k in by}, "_sum": {}})
|
||||
for field in sum:
|
||||
bucket["_sum"][field] = bucket["_sum"].get(field, 0) + row[field]
|
||||
return list(grouped.values())
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_model_insights_reads_back_what_the_rollup_wrote() -> None:
|
||||
table = _InMemoryUsageTable()
|
||||
prisma = MagicMock()
|
||||
prisma.db.litellm_dailymodelusage = table
|
||||
payload = {
|
||||
"call_type": "acompletion",
|
||||
"spend": 0.5,
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 20,
|
||||
"startTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
|
||||
"model": "gpt-5",
|
||||
"model_group": "gpt-5",
|
||||
"metadata": "{}",
|
||||
"request_tags": '["task:debugging"]',
|
||||
"custom_llm_provider": "openai",
|
||||
"status": "success",
|
||||
}
|
||||
|
||||
await increment_daily_model_usage(prisma, payload)
|
||||
await increment_daily_model_usage(prisma, {**payload, "request_tags": "[]"})
|
||||
|
||||
body = _call(table, "metric=requests").json()
|
||||
|
||||
assert [(m["model_group"], m["requests"], m["prompt_tokens"]) for m in body["top_models"]] == [("gpt-5", 2, 20)]
|
||||
tasks = _call(table, "metric=requests", path="/model-insights/tasks").json()["tasks"]
|
||||
assert sorted((t["task_type"], t["value"]) for t in tasks) == [("debugging", 1), ("uncategorized", 1)]
|
||||
assert [(d["date"], d["requests"]) for d in body["daily"]] == [("2026-09-28", 2)]
|
||||
|
|
@ -35,6 +35,7 @@ const LEGACY_PAGE_ROUTES: ReadonlyMap<string, string> = new Map(
|
|||
new_usage: "usage",
|
||||
usage: "old-usage",
|
||||
"cost-optimization": "cost-optimization",
|
||||
"model-insights": "model-insights",
|
||||
agents: "agents",
|
||||
"router-settings": "router-settings",
|
||||
users: "users",
|
||||
|
|
|
|||
|
|
@ -0,0 +1,148 @@
|
|||
import { render, screen, waitFor } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import type React from "react";
|
||||
import { beforeEach, describe, expect, it, vi } from "vitest";
|
||||
|
||||
import ModelInsightsView from "./ModelInsightsView";
|
||||
import { apiClient } from "@/components/networking";
|
||||
|
||||
vi.mock("@/components/networking", () => ({ apiClient: { get: vi.fn() } }));
|
||||
vi.mock("@/components/ui/chart", () => ({
|
||||
ChartContainer: ({ children }: { children: React.ReactNode }) => <div>{children}</div>,
|
||||
ChartTooltip: () => null,
|
||||
ChartTooltipContent: () => null,
|
||||
}));
|
||||
vi.mock("recharts", () => ({
|
||||
Bar: () => null,
|
||||
BarChart: ({ children }: { children: React.ReactNode }) => <div>{children}</div>,
|
||||
CartesianGrid: () => null,
|
||||
Treemap: () => null,
|
||||
XAxis: () => null,
|
||||
YAxis: () => null,
|
||||
}));
|
||||
|
||||
const metrics = {
|
||||
model_group: "fast-chat",
|
||||
model: "openai/gpt-5.4-mini",
|
||||
provider: "openai",
|
||||
spend: 2.5,
|
||||
prompt_tokens: 1000,
|
||||
completion_tokens: 2000,
|
||||
requests: 12,
|
||||
successful_requests: 12,
|
||||
failed_requests: 0,
|
||||
};
|
||||
|
||||
const response = {
|
||||
start_date: "2025-09-29",
|
||||
end_date: "2026-09-28",
|
||||
top_models: [metrics],
|
||||
daily: [{ ...metrics, date: "2026-09-28" }],
|
||||
};
|
||||
|
||||
const taskResponse = {
|
||||
start_date: "2025-09-29",
|
||||
end_date: "2026-09-28",
|
||||
tasks: [
|
||||
{
|
||||
task_type: "code_generation",
|
||||
label: "Code Generation",
|
||||
category: "Code",
|
||||
value: 2.5,
|
||||
share: 100,
|
||||
leader: "fast-chat",
|
||||
provider: "openai",
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
const mockApi = (tasks: unknown = taskResponse) =>
|
||||
vi
|
||||
.mocked(apiClient.get)
|
||||
.mockImplementation((path: string) =>
|
||||
path === "/model-insights/tasks" ? (tasks as Promise<unknown>) : Promise.resolve(response),
|
||||
);
|
||||
|
||||
describe("ModelInsightsView", () => {
|
||||
beforeEach(() => {
|
||||
vi.mocked(apiClient.get).mockReset();
|
||||
mockApi(Promise.resolve(taskResponse));
|
||||
});
|
||||
|
||||
it("shows the ranking with share and the task legend from the API response", async () => {
|
||||
render(<ModelInsightsView accessToken="token" />);
|
||||
|
||||
expect(await screen.findByText("fast-chat")).toBeInTheDocument();
|
||||
expect(screen.getByText("by openai")).toBeInTheDocument();
|
||||
expect(await screen.findByText("Code")).toBeInTheDocument();
|
||||
expect(screen.getAllByText("100.0%")).toHaveLength(2);
|
||||
expect(screen.getByRole("tab", { name: "tokens" })).toHaveAttribute("aria-selected", "true");
|
||||
expect(screen.getByRole("tab", { name: "log" })).toBeInTheDocument();
|
||||
expect(apiClient.get).toHaveBeenCalledWith("/model-insights", {
|
||||
accessToken: "token",
|
||||
query: { metric: "tokens" },
|
||||
});
|
||||
});
|
||||
|
||||
it("refetches with the selected metric so top models are ranked by it", async () => {
|
||||
render(<ModelInsightsView accessToken="token" />);
|
||||
await screen.findByText("fast-chat");
|
||||
|
||||
await userEvent.click(screen.getByRole("tab", { name: "requests" }));
|
||||
|
||||
await waitFor(() =>
|
||||
expect(apiClient.get).toHaveBeenCalledWith("/model-insights", {
|
||||
accessToken: "token",
|
||||
query: { metric: "requests" },
|
||||
}),
|
||||
);
|
||||
});
|
||||
|
||||
it("does not refetch the task breakdown when the chart metric changes", async () => {
|
||||
render(<ModelInsightsView accessToken="token" />);
|
||||
await screen.findByText("Code");
|
||||
const taskCalls = () =>
|
||||
vi.mocked(apiClient.get).mock.calls.filter(([path]) => path === "/model-insights/tasks").length;
|
||||
const before = taskCalls();
|
||||
|
||||
await userEvent.click(screen.getByRole("tab", { name: "requests" }));
|
||||
await waitFor(() =>
|
||||
expect(apiClient.get).toHaveBeenCalledWith("/model-insights", {
|
||||
accessToken: "token",
|
||||
query: { metric: "requests" },
|
||||
}),
|
||||
);
|
||||
|
||||
expect(taskCalls()).toBe(before);
|
||||
});
|
||||
|
||||
it("shows the API error instead of loading forever", async () => {
|
||||
vi.mocked(apiClient.get).mockRejectedValue(new Error("Only proxy admins can view deployment-wide model insights"));
|
||||
render(<ModelInsightsView accessToken="token" />);
|
||||
|
||||
expect(await screen.findByText("Could not load model insights")).toBeInTheDocument();
|
||||
expect(screen.getByText("Only proxy admins can view deployment-wide model insights")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("keeps the previous ranking, dimmed, until the new metric's data arrives", async () => {
|
||||
render(<ModelInsightsView accessToken="token" />);
|
||||
await screen.findByText("fast-chat");
|
||||
let resolve: (value: typeof response) => void = () => {};
|
||||
vi.mocked(apiClient.get).mockImplementation((path: string) =>
|
||||
path === "/model-insights/tasks"
|
||||
? Promise.resolve(taskResponse)
|
||||
: new Promise((done) => (resolve = done as typeof resolve)),
|
||||
);
|
||||
|
||||
await userEvent.click(screen.getByRole("tab", { name: "spend" }));
|
||||
|
||||
expect(
|
||||
screen.getByText("Share of tokens, with the change between the first and second half of the period"),
|
||||
).toBeInTheDocument();
|
||||
|
||||
resolve(response);
|
||||
expect(
|
||||
await screen.findByText("Share of spend, with the change between the first and second half of the period"),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,352 @@
|
|||
"use client";
|
||||
|
||||
import React from "react";
|
||||
import { Bar, BarChart, CartesianGrid, Treemap, XAxis, YAxis } from "recharts";
|
||||
import { ArrowDownRight, ArrowUpRight, BarChart3, Layers, Minus } from "lucide-react";
|
||||
|
||||
import { apiClient } from "@/components/networking";
|
||||
import { extractErrorMessage } from "@/utils/errorUtils";
|
||||
import { ProviderLogo } from "@/components/molecules/models/ProviderLogo";
|
||||
import { PageHeader } from "@/components/shared/PageHeader";
|
||||
import { Alert, AlertDescription, AlertTitle } from "@/components/ui/alert";
|
||||
import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
|
||||
import { ChartConfig, ChartContainer, ChartTooltip, ChartTooltipContent } from "@/components/ui/chart";
|
||||
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select";
|
||||
import { Skeleton } from "@/components/ui/skeleton";
|
||||
import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs";
|
||||
import {
|
||||
buildWeeklySeries,
|
||||
formatMetric,
|
||||
Metric,
|
||||
ModelInsightsResponse,
|
||||
ModelInsightTasksResponse,
|
||||
TaskSummary,
|
||||
modelOrder,
|
||||
rankModels,
|
||||
RankedModel,
|
||||
} from "./modelInsightsData";
|
||||
|
||||
const PALETTE = [
|
||||
"#ec4899",
|
||||
"#a855f7",
|
||||
"#f59e0b",
|
||||
"#3b82f6",
|
||||
"#10b981",
|
||||
"#ef4444",
|
||||
"#14b8a6",
|
||||
"#84cc16",
|
||||
"#6366f1",
|
||||
"#f97316",
|
||||
];
|
||||
const FALLBACK_COLOR = "#64748b";
|
||||
const CATEGORY_COLORS: Record<string, string> = {
|
||||
General: "#ee8650",
|
||||
Agent: "#7666e4",
|
||||
Code: "#5fb074",
|
||||
Data: "#3b82f6",
|
||||
};
|
||||
const SCALES = ["linear", "log"] as const;
|
||||
const METRIC_LABELS: Record<Metric, string> = { requests: "requests", spend: "spend", tokens: "tokens" };
|
||||
const RANKING_ROWS = 5;
|
||||
|
||||
type Scale = (typeof SCALES)[number];
|
||||
|
||||
const formatDelta = (value: number) => `${value > 0 ? "+" : ""}${value.toFixed(1)}`;
|
||||
|
||||
const DeltaBadge = ({ value }: { value: number }) => {
|
||||
if (Math.abs(value) < 0.05) {
|
||||
return (
|
||||
<span className="flex items-center justify-end gap-1 text-xs text-muted-foreground">
|
||||
<Minus className="size-3" /> 0.0
|
||||
</span>
|
||||
);
|
||||
}
|
||||
const up = value > 0;
|
||||
const Icon = up ? ArrowUpRight : ArrowDownRight;
|
||||
return (
|
||||
<span className={`flex items-center justify-end gap-1 text-xs ${up ? "text-emerald-600" : "text-red-600"}`}>
|
||||
<Icon className="size-3" /> {formatDelta(value)}
|
||||
</span>
|
||||
);
|
||||
};
|
||||
|
||||
const RankingRow = ({ model, rank }: { model: RankedModel; rank: number }) => (
|
||||
<li className="grid grid-cols-[1.5rem_2.5rem_1fr_auto] items-center gap-3 py-2">
|
||||
<span className="text-sm tabular-nums text-muted-foreground">{rank}</span>
|
||||
<ProviderLogo provider={model.provider} className="size-9 rounded-md border p-1" />
|
||||
<div className="min-w-0">
|
||||
<p className="truncate font-medium">{model.model_group}</p>
|
||||
<p className="truncate text-sm text-muted-foreground">by {model.provider}</p>
|
||||
</div>
|
||||
<div className="text-right">
|
||||
<p className="font-medium tabular-nums">{model.share.toFixed(1)}%</p>
|
||||
<DeltaBadge value={model.delta} />
|
||||
</div>
|
||||
</li>
|
||||
);
|
||||
|
||||
type TileProps = TaskSummary & { x: number; y: number; width: number; height: number; index: number };
|
||||
|
||||
const TaskTileContent = ({ x, y, width, height, category, label, leader }: TileProps) => {
|
||||
if (width <= 0 || height <= 0) return null;
|
||||
const color = CATEGORY_COLORS[category] ?? FALLBACK_COLOR;
|
||||
const fits = width > 90 && height > 44;
|
||||
return (
|
||||
<g>
|
||||
<rect x={x} y={y} width={width} height={height} fill={color} stroke="#fff" strokeWidth={2} />
|
||||
{fits && (
|
||||
<>
|
||||
<text x={x + 12} y={y + 26} fill="#fff" fontSize={16} fontWeight={500}>
|
||||
{label}
|
||||
</text>
|
||||
<text x={x + 12} y={y + 46} fill="#ffffffcc" fontSize={12}>
|
||||
{leader}
|
||||
</text>
|
||||
</>
|
||||
)}
|
||||
</g>
|
||||
);
|
||||
};
|
||||
|
||||
export default function ModelInsightsView({ accessToken }: { accessToken: string | null }) {
|
||||
const [loaded, setLoaded] = React.useState<{ metric: Metric; response: ModelInsightsResponse } | null>(null);
|
||||
const [metric, setMetric] = React.useState<Metric>("tokens");
|
||||
const [scale, setScale] = React.useState<Scale>("linear");
|
||||
const [taskMetric, setTaskMetric] = React.useState<Metric>("spend");
|
||||
const [taskData, setTaskData] = React.useState<ModelInsightTasksResponse | null>(null);
|
||||
const [taskError, setTaskError] = React.useState<string | null>(null);
|
||||
const [error, setError] = React.useState<string | null>(null);
|
||||
|
||||
React.useEffect(() => {
|
||||
if (!accessToken) return;
|
||||
let cancelled = false;
|
||||
apiClient
|
||||
.get<ModelInsightsResponse>("/model-insights", { accessToken, query: { metric } })
|
||||
.then((response) => {
|
||||
if (cancelled) return;
|
||||
setError(null);
|
||||
setLoaded({ metric, response });
|
||||
})
|
||||
.catch((err: unknown) => {
|
||||
if (!cancelled) setError(extractErrorMessage(err));
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [accessToken, metric]);
|
||||
|
||||
React.useEffect(() => {
|
||||
if (!accessToken) return;
|
||||
let cancelled = false;
|
||||
apiClient
|
||||
.get<ModelInsightTasksResponse>("/model-insights/tasks", { accessToken, query: { metric: taskMetric } })
|
||||
.then((response) => {
|
||||
if (cancelled) return;
|
||||
setTaskError(null);
|
||||
setTaskData(response);
|
||||
})
|
||||
.catch((err: unknown) => {
|
||||
if (!cancelled) setTaskError(extractErrorMessage(err));
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [accessToken, taskMetric]);
|
||||
|
||||
const data = loaded?.response ?? null;
|
||||
const shown = loaded?.metric ?? metric;
|
||||
const isStale = loaded !== null && loaded.metric !== metric;
|
||||
const range = React.useMemo(() => ({ start: data?.start_date ?? "", end: data?.end_date ?? "" }), [data]);
|
||||
const models = React.useMemo(() => (data ? modelOrder(data.daily, shown) : []), [data, shown]);
|
||||
const series = React.useMemo(
|
||||
() => (data ? buildWeeklySeries(data.daily, models, shown, range) : []),
|
||||
[data, models, shown, range],
|
||||
);
|
||||
const ranking = React.useMemo(
|
||||
() => (data ? rankModels(data.top_models, data.daily, shown, range) : []),
|
||||
[data, shown, range],
|
||||
);
|
||||
const tiles = React.useMemo(() => taskData?.tasks ?? [], [taskData]);
|
||||
const categoryShares = React.useMemo(
|
||||
() =>
|
||||
[...new Set(tiles.map((tile) => tile.category))].map((category) => ({
|
||||
category,
|
||||
share: tiles.filter((tile) => tile.category === category).reduce((sum, tile) => sum + tile.share, 0),
|
||||
})),
|
||||
[tiles],
|
||||
);
|
||||
|
||||
if (error) {
|
||||
return (
|
||||
<div className="p-8">
|
||||
<Alert variant="destructive">
|
||||
<AlertTitle>Could not load model insights</AlertTitle>
|
||||
<AlertDescription>{error}</AlertDescription>
|
||||
</Alert>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
if (!data) {
|
||||
return (
|
||||
<div className="space-y-6 p-8">
|
||||
<Skeleton className="h-16 w-96" />
|
||||
<Skeleton className="h-96 w-full" />
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
const chartConfig = Object.fromEntries(
|
||||
models.map((model, index) => [model, { label: model, color: PALETTE[index % PALETTE.length] }]),
|
||||
) satisfies ChartConfig;
|
||||
|
||||
return (
|
||||
<main className="w-full space-y-6 p-8">
|
||||
<PageHeader
|
||||
icon={<BarChart3 />}
|
||||
title="Model Leaderboard"
|
||||
subtitle={`See which models your gateway used from ${data.start_date} through ${data.end_date}`}
|
||||
/>
|
||||
|
||||
<Card aria-busy={isStale} className={isStale ? "opacity-60 transition-opacity" : "transition-opacity"}>
|
||||
<CardHeader className="flex-row items-start justify-between space-y-0">
|
||||
<div>
|
||||
<CardTitle>Top models</CardTitle>
|
||||
<CardDescription>Weekly {METRIC_LABELS[shown]} across your gateway</CardDescription>
|
||||
</div>
|
||||
<div className="flex items-center gap-3">
|
||||
<Tabs value={metric} onValueChange={(value) => setMetric(value as Metric)}>
|
||||
<TabsList>
|
||||
{(["requests", "spend", "tokens"] as const).map((value) => (
|
||||
<TabsTrigger key={value} value={value} className="capitalize">
|
||||
{value}
|
||||
</TabsTrigger>
|
||||
))}
|
||||
</TabsList>
|
||||
</Tabs>
|
||||
<Tabs value={scale} onValueChange={(value) => setScale(value as Scale)}>
|
||||
<TabsList>
|
||||
{SCALES.map((value) => (
|
||||
<TabsTrigger key={value} value={value} className="capitalize">
|
||||
{value}
|
||||
</TabsTrigger>
|
||||
))}
|
||||
</TabsList>
|
||||
</Tabs>
|
||||
</div>
|
||||
</CardHeader>
|
||||
<CardContent>
|
||||
<ChartContainer config={chartConfig} className="h-[380px] w-full aspect-auto">
|
||||
<BarChart data={series} margin={{ left: 8, right: 8 }} barCategoryGap={2}>
|
||||
<CartesianGrid vertical={false} />
|
||||
<XAxis dataKey="date" tickLine={false} axisLine={false} minTickGap={48} />
|
||||
<YAxis
|
||||
scale={scale}
|
||||
domain={scale === "log" ? [1, "auto"] : [0, "auto"]}
|
||||
allowDataOverflow
|
||||
tickLine={false}
|
||||
axisLine={false}
|
||||
tickFormatter={(value) => formatMetric(Number(value), shown)}
|
||||
/>
|
||||
<ChartTooltip content={<ChartTooltipContent />} />
|
||||
{models.map((model, index) => (
|
||||
<Bar
|
||||
key={model}
|
||||
dataKey={model}
|
||||
stackId="usage"
|
||||
fill={PALETTE[index % PALETTE.length]}
|
||||
isAnimationActive={false}
|
||||
/>
|
||||
))}
|
||||
</BarChart>
|
||||
</ChartContainer>
|
||||
</CardContent>
|
||||
</Card>
|
||||
|
||||
<Card aria-busy={isStale} className={isStale ? "opacity-60 transition-opacity" : "transition-opacity"}>
|
||||
<CardHeader>
|
||||
<CardTitle>Leaderboard</CardTitle>
|
||||
<CardDescription>
|
||||
Share of {METRIC_LABELS[shown]}, with the change between the first and second half of the period
|
||||
</CardDescription>
|
||||
</CardHeader>
|
||||
<CardContent className="grid gap-x-12 md:grid-cols-2">
|
||||
<ol className="divide-y">
|
||||
{ranking.slice(0, RANKING_ROWS).map((model, index) => (
|
||||
<RankingRow key={model.model_group} model={model} rank={index + 1} />
|
||||
))}
|
||||
</ol>
|
||||
<ol className="divide-y">
|
||||
{ranking.slice(RANKING_ROWS).map((model, index) => (
|
||||
<RankingRow key={model.model_group} model={model} rank={RANKING_ROWS + index + 1} />
|
||||
))}
|
||||
</ol>
|
||||
</CardContent>
|
||||
</Card>
|
||||
|
||||
<Card>
|
||||
<CardHeader className="flex-row items-start justify-between space-y-0">
|
||||
<div>
|
||||
<CardTitle className="flex items-center gap-2">
|
||||
<Layers className="size-5" /> Top models by task
|
||||
</CardTitle>
|
||||
<CardDescription>
|
||||
Each task's share of {METRIC_LABELS[taskMetric]}, labelled with its leading model
|
||||
</CardDescription>
|
||||
</div>
|
||||
<Select value={taskMetric} onValueChange={(value) => setTaskMetric(value as Metric)}>
|
||||
<SelectTrigger className="w-44" aria-label="Task metric">
|
||||
<SelectValue />
|
||||
</SelectTrigger>
|
||||
<SelectContent>
|
||||
<SelectItem value="spend">Share of spend</SelectItem>
|
||||
<SelectItem value="requests">Share of requests</SelectItem>
|
||||
<SelectItem value="tokens">Share of tokens</SelectItem>
|
||||
</SelectContent>
|
||||
</Select>
|
||||
</CardHeader>
|
||||
<CardContent className="space-y-4">
|
||||
{taskError && (
|
||||
<Alert variant="destructive">
|
||||
<AlertTitle>Could not load tasks</AlertTitle>
|
||||
<AlertDescription>{taskError}</AlertDescription>
|
||||
</Alert>
|
||||
)}
|
||||
<ChartContainer config={{}} className="h-[360px] w-full aspect-auto">
|
||||
<Treemap
|
||||
data={tiles.map((tile) => ({ ...tile, name: tile.task_type }))}
|
||||
dataKey="value"
|
||||
isAnimationActive={false}
|
||||
content={<TaskTileContent {...({} as TileProps)} />}
|
||||
/>
|
||||
</ChartContainer>
|
||||
<ul className="flex flex-wrap gap-x-6 gap-y-2">
|
||||
{categoryShares.map(({ category, share }) => (
|
||||
<li key={category} className="flex items-center gap-2 text-sm">
|
||||
<span
|
||||
className="size-3 rounded-full"
|
||||
style={{ backgroundColor: CATEGORY_COLORS[category] ?? FALLBACK_COLOR }}
|
||||
/>
|
||||
<span className="text-muted-foreground">{category}</span>
|
||||
<span className="font-medium tabular-nums">{share.toFixed(1)}%</span>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</CardContent>
|
||||
</Card>
|
||||
|
||||
<Card>
|
||||
<CardHeader>
|
||||
<CardTitle>Cost per session</CardTitle>
|
||||
<CardDescription>Session cost is not estimated from request counts</CardDescription>
|
||||
</CardHeader>
|
||||
<CardContent>
|
||||
<p className="text-sm text-muted-foreground">
|
||||
Add a stable session_id to requests to unlock accurate session-level model comparisons in a future bounded
|
||||
session rollup
|
||||
</p>
|
||||
</CardContent>
|
||||
</Card>
|
||||
</main>
|
||||
);
|
||||
}
|
||||
|
|
@ -0,0 +1,92 @@
|
|||
import { describe, expect, it } from "vitest";
|
||||
|
||||
import { buildWeeklySeries, DailyMetric, formatMetric, modelOrder, rankModels } from "./modelInsightsData";
|
||||
|
||||
const row = (over: Partial<DailyMetric>): DailyMetric => ({
|
||||
model_group: "a",
|
||||
model: "a",
|
||||
provider: "openai",
|
||||
date: "2026-01-01",
|
||||
spend: 0,
|
||||
prompt_tokens: 0,
|
||||
completion_tokens: 0,
|
||||
requests: 0,
|
||||
successful_requests: 0,
|
||||
failed_requests: 0,
|
||||
...over,
|
||||
});
|
||||
|
||||
describe("buildWeeklySeries", () => {
|
||||
const range = { start: "2026-01-01", end: "2026-01-15" };
|
||||
|
||||
it("sums days into 7-day buckets per model", () => {
|
||||
const rows = [
|
||||
row({ date: "2026-01-01", requests: 1 }),
|
||||
row({ date: "2026-01-07", requests: 2 }),
|
||||
row({ date: "2026-01-08", requests: 4 }),
|
||||
row({ date: "2026-01-02", model_group: "b", requests: 8 }),
|
||||
];
|
||||
expect(buildWeeklySeries(rows, ["a", "b"], "requests", range)).toEqual([
|
||||
{ date: "2026-01-01", a: 3, b: 8 },
|
||||
{ date: "2026-01-08", a: 4, b: 0 },
|
||||
{ date: "2026-01-15", a: 0, b: 0 },
|
||||
]);
|
||||
});
|
||||
|
||||
it("keeps weeks with no usage as zero instead of dropping them", () => {
|
||||
const rows = [row({ date: "2026-01-01", requests: 1 }), row({ date: "2026-01-15", requests: 2 })];
|
||||
expect(buildWeeklySeries(rows, ["a"], "requests", range).map((week) => [week.date, week.a])).toEqual([
|
||||
["2026-01-01", 1],
|
||||
["2026-01-08", 0],
|
||||
["2026-01-15", 2],
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("modelOrder", () => {
|
||||
it("orders models by the selected metric, largest first", () => {
|
||||
const rows = [row({ model_group: "a", spend: 1, requests: 9 }), row({ model_group: "b", spend: 5, requests: 1 })];
|
||||
expect(modelOrder(rows, "spend")).toEqual(["b", "a"]);
|
||||
expect(modelOrder(rows, "requests")).toEqual(["a", "b"]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("rankModels", () => {
|
||||
const range = { start: "2026-01-01", end: "2026-01-10" };
|
||||
const totals = [row({ model_group: "a", requests: 40 }), row({ model_group: "b", requests: 40 })];
|
||||
|
||||
it("computes share and the change in share between the first and second half of the range", () => {
|
||||
const daily = [
|
||||
row({ date: "2026-01-01", model_group: "a", requests: 30 }),
|
||||
row({ date: "2026-01-01", model_group: "b", requests: 10 }),
|
||||
row({ date: "2026-01-10", model_group: "a", requests: 10 }),
|
||||
row({ date: "2026-01-10", model_group: "b", requests: 30 }),
|
||||
];
|
||||
const ranked = rankModels(totals, daily, "requests", range);
|
||||
expect(ranked.find((m) => m.model_group === "a")).toMatchObject({ share: 50, delta: -50 });
|
||||
expect(ranked.find((m) => m.model_group === "b")).toMatchObject({ share: 50, delta: 50 });
|
||||
});
|
||||
|
||||
it("splits at the middle of the range, not the middle of the days that had usage", () => {
|
||||
const daily = [
|
||||
row({ date: "2026-01-01", model_group: "a", requests: 10 }),
|
||||
row({ date: "2026-01-02", model_group: "b", requests: 10 }),
|
||||
row({ date: "2026-01-03", model_group: "b", requests: 10 }),
|
||||
];
|
||||
const ranked = rankModels(totals, daily, "requests", range);
|
||||
expect(ranked.find((m) => m.model_group === "a")?.delta).toBe(0);
|
||||
});
|
||||
|
||||
it("shows no change when one half of the range has no usage to compare against", () => {
|
||||
const daily = [row({ date: "2026-01-10", model_group: "a", requests: 10 })];
|
||||
const ranked = rankModels(totals, daily, "requests", range);
|
||||
expect(ranked.map((m) => m.delta)).toEqual([0, 0]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("formatMetric", () => {
|
||||
it("formats spend as currency and counts compactly", () => {
|
||||
expect(formatMetric(12.5, "spend")).toBe("$12.50");
|
||||
expect(formatMetric(1_500_000, "tokens")).toBe("1.5M");
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,132 @@
|
|||
export type Metric = "requests" | "spend" | "tokens";
|
||||
|
||||
export type ModelMetric = {
|
||||
model_group: string;
|
||||
model: string;
|
||||
provider: string;
|
||||
spend: number;
|
||||
prompt_tokens: number;
|
||||
completion_tokens: number;
|
||||
requests: number;
|
||||
successful_requests: number;
|
||||
failed_requests: number;
|
||||
};
|
||||
export type DailyMetric = ModelMetric & { date: string };
|
||||
export type ModelInsightsResponse = {
|
||||
start_date: string;
|
||||
end_date: string;
|
||||
daily: DailyMetric[];
|
||||
top_models: ModelMetric[];
|
||||
};
|
||||
export type TaskSummary = {
|
||||
task_type: string;
|
||||
label: string;
|
||||
category: string;
|
||||
value: number;
|
||||
share: number;
|
||||
leader: string;
|
||||
provider: string;
|
||||
};
|
||||
export type ModelInsightTasksResponse = { start_date: string; end_date: string; tasks: TaskSummary[] };
|
||||
|
||||
export type RankedModel = { model_group: string; provider: string; share: number; delta: number };
|
||||
const DAY_MS = 86_400_000;
|
||||
const WEEK_DAYS = 7;
|
||||
|
||||
export const metricValue = (row: ModelMetric, metric: Metric) => {
|
||||
if (metric === "requests") return row.requests;
|
||||
if (metric === "spend") return row.spend;
|
||||
return row.prompt_tokens + row.completion_tokens;
|
||||
};
|
||||
|
||||
const COMPACT_SPEND_FROM = 10_000;
|
||||
|
||||
export const formatMetric = (value: number, metric: Metric) => {
|
||||
if (metric === "spend") {
|
||||
const compact = value >= COMPACT_SPEND_FROM;
|
||||
const options: Intl.NumberFormatOptions = {
|
||||
style: "currency",
|
||||
currency: "USD",
|
||||
notation: compact ? "compact" : "standard",
|
||||
maximumFractionDigits: compact ? 1 : 2,
|
||||
};
|
||||
return new Intl.NumberFormat("en-US", options).format(value);
|
||||
}
|
||||
return new Intl.NumberFormat("en-US", { notation: "compact", maximumFractionDigits: 1 }).format(value);
|
||||
};
|
||||
|
||||
const toDay = (date: string) => Date.parse(`${date}T00:00:00Z`);
|
||||
const isoDay = (ms: number) => new Date(ms).toISOString().slice(0, 10);
|
||||
|
||||
export type DateRange = { start: string; end: string };
|
||||
|
||||
export const modelOrder = (rows: DailyMetric[], metric: Metric) => {
|
||||
const totals = new Map<string, number>();
|
||||
for (const row of rows) totals.set(row.model_group, (totals.get(row.model_group) ?? 0) + metricValue(row, metric));
|
||||
return [...totals.entries()].sort((a, b) => b[1] - a[1]).map(([model]) => model);
|
||||
};
|
||||
|
||||
export const buildWeeklySeries = (rows: DailyMetric[], models: string[], metric: Metric, range: DateRange) => {
|
||||
const weekMs = WEEK_DAYS * DAY_MS;
|
||||
const origin = toDay(range.start);
|
||||
const weekCount = Math.floor((toDay(range.end) - origin) / weekMs) + 1;
|
||||
const buckets = Array.from({ length: weekCount }, (_, week) => ({
|
||||
date: isoDay(origin + week * weekMs),
|
||||
...Object.fromEntries(models.map((model) => [model, 0])),
|
||||
})) as Record<string, number | string>[];
|
||||
for (const row of rows) {
|
||||
const bucket = buckets[Math.floor((toDay(row.date) - origin) / weekMs)];
|
||||
if (bucket) bucket[row.model_group] = Number(bucket[row.model_group] ?? 0) + metricValue(row, metric);
|
||||
}
|
||||
return buckets;
|
||||
};
|
||||
|
||||
const shareByModel = (rows: { model_group: string; provider: string }[], values: number[]) => {
|
||||
const totals = new Map<string, { provider: string; value: number }>();
|
||||
rows.forEach((row, index) => {
|
||||
const current = totals.get(row.model_group) ?? { provider: row.provider, value: 0 };
|
||||
totals.set(row.model_group, { provider: row.provider, value: current.value + values[index] });
|
||||
});
|
||||
const grand = [...totals.values()].reduce((sum, entry) => sum + entry.value, 0);
|
||||
return { totals, grand };
|
||||
};
|
||||
|
||||
const halfShares = (daily: DailyMetric[], metric: Metric, range: DateRange) => {
|
||||
const midpoint = isoDay(toDay(range.start) + Math.floor((toDay(range.end) - toDay(range.start)) / 2 + DAY_MS / 2));
|
||||
const share = (rows: DailyMetric[]) => {
|
||||
const { totals, grand } = shareByModel(
|
||||
rows,
|
||||
rows.map((row) => metricValue(row, metric)),
|
||||
);
|
||||
return {
|
||||
hasUsage: grand > 0,
|
||||
of: (model: string) => (grand === 0 ? 0 : ((totals.get(model)?.value ?? 0) / grand) * 100),
|
||||
};
|
||||
};
|
||||
return {
|
||||
earlier: share(daily.filter((row) => row.date < midpoint)),
|
||||
later: share(daily.filter((row) => row.date >= midpoint)),
|
||||
};
|
||||
};
|
||||
|
||||
export const rankModels = (
|
||||
rows: ModelMetric[],
|
||||
daily: DailyMetric[],
|
||||
metric: Metric,
|
||||
range: DateRange,
|
||||
): RankedModel[] => {
|
||||
const { totals, grand } = shareByModel(
|
||||
rows,
|
||||
rows.map((row) => metricValue(row, metric)),
|
||||
);
|
||||
const { earlier, later } = halfShares(daily, metric, range);
|
||||
const comparable = earlier.hasUsage && later.hasUsage;
|
||||
return [...totals.entries()]
|
||||
.sort((a, b) => b[1].value - a[1].value)
|
||||
.map(([model_group, entry]) => ({
|
||||
model_group,
|
||||
provider: entry.provider,
|
||||
share: grand === 0 ? 0 : (entry.value / grand) * 100,
|
||||
delta: comparable ? later.of(model_group) - earlier.of(model_group) : 0,
|
||||
}));
|
||||
};
|
||||
|
|
@ -0,0 +1,9 @@
|
|||
"use client";
|
||||
|
||||
import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
|
||||
import ModelInsightsView from "./_components/ModelInsightsView";
|
||||
|
||||
export default function ModelInsightsPage() {
|
||||
const { accessToken } = useAuthorized();
|
||||
return <ModelInsightsView accessToken={accessToken} />;
|
||||
}
|
||||
|
|
@ -204,6 +204,17 @@ const menuGroups: MenuGroup[] = [
|
|||
roles: [...all_admin_roles, ...internalUserRoles],
|
||||
label: "Usage",
|
||||
},
|
||||
{
|
||||
key: "model-insights",
|
||||
page: "model-insights",
|
||||
icon: <BarChart3 {...ICON} />,
|
||||
roles: all_admin_roles,
|
||||
label: (
|
||||
<span className="flex items-center gap-2">
|
||||
Model Leaderboard <BetaBadge />
|
||||
</span>
|
||||
),
|
||||
},
|
||||
{
|
||||
key: "cost-optimization",
|
||||
page: "cost-optimization",
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ export const pageDescriptions: Record<string, string> = {
|
|||
"vector-stores": "Manage vector databases for embeddings",
|
||||
new_usage: "View usage analytics and metrics",
|
||||
"cost-optimization": "Track and configure cost-saving features: prompt compression, caching, and auto routing",
|
||||
"model-insights": "Model Leaderboard: compare usage, spend, tokens, and task mix across this gateway",
|
||||
logs: "Access request and response logs",
|
||||
"guardrails-monitor": "Monitor guardrail performance and view logs",
|
||||
users: "Manage internal user accounts and permissions",
|
||||
|
|
|
|||
187
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
187
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -9189,6 +9189,40 @@ export interface paths {
|
|||
patch: operations["mistral_proxy_route_mistral__endpoint__patch"];
|
||||
trace?: never;
|
||||
};
|
||||
"/model-insights": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/** Get Model Insights */
|
||||
get: operations["get_model_insights_model_insights_get"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/model-insights/tasks": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
/** Get Model Insight Tasks */
|
||||
get: operations["get_model_insight_tasks_model_insights_tasks_get"];
|
||||
put?: never;
|
||||
post?: never;
|
||||
delete?: never;
|
||||
options?: never;
|
||||
head?: never;
|
||||
patch?: never;
|
||||
trace?: never;
|
||||
};
|
||||
"/model/block": {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
@ -35974,6 +36008,87 @@ export interface components {
|
|||
/** Id */
|
||||
id: string;
|
||||
};
|
||||
/** ModelInsightDailyMetric */
|
||||
ModelInsightDailyMetric: {
|
||||
/** Completion Tokens */
|
||||
completion_tokens: number;
|
||||
/** Date */
|
||||
date: string;
|
||||
/** Failed Requests */
|
||||
failed_requests: number;
|
||||
/** Model */
|
||||
model: string;
|
||||
/** Model Group */
|
||||
model_group: string;
|
||||
/** Prompt Tokens */
|
||||
prompt_tokens: number;
|
||||
/** Provider */
|
||||
provider: string;
|
||||
/** Requests */
|
||||
requests: number;
|
||||
/** Spend */
|
||||
spend: number;
|
||||
/** Successful Requests */
|
||||
successful_requests: number;
|
||||
};
|
||||
/** ModelInsightMetric */
|
||||
ModelInsightMetric: {
|
||||
/** Completion Tokens */
|
||||
completion_tokens: number;
|
||||
/** Failed Requests */
|
||||
failed_requests: number;
|
||||
/** Model */
|
||||
model: string;
|
||||
/** Model Group */
|
||||
model_group: string;
|
||||
/** Prompt Tokens */
|
||||
prompt_tokens: number;
|
||||
/** Provider */
|
||||
provider: string;
|
||||
/** Requests */
|
||||
requests: number;
|
||||
/** Spend */
|
||||
spend: number;
|
||||
/** Successful Requests */
|
||||
successful_requests: number;
|
||||
};
|
||||
/** ModelInsightTaskSummary */
|
||||
ModelInsightTaskSummary: {
|
||||
/** Category */
|
||||
category: string;
|
||||
/** Label */
|
||||
label: string;
|
||||
/** Leader */
|
||||
leader: string;
|
||||
/** Provider */
|
||||
provider: string;
|
||||
/** Share */
|
||||
share: number;
|
||||
/** Task Type */
|
||||
task_type: string;
|
||||
/** Value */
|
||||
value: number;
|
||||
};
|
||||
/** ModelInsightTasksResponse */
|
||||
ModelInsightTasksResponse: {
|
||||
/** End Date */
|
||||
end_date: string;
|
||||
/** Start Date */
|
||||
start_date: string;
|
||||
/** Tasks */
|
||||
tasks: components["schemas"]["ModelInsightTaskSummary"][];
|
||||
};
|
||||
/** ModelInsightsResponse */
|
||||
ModelInsightsResponse: {
|
||||
/** Daily */
|
||||
daily: components["schemas"]["ModelInsightDailyMetric"][];
|
||||
/** End Date */
|
||||
end_date: string;
|
||||
/** Start Date */
|
||||
start_date: string;
|
||||
/** Top Models */
|
||||
top_models: components["schemas"]["ModelInsightMetric"][];
|
||||
};
|
||||
/** ModelParams */
|
||||
ModelParams: {
|
||||
/** Litellm Params */
|
||||
|
|
@ -59332,6 +59447,78 @@ export interface operations {
|
|||
};
|
||||
};
|
||||
};
|
||||
get_model_insights_model_insights_get: {
|
||||
parameters: {
|
||||
query?: {
|
||||
/** @description YYYY-MM-DD, defaults to 365 days ago */
|
||||
start_date?: string | null;
|
||||
/** @description YYYY-MM-DD, defaults to today */
|
||||
end_date?: string | null;
|
||||
/** @description Metric the top models are ranked by */
|
||||
metric?: "requests" | "spend" | "tokens";
|
||||
};
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["ModelInsightsResponse"];
|
||||
};
|
||||
};
|
||||
/** @description Validation Error */
|
||||
422: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["HTTPValidationError"];
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
get_model_insight_tasks_model_insights_tasks_get: {
|
||||
parameters: {
|
||||
query?: {
|
||||
/** @description YYYY-MM-DD, defaults to 365 days ago */
|
||||
start_date?: string | null;
|
||||
/** @description YYYY-MM-DD, defaults to today */
|
||||
end_date?: string | null;
|
||||
/** @description Metric task shares are computed from */
|
||||
metric?: "requests" | "spend" | "tokens";
|
||||
};
|
||||
header?: never;
|
||||
path?: never;
|
||||
cookie?: never;
|
||||
};
|
||||
requestBody?: never;
|
||||
responses: {
|
||||
/** @description Successful Response */
|
||||
200: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["ModelInsightTasksResponse"];
|
||||
};
|
||||
};
|
||||
/** @description Validation Error */
|
||||
422: {
|
||||
headers: {
|
||||
[name: string]: unknown;
|
||||
};
|
||||
content: {
|
||||
"application/json": components["schemas"]["HTTPValidationError"];
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
block_model_model_block_post: {
|
||||
parameters: {
|
||||
query?: never;
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue