feat: add model leaderboard page (#43649)

* feat(proxy): add model leaderboard analytics

* feat: add model insights task and range constants

* feat: record task type from task tags in model usage rollup

* feat: serve 365 days of model insights by UTC date

* test: cover task tag resolution in model usage rollup

* test: update model insights range limit test to 365 days

* chore: regenerate dashboard api types for model insights

* feat: add model insights aggregation helpers

* test: cover model insights aggregation helpers

* feat: redesign model leaderboard with stacked bars, treemap and ranking

* test: update model leaderboard view test

* feat: mark model leaderboard as beta in sidebar

* chore: sync schema.prisma copies from root

* fix: only treat task: prefixed tags as model insight tasks

* feat: add metric type for model insights ranking

* fix: rank model insights by selected metric and scope detail queries to ranked deployments

* test: plain tags are not model insight tasks

* test: cover metric ranking, deployment scoping and rollup round trip

* fix: build model insights weeks and halves from the requested date range

* test: cover empty weeks and range-based change comparison

* fix: refetch by metric, show load errors and ignore stale responses

* test: cover metric refetch and error state

* feat: define model insight tasks in a JSON file

* feat: return task labels and categories from model insights

* feat: load model insight tasks from JSON

* refactor: validate rollup task tags against the JSON task list

* feat: serve the task list with model insights

* refactor: drop hardcoded task list from constants

* build: ship model insight tasks JSON in the wheel

* test: cover model insight task JSON

* refactor: take task labels and categories from the API

* test: pass task info to task tile builder

* refactor: color treemap by API-provided category

* test: include tasks in model leaderboard fixture

* fix: make daily model usage migration idempotent

* feat: bound the model insights task query size

* fix: compute task breakdown independent of the chart metric

* test: task breakdown is stable across chart metrics

* chore: regenerate lazy openapi snapshot for model insights

* chore: regenerate dashboard api types for model insights

* fix: keep previous ranking dimmed while a new metric loads

* test: cover stale metric state in model leaderboard

* refactor: drop task row cap constant

* fix: return the full task breakdown instead of a truncated one

* test: task query is not truncated

* feat: add task summary types for model insights

* feat: summarise tasks server-side on a separate model insights endpoint

* test: cover the model insights tasks endpoint

* chore: regenerate lazy openapi snapshot for model insights tasks

* chore: regenerate dashboard api types for model insights tasks

* refactor: drop client-side task aggregation

* test: remove client-side task aggregation tests

* feat: load task breakdown separately from the chart metric

* test: task breakdown is not refetched on chart metric change

---------

Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
This commit is contained in:
ishaan-berri 2026-09-28 19:40:04 -07:00 • committed by GitHub
parent 39d14bd855
commit 118ce3cc91
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
28 changed files with 2225 additions and 0 deletions

View file

@ -0,0 +1,19 @@
CREATE TABLE IF NOT EXISTS "LiteLLM_DailyModelUsage" (
"date" TEXT NOT NULL,
"model_group" TEXT NOT NULL,
"model" TEXT NOT NULL,
"custom_llm_provider" TEXT NOT NULL,
"task_type" TEXT NOT NULL,
"spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0,
"prompt_tokens" BIGINT NOT NULL DEFAULT 0,
"completion_tokens" BIGINT NOT NULL DEFAULT 0,
"request_count" BIGINT NOT NULL DEFAULT 0,
"successful_requests" BIGINT NOT NULL DEFAULT 0,
"failed_requests" BIGINT NOT NULL DEFAULT 0,
"created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP,
"updated_at" TIMESTAMP(3) NOT NULL,
CONSTRAINT "LiteLLM_DailyModelUsage_pkey" PRIMARY KEY ("date", "model_group", "model", "custom_llm_provider", "task_type")
);
CREATE INDEX IF NOT EXISTS "LiteLLM_DailyModelUsage_date_idx" ON "LiteLLM_DailyModelUsage"("date");
CREATE INDEX IF NOT EXISTS "LiteLLM_DailyModelUsage_model_group_idx" ON "LiteLLM_DailyModelUsage"("model_group");

View file

@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend {
@@id([date, tool_name])
}
model LiteLLM_DailyModelUsage {
date String
model_group String
model String
custom_llm_provider String
task_type String
spend Float @default(0.0)
prompt_tokens BigInt @default(0)
completion_tokens BigInt @default(0)
request_count BigInt @default(0)
successful_requests BigInt @default(0)
failed_requests BigInt @default(0)
created_at DateTime @default(now())
updated_at DateTime @updatedAt
@@id([date, model_group, model, custom_llm_provider, task_type])
@@index([date])
@@index([model_group])
}
// Gateway request counts recorded at the ASGI edge by
// BillableRequestMetricsMiddleware. This is the source of truth for SGR
// (successful gateway requests): it counts what the proxy actually answered,

View file

@ -1777,6 +1777,10 @@ SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS: Final = float(
os.getenv("SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS", "5")
)
TOOL_SPEND_TOP_TOOLS: Final = 100
MODEL_INSIGHTS_TOP_MODELS: Final = 10
MODEL_INSIGHTS_MAX_RANGE_DAYS: Final = 365
MODEL_INSIGHTS_DEFAULT_TASK: Final = "uncategorized"
MODEL_INSIGHTS_TASK_TAG_PREFIX: Final = "task:"
SPEND_LOG_PARTITION_INTERVAL: Final = os.getenv("SPEND_LOG_PARTITION_INTERVAL", "day")
SPEND_LOG_PARTITION_PRECREATE_AHEAD: Final = int(os.getenv("SPEND_LOG_PARTITION_PRECREATE_AHEAD", 7))
SPEND_LOG_WRITE_BATCH_MAX_BYTES: Final = max(1, int(os.getenv("SPEND_LOG_WRITE_BATCH_MAX_BYTES", 2_000_000)))

View file

@ -128,6 +128,11 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = (
module_path="litellm.proxy.management_endpoints.tool_management_endpoints",
path_prefixes=("/v1/tool", "/tool"),
),
LazyFeature(
name="model_insights",
module_path="litellm.proxy.management_endpoints.model_insights_endpoints",
path_prefixes=("/model-insights",),
),
LazyFeature(
name="search_tools",
module_path="litellm.proxy.search_endpoints.search_tool_management",

View file

@ -40700,6 +40700,463 @@
}
}
},
"model_insights": {
"components": {
"schemas": {
"HTTPValidationError": {
"properties": {
"detail": {
"items": {
"$ref": "#/components/schemas/ValidationError"
},
"title": "Detail",
"type": "array"
}
},
"title": "HTTPValidationError",
"type": "object"
},
"ModelInsightDailyMetric": {
"properties": {
"completion_tokens": {
"title": "Completion Tokens",
"type": "integer"
},
"date": {
"title": "Date",
"type": "string"
},
"failed_requests": {
"title": "Failed Requests",
"type": "integer"
},
"model": {
"title": "Model",
"type": "string"
},
"model_group": {
"title": "Model Group",
"type": "string"
},
"prompt_tokens": {
"title": "Prompt Tokens",
"type": "integer"
},
"provider": {
"title": "Provider",
"type": "string"
},
"requests": {
"title": "Requests",
"type": "integer"
},
"spend": {
"title": "Spend",
"type": "number"
},
"successful_requests": {
"title": "Successful Requests",
"type": "integer"
}
},
"required": [
"model_group",
"model",
"provider",
"spend",
"prompt_tokens",
"completion_tokens",
"requests",
"successful_requests",
"failed_requests",
"date"
],
"title": "ModelInsightDailyMetric",
"type": "object"
},
"ModelInsightMetric": {
"properties": {
"completion_tokens": {
"title": "Completion Tokens",
"type": "integer"
},
"failed_requests": {
"title": "Failed Requests",
"type": "integer"
},
"model": {
"title": "Model",
"type": "string"
},
"model_group": {
"title": "Model Group",
"type": "string"
},
"prompt_tokens": {
"title": "Prompt Tokens",
"type": "integer"
},
"provider": {
"title": "Provider",
"type": "string"
},
"requests": {
"title": "Requests",
"type": "integer"
},
"spend": {
"title": "Spend",
"type": "number"
},
"successful_requests": {
"title": "Successful Requests",
"type": "integer"
}
},
"required": [
"model_group",
"model",
"provider",
"spend",
"prompt_tokens",
"completion_tokens",
"requests",
"successful_requests",
"failed_requests"
],
"title": "ModelInsightMetric",
"type": "object"
},
"ModelInsightTaskSummary": {
"properties": {
"category": {
"title": "Category",
"type": "string"
},
"label": {
"title": "Label",
"type": "string"
},
"leader": {
"title": "Leader",
"type": "string"
},
"provider": {
"title": "Provider",
"type": "string"
},
"share": {
"title": "Share",
"type": "number"
},
"task_type": {
"title": "Task Type",
"type": "string"
},
"value": {
"title": "Value",
"type": "number"
}
},
"required": [
"task_type",
"label",
"category",
"value",
"share",
"leader",
"provider"
],
"title": "ModelInsightTaskSummary",
"type": "object"
},
"ModelInsightTasksResponse": {
"properties": {
"end_date": {
"title": "End Date",
"type": "string"
},
"start_date": {
"title": "Start Date",
"type": "string"
},
"tasks": {
"items": {
"$ref": "#/components/schemas/ModelInsightTaskSummary"
},
"title": "Tasks",
"type": "array"
}
},
"required": [
"start_date",
"end_date",
"tasks"
],
"title": "ModelInsightTasksResponse",
"type": "object"
},
"ModelInsightsResponse": {
"properties": {
"daily": {
"items": {
"$ref": "#/components/schemas/ModelInsightDailyMetric"
},
"title": "Daily",
"type": "array"
},
"end_date": {
"title": "End Date",
"type": "string"
},
"start_date": {
"title": "Start Date",
"type": "string"
},
"top_models": {
"items": {
"$ref": "#/components/schemas/ModelInsightMetric"
},
"title": "Top Models",
"type": "array"
}
},
"required": [
"start_date",
"end_date",
"daily",
"top_models"
],
"title": "ModelInsightsResponse",
"type": "object"
},
"ValidationError": {
"properties": {
"ctx": {
"title": "Context",
"type": "object"
},
"input": {
"title": "Input"
},
"loc": {
"items": {
"anyOf": [
{
"type": "string"
},
{
"type": "integer"
}
]
},
"title": "Location",
"type": "array"
},
"msg": {
"title": "Message",
"type": "string"
},
"type": {
"title": "Error Type",
"type": "string"
}
},
"required": [
"loc",
"msg",
"type"
],
"title": "ValidationError",
"type": "object"
}
}
},
"paths": {
"/model-insights": {
"get": {
"operationId": "get_model_insights_model_insights_get",
"parameters": [
{
"description": "YYYY-MM-DD, defaults to 365 days ago",
"in": "query",
"name": "start_date",
"required": false,
"schema": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"description": "YYYY-MM-DD, defaults to 365 days ago",
"title": "Start Date"
}
},
{
"description": "YYYY-MM-DD, defaults to today",
"in": "query",
"name": "end_date",
"required": false,
"schema": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"description": "YYYY-MM-DD, defaults to today",
"title": "End Date"
}
},
{
"description": "Metric the top models are ranked by",
"in": "query",
"name": "metric",
"required": false,
"schema": {
"default": "tokens",
"description": "Metric the top models are ranked by",
"enum": [
"requests",
"spend",
"tokens"
],
"title": "Metric",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/ModelInsightsResponse"
}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Get Model Insights",
"tags": [
"model_insights"
]
}
},
"/model-insights/tasks": {
"get": {
"operationId": "get_model_insight_tasks_model_insights_tasks_get",
"parameters": [
{
"description": "YYYY-MM-DD, defaults to 365 days ago",
"in": "query",
"name": "start_date",
"required": false,
"schema": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"description": "YYYY-MM-DD, defaults to 365 days ago",
"title": "Start Date"
}
},
{
"description": "YYYY-MM-DD, defaults to today",
"in": "query",
"name": "end_date",
"required": false,
"schema": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"description": "YYYY-MM-DD, defaults to today",
"title": "End Date"
}
},
{
"description": "Metric task shares are computed from",
"in": "query",
"name": "metric",
"required": false,
"schema": {
"default": "spend",
"description": "Metric task shares are computed from",
"enum": [
"requests",
"spend",
"tokens"
],
"title": "Metric",
"type": "string"
}
}
],
"responses": {
"200": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/ModelInsightTasksResponse"
}
}
},
"description": "Successful Response"
},
"422": {
"content": {
"application/json": {
"schema": {
"$ref": "#/components/schemas/HTTPValidationError"
}
}
},
"description": "Validation Error"
}
},
"security": [
{
"APIKeyHeader": []
}
],
"summary": "Get Model Insight Tasks",
"tags": [
"model_insights"
]
}
}
}
},
"policies": {
"components": {
"schemas": {

View file

@ -1148,6 +1148,16 @@ class DBSpendUpdateWriter:
traceback.format_exc(),
)
try:
from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage
await increment_daily_model_usage(prisma_client=prisma_client, payload=payload_copy)
except Exception:
verbose_proxy_logger.debug(
"_batch_database_updates: increment_daily_model_usage failed: %s",
traceback.format_exc(),
)
async def _update_key_db(
self,
response_cost: float | None,

View file

@ -0,0 +1,14 @@
import json
from functools import lru_cache
from pathlib import Path
from typing import Final
from litellm.types.model_insights import ModelInsightTask
_TASKS_FILE: Final = Path(__file__).resolve().parent.parent / "model_insights_tasks.json"
@lru_cache(maxsize=1)
def load_model_insight_tasks() -> dict[str, ModelInsightTask]:
raw: Final = json.loads(_TASKS_FILE.read_text())
return {name: ModelInsightTask(task_type=name, **entry) for name, entry in raw.items()}

View file

@ -0,0 +1,86 @@
from datetime import datetime
from typing import Final
from pydantic import TypeAdapter, ValidationError
from litellm.constants import (
INTERNAL_CALL_ORIGIN_METADATA_KEY,
MODEL_INSIGHTS_DEFAULT_TASK,
MODEL_INSIGHTS_TASK_TAG_PREFIX,
)
from litellm.proxy._types import SpendLogsPayload
from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks
from litellm.proxy.utils import PrismaClient
from litellm.repositories.table_repositories import DailyModelUsageRepository
_METADATA: Final = TypeAdapter(dict[str, object])
_TAGS: Final = TypeAdapter(list[object])
def model_usage_task_type(request_tags: str) -> str:
try:
tags: Final = _TAGS.validate_json(request_tags)
except ValidationError:
return MODEL_INSIGHTS_DEFAULT_TASK
for tag in tags:
if isinstance(tag, str) and tag.startswith(MODEL_INSIGHTS_TASK_TAG_PREFIX):
task = tag.removeprefix(MODEL_INSIGHTS_TASK_TAG_PREFIX)
if task in load_model_insight_tasks():
return task
return MODEL_INSIGHTS_DEFAULT_TASK
def _is_internal_call(metadata: str) -> bool:
try:
decoded: Final = _METADATA.validate_json(metadata)
except ValidationError:
return False
return bool(decoded.get(INTERNAL_CALL_ORIGIN_METADATA_KEY))
def _date_from_start_time(start_time: datetime | str) -> str | None:
if isinstance(start_time, datetime):
return start_time.date().isoformat()
return start_time[:10] if len(start_time) >= 10 else None
async def increment_daily_model_usage(prisma_client: PrismaClient, payload: SpendLogsPayload) -> None:
date: Final = _date_from_start_time(payload["startTime"])
if date is None or _is_internal_call(payload["metadata"]):
return
model: Final = payload["model"] or "unknown"
model_group: Final = payload["model_group"] or model
provider: Final = payload["custom_llm_provider"] or "unknown"
task_type: Final = model_usage_task_type(payload["request_tags"])
successful: Final = 1 if payload["status"] == "success" else 0
failed: Final = 1 - successful
key: Final = {
"date": date,
"model_group": model_group,
"model": model,
"custom_llm_provider": provider,
"task_type": task_type,
}
await DailyModelUsageRepository(prisma_client).table.upsert(
where={"date_model_group_model_custom_llm_provider_task_type": key},
data={
"create": {
**key,
"spend": payload["spend"],
"prompt_tokens": payload["prompt_tokens"],
"completion_tokens": payload["completion_tokens"],
"request_count": 1,
"successful_requests": successful,
"failed_requests": failed,
},
"update": {
"spend": {"increment": payload["spend"]},
"prompt_tokens": {"increment": payload["prompt_tokens"]},
"completion_tokens": {"increment": payload["completion_tokens"]},
"request_count": {"increment": 1},
"successful_requests": {"increment": successful},
"failed_requests": {"increment": failed},
},
},
)

View file

@ -0,0 +1,220 @@
from collections.abc import Mapping
from datetime import date, datetime, timedelta, timezone
from typing import Annotated, Final
from fastapi import APIRouter, Depends, HTTPException, Query
from pydantic import BaseModel, Field, TypeAdapter
from litellm.constants import MODEL_INSIGHTS_DEFAULT_TASK, MODEL_INSIGHTS_MAX_RANGE_DAYS, MODEL_INSIGHTS_TOP_MODELS
from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks
from litellm.repositories.table_repositories import DailyModelUsageRepository
from litellm.types.model_insights import (
ModelInsightDailyMetric,
ModelInsightMetric,
ModelInsightsMetric,
ModelInsightsResponse,
ModelInsightTask,
ModelInsightTasksResponse,
ModelInsightTaskSummary,
)
router: Final = APIRouter()
class _Sums(BaseModel):
spend: float = 0.0
prompt_tokens: int = 0
completion_tokens: int = 0
request_count: int = 0
successful_requests: int = 0
failed_requests: int = 0
class _GroupedModel(BaseModel):
model_group: str
model: str
custom_llm_provider: str
sums: _Sums = Field(alias="_sum")
class _GroupedDaily(_GroupedModel):
date: str
class _GroupedTask(_GroupedModel):
task_type: str
_MODEL_ROWS: Final = TypeAdapter(list[_GroupedModel])
_DAILY_ROWS: Final = TypeAdapter(list[_GroupedDaily])
_TASK_ROWS: Final = TypeAdapter(list[_GroupedTask])
_UNCATEGORIZED_TASK: Final = ModelInsightTask(
task_type=MODEL_INSIGHTS_DEFAULT_TASK, label="Uncategorized", category="General"
)
_SUM_FIELDS: Final = {
"spend": True,
"prompt_tokens": True,
"completion_tokens": True,
"request_count": True,
"successful_requests": True,
"failed_requests": True,
}
def _parse_date(value: str | None, fallback: date) -> date:
if value is None:
return fallback
try:
return date.fromisoformat(value)
except ValueError as exc:
raise HTTPException(status_code=400, detail="Dates must use YYYY-MM-DD") from exc
def _metric(row: _GroupedModel) -> ModelInsightMetric:
return ModelInsightMetric(
model_group=row.model_group,
model=row.model,
provider=row.custom_llm_provider,
spend=row.sums.spend,
prompt_tokens=row.sums.prompt_tokens,
completion_tokens=row.sums.completion_tokens,
requests=row.sums.request_count,
successful_requests=row.sums.successful_requests,
failed_requests=row.sums.failed_requests,
)
def _rank_value(row: _GroupedModel, metric: ModelInsightsMetric) -> float:
if metric == "requests":
return row.sums.request_count
if metric == "spend":
return row.sums.spend
return row.sums.prompt_tokens + row.sums.completion_tokens
def _top_model_rows(rows: list[_GroupedModel], metric: ModelInsightsMetric) -> list[_GroupedModel]:
return sorted(rows, key=lambda row: _rank_value(row, metric), reverse=True)[:MODEL_INSIGHTS_TOP_MODELS]
def _deployment_filter(rows: list[_GroupedModel]) -> list[dict[str, str]]:
return [
{"model_group": row.model_group, "model": row.model, "custom_llm_provider": row.custom_llm_provider}
for row in rows
]
def _daily_metric(row: _GroupedDaily) -> ModelInsightDailyMetric:
return ModelInsightDailyMetric(date=row.date, **_metric(row).model_dump())
def _summarize_tasks(rows: list[_GroupedTask], metric: ModelInsightsMetric) -> list[ModelInsightTaskSummary]:
catalog: Final = load_model_insight_tasks()
totals: Final[dict[str, float]] = {}
leaders: Final[dict[str, _GroupedTask]] = {}
for row in rows:
value = _rank_value(row, metric)
totals[row.task_type] = totals.get(row.task_type, 0.0) + value
leader = leaders.get(row.task_type)
if leader is None or value > _rank_value(leader, metric):
leaders[row.task_type] = row
grand: Final = sum(totals.values())
return [
ModelInsightTaskSummary(
**(catalog.get(task) or _UNCATEGORIZED_TASK).model_copy(update={"task_type": task}).model_dump(),
value=value,
share=value / grand * 100 if grand else 0.0,
leader=leaders[task].model_group,
provider=leaders[task].custom_llm_provider,
)
for task, value in sorted(totals.items(), key=lambda item: item[1], reverse=True)
]
def _resolve_window(
user_api_key_dict: UserAPIKeyAuth, start_date: str | None, end_date: str | None
) -> tuple[date, date, Mapping[str, object], DailyModelUsageRepository]:
from litellm.proxy.proxy_server import prisma_client
if user_api_key_dict.user_role not in (LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY):
raise HTTPException(status_code=403, detail="Only proxy admins can view deployment-wide model insights")
if prisma_client is None:
raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value)
end_day: Final = _parse_date(end_date, datetime.now(timezone.utc).date())
start_day: Final = _parse_date(start_date, end_day - timedelta(days=MODEL_INSIGHTS_MAX_RANGE_DAYS - 1))
if start_day > end_day or (end_day - start_day).days >= MODEL_INSIGHTS_MAX_RANGE_DAYS:
raise HTTPException(
status_code=400, detail=f"Date range must be between 1 and {MODEL_INSIGHTS_MAX_RANGE_DAYS} days"
)
date_window: Final[Mapping[str, object]] = {"date": {"gte": start_day.isoformat(), "lte": end_day.isoformat()}}
return start_day, end_day, date_window, DailyModelUsageRepository(prisma_client)
@router.get(
"/model-insights",
tags=["model insights"],
dependencies=[Depends(user_api_key_auth)],
response_model=ModelInsightsResponse,
)
async def get_model_insights(
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
start_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to 365 days ago")] = None,
end_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to today")] = None,
metric: Annotated[ModelInsightsMetric, Query(description="Metric the top models are ranked by")] = "tokens",
) -> ModelInsightsResponse:
start_day, end_day, date_window, repository = _resolve_window(user_api_key_dict, start_date, end_date)
table: Final = repository.table
grouped_model_rows: Final = _MODEL_ROWS.validate_python(
await table.group_by(
by=["model_group", "model", "custom_llm_provider"],
sum=_SUM_FIELDS,
where=date_window,
)
)
model_rows: Final = _top_model_rows(grouped_model_rows, metric)
selected_window: Final = {**date_window, "OR": _deployment_filter(model_rows)}
daily_rows: Final = _DAILY_ROWS.validate_python(
await table.group_by(
by=["date", "model_group", "model", "custom_llm_provider"],
sum=_SUM_FIELDS,
where=selected_window,
order={"date": "asc"},
)
if model_rows
else []
)
return ModelInsightsResponse(
start_date=start_day.isoformat(),
end_date=end_day.isoformat(),
top_models=[_metric(row) for row in model_rows],
daily=[_daily_metric(row) for row in daily_rows],
)
@router.get(
"/model-insights/tasks",
tags=["model insights"],
dependencies=[Depends(user_api_key_auth)],
response_model=ModelInsightTasksResponse,
)
async def get_model_insight_tasks(
user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)],
start_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to 365 days ago")] = None,
end_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to today")] = None,
metric: Annotated[ModelInsightsMetric, Query(description="Metric task shares are computed from")] = "spend",
) -> ModelInsightTasksResponse:
start_day, end_day, date_window, repository = _resolve_window(user_api_key_dict, start_date, end_date)
task_rows: Final = _TASK_ROWS.validate_python(
await repository.table.group_by(
by=["task_type", "model_group", "model", "custom_llm_provider"],
sum=_SUM_FIELDS,
where=date_window,
)
)
return ModelInsightTasksResponse(
start_date=start_day.isoformat(),
end_date=end_day.isoformat(),
tasks=_summarize_tasks(task_rows, metric),
)

View file

@ -0,0 +1,22 @@
{
"classification": {"label": "Classification", "category": "General"},
"content_writing": {"label": "Content Writing", "category": "General"},
"roleplay_fiction": {"label": "Roleplay & Fiction", "category": "General"},
"conversation": {"label": "Conversation", "category": "General"},
"research_reports": {"label": "Research & Reports", "category": "General"},
"qa_knowledge": {"label": "Q&A & Knowledge", "category": "General"},
"customer_support": {"label": "Customer Support", "category": "General"},
"summarization": {"label": "Summarization", "category": "General"},
"translation": {"label": "Translation", "category": "General"},
"workflow_execution": {"label": "Workflow Execution", "category": "Agent"},
"multi_step_planning": {"label": "Multi-step Planning", "category": "Agent"},
"tool_dispatch": {"label": "Tool Dispatch", "category": "Agent"},
"code_generation": {"label": "Code Generation", "category": "Code"},
"debugging": {"label": "Debugging", "category": "Code"},
"code_review": {"label": "Code Review", "category": "Code"},
"frontend_ui": {"label": "Frontend & UI", "category": "Code"},
"file_io": {"label": "File I/O", "category": "Code"},
"shell_execution": {"label": "Shell Execution", "category": "Code"},
"data_extraction": {"label": "Data Extraction", "category": "Data"},
"data_transformation": {"label": "Data Transformation", "category": "Data"}
}

View file

@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend {
@@id([date, tool_name])
}
model LiteLLM_DailyModelUsage {
date String
model_group String
model String
custom_llm_provider String
task_type String
spend Float @default(0.0)
prompt_tokens BigInt @default(0)
completion_tokens BigInt @default(0)
request_count BigInt @default(0)
successful_requests BigInt @default(0)
failed_requests BigInt @default(0)
created_at DateTime @default(now())
updated_at DateTime @updatedAt
@@id([date, model_group, model, custom_llm_provider, task_type])
@@index([date])
@@index([model_group])
}
// Gateway request counts recorded at the ASGI edge by
// BillableRequestMetricsMiddleware. This is the source of truth for SGR
// (successful gateway requests): it counts what the proxy actually answered,

View file

@ -30,6 +30,7 @@ from litellm.repositories.table_repositories import (
ConfigOverridesRepository,
DailyGuardrailMetricsRepository,
DailyGuardrailUsageUnitsRepository,
DailyModelUsageRepository,
DailyPolicyMetricsRepository,
DailyTagSpendRepository,
DailyToolSpendRepository,
@ -105,6 +106,7 @@ __all__ = [
"CredentialsRepository",
"DailyGuardrailMetricsRepository",
"DailyGuardrailUsageUnitsRepository",
"DailyModelUsageRepository",
"DailyPolicyMetricsRepository",
"DailyTagSpendRepository",
"DailyToolSpendRepository",

View file

@ -212,6 +212,10 @@ class DailyToolSpendRepository(PrismaTableRepository["prisma_models.LiteLLM_Dail
table_name = "litellm_dailytoolspend"
class DailyModelUsageRepository(PrismaTableRepository["prisma_models.LiteLLM_DailyModelUsage"]):
table_name = "litellm_dailymodelusage"
class SpendLogGuardrailIndexRepository(PrismaTableRepository["prisma_models.LiteLLM_SpendLogGuardrailIndex"]):
table_name = "litellm_spendlogguardrailindex"

View file

@ -0,0 +1,47 @@
from typing import Literal
from pydantic import BaseModel
ModelInsightsMetric = Literal["requests", "spend", "tokens"]
class ModelInsightMetric(BaseModel):
model_group: str
model: str
provider: str
spend: float
prompt_tokens: int
completion_tokens: int
requests: int
successful_requests: int
failed_requests: int
class ModelInsightDailyMetric(ModelInsightMetric):
date: str
class ModelInsightTask(BaseModel):
task_type: str
label: str
category: str
class ModelInsightTaskSummary(ModelInsightTask):
value: float
share: float
leader: str
provider: str
class ModelInsightsResponse(BaseModel):
start_date: str
end_date: str
daily: list[ModelInsightDailyMetric]
top_models: list[ModelInsightMetric]
class ModelInsightTasksResponse(BaseModel):
start_date: str
end_date: str
tasks: list[ModelInsightTaskSummary]

View file

@ -317,6 +317,7 @@ include = [
"litellm/proxy/_experimental/out/**",
"litellm/router_strategy/complexity_router/artifacts/*.json",
"litellm/router_strategy/complexity_router/fuse_presets.json",
"litellm/proxy/model_insights_tasks.json",
"litellm/proxy/client/cli/commands/codex_base_instructions.md",
]
exclude = [

View file

@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend {
@@id([date, tool_name])
}
model LiteLLM_DailyModelUsage {
date String
model_group String
model String
custom_llm_provider String
task_type String
spend Float @default(0.0)
prompt_tokens BigInt @default(0)
completion_tokens BigInt @default(0)
request_count BigInt @default(0)
successful_requests BigInt @default(0)
failed_requests BigInt @default(0)
created_at DateTime @default(now())
updated_at DateTime @updatedAt
@@id([date, model_group, model, custom_llm_provider, task_type])
@@index([date])
@@index([model_group])
}
// Gateway request counts recorded at the ASGI edge by
// BillableRequestMetricsMiddleware. This is the source of truth for SGR
// (successful gateway requests): it counts what the proxy actually answered,

View file

@ -0,0 +1,19 @@
from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks
from litellm.proxy.db.model_usage_rollup import model_usage_task_type
def test_every_task_has_a_label_and_a_category() -> None:
tasks = load_model_insight_tasks()
assert tasks
for name, task in tasks.items():
assert task.task_type == name
assert task.label
assert task.category in {"General", "Agent", "Code", "Data"}
def test_tasks_in_the_json_file_are_the_ones_the_rollup_accepts() -> None:
for name in load_model_insight_tasks():
assert model_usage_task_type(f'["task:{name}"]') == name
assert model_usage_task_type('["task:not_in_the_file"]') == "uncategorized"

View file

@ -0,0 +1,89 @@
from datetime import datetime, timezone
from unittest.mock import AsyncMock, MagicMock
import pytest
from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage, model_usage_task_type
def test_model_usage_task_type_reads_task_tag_or_defaults() -> None:
assert model_usage_task_type('["team-a", "task:classification"]') == "classification"
assert model_usage_task_type('["task:made-up"]') == "uncategorized"
assert model_usage_task_type('["debugging"]') == "uncategorized"
assert model_usage_task_type("[]") == "uncategorized"
assert model_usage_task_type("not json") == "uncategorized"
@pytest.mark.asyncio
async def test_increment_daily_model_usage_uses_atomic_prisma_upsert() -> None:
table = MagicMock()
table.upsert = AsyncMock()
prisma_client = MagicMock()
prisma_client.db.litellm_dailymodelusage = table
payload = {
"request_id": "request-1",
"call_type": "acompletion",
"api_key": "key",
"spend": 0.25,
"total_tokens": 30,
"prompt_tokens": 10,
"completion_tokens": 20,
"startTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
"endTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
"completionStartTime": None,
"model": "openai/gpt-5.4-mini",
"model_id": None,
"model_group": "fast-chat",
"mcp_namespaced_tool_name": None,
"agent_id": None,
"api_base": "",
"user": "user",
"metadata": "{}",
"cache_hit": "False",
"cache_key": "",
"request_tags": "[]",
"team_id": None,
"organization_id": None,
"end_user": None,
"requester_ip_address": None,
"custom_llm_provider": "openai",
"messages": None,
"response": None,
"proxy_server_request": None,
"session_id": None,
"request_duration_ms": 20,
"status": "success",
"litellm_call_id": None,
}
await increment_daily_model_usage(prisma_client, payload)
call = table.upsert.await_args.kwargs
assert call["data"]["create"]["request_count"] == 1
assert call["data"]["update"]["completion_tokens"] == {"increment": 20}
assert call["data"]["create"]["task_type"] == "uncategorized"
@pytest.mark.asyncio
async def test_increment_daily_model_usage_records_task_from_request_tags() -> None:
table = MagicMock()
table.upsert = AsyncMock()
prisma_client = MagicMock()
prisma_client.db.litellm_dailymodelusage = table
payload = {
"call_type": "acompletion",
"spend": 0.1,
"prompt_tokens": 1,
"completion_tokens": 2,
"startTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
"model": "gpt-5",
"model_group": "gpt-5",
"metadata": "{}",
"request_tags": '["task:debugging"]',
"custom_llm_provider": "openai",
"status": "success",
}
await increment_daily_model_usage(prisma_client, payload)
assert table.upsert.await_args.kwargs["data"]["create"]["task_type"] == "debugging"

View file

@ -0,0 +1,233 @@
from datetime import datetime, timezone
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from fastapi import FastAPI
from fastapi.testclient import TestClient
from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage
from litellm.proxy.management_endpoints.model_insights_endpoints import router
def _override_auth() -> UserAPIKeyAuth:
return UserAPIKeyAuth(api_key="sk-test", user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN)
def _grouped_row(*, prompt_tokens: str = "100", completion_tokens: str = "200", **dimensions: str) -> dict[str, object]:
return {
**dimensions,
"_sum": {
"spend": 1.25,
"prompt_tokens": prompt_tokens,
"completion_tokens": completion_tokens,
"request_count": "3",
"successful_requests": "3",
"failed_requests": "0",
},
}
def test_model_insights_reads_only_bounded_rollup() -> None:
model = _grouped_row(model_group="fast-chat", model="openai/gpt-5.4-mini", custom_llm_provider="openai")
prompt_heavy_model = _grouped_row(
prompt_tokens="500",
completion_tokens="10",
model_group="long-context",
model="anthropic/claude-sonnet-4-5",
custom_llm_provider="anthropic",
)
daily = _grouped_row(
date="2026-09-28",
model_group="fast-chat",
model="openai/gpt-5.4-mini",
custom_llm_provider="openai",
)
table = MagicMock()
table.group_by = AsyncMock(side_effect=[[model, prompt_heavy_model], [daily]])
prisma = MagicMock()
prisma.db.litellm_dailymodelusage = table
prisma.db.query_raw = AsyncMock()
prisma.db.litellm_spendlogs.find_many = AsyncMock()
app = FastAPI()
app.include_router(router)
app.dependency_overrides[user_api_key_auth] = _override_auth
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
response = TestClient(app).get("/model-insights?start_date=2026-09-09&end_date=2026-09-28")
assert response.status_code == 200
assert response.json()["top_models"][0]["model_group"] == "long-context"
assert "by_task" not in response.json()
assert table.group_by.await_count == 2
prisma.db.query_raw.assert_not_awaited()
prisma.db.litellm_spendlogs.find_many.assert_not_awaited()
def test_model_insights_rejects_ranges_over_365_days() -> None:
prisma = MagicMock()
app = FastAPI()
app.include_router(router)
app.dependency_overrides[user_api_key_auth] = _override_auth
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
response = TestClient(app).get("/model-insights?start_date=2025-09-01&end_date=2026-09-28")
assert response.status_code == 400
def _call(table: MagicMock, query: str, path: str = "/model-insights") -> object:
prisma = MagicMock()
prisma.db.litellm_dailymodelusage = table
app = FastAPI()
app.include_router(router)
app.dependency_overrides[user_api_key_auth] = _override_auth
with patch("litellm.proxy.proxy_server.prisma_client", prisma):
return TestClient(app).get(f"{path}?start_date=2026-09-01&end_date=2026-09-28&{query}")
def test_model_insights_ranks_top_models_by_selected_metric() -> None:
token_heavy = _grouped_row(
prompt_tokens="9000", completion_tokens="9000", model_group="big", model="m1", custom_llm_provider="openai"
)
request_heavy = _grouped_row(
prompt_tokens="1", completion_tokens="1", model_group="busy", model="m2", custom_llm_provider="openai"
)
request_heavy["_sum"]["request_count"] = "500"
table = MagicMock()
table.group_by = AsyncMock(side_effect=[[token_heavy, request_heavy], []])
by_requests = _call(table, "metric=requests").json()
by_tokens = _call(
MagicMock(group_by=AsyncMock(side_effect=[[token_heavy, request_heavy], []])), "metric=tokens"
).json()
assert by_requests["top_models"][0]["model_group"] == "busy"
assert by_tokens["top_models"][0]["model_group"] == "big"
def test_model_insights_scopes_daily_to_ranked_deployments() -> None:
ranked = _grouped_row(model_group="shared", model="m1", custom_llm_provider="openai")
table = MagicMock()
table.group_by = AsyncMock(side_effect=[[ranked], []])
_call(table, "metric=tokens")
daily_where = table.group_by.await_args_list[1].kwargs["where"]
assert daily_where["OR"] == [{"model_group": "shared", "model": "m1", "custom_llm_provider": "openai"}]
assert "model_group" not in daily_where
def _task_rows() -> list[dict[str, object]]:
def row(task: str, group: str, requests: str, spend: float) -> dict[str, object]:
base = _grouped_row(task_type=task, model_group=group, model=group, custom_llm_provider="openai")
base["_sum"].update({"request_count": requests, "spend": spend}) # type: ignore[union-attr]
return base
return [
row("debugging", "big", "1", 9.0),
row("debugging", "busy", "50", 1.0),
row("classification", "busy", "10", 1.0),
]
def test_model_insight_tasks_are_summarised_on_the_server() -> None:
table = MagicMock(group_by=AsyncMock(return_value=_task_rows()))
body = _call(table, "metric=spend", path="/model-insights/tasks").json()
assert [(t["task_type"], t["label"], t["category"], t["leader"]) for t in body["tasks"]] == [
("debugging", "Debugging", "Code", "big"),
("classification", "Classification", "General", "busy"),
]
assert [round(t["share"], 1) for t in body["tasks"]] == [90.9, 9.1]
assert "OR" not in table.group_by.await_args.kwargs["where"]
assert "take" not in table.group_by.await_args.kwargs
def test_model_insight_tasks_leader_follows_the_selected_metric() -> None:
by_spend = _call(MagicMock(group_by=AsyncMock(return_value=_task_rows())), "metric=spend", "/model-insights/tasks")
by_requests = _call(
MagicMock(group_by=AsyncMock(return_value=_task_rows())), "metric=requests", "/model-insights/tasks"
)
assert by_spend.json()["tasks"][0]["leader"] == "big"
assert by_requests.json()["tasks"][0]["leader"] == "busy"
def test_model_insight_tasks_unknown_task_shows_as_uncategorized() -> None:
row = _grouped_row(task_type="uncategorized", model_group="a", model="a", custom_llm_provider="openai")
body = _call(MagicMock(group_by=AsyncMock(return_value=[row])), "metric=spend", "/model-insights/tasks").json()
assert [(t["label"], t["category"]) for t in body["tasks"]] == [("Uncategorized", "General")]
def test_model_insight_tasks_require_an_admin() -> None:
app = FastAPI()
app.include_router(router)
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(
api_key="sk-test", user_id="u", user_role=LitellmUserRoles.INTERNAL_USER
)
with patch("litellm.proxy.proxy_server.prisma_client", MagicMock()):
assert TestClient(app).get("/model-insights/tasks").status_code == 403
def test_model_insights_rejects_unknown_metric() -> None:
assert _call(MagicMock(group_by=AsyncMock()), "metric=bogus").status_code == 422
class _InMemoryUsageTable:
def __init__(self) -> None:
self.rows: dict[tuple[str, ...], dict[str, float]] = {}
async def upsert(self, where: dict, data: dict) -> None:
key_fields = where["date_model_group_model_custom_llm_provider_task_type"]
key = tuple(key_fields.values())
if key not in self.rows:
self.rows[key] = {**key_fields, **{k: v for k, v in data["create"].items() if k not in key_fields}}
return
for field, change in data["update"].items():
self.rows[key][field] += change["increment"]
async def group_by(self, by: list[str], sum: dict, where: dict, **_: object) -> list[dict]:
grouped: dict[tuple, dict] = {}
for row in self.rows.values():
if not where["date"]["gte"] <= row["date"] <= where["date"]["lte"]:
continue
if where.get("OR") and not any(all(row[k] == v for k, v in option.items()) for option in where["OR"]):
continue
bucket = grouped.setdefault(tuple(row[k] for k in by), {**{k: row[k] for k in by}, "_sum": {}})
for field in sum:
bucket["_sum"][field] = bucket["_sum"].get(field, 0) + row[field]
return list(grouped.values())
@pytest.mark.asyncio
async def test_model_insights_reads_back_what_the_rollup_wrote() -> None:
table = _InMemoryUsageTable()
prisma = MagicMock()
prisma.db.litellm_dailymodelusage = table
payload = {
"call_type": "acompletion",
"spend": 0.5,
"prompt_tokens": 10,
"completion_tokens": 20,
"startTime": datetime(2026, 9, 28, tzinfo=timezone.utc),
"model": "gpt-5",
"model_group": "gpt-5",
"metadata": "{}",
"request_tags": '["task:debugging"]',
"custom_llm_provider": "openai",
"status": "success",
}
await increment_daily_model_usage(prisma, payload)
await increment_daily_model_usage(prisma, {**payload, "request_tags": "[]"})
body = _call(table, "metric=requests").json()
assert [(m["model_group"], m["requests"], m["prompt_tokens"]) for m in body["top_models"]] == [("gpt-5", 2, 20)]
tasks = _call(table, "metric=requests", path="/model-insights/tasks").json()["tasks"]
assert sorted((t["task_type"], t["value"]) for t in tasks) == [("debugging", 1), ("uncategorized", 1)]
assert [(d["date"], d["requests"]) for d in body["daily"]] == [("2026-09-28", 2)]

View file

@ -35,6 +35,7 @@ const LEGACY_PAGE_ROUTES: ReadonlyMap<string, string> = new Map(
new_usage: "usage",
usage: "old-usage",
"cost-optimization": "cost-optimization",
"model-insights": "model-insights",
agents: "agents",
"router-settings": "router-settings",
users: "users",

View file

@ -0,0 +1,148 @@
import { render, screen, waitFor } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import type React from "react";
import { beforeEach, describe, expect, it, vi } from "vitest";
import ModelInsightsView from "./ModelInsightsView";
import { apiClient } from "@/components/networking";
vi.mock("@/components/networking", () => ({ apiClient: { get: vi.fn() } }));
vi.mock("@/components/ui/chart", () => ({
ChartContainer: ({ children }: { children: React.ReactNode }) => <div>{children}</div>,
ChartTooltip: () => null,
ChartTooltipContent: () => null,
}));
vi.mock("recharts", () => ({
Bar: () => null,
BarChart: ({ children }: { children: React.ReactNode }) => <div>{children}</div>,
CartesianGrid: () => null,
Treemap: () => null,
XAxis: () => null,
YAxis: () => null,
}));
const metrics = {
model_group: "fast-chat",
model: "openai/gpt-5.4-mini",
provider: "openai",
spend: 2.5,
prompt_tokens: 1000,
completion_tokens: 2000,
requests: 12,
successful_requests: 12,
failed_requests: 0,
};
const response = {
start_date: "2025-09-29",
end_date: "2026-09-28",
top_models: [metrics],
daily: [{ ...metrics, date: "2026-09-28" }],
};
const taskResponse = {
start_date: "2025-09-29",
end_date: "2026-09-28",
tasks: [
{
task_type: "code_generation",
label: "Code Generation",
category: "Code",
value: 2.5,
share: 100,
leader: "fast-chat",
provider: "openai",
},
],
};
const mockApi = (tasks: unknown = taskResponse) =>
vi
.mocked(apiClient.get)
.mockImplementation((path: string) =>
path === "/model-insights/tasks" ? (tasks as Promise<unknown>) : Promise.resolve(response),
);
describe("ModelInsightsView", () => {
beforeEach(() => {
vi.mocked(apiClient.get).mockReset();
mockApi(Promise.resolve(taskResponse));
});
it("shows the ranking with share and the task legend from the API response", async () => {
render(<ModelInsightsView accessToken="token" />);
expect(await screen.findByText("fast-chat")).toBeInTheDocument();
expect(screen.getByText("by openai")).toBeInTheDocument();
expect(await screen.findByText("Code")).toBeInTheDocument();
expect(screen.getAllByText("100.0%")).toHaveLength(2);
expect(screen.getByRole("tab", { name: "tokens" })).toHaveAttribute("aria-selected", "true");
expect(screen.getByRole("tab", { name: "log" })).toBeInTheDocument();
expect(apiClient.get).toHaveBeenCalledWith("/model-insights", {
accessToken: "token",
query: { metric: "tokens" },
});
});
it("refetches with the selected metric so top models are ranked by it", async () => {
render(<ModelInsightsView accessToken="token" />);
await screen.findByText("fast-chat");
await userEvent.click(screen.getByRole("tab", { name: "requests" }));
await waitFor(() =>
expect(apiClient.get).toHaveBeenCalledWith("/model-insights", {
accessToken: "token",
query: { metric: "requests" },
}),
);
});
it("does not refetch the task breakdown when the chart metric changes", async () => {
render(<ModelInsightsView accessToken="token" />);
await screen.findByText("Code");
const taskCalls = () =>
vi.mocked(apiClient.get).mock.calls.filter(([path]) => path === "/model-insights/tasks").length;
const before = taskCalls();
await userEvent.click(screen.getByRole("tab", { name: "requests" }));
await waitFor(() =>
expect(apiClient.get).toHaveBeenCalledWith("/model-insights", {
accessToken: "token",
query: { metric: "requests" },
}),
);
expect(taskCalls()).toBe(before);
});
it("shows the API error instead of loading forever", async () => {
vi.mocked(apiClient.get).mockRejectedValue(new Error("Only proxy admins can view deployment-wide model insights"));
render(<ModelInsightsView accessToken="token" />);
expect(await screen.findByText("Could not load model insights")).toBeInTheDocument();
expect(screen.getByText("Only proxy admins can view deployment-wide model insights")).toBeInTheDocument();
});
it("keeps the previous ranking, dimmed, until the new metric's data arrives", async () => {
render(<ModelInsightsView accessToken="token" />);
await screen.findByText("fast-chat");
let resolve: (value: typeof response) => void = () => {};
vi.mocked(apiClient.get).mockImplementation((path: string) =>
path === "/model-insights/tasks"
? Promise.resolve(taskResponse)
: new Promise((done) => (resolve = done as typeof resolve)),
);
await userEvent.click(screen.getByRole("tab", { name: "spend" }));
expect(
screen.getByText("Share of tokens, with the change between the first and second half of the period"),
).toBeInTheDocument();
resolve(response);
expect(
await screen.findByText("Share of spend, with the change between the first and second half of the period"),
).toBeInTheDocument();
});
});

View file

@ -0,0 +1,352 @@
"use client";
import React from "react";
import { Bar, BarChart, CartesianGrid, Treemap, XAxis, YAxis } from "recharts";
import { ArrowDownRight, ArrowUpRight, BarChart3, Layers, Minus } from "lucide-react";
import { apiClient } from "@/components/networking";
import { extractErrorMessage } from "@/utils/errorUtils";
import { ProviderLogo } from "@/components/molecules/models/ProviderLogo";
import { PageHeader } from "@/components/shared/PageHeader";
import { Alert, AlertDescription, AlertTitle } from "@/components/ui/alert";
import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
import { ChartConfig, ChartContainer, ChartTooltip, ChartTooltipContent } from "@/components/ui/chart";
import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select";
import { Skeleton } from "@/components/ui/skeleton";
import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs";
import {
buildWeeklySeries,
formatMetric,
Metric,
ModelInsightsResponse,
ModelInsightTasksResponse,
TaskSummary,
modelOrder,
rankModels,
RankedModel,
} from "./modelInsightsData";
const PALETTE = [
"#ec4899",
"#a855f7",
"#f59e0b",
"#3b82f6",
"#10b981",
"#ef4444",
"#14b8a6",
"#84cc16",
"#6366f1",
"#f97316",
];
const FALLBACK_COLOR = "#64748b";
const CATEGORY_COLORS: Record<string, string> = {
General: "#ee8650",
Agent: "#7666e4",
Code: "#5fb074",
Data: "#3b82f6",
};
const SCALES = ["linear", "log"] as const;
const METRIC_LABELS: Record<Metric, string> = { requests: "requests", spend: "spend", tokens: "tokens" };
const RANKING_ROWS = 5;
type Scale = (typeof SCALES)[number];
const formatDelta = (value: number) => `${value > 0 ? "+" : ""}${value.toFixed(1)}`;
const DeltaBadge = ({ value }: { value: number }) => {
if (Math.abs(value) < 0.05) {
return (
<span className="flex items-center justify-end gap-1 text-xs text-muted-foreground">
<Minus className="size-3" /> 0.0
</span>
);
}
const up = value > 0;
const Icon = up ? ArrowUpRight : ArrowDownRight;
return (
<span className={`flex items-center justify-end gap-1 text-xs ${up ? "text-emerald-600" : "text-red-600"}`}>
<Icon className="size-3" /> {formatDelta(value)}
</span>
);
};
const RankingRow = ({ model, rank }: { model: RankedModel; rank: number }) => (
<li className="grid grid-cols-[1.5rem_2.5rem_1fr_auto] items-center gap-3 py-2">
<span className="text-sm tabular-nums text-muted-foreground">{rank}</span>
<ProviderLogo provider={model.provider} className="size-9 rounded-md border p-1" />
<div className="min-w-0">
<p className="truncate font-medium">{model.model_group}</p>
<p className="truncate text-sm text-muted-foreground">by {model.provider}</p>
</div>
<div className="text-right">
<p className="font-medium tabular-nums">{model.share.toFixed(1)}%</p>
<DeltaBadge value={model.delta} />
</div>
</li>
);
type TileProps = TaskSummary & { x: number; y: number; width: number; height: number; index: number };
const TaskTileContent = ({ x, y, width, height, category, label, leader }: TileProps) => {
if (width <= 0 || height <= 0) return null;
const color = CATEGORY_COLORS[category] ?? FALLBACK_COLOR;
const fits = width > 90 && height > 44;
return (
<g>
<rect x={x} y={y} width={width} height={height} fill={color} stroke="#fff" strokeWidth={2} />
{fits && (
<>
<text x={x + 12} y={y + 26} fill="#fff" fontSize={16} fontWeight={500}>
{label}
</text>
<text x={x + 12} y={y + 46} fill="#ffffffcc" fontSize={12}>
{leader}
</text>
</>
)}
</g>
);
};
export default function ModelInsightsView({ accessToken }: { accessToken: string | null }) {
const [loaded, setLoaded] = React.useState<{ metric: Metric; response: ModelInsightsResponse } | null>(null);
const [metric, setMetric] = React.useState<Metric>("tokens");
const [scale, setScale] = React.useState<Scale>("linear");
const [taskMetric, setTaskMetric] = React.useState<Metric>("spend");
const [taskData, setTaskData] = React.useState<ModelInsightTasksResponse | null>(null);
const [taskError, setTaskError] = React.useState<string | null>(null);
const [error, setError] = React.useState<string | null>(null);
React.useEffect(() => {
if (!accessToken) return;
let cancelled = false;
apiClient
.get<ModelInsightsResponse>("/model-insights", { accessToken, query: { metric } })
.then((response) => {
if (cancelled) return;
setError(null);
setLoaded({ metric, response });
})
.catch((err: unknown) => {
if (!cancelled) setError(extractErrorMessage(err));
});
return () => {
cancelled = true;
};
}, [accessToken, metric]);
React.useEffect(() => {
if (!accessToken) return;
let cancelled = false;
apiClient
.get<ModelInsightTasksResponse>("/model-insights/tasks", { accessToken, query: { metric: taskMetric } })
.then((response) => {
if (cancelled) return;
setTaskError(null);
setTaskData(response);
})
.catch((err: unknown) => {
if (!cancelled) setTaskError(extractErrorMessage(err));
});
return () => {
cancelled = true;
};
}, [accessToken, taskMetric]);
const data = loaded?.response ?? null;
const shown = loaded?.metric ?? metric;
const isStale = loaded !== null && loaded.metric !== metric;
const range = React.useMemo(() => ({ start: data?.start_date ?? "", end: data?.end_date ?? "" }), [data]);
const models = React.useMemo(() => (data ? modelOrder(data.daily, shown) : []), [data, shown]);
const series = React.useMemo(
() => (data ? buildWeeklySeries(data.daily, models, shown, range) : []),
[data, models, shown, range],
);
const ranking = React.useMemo(
() => (data ? rankModels(data.top_models, data.daily, shown, range) : []),
[data, shown, range],
);
const tiles = React.useMemo(() => taskData?.tasks ?? [], [taskData]);
const categoryShares = React.useMemo(
() =>
[...new Set(tiles.map((tile) => tile.category))].map((category) => ({
category,
share: tiles.filter((tile) => tile.category === category).reduce((sum, tile) => sum + tile.share, 0),
})),
[tiles],
);
if (error) {
return (
<div className="p-8">
<Alert variant="destructive">
<AlertTitle>Could not load model insights</AlertTitle>
<AlertDescription>{error}</AlertDescription>
</Alert>
</div>
);
}
if (!data) {
return (
<div className="space-y-6 p-8">
<Skeleton className="h-16 w-96" />
<Skeleton className="h-96 w-full" />
</div>
);
}
const chartConfig = Object.fromEntries(
models.map((model, index) => [model, { label: model, color: PALETTE[index % PALETTE.length] }]),
) satisfies ChartConfig;
return (
<main className="w-full space-y-6 p-8">
<PageHeader
icon={<BarChart3 />}
title="Model Leaderboard"
subtitle={`See which models your gateway used from ${data.start_date} through ${data.end_date}`}
/>
<Card aria-busy={isStale} className={isStale ? "opacity-60 transition-opacity" : "transition-opacity"}>
<CardHeader className="flex-row items-start justify-between space-y-0">
<div>
<CardTitle>Top models</CardTitle>
<CardDescription>Weekly {METRIC_LABELS[shown]} across your gateway</CardDescription>
</div>
<div className="flex items-center gap-3">
<Tabs value={metric} onValueChange={(value) => setMetric(value as Metric)}>
<TabsList>
{(["requests", "spend", "tokens"] as const).map((value) => (
<TabsTrigger key={value} value={value} className="capitalize">
{value}
</TabsTrigger>
))}
</TabsList>
</Tabs>
<Tabs value={scale} onValueChange={(value) => setScale(value as Scale)}>
<TabsList>
{SCALES.map((value) => (
<TabsTrigger key={value} value={value} className="capitalize">
{value}
</TabsTrigger>
))}
</TabsList>
</Tabs>
</div>
</CardHeader>
<CardContent>
<ChartContainer config={chartConfig} className="h-[380px] w-full aspect-auto">
<BarChart data={series} margin={{ left: 8, right: 8 }} barCategoryGap={2}>
<CartesianGrid vertical={false} />
<XAxis dataKey="date" tickLine={false} axisLine={false} minTickGap={48} />
<YAxis
scale={scale}
domain={scale === "log" ? [1, "auto"] : [0, "auto"]}
allowDataOverflow
tickLine={false}
axisLine={false}
tickFormatter={(value) => formatMetric(Number(value), shown)}
/>
<ChartTooltip content={<ChartTooltipContent />} />
{models.map((model, index) => (
<Bar
key={model}
dataKey={model}
stackId="usage"
fill={PALETTE[index % PALETTE.length]}
isAnimationActive={false}
/>
))}
</BarChart>
</ChartContainer>
</CardContent>
</Card>
<Card aria-busy={isStale} className={isStale ? "opacity-60 transition-opacity" : "transition-opacity"}>
<CardHeader>
<CardTitle>Leaderboard</CardTitle>
<CardDescription>
Share of {METRIC_LABELS[shown]}, with the change between the first and second half of the period
</CardDescription>
</CardHeader>
<CardContent className="grid gap-x-12 md:grid-cols-2">
<ol className="divide-y">
{ranking.slice(0, RANKING_ROWS).map((model, index) => (
<RankingRow key={model.model_group} model={model} rank={index + 1} />
))}
</ol>
<ol className="divide-y">
{ranking.slice(RANKING_ROWS).map((model, index) => (
<RankingRow key={model.model_group} model={model} rank={RANKING_ROWS + index + 1} />
))}
</ol>
</CardContent>
</Card>
<Card>
<CardHeader className="flex-row items-start justify-between space-y-0">
<div>
<CardTitle className="flex items-center gap-2">
<Layers className="size-5" /> Top models by task
</CardTitle>
<CardDescription>
Each task&apos;s share of {METRIC_LABELS[taskMetric]}, labelled with its leading model
</CardDescription>
</div>
<Select value={taskMetric} onValueChange={(value) => setTaskMetric(value as Metric)}>
<SelectTrigger className="w-44" aria-label="Task metric">
<SelectValue />
</SelectTrigger>
<SelectContent>
<SelectItem value="spend">Share of spend</SelectItem>
<SelectItem value="requests">Share of requests</SelectItem>
<SelectItem value="tokens">Share of tokens</SelectItem>
</SelectContent>
</Select>
</CardHeader>
<CardContent className="space-y-4">
{taskError && (
<Alert variant="destructive">
<AlertTitle>Could not load tasks</AlertTitle>
<AlertDescription>{taskError}</AlertDescription>
</Alert>
)}
<ChartContainer config={{}} className="h-[360px] w-full aspect-auto">
<Treemap
data={tiles.map((tile) => ({ ...tile, name: tile.task_type }))}
dataKey="value"
isAnimationActive={false}
content={<TaskTileContent {...({} as TileProps)} />}
/>
</ChartContainer>
<ul className="flex flex-wrap gap-x-6 gap-y-2">
{categoryShares.map(({ category, share }) => (
<li key={category} className="flex items-center gap-2 text-sm">
<span
className="size-3 rounded-full"
style={{ backgroundColor: CATEGORY_COLORS[category] ?? FALLBACK_COLOR }}
/>
<span className="text-muted-foreground">{category}</span>
<span className="font-medium tabular-nums">{share.toFixed(1)}%</span>
</li>
))}
</ul>
</CardContent>
</Card>
<Card>
<CardHeader>
<CardTitle>Cost per session</CardTitle>
<CardDescription>Session cost is not estimated from request counts</CardDescription>
</CardHeader>
<CardContent>
<p className="text-sm text-muted-foreground">
Add a stable session_id to requests to unlock accurate session-level model comparisons in a future bounded
session rollup
</p>
</CardContent>
</Card>
</main>
);
}

View file

@ -0,0 +1,92 @@
import { describe, expect, it } from "vitest";
import { buildWeeklySeries, DailyMetric, formatMetric, modelOrder, rankModels } from "./modelInsightsData";
const row = (over: Partial<DailyMetric>): DailyMetric => ({
model_group: "a",
model: "a",
provider: "openai",
date: "2026-01-01",
spend: 0,
prompt_tokens: 0,
completion_tokens: 0,
requests: 0,
successful_requests: 0,
failed_requests: 0,
...over,
});
describe("buildWeeklySeries", () => {
const range = { start: "2026-01-01", end: "2026-01-15" };
it("sums days into 7-day buckets per model", () => {
const rows = [
row({ date: "2026-01-01", requests: 1 }),
row({ date: "2026-01-07", requests: 2 }),
row({ date: "2026-01-08", requests: 4 }),
row({ date: "2026-01-02", model_group: "b", requests: 8 }),
];
expect(buildWeeklySeries(rows, ["a", "b"], "requests", range)).toEqual([
{ date: "2026-01-01", a: 3, b: 8 },
{ date: "2026-01-08", a: 4, b: 0 },
{ date: "2026-01-15", a: 0, b: 0 },
]);
});
it("keeps weeks with no usage as zero instead of dropping them", () => {
const rows = [row({ date: "2026-01-01", requests: 1 }), row({ date: "2026-01-15", requests: 2 })];
expect(buildWeeklySeries(rows, ["a"], "requests", range).map((week) => [week.date, week.a])).toEqual([
["2026-01-01", 1],
["2026-01-08", 0],
["2026-01-15", 2],
]);
});
});
describe("modelOrder", () => {
it("orders models by the selected metric, largest first", () => {
const rows = [row({ model_group: "a", spend: 1, requests: 9 }), row({ model_group: "b", spend: 5, requests: 1 })];
expect(modelOrder(rows, "spend")).toEqual(["b", "a"]);
expect(modelOrder(rows, "requests")).toEqual(["a", "b"]);
});
});
describe("rankModels", () => {
const range = { start: "2026-01-01", end: "2026-01-10" };
const totals = [row({ model_group: "a", requests: 40 }), row({ model_group: "b", requests: 40 })];
it("computes share and the change in share between the first and second half of the range", () => {
const daily = [
row({ date: "2026-01-01", model_group: "a", requests: 30 }),
row({ date: "2026-01-01", model_group: "b", requests: 10 }),
row({ date: "2026-01-10", model_group: "a", requests: 10 }),
row({ date: "2026-01-10", model_group: "b", requests: 30 }),
];
const ranked = rankModels(totals, daily, "requests", range);
expect(ranked.find((m) => m.model_group === "a")).toMatchObject({ share: 50, delta: -50 });
expect(ranked.find((m) => m.model_group === "b")).toMatchObject({ share: 50, delta: 50 });
});
it("splits at the middle of the range, not the middle of the days that had usage", () => {
const daily = [
row({ date: "2026-01-01", model_group: "a", requests: 10 }),
row({ date: "2026-01-02", model_group: "b", requests: 10 }),
row({ date: "2026-01-03", model_group: "b", requests: 10 }),
];
const ranked = rankModels(totals, daily, "requests", range);
expect(ranked.find((m) => m.model_group === "a")?.delta).toBe(0);
});
it("shows no change when one half of the range has no usage to compare against", () => {
const daily = [row({ date: "2026-01-10", model_group: "a", requests: 10 })];
const ranked = rankModels(totals, daily, "requests", range);
expect(ranked.map((m) => m.delta)).toEqual([0, 0]);
});
});
describe("formatMetric", () => {
it("formats spend as currency and counts compactly", () => {
expect(formatMetric(12.5, "spend")).toBe("$12.50");
expect(formatMetric(1_500_000, "tokens")).toBe("1.5M");
});
});

View file

@ -0,0 +1,132 @@
export type Metric = "requests" | "spend" | "tokens";
export type ModelMetric = {
model_group: string;
model: string;
provider: string;
spend: number;
prompt_tokens: number;
completion_tokens: number;
requests: number;
successful_requests: number;
failed_requests: number;
};
export type DailyMetric = ModelMetric & { date: string };
export type ModelInsightsResponse = {
start_date: string;
end_date: string;
daily: DailyMetric[];
top_models: ModelMetric[];
};
export type TaskSummary = {
task_type: string;
label: string;
category: string;
value: number;
share: number;
leader: string;
provider: string;
};
export type ModelInsightTasksResponse = { start_date: string; end_date: string; tasks: TaskSummary[] };
export type RankedModel = { model_group: string; provider: string; share: number; delta: number };
const DAY_MS = 86_400_000;
const WEEK_DAYS = 7;
export const metricValue = (row: ModelMetric, metric: Metric) => {
if (metric === "requests") return row.requests;
if (metric === "spend") return row.spend;
return row.prompt_tokens + row.completion_tokens;
};
const COMPACT_SPEND_FROM = 10_000;
export const formatMetric = (value: number, metric: Metric) => {
if (metric === "spend") {
const compact = value >= COMPACT_SPEND_FROM;
const options: Intl.NumberFormatOptions = {
style: "currency",
currency: "USD",
notation: compact ? "compact" : "standard",
maximumFractionDigits: compact ? 1 : 2,
};
return new Intl.NumberFormat("en-US", options).format(value);
}
return new Intl.NumberFormat("en-US", { notation: "compact", maximumFractionDigits: 1 }).format(value);
};
const toDay = (date: string) => Date.parse(`${date}T00:00:00Z`);
const isoDay = (ms: number) => new Date(ms).toISOString().slice(0, 10);
export type DateRange = { start: string; end: string };
export const modelOrder = (rows: DailyMetric[], metric: Metric) => {
const totals = new Map<string, number>();
for (const row of rows) totals.set(row.model_group, (totals.get(row.model_group) ?? 0) + metricValue(row, metric));
return [...totals.entries()].sort((a, b) => b[1] - a[1]).map(([model]) => model);
};
export const buildWeeklySeries = (rows: DailyMetric[], models: string[], metric: Metric, range: DateRange) => {
const weekMs = WEEK_DAYS * DAY_MS;
const origin = toDay(range.start);
const weekCount = Math.floor((toDay(range.end) - origin) / weekMs) + 1;
const buckets = Array.from({ length: weekCount }, (_, week) => ({
date: isoDay(origin + week * weekMs),
...Object.fromEntries(models.map((model) => [model, 0])),
})) as Record<string, number | string>[];
for (const row of rows) {
const bucket = buckets[Math.floor((toDay(row.date) - origin) / weekMs)];
if (bucket) bucket[row.model_group] = Number(bucket[row.model_group] ?? 0) + metricValue(row, metric);
}
return buckets;
};
const shareByModel = (rows: { model_group: string; provider: string }[], values: number[]) => {
const totals = new Map<string, { provider: string; value: number }>();
rows.forEach((row, index) => {
const current = totals.get(row.model_group) ?? { provider: row.provider, value: 0 };
totals.set(row.model_group, { provider: row.provider, value: current.value + values[index] });
});
const grand = [...totals.values()].reduce((sum, entry) => sum + entry.value, 0);
return { totals, grand };
};
const halfShares = (daily: DailyMetric[], metric: Metric, range: DateRange) => {
const midpoint = isoDay(toDay(range.start) + Math.floor((toDay(range.end) - toDay(range.start)) / 2 + DAY_MS / 2));
const share = (rows: DailyMetric[]) => {
const { totals, grand } = shareByModel(
rows,
rows.map((row) => metricValue(row, metric)),
);
return {
hasUsage: grand > 0,
of: (model: string) => (grand === 0 ? 0 : ((totals.get(model)?.value ?? 0) / grand) * 100),
};
};
return {
earlier: share(daily.filter((row) => row.date < midpoint)),
later: share(daily.filter((row) => row.date >= midpoint)),
};
};
export const rankModels = (
rows: ModelMetric[],
daily: DailyMetric[],
metric: Metric,
range: DateRange,
): RankedModel[] => {
const { totals, grand } = shareByModel(
rows,
rows.map((row) => metricValue(row, metric)),
);
const { earlier, later } = halfShares(daily, metric, range);
const comparable = earlier.hasUsage && later.hasUsage;
return [...totals.entries()]
.sort((a, b) => b[1].value - a[1].value)
.map(([model_group, entry]) => ({
model_group,
provider: entry.provider,
share: grand === 0 ? 0 : (entry.value / grand) * 100,
delta: comparable ? later.of(model_group) - earlier.of(model_group) : 0,
}));
};

View file

@ -0,0 +1,9 @@
"use client";
import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized";
import ModelInsightsView from "./_components/ModelInsightsView";
export default function ModelInsightsPage() {
const { accessToken } = useAuthorized();
return <ModelInsightsView accessToken={accessToken} />;
}

View file

@ -204,6 +204,17 @@ const menuGroups: MenuGroup[] = [
roles: [...all_admin_roles, ...internalUserRoles],
label: "Usage",
},
{
key: "model-insights",
page: "model-insights",
icon: <BarChart3 {...ICON} />,
roles: all_admin_roles,
label: (
<span className="flex items-center gap-2">
Model Leaderboard <BetaBadge />
</span>
),
},
{
key: "cost-optimization",
page: "cost-optimization",

View file

@ -20,6 +20,7 @@ export const pageDescriptions: Record<string, string> = {
"vector-stores": "Manage vector databases for embeddings",
new_usage: "View usage analytics and metrics",
"cost-optimization": "Track and configure cost-saving features: prompt compression, caching, and auto routing",
"model-insights": "Model Leaderboard: compare usage, spend, tokens, and task mix across this gateway",
logs: "Access request and response logs",
"guardrails-monitor": "Monitor guardrail performance and view logs",
users: "Manage internal user accounts and permissions",

View file

@ -9189,6 +9189,40 @@ export interface paths {
patch: operations["mistral_proxy_route_mistral__endpoint__patch"];
trace?: never;
};
"/model-insights": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/** Get Model Insights */
get: operations["get_model_insights_model_insights_get"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/model-insights/tasks": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/** Get Model Insight Tasks */
get: operations["get_model_insight_tasks_model_insights_tasks_get"];
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/model/block": {
parameters: {
query?: never;
@ -35974,6 +36008,87 @@ export interface components {
/** Id */
id: string;
};
/** ModelInsightDailyMetric */
ModelInsightDailyMetric: {
/** Completion Tokens */
completion_tokens: number;
/** Date */
date: string;
/** Failed Requests */
failed_requests: number;
/** Model */
model: string;
/** Model Group */
model_group: string;
/** Prompt Tokens */
prompt_tokens: number;
/** Provider */
provider: string;
/** Requests */
requests: number;
/** Spend */
spend: number;
/** Successful Requests */
successful_requests: number;
};
/** ModelInsightMetric */
ModelInsightMetric: {
/** Completion Tokens */
completion_tokens: number;
/** Failed Requests */
failed_requests: number;
/** Model */
model: string;
/** Model Group */
model_group: string;
/** Prompt Tokens */
prompt_tokens: number;
/** Provider */
provider: string;
/** Requests */
requests: number;
/** Spend */
spend: number;
/** Successful Requests */
successful_requests: number;
};
/** ModelInsightTaskSummary */
ModelInsightTaskSummary: {
/** Category */
category: string;
/** Label */
label: string;
/** Leader */
leader: string;
/** Provider */
provider: string;
/** Share */
share: number;
/** Task Type */
task_type: string;
/** Value */
value: number;
};
/** ModelInsightTasksResponse */
ModelInsightTasksResponse: {
/** End Date */
end_date: string;
/** Start Date */
start_date: string;
/** Tasks */
tasks: components["schemas"]["ModelInsightTaskSummary"][];
};
/** ModelInsightsResponse */
ModelInsightsResponse: {
/** Daily */
daily: components["schemas"]["ModelInsightDailyMetric"][];
/** End Date */
end_date: string;
/** Start Date */
start_date: string;
/** Top Models */
top_models: components["schemas"]["ModelInsightMetric"][];
};
/** ModelParams */
ModelParams: {
/** Litellm Params */
@ -59332,6 +59447,78 @@ export interface operations {
};
};
};
get_model_insights_model_insights_get: {
parameters: {
query?: {
/** @description YYYY-MM-DD, defaults to 365 days ago */
start_date?: string | null;
/** @description YYYY-MM-DD, defaults to today */
end_date?: string | null;
/** @description Metric the top models are ranked by */
metric?: "requests" | "spend" | "tokens";
};
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["ModelInsightsResponse"];
};
};
/** @description Validation Error */
422: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["HTTPValidationError"];
};
};
};
};
get_model_insight_tasks_model_insights_tasks_get: {
parameters: {
query?: {
/** @description YYYY-MM-DD, defaults to 365 days ago */
start_date?: string | null;
/** @description YYYY-MM-DD, defaults to today */
end_date?: string | null;
/** @description Metric task shares are computed from */
metric?: "requests" | "spend" | "tokens";
};
header?: never;
path?: never;
cookie?: never;
};
requestBody?: never;
responses: {
/** @description Successful Response */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["ModelInsightTasksResponse"];
};
};
/** @description Validation Error */
422: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["HTTPValidationError"];
};
};
};
};
block_model_model_block_post: {
parameters: {
query?: never;