From 118ce3cc916d78490ff9ae9721fc87ef78d05e84 Mon Sep 17 00:00:00 2001 From: ishaan-berri <155045088+ishaan-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 19:40:04 -0700 Subject: [PATCH] feat: add model leaderboard page (#43649) * feat(proxy): add model leaderboard analytics * feat: add model insights task and range constants * feat: record task type from task tags in model usage rollup * feat: serve 365 days of model insights by UTC date * test: cover task tag resolution in model usage rollup * test: update model insights range limit test to 365 days * chore: regenerate dashboard api types for model insights * feat: add model insights aggregation helpers * test: cover model insights aggregation helpers * feat: redesign model leaderboard with stacked bars, treemap and ranking * test: update model leaderboard view test * feat: mark model leaderboard as beta in sidebar * chore: sync schema.prisma copies from root * fix: only treat task: prefixed tags as model insight tasks * feat: add metric type for model insights ranking * fix: rank model insights by selected metric and scope detail queries to ranked deployments * test: plain tags are not model insight tasks * test: cover metric ranking, deployment scoping and rollup round trip * fix: build model insights weeks and halves from the requested date range * test: cover empty weeks and range-based change comparison * fix: refetch by metric, show load errors and ignore stale responses * test: cover metric refetch and error state * feat: define model insight tasks in a JSON file * feat: return task labels and categories from model insights * feat: load model insight tasks from JSON * refactor: validate rollup task tags against the JSON task list * feat: serve the task list with model insights * refactor: drop hardcoded task list from constants * build: ship model insight tasks JSON in the wheel * test: cover model insight task JSON * refactor: take task labels and categories from the API * test: pass task info to task tile builder * refactor: color treemap by API-provided category * test: include tasks in model leaderboard fixture * fix: make daily model usage migration idempotent * feat: bound the model insights task query size * fix: compute task breakdown independent of the chart metric * test: task breakdown is stable across chart metrics * chore: regenerate lazy openapi snapshot for model insights * chore: regenerate dashboard api types for model insights * fix: keep previous ranking dimmed while a new metric loads * test: cover stale metric state in model leaderboard * refactor: drop task row cap constant * fix: return the full task breakdown instead of a truncated one * test: task query is not truncated * feat: add task summary types for model insights * feat: summarise tasks server-side on a separate model insights endpoint * test: cover the model insights tasks endpoint * chore: regenerate lazy openapi snapshot for model insights tasks * chore: regenerate dashboard api types for model insights tasks * refactor: drop client-side task aggregation * test: remove client-side task aggregation tests * feat: load task breakdown separately from the chart metric * test: task breakdown is not refetched on chart metric change --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .../migration.sql | 19 + .../litellm_proxy_extras/schema.prisma | 20 + litellm/constants.py | 4 + litellm/proxy/_lazy_features.py | 5 + litellm/proxy/_lazy_openapi_snapshot.json | 457 ++++++++++++++++++ litellm/proxy/db/db_spend_update_writer.py | 10 + litellm/proxy/db/model_insights_tasks.py | 14 + litellm/proxy/db/model_usage_rollup.py | 86 ++++ .../model_insights_endpoints.py | 220 +++++++++ litellm/proxy/model_insights_tasks.json | 22 + litellm/proxy/schema.prisma | 20 + litellm/repositories/__init__.py | 2 + litellm/repositories/table_repositories.py | 4 + litellm/types/model_insights.py | 47 ++ pyproject.toml | 1 + schema.prisma | 20 + .../proxy/db/test_model_insights_tasks.py | 19 + .../proxy/db/test_model_usage_rollup.py | 89 ++++ .../test_model_insights_endpoints.py | 233 +++++++++ .../src/app/(dashboard)/legacyPageRoutes.ts | 1 + .../_components/ModelInsightsView.test.tsx | 148 ++++++ .../_components/ModelInsightsView.tsx | 352 ++++++++++++++ .../_components/modelInsightsData.test.ts | 92 ++++ .../_components/modelInsightsData.ts | 132 +++++ .../app/(dashboard)/model-insights/page.tsx | 9 + .../src/components/leftnav.tsx | 11 + .../src/components/page_metadata.ts | 1 + ui/litellm-dashboard/src/lib/http/schema.d.ts | 187 +++++++ 28 files changed, 2225 insertions(+) create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20260928000000_add_daily_model_usage/migration.sql create mode 100644 litellm/proxy/db/model_insights_tasks.py create mode 100644 litellm/proxy/db/model_usage_rollup.py create mode 100644 litellm/proxy/management_endpoints/model_insights_endpoints.py create mode 100644 litellm/proxy/model_insights_tasks.json create mode 100644 litellm/types/model_insights.py create mode 100644 tests/test_litellm/proxy/db/test_model_insights_tasks.py create mode 100644 tests/test_litellm/proxy/db/test_model_usage_rollup.py create mode 100644 tests/test_litellm/proxy/management_endpoints/test_model_insights_endpoints.py create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.test.tsx create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.tsx create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.test.ts create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.ts create mode 100644 ui/litellm-dashboard/src/app/(dashboard)/model-insights/page.tsx diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20260928000000_add_daily_model_usage/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260928000000_add_daily_model_usage/migration.sql new file mode 100644 index 00000000000..1395296ea61 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20260928000000_add_daily_model_usage/migration.sql @@ -0,0 +1,19 @@ +CREATE TABLE IF NOT EXISTS "LiteLLM_DailyModelUsage" ( + "date" TEXT NOT NULL, + "model_group" TEXT NOT NULL, + "model" TEXT NOT NULL, + "custom_llm_provider" TEXT NOT NULL, + "task_type" TEXT NOT NULL, + "spend" DOUBLE PRECISION NOT NULL DEFAULT 0.0, + "prompt_tokens" BIGINT NOT NULL DEFAULT 0, + "completion_tokens" BIGINT NOT NULL DEFAULT 0, + "request_count" BIGINT NOT NULL DEFAULT 0, + "successful_requests" BIGINT NOT NULL DEFAULT 0, + "failed_requests" BIGINT NOT NULL DEFAULT 0, + "created_at" TIMESTAMP(3) NOT NULL DEFAULT CURRENT_TIMESTAMP, + "updated_at" TIMESTAMP(3) NOT NULL, + CONSTRAINT "LiteLLM_DailyModelUsage_pkey" PRIMARY KEY ("date", "model_group", "model", "custom_llm_provider", "task_type") +); + +CREATE INDEX IF NOT EXISTS "LiteLLM_DailyModelUsage_date_idx" ON "LiteLLM_DailyModelUsage"("date"); +CREATE INDEX IF NOT EXISTS "LiteLLM_DailyModelUsage_model_group_idx" ON "LiteLLM_DailyModelUsage"("model_group"); diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 7edc565879a..03e59257f76 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend { @@id([date, tool_name]) } +model LiteLLM_DailyModelUsage { + date String + model_group String + model String + custom_llm_provider String + task_type String + spend Float @default(0.0) + prompt_tokens BigInt @default(0) + completion_tokens BigInt @default(0) + request_count BigInt @default(0) + successful_requests BigInt @default(0) + failed_requests BigInt @default(0) + created_at DateTime @default(now()) + updated_at DateTime @updatedAt + + @@id([date, model_group, model, custom_llm_provider, task_type]) + @@index([date]) + @@index([model_group]) +} + // Gateway request counts recorded at the ASGI edge by // BillableRequestMetricsMiddleware. This is the source of truth for SGR // (successful gateway requests): it counts what the proxy actually answered, diff --git a/litellm/constants.py b/litellm/constants.py index 10c943656f7..39c10d71709 100644 --- a/litellm/constants.py +++ b/litellm/constants.py @@ -1777,6 +1777,10 @@ SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS: Final = float( os.getenv("SCHEDULED_JOB_SHUTDOWN_CANCEL_TIMEOUT_SECONDS", "5") ) TOOL_SPEND_TOP_TOOLS: Final = 100 +MODEL_INSIGHTS_TOP_MODELS: Final = 10 +MODEL_INSIGHTS_MAX_RANGE_DAYS: Final = 365 +MODEL_INSIGHTS_DEFAULT_TASK: Final = "uncategorized" +MODEL_INSIGHTS_TASK_TAG_PREFIX: Final = "task:" SPEND_LOG_PARTITION_INTERVAL: Final = os.getenv("SPEND_LOG_PARTITION_INTERVAL", "day") SPEND_LOG_PARTITION_PRECREATE_AHEAD: Final = int(os.getenv("SPEND_LOG_PARTITION_PRECREATE_AHEAD", 7)) SPEND_LOG_WRITE_BATCH_MAX_BYTES: Final = max(1, int(os.getenv("SPEND_LOG_WRITE_BATCH_MAX_BYTES", 2_000_000))) diff --git a/litellm/proxy/_lazy_features.py b/litellm/proxy/_lazy_features.py index 5be87a8bf4d..98cf3a4ba23 100644 --- a/litellm/proxy/_lazy_features.py +++ b/litellm/proxy/_lazy_features.py @@ -128,6 +128,11 @@ LAZY_FEATURES: Final[tuple[LazyFeature, ...]] = ( module_path="litellm.proxy.management_endpoints.tool_management_endpoints", path_prefixes=("/v1/tool", "/tool"), ), + LazyFeature( + name="model_insights", + module_path="litellm.proxy.management_endpoints.model_insights_endpoints", + path_prefixes=("/model-insights",), + ), LazyFeature( name="search_tools", module_path="litellm.proxy.search_endpoints.search_tool_management", diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 3e0d623375c..75dce43c84a 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -40700,6 +40700,463 @@ } } }, + "model_insights": { + "components": { + "schemas": { + "HTTPValidationError": { + "properties": { + "detail": { + "items": { + "$ref": "#/components/schemas/ValidationError" + }, + "title": "Detail", + "type": "array" + } + }, + "title": "HTTPValidationError", + "type": "object" + }, + "ModelInsightDailyMetric": { + "properties": { + "completion_tokens": { + "title": "Completion Tokens", + "type": "integer" + }, + "date": { + "title": "Date", + "type": "string" + }, + "failed_requests": { + "title": "Failed Requests", + "type": "integer" + }, + "model": { + "title": "Model", + "type": "string" + }, + "model_group": { + "title": "Model Group", + "type": "string" + }, + "prompt_tokens": { + "title": "Prompt Tokens", + "type": "integer" + }, + "provider": { + "title": "Provider", + "type": "string" + }, + "requests": { + "title": "Requests", + "type": "integer" + }, + "spend": { + "title": "Spend", + "type": "number" + }, + "successful_requests": { + "title": "Successful Requests", + "type": "integer" + } + }, + "required": [ + "model_group", + "model", + "provider", + "spend", + "prompt_tokens", + "completion_tokens", + "requests", + "successful_requests", + "failed_requests", + "date" + ], + "title": "ModelInsightDailyMetric", + "type": "object" + }, + "ModelInsightMetric": { + "properties": { + "completion_tokens": { + "title": "Completion Tokens", + "type": "integer" + }, + "failed_requests": { + "title": "Failed Requests", + "type": "integer" + }, + "model": { + "title": "Model", + "type": "string" + }, + "model_group": { + "title": "Model Group", + "type": "string" + }, + "prompt_tokens": { + "title": "Prompt Tokens", + "type": "integer" + }, + "provider": { + "title": "Provider", + "type": "string" + }, + "requests": { + "title": "Requests", + "type": "integer" + }, + "spend": { + "title": "Spend", + "type": "number" + }, + "successful_requests": { + "title": "Successful Requests", + "type": "integer" + } + }, + "required": [ + "model_group", + "model", + "provider", + "spend", + "prompt_tokens", + "completion_tokens", + "requests", + "successful_requests", + "failed_requests" + ], + "title": "ModelInsightMetric", + "type": "object" + }, + "ModelInsightTaskSummary": { + "properties": { + "category": { + "title": "Category", + "type": "string" + }, + "label": { + "title": "Label", + "type": "string" + }, + "leader": { + "title": "Leader", + "type": "string" + }, + "provider": { + "title": "Provider", + "type": "string" + }, + "share": { + "title": "Share", + "type": "number" + }, + "task_type": { + "title": "Task Type", + "type": "string" + }, + "value": { + "title": "Value", + "type": "number" + } + }, + "required": [ + "task_type", + "label", + "category", + "value", + "share", + "leader", + "provider" + ], + "title": "ModelInsightTaskSummary", + "type": "object" + }, + "ModelInsightTasksResponse": { + "properties": { + "end_date": { + "title": "End Date", + "type": "string" + }, + "start_date": { + "title": "Start Date", + "type": "string" + }, + "tasks": { + "items": { + "$ref": "#/components/schemas/ModelInsightTaskSummary" + }, + "title": "Tasks", + "type": "array" + } + }, + "required": [ + "start_date", + "end_date", + "tasks" + ], + "title": "ModelInsightTasksResponse", + "type": "object" + }, + "ModelInsightsResponse": { + "properties": { + "daily": { + "items": { + "$ref": "#/components/schemas/ModelInsightDailyMetric" + }, + "title": "Daily", + "type": "array" + }, + "end_date": { + "title": "End Date", + "type": "string" + }, + "start_date": { + "title": "Start Date", + "type": "string" + }, + "top_models": { + "items": { + "$ref": "#/components/schemas/ModelInsightMetric" + }, + "title": "Top Models", + "type": "array" + } + }, + "required": [ + "start_date", + "end_date", + "daily", + "top_models" + ], + "title": "ModelInsightsResponse", + "type": "object" + }, + "ValidationError": { + "properties": { + "ctx": { + "title": "Context", + "type": "object" + }, + "input": { + "title": "Input" + }, + "loc": { + "items": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "integer" + } + ] + }, + "title": "Location", + "type": "array" + }, + "msg": { + "title": "Message", + "type": "string" + }, + "type": { + "title": "Error Type", + "type": "string" + } + }, + "required": [ + "loc", + "msg", + "type" + ], + "title": "ValidationError", + "type": "object" + } + } + }, + "paths": { + "/model-insights": { + "get": { + "operationId": "get_model_insights_model_insights_get", + "parameters": [ + { + "description": "YYYY-MM-DD, defaults to 365 days ago", + "in": "query", + "name": "start_date", + "required": false, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "description": "YYYY-MM-DD, defaults to 365 days ago", + "title": "Start Date" + } + }, + { + "description": "YYYY-MM-DD, defaults to today", + "in": "query", + "name": "end_date", + "required": false, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "description": "YYYY-MM-DD, defaults to today", + "title": "End Date" + } + }, + { + "description": "Metric the top models are ranked by", + "in": "query", + "name": "metric", + "required": false, + "schema": { + "default": "tokens", + "description": "Metric the top models are ranked by", + "enum": [ + "requests", + "spend", + "tokens" + ], + "title": "Metric", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ModelInsightsResponse" + } + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Get Model Insights", + "tags": [ + "model_insights" + ] + } + }, + "/model-insights/tasks": { + "get": { + "operationId": "get_model_insight_tasks_model_insights_tasks_get", + "parameters": [ + { + "description": "YYYY-MM-DD, defaults to 365 days ago", + "in": "query", + "name": "start_date", + "required": false, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "description": "YYYY-MM-DD, defaults to 365 days ago", + "title": "Start Date" + } + }, + { + "description": "YYYY-MM-DD, defaults to today", + "in": "query", + "name": "end_date", + "required": false, + "schema": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "description": "YYYY-MM-DD, defaults to today", + "title": "End Date" + } + }, + { + "description": "Metric task shares are computed from", + "in": "query", + "name": "metric", + "required": false, + "schema": { + "default": "spend", + "description": "Metric task shares are computed from", + "enum": [ + "requests", + "spend", + "tokens" + ], + "title": "Metric", + "type": "string" + } + } + ], + "responses": { + "200": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ModelInsightTasksResponse" + } + } + }, + "description": "Successful Response" + }, + "422": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + }, + "description": "Validation Error" + } + }, + "security": [ + { + "APIKeyHeader": [] + } + ], + "summary": "Get Model Insight Tasks", + "tags": [ + "model_insights" + ] + } + } + } + }, "policies": { "components": { "schemas": { diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py index 17e6152bef6..72553e82283 100644 --- a/litellm/proxy/db/db_spend_update_writer.py +++ b/litellm/proxy/db/db_spend_update_writer.py @@ -1148,6 +1148,16 @@ class DBSpendUpdateWriter: traceback.format_exc(), ) + try: + from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage + + await increment_daily_model_usage(prisma_client=prisma_client, payload=payload_copy) + except Exception: + verbose_proxy_logger.debug( + "_batch_database_updates: increment_daily_model_usage failed: %s", + traceback.format_exc(), + ) + async def _update_key_db( self, response_cost: float | None, diff --git a/litellm/proxy/db/model_insights_tasks.py b/litellm/proxy/db/model_insights_tasks.py new file mode 100644 index 00000000000..865965dcf75 --- /dev/null +++ b/litellm/proxy/db/model_insights_tasks.py @@ -0,0 +1,14 @@ +import json +from functools import lru_cache +from pathlib import Path +from typing import Final + +from litellm.types.model_insights import ModelInsightTask + +_TASKS_FILE: Final = Path(__file__).resolve().parent.parent / "model_insights_tasks.json" + + +@lru_cache(maxsize=1) +def load_model_insight_tasks() -> dict[str, ModelInsightTask]: + raw: Final = json.loads(_TASKS_FILE.read_text()) + return {name: ModelInsightTask(task_type=name, **entry) for name, entry in raw.items()} diff --git a/litellm/proxy/db/model_usage_rollup.py b/litellm/proxy/db/model_usage_rollup.py new file mode 100644 index 00000000000..808c3528051 --- /dev/null +++ b/litellm/proxy/db/model_usage_rollup.py @@ -0,0 +1,86 @@ +from datetime import datetime +from typing import Final + +from pydantic import TypeAdapter, ValidationError + +from litellm.constants import ( + INTERNAL_CALL_ORIGIN_METADATA_KEY, + MODEL_INSIGHTS_DEFAULT_TASK, + MODEL_INSIGHTS_TASK_TAG_PREFIX, +) +from litellm.proxy._types import SpendLogsPayload +from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks +from litellm.proxy.utils import PrismaClient +from litellm.repositories.table_repositories import DailyModelUsageRepository + +_METADATA: Final = TypeAdapter(dict[str, object]) +_TAGS: Final = TypeAdapter(list[object]) + + +def model_usage_task_type(request_tags: str) -> str: + try: + tags: Final = _TAGS.validate_json(request_tags) + except ValidationError: + return MODEL_INSIGHTS_DEFAULT_TASK + for tag in tags: + if isinstance(tag, str) and tag.startswith(MODEL_INSIGHTS_TASK_TAG_PREFIX): + task = tag.removeprefix(MODEL_INSIGHTS_TASK_TAG_PREFIX) + if task in load_model_insight_tasks(): + return task + return MODEL_INSIGHTS_DEFAULT_TASK + + +def _is_internal_call(metadata: str) -> bool: + try: + decoded: Final = _METADATA.validate_json(metadata) + except ValidationError: + return False + return bool(decoded.get(INTERNAL_CALL_ORIGIN_METADATA_KEY)) + + +def _date_from_start_time(start_time: datetime | str) -> str | None: + if isinstance(start_time, datetime): + return start_time.date().isoformat() + return start_time[:10] if len(start_time) >= 10 else None + + +async def increment_daily_model_usage(prisma_client: PrismaClient, payload: SpendLogsPayload) -> None: + date: Final = _date_from_start_time(payload["startTime"]) + if date is None or _is_internal_call(payload["metadata"]): + return + + model: Final = payload["model"] or "unknown" + model_group: Final = payload["model_group"] or model + provider: Final = payload["custom_llm_provider"] or "unknown" + task_type: Final = model_usage_task_type(payload["request_tags"]) + successful: Final = 1 if payload["status"] == "success" else 0 + failed: Final = 1 - successful + key: Final = { + "date": date, + "model_group": model_group, + "model": model, + "custom_llm_provider": provider, + "task_type": task_type, + } + await DailyModelUsageRepository(prisma_client).table.upsert( + where={"date_model_group_model_custom_llm_provider_task_type": key}, + data={ + "create": { + **key, + "spend": payload["spend"], + "prompt_tokens": payload["prompt_tokens"], + "completion_tokens": payload["completion_tokens"], + "request_count": 1, + "successful_requests": successful, + "failed_requests": failed, + }, + "update": { + "spend": {"increment": payload["spend"]}, + "prompt_tokens": {"increment": payload["prompt_tokens"]}, + "completion_tokens": {"increment": payload["completion_tokens"]}, + "request_count": {"increment": 1}, + "successful_requests": {"increment": successful}, + "failed_requests": {"increment": failed}, + }, + }, + ) diff --git a/litellm/proxy/management_endpoints/model_insights_endpoints.py b/litellm/proxy/management_endpoints/model_insights_endpoints.py new file mode 100644 index 00000000000..dbdaa59d7d4 --- /dev/null +++ b/litellm/proxy/management_endpoints/model_insights_endpoints.py @@ -0,0 +1,220 @@ +from collections.abc import Mapping +from datetime import date, datetime, timedelta, timezone +from typing import Annotated, Final + +from fastapi import APIRouter, Depends, HTTPException, Query +from pydantic import BaseModel, Field, TypeAdapter + +from litellm.constants import MODEL_INSIGHTS_DEFAULT_TASK, MODEL_INSIGHTS_MAX_RANGE_DAYS, MODEL_INSIGHTS_TOP_MODELS +from litellm.proxy._types import CommonProxyErrors, LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks +from litellm.repositories.table_repositories import DailyModelUsageRepository +from litellm.types.model_insights import ( + ModelInsightDailyMetric, + ModelInsightMetric, + ModelInsightsMetric, + ModelInsightsResponse, + ModelInsightTask, + ModelInsightTasksResponse, + ModelInsightTaskSummary, +) + +router: Final = APIRouter() + + +class _Sums(BaseModel): + spend: float = 0.0 + prompt_tokens: int = 0 + completion_tokens: int = 0 + request_count: int = 0 + successful_requests: int = 0 + failed_requests: int = 0 + + +class _GroupedModel(BaseModel): + model_group: str + model: str + custom_llm_provider: str + sums: _Sums = Field(alias="_sum") + + +class _GroupedDaily(_GroupedModel): + date: str + + +class _GroupedTask(_GroupedModel): + task_type: str + + +_MODEL_ROWS: Final = TypeAdapter(list[_GroupedModel]) +_DAILY_ROWS: Final = TypeAdapter(list[_GroupedDaily]) +_TASK_ROWS: Final = TypeAdapter(list[_GroupedTask]) +_UNCATEGORIZED_TASK: Final = ModelInsightTask( + task_type=MODEL_INSIGHTS_DEFAULT_TASK, label="Uncategorized", category="General" +) +_SUM_FIELDS: Final = { + "spend": True, + "prompt_tokens": True, + "completion_tokens": True, + "request_count": True, + "successful_requests": True, + "failed_requests": True, +} + + +def _parse_date(value: str | None, fallback: date) -> date: + if value is None: + return fallback + try: + return date.fromisoformat(value) + except ValueError as exc: + raise HTTPException(status_code=400, detail="Dates must use YYYY-MM-DD") from exc + + +def _metric(row: _GroupedModel) -> ModelInsightMetric: + return ModelInsightMetric( + model_group=row.model_group, + model=row.model, + provider=row.custom_llm_provider, + spend=row.sums.spend, + prompt_tokens=row.sums.prompt_tokens, + completion_tokens=row.sums.completion_tokens, + requests=row.sums.request_count, + successful_requests=row.sums.successful_requests, + failed_requests=row.sums.failed_requests, + ) + + +def _rank_value(row: _GroupedModel, metric: ModelInsightsMetric) -> float: + if metric == "requests": + return row.sums.request_count + if metric == "spend": + return row.sums.spend + return row.sums.prompt_tokens + row.sums.completion_tokens + + +def _top_model_rows(rows: list[_GroupedModel], metric: ModelInsightsMetric) -> list[_GroupedModel]: + return sorted(rows, key=lambda row: _rank_value(row, metric), reverse=True)[:MODEL_INSIGHTS_TOP_MODELS] + + +def _deployment_filter(rows: list[_GroupedModel]) -> list[dict[str, str]]: + return [ + {"model_group": row.model_group, "model": row.model, "custom_llm_provider": row.custom_llm_provider} + for row in rows + ] + + +def _daily_metric(row: _GroupedDaily) -> ModelInsightDailyMetric: + return ModelInsightDailyMetric(date=row.date, **_metric(row).model_dump()) + + +def _summarize_tasks(rows: list[_GroupedTask], metric: ModelInsightsMetric) -> list[ModelInsightTaskSummary]: + catalog: Final = load_model_insight_tasks() + totals: Final[dict[str, float]] = {} + leaders: Final[dict[str, _GroupedTask]] = {} + for row in rows: + value = _rank_value(row, metric) + totals[row.task_type] = totals.get(row.task_type, 0.0) + value + leader = leaders.get(row.task_type) + if leader is None or value > _rank_value(leader, metric): + leaders[row.task_type] = row + grand: Final = sum(totals.values()) + return [ + ModelInsightTaskSummary( + **(catalog.get(task) or _UNCATEGORIZED_TASK).model_copy(update={"task_type": task}).model_dump(), + value=value, + share=value / grand * 100 if grand else 0.0, + leader=leaders[task].model_group, + provider=leaders[task].custom_llm_provider, + ) + for task, value in sorted(totals.items(), key=lambda item: item[1], reverse=True) + ] + + +def _resolve_window( + user_api_key_dict: UserAPIKeyAuth, start_date: str | None, end_date: str | None +) -> tuple[date, date, Mapping[str, object], DailyModelUsageRepository]: + from litellm.proxy.proxy_server import prisma_client + + if user_api_key_dict.user_role not in (LitellmUserRoles.PROXY_ADMIN, LitellmUserRoles.PROXY_ADMIN_VIEW_ONLY): + raise HTTPException(status_code=403, detail="Only proxy admins can view deployment-wide model insights") + if prisma_client is None: + raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value) + + end_day: Final = _parse_date(end_date, datetime.now(timezone.utc).date()) + start_day: Final = _parse_date(start_date, end_day - timedelta(days=MODEL_INSIGHTS_MAX_RANGE_DAYS - 1)) + if start_day > end_day or (end_day - start_day).days >= MODEL_INSIGHTS_MAX_RANGE_DAYS: + raise HTTPException( + status_code=400, detail=f"Date range must be between 1 and {MODEL_INSIGHTS_MAX_RANGE_DAYS} days" + ) + date_window: Final[Mapping[str, object]] = {"date": {"gte": start_day.isoformat(), "lte": end_day.isoformat()}} + return start_day, end_day, date_window, DailyModelUsageRepository(prisma_client) + + +@router.get( + "/model-insights", + tags=["model insights"], + dependencies=[Depends(user_api_key_auth)], + response_model=ModelInsightsResponse, +) +async def get_model_insights( + user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], + start_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to 365 days ago")] = None, + end_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to today")] = None, + metric: Annotated[ModelInsightsMetric, Query(description="Metric the top models are ranked by")] = "tokens", +) -> ModelInsightsResponse: + start_day, end_day, date_window, repository = _resolve_window(user_api_key_dict, start_date, end_date) + table: Final = repository.table + grouped_model_rows: Final = _MODEL_ROWS.validate_python( + await table.group_by( + by=["model_group", "model", "custom_llm_provider"], + sum=_SUM_FIELDS, + where=date_window, + ) + ) + model_rows: Final = _top_model_rows(grouped_model_rows, metric) + selected_window: Final = {**date_window, "OR": _deployment_filter(model_rows)} + daily_rows: Final = _DAILY_ROWS.validate_python( + await table.group_by( + by=["date", "model_group", "model", "custom_llm_provider"], + sum=_SUM_FIELDS, + where=selected_window, + order={"date": "asc"}, + ) + if model_rows + else [] + ) + return ModelInsightsResponse( + start_date=start_day.isoformat(), + end_date=end_day.isoformat(), + top_models=[_metric(row) for row in model_rows], + daily=[_daily_metric(row) for row in daily_rows], + ) + + +@router.get( + "/model-insights/tasks", + tags=["model insights"], + dependencies=[Depends(user_api_key_auth)], + response_model=ModelInsightTasksResponse, +) +async def get_model_insight_tasks( + user_api_key_dict: Annotated[UserAPIKeyAuth, Depends(user_api_key_auth)], + start_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to 365 days ago")] = None, + end_date: Annotated[str | None, Query(description="YYYY-MM-DD, defaults to today")] = None, + metric: Annotated[ModelInsightsMetric, Query(description="Metric task shares are computed from")] = "spend", +) -> ModelInsightTasksResponse: + start_day, end_day, date_window, repository = _resolve_window(user_api_key_dict, start_date, end_date) + task_rows: Final = _TASK_ROWS.validate_python( + await repository.table.group_by( + by=["task_type", "model_group", "model", "custom_llm_provider"], + sum=_SUM_FIELDS, + where=date_window, + ) + ) + return ModelInsightTasksResponse( + start_date=start_day.isoformat(), + end_date=end_day.isoformat(), + tasks=_summarize_tasks(task_rows, metric), + ) diff --git a/litellm/proxy/model_insights_tasks.json b/litellm/proxy/model_insights_tasks.json new file mode 100644 index 00000000000..17f9dabd24e --- /dev/null +++ b/litellm/proxy/model_insights_tasks.json @@ -0,0 +1,22 @@ +{ + "classification": {"label": "Classification", "category": "General"}, + "content_writing": {"label": "Content Writing", "category": "General"}, + "roleplay_fiction": {"label": "Roleplay & Fiction", "category": "General"}, + "conversation": {"label": "Conversation", "category": "General"}, + "research_reports": {"label": "Research & Reports", "category": "General"}, + "qa_knowledge": {"label": "Q&A & Knowledge", "category": "General"}, + "customer_support": {"label": "Customer Support", "category": "General"}, + "summarization": {"label": "Summarization", "category": "General"}, + "translation": {"label": "Translation", "category": "General"}, + "workflow_execution": {"label": "Workflow Execution", "category": "Agent"}, + "multi_step_planning": {"label": "Multi-step Planning", "category": "Agent"}, + "tool_dispatch": {"label": "Tool Dispatch", "category": "Agent"}, + "code_generation": {"label": "Code Generation", "category": "Code"}, + "debugging": {"label": "Debugging", "category": "Code"}, + "code_review": {"label": "Code Review", "category": "Code"}, + "frontend_ui": {"label": "Frontend & UI", "category": "Code"}, + "file_io": {"label": "File I/O", "category": "Code"}, + "shell_execution": {"label": "Shell Execution", "category": "Code"}, + "data_extraction": {"label": "Data Extraction", "category": "Data"}, + "data_transformation": {"label": "Data Transformation", "category": "Data"} +} diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 7edc565879a..03e59257f76 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend { @@id([date, tool_name]) } +model LiteLLM_DailyModelUsage { + date String + model_group String + model String + custom_llm_provider String + task_type String + spend Float @default(0.0) + prompt_tokens BigInt @default(0) + completion_tokens BigInt @default(0) + request_count BigInt @default(0) + successful_requests BigInt @default(0) + failed_requests BigInt @default(0) + created_at DateTime @default(now()) + updated_at DateTime @updatedAt + + @@id([date, model_group, model, custom_llm_provider, task_type]) + @@index([date]) + @@index([model_group]) +} + // Gateway request counts recorded at the ASGI edge by // BillableRequestMetricsMiddleware. This is the source of truth for SGR // (successful gateway requests): it counts what the proxy actually answered, diff --git a/litellm/repositories/__init__.py b/litellm/repositories/__init__.py index dcf9ddfc32a..7ffdcfa5ce6 100644 --- a/litellm/repositories/__init__.py +++ b/litellm/repositories/__init__.py @@ -30,6 +30,7 @@ from litellm.repositories.table_repositories import ( ConfigOverridesRepository, DailyGuardrailMetricsRepository, DailyGuardrailUsageUnitsRepository, + DailyModelUsageRepository, DailyPolicyMetricsRepository, DailyTagSpendRepository, DailyToolSpendRepository, @@ -105,6 +106,7 @@ __all__ = [ "CredentialsRepository", "DailyGuardrailMetricsRepository", "DailyGuardrailUsageUnitsRepository", + "DailyModelUsageRepository", "DailyPolicyMetricsRepository", "DailyTagSpendRepository", "DailyToolSpendRepository", diff --git a/litellm/repositories/table_repositories.py b/litellm/repositories/table_repositories.py index 1ad7a735d96..ab68f1a2bc7 100644 --- a/litellm/repositories/table_repositories.py +++ b/litellm/repositories/table_repositories.py @@ -212,6 +212,10 @@ class DailyToolSpendRepository(PrismaTableRepository["prisma_models.LiteLLM_Dail table_name = "litellm_dailytoolspend" +class DailyModelUsageRepository(PrismaTableRepository["prisma_models.LiteLLM_DailyModelUsage"]): + table_name = "litellm_dailymodelusage" + + class SpendLogGuardrailIndexRepository(PrismaTableRepository["prisma_models.LiteLLM_SpendLogGuardrailIndex"]): table_name = "litellm_spendlogguardrailindex" diff --git a/litellm/types/model_insights.py b/litellm/types/model_insights.py new file mode 100644 index 00000000000..6b7939386a7 --- /dev/null +++ b/litellm/types/model_insights.py @@ -0,0 +1,47 @@ +from typing import Literal + +from pydantic import BaseModel + +ModelInsightsMetric = Literal["requests", "spend", "tokens"] + + +class ModelInsightMetric(BaseModel): + model_group: str + model: str + provider: str + spend: float + prompt_tokens: int + completion_tokens: int + requests: int + successful_requests: int + failed_requests: int + + +class ModelInsightDailyMetric(ModelInsightMetric): + date: str + + +class ModelInsightTask(BaseModel): + task_type: str + label: str + category: str + + +class ModelInsightTaskSummary(ModelInsightTask): + value: float + share: float + leader: str + provider: str + + +class ModelInsightsResponse(BaseModel): + start_date: str + end_date: str + daily: list[ModelInsightDailyMetric] + top_models: list[ModelInsightMetric] + + +class ModelInsightTasksResponse(BaseModel): + start_date: str + end_date: str + tasks: list[ModelInsightTaskSummary] diff --git a/pyproject.toml b/pyproject.toml index 28b00379cc7..fb21d8fa23b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -317,6 +317,7 @@ include = [ "litellm/proxy/_experimental/out/**", "litellm/router_strategy/complexity_router/artifacts/*.json", "litellm/router_strategy/complexity_router/fuse_presets.json", + "litellm/proxy/model_insights_tasks.json", "litellm/proxy/client/cli/commands/codex_base_instructions.md", ] exclude = [ diff --git a/schema.prisma b/schema.prisma index 7edc565879a..03e59257f76 100644 --- a/schema.prisma +++ b/schema.prisma @@ -1260,6 +1260,26 @@ model LiteLLM_DailyToolSpend { @@id([date, tool_name]) } +model LiteLLM_DailyModelUsage { + date String + model_group String + model String + custom_llm_provider String + task_type String + spend Float @default(0.0) + prompt_tokens BigInt @default(0) + completion_tokens BigInt @default(0) + request_count BigInt @default(0) + successful_requests BigInt @default(0) + failed_requests BigInt @default(0) + created_at DateTime @default(now()) + updated_at DateTime @updatedAt + + @@id([date, model_group, model, custom_llm_provider, task_type]) + @@index([date]) + @@index([model_group]) +} + // Gateway request counts recorded at the ASGI edge by // BillableRequestMetricsMiddleware. This is the source of truth for SGR // (successful gateway requests): it counts what the proxy actually answered, diff --git a/tests/test_litellm/proxy/db/test_model_insights_tasks.py b/tests/test_litellm/proxy/db/test_model_insights_tasks.py new file mode 100644 index 00000000000..5c0786deaf5 --- /dev/null +++ b/tests/test_litellm/proxy/db/test_model_insights_tasks.py @@ -0,0 +1,19 @@ +from litellm.proxy.db.model_insights_tasks import load_model_insight_tasks +from litellm.proxy.db.model_usage_rollup import model_usage_task_type + + +def test_every_task_has_a_label_and_a_category() -> None: + tasks = load_model_insight_tasks() + + assert tasks + for name, task in tasks.items(): + assert task.task_type == name + assert task.label + assert task.category in {"General", "Agent", "Code", "Data"} + + +def test_tasks_in_the_json_file_are_the_ones_the_rollup_accepts() -> None: + for name in load_model_insight_tasks(): + assert model_usage_task_type(f'["task:{name}"]') == name + + assert model_usage_task_type('["task:not_in_the_file"]') == "uncategorized" diff --git a/tests/test_litellm/proxy/db/test_model_usage_rollup.py b/tests/test_litellm/proxy/db/test_model_usage_rollup.py new file mode 100644 index 00000000000..f54856129dc --- /dev/null +++ b/tests/test_litellm/proxy/db/test_model_usage_rollup.py @@ -0,0 +1,89 @@ +from datetime import datetime, timezone +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage, model_usage_task_type + + +def test_model_usage_task_type_reads_task_tag_or_defaults() -> None: + assert model_usage_task_type('["team-a", "task:classification"]') == "classification" + assert model_usage_task_type('["task:made-up"]') == "uncategorized" + assert model_usage_task_type('["debugging"]') == "uncategorized" + assert model_usage_task_type("[]") == "uncategorized" + assert model_usage_task_type("not json") == "uncategorized" + + +@pytest.mark.asyncio +async def test_increment_daily_model_usage_uses_atomic_prisma_upsert() -> None: + table = MagicMock() + table.upsert = AsyncMock() + prisma_client = MagicMock() + prisma_client.db.litellm_dailymodelusage = table + payload = { + "request_id": "request-1", + "call_type": "acompletion", + "api_key": "key", + "spend": 0.25, + "total_tokens": 30, + "prompt_tokens": 10, + "completion_tokens": 20, + "startTime": datetime(2026, 9, 28, tzinfo=timezone.utc), + "endTime": datetime(2026, 9, 28, tzinfo=timezone.utc), + "completionStartTime": None, + "model": "openai/gpt-5.4-mini", + "model_id": None, + "model_group": "fast-chat", + "mcp_namespaced_tool_name": None, + "agent_id": None, + "api_base": "", + "user": "user", + "metadata": "{}", + "cache_hit": "False", + "cache_key": "", + "request_tags": "[]", + "team_id": None, + "organization_id": None, + "end_user": None, + "requester_ip_address": None, + "custom_llm_provider": "openai", + "messages": None, + "response": None, + "proxy_server_request": None, + "session_id": None, + "request_duration_ms": 20, + "status": "success", + "litellm_call_id": None, + } + + await increment_daily_model_usage(prisma_client, payload) + + call = table.upsert.await_args.kwargs + assert call["data"]["create"]["request_count"] == 1 + assert call["data"]["update"]["completion_tokens"] == {"increment": 20} + assert call["data"]["create"]["task_type"] == "uncategorized" + + +@pytest.mark.asyncio +async def test_increment_daily_model_usage_records_task_from_request_tags() -> None: + table = MagicMock() + table.upsert = AsyncMock() + prisma_client = MagicMock() + prisma_client.db.litellm_dailymodelusage = table + payload = { + "call_type": "acompletion", + "spend": 0.1, + "prompt_tokens": 1, + "completion_tokens": 2, + "startTime": datetime(2026, 9, 28, tzinfo=timezone.utc), + "model": "gpt-5", + "model_group": "gpt-5", + "metadata": "{}", + "request_tags": '["task:debugging"]', + "custom_llm_provider": "openai", + "status": "success", + } + + await increment_daily_model_usage(prisma_client, payload) + + assert table.upsert.await_args.kwargs["data"]["create"]["task_type"] == "debugging" diff --git a/tests/test_litellm/proxy/management_endpoints/test_model_insights_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_model_insights_endpoints.py new file mode 100644 index 00000000000..af58c2d9884 --- /dev/null +++ b/tests/test_litellm/proxy/management_endpoints/test_model_insights_endpoints.py @@ -0,0 +1,233 @@ +from datetime import datetime, timezone +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest +from fastapi import FastAPI +from fastapi.testclient import TestClient + +from litellm.proxy._types import LitellmUserRoles, UserAPIKeyAuth +from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.db.model_usage_rollup import increment_daily_model_usage +from litellm.proxy.management_endpoints.model_insights_endpoints import router + + +def _override_auth() -> UserAPIKeyAuth: + return UserAPIKeyAuth(api_key="sk-test", user_id="admin", user_role=LitellmUserRoles.PROXY_ADMIN) + + +def _grouped_row(*, prompt_tokens: str = "100", completion_tokens: str = "200", **dimensions: str) -> dict[str, object]: + return { + **dimensions, + "_sum": { + "spend": 1.25, + "prompt_tokens": prompt_tokens, + "completion_tokens": completion_tokens, + "request_count": "3", + "successful_requests": "3", + "failed_requests": "0", + }, + } + + +def test_model_insights_reads_only_bounded_rollup() -> None: + model = _grouped_row(model_group="fast-chat", model="openai/gpt-5.4-mini", custom_llm_provider="openai") + prompt_heavy_model = _grouped_row( + prompt_tokens="500", + completion_tokens="10", + model_group="long-context", + model="anthropic/claude-sonnet-4-5", + custom_llm_provider="anthropic", + ) + daily = _grouped_row( + date="2026-09-28", + model_group="fast-chat", + model="openai/gpt-5.4-mini", + custom_llm_provider="openai", + ) + table = MagicMock() + table.group_by = AsyncMock(side_effect=[[model, prompt_heavy_model], [daily]]) + prisma = MagicMock() + prisma.db.litellm_dailymodelusage = table + prisma.db.query_raw = AsyncMock() + prisma.db.litellm_spendlogs.find_many = AsyncMock() + app = FastAPI() + app.include_router(router) + app.dependency_overrides[user_api_key_auth] = _override_auth + + with patch("litellm.proxy.proxy_server.prisma_client", prisma): + response = TestClient(app).get("/model-insights?start_date=2026-09-09&end_date=2026-09-28") + + assert response.status_code == 200 + assert response.json()["top_models"][0]["model_group"] == "long-context" + assert "by_task" not in response.json() + assert table.group_by.await_count == 2 + prisma.db.query_raw.assert_not_awaited() + prisma.db.litellm_spendlogs.find_many.assert_not_awaited() + + +def test_model_insights_rejects_ranges_over_365_days() -> None: + prisma = MagicMock() + app = FastAPI() + app.include_router(router) + app.dependency_overrides[user_api_key_auth] = _override_auth + + with patch("litellm.proxy.proxy_server.prisma_client", prisma): + response = TestClient(app).get("/model-insights?start_date=2025-09-01&end_date=2026-09-28") + + assert response.status_code == 400 + + +def _call(table: MagicMock, query: str, path: str = "/model-insights") -> object: + prisma = MagicMock() + prisma.db.litellm_dailymodelusage = table + app = FastAPI() + app.include_router(router) + app.dependency_overrides[user_api_key_auth] = _override_auth + with patch("litellm.proxy.proxy_server.prisma_client", prisma): + return TestClient(app).get(f"{path}?start_date=2026-09-01&end_date=2026-09-28&{query}") + + +def test_model_insights_ranks_top_models_by_selected_metric() -> None: + token_heavy = _grouped_row( + prompt_tokens="9000", completion_tokens="9000", model_group="big", model="m1", custom_llm_provider="openai" + ) + request_heavy = _grouped_row( + prompt_tokens="1", completion_tokens="1", model_group="busy", model="m2", custom_llm_provider="openai" + ) + request_heavy["_sum"]["request_count"] = "500" + table = MagicMock() + table.group_by = AsyncMock(side_effect=[[token_heavy, request_heavy], []]) + + by_requests = _call(table, "metric=requests").json() + by_tokens = _call( + MagicMock(group_by=AsyncMock(side_effect=[[token_heavy, request_heavy], []])), "metric=tokens" + ).json() + + assert by_requests["top_models"][0]["model_group"] == "busy" + assert by_tokens["top_models"][0]["model_group"] == "big" + + +def test_model_insights_scopes_daily_to_ranked_deployments() -> None: + ranked = _grouped_row(model_group="shared", model="m1", custom_llm_provider="openai") + table = MagicMock() + table.group_by = AsyncMock(side_effect=[[ranked], []]) + + _call(table, "metric=tokens") + + daily_where = table.group_by.await_args_list[1].kwargs["where"] + assert daily_where["OR"] == [{"model_group": "shared", "model": "m1", "custom_llm_provider": "openai"}] + assert "model_group" not in daily_where + + +def _task_rows() -> list[dict[str, object]]: + def row(task: str, group: str, requests: str, spend: float) -> dict[str, object]: + base = _grouped_row(task_type=task, model_group=group, model=group, custom_llm_provider="openai") + base["_sum"].update({"request_count": requests, "spend": spend}) # type: ignore[union-attr] + return base + + return [ + row("debugging", "big", "1", 9.0), + row("debugging", "busy", "50", 1.0), + row("classification", "busy", "10", 1.0), + ] + + +def test_model_insight_tasks_are_summarised_on_the_server() -> None: + table = MagicMock(group_by=AsyncMock(return_value=_task_rows())) + + body = _call(table, "metric=spend", path="/model-insights/tasks").json() + + assert [(t["task_type"], t["label"], t["category"], t["leader"]) for t in body["tasks"]] == [ + ("debugging", "Debugging", "Code", "big"), + ("classification", "Classification", "General", "busy"), + ] + assert [round(t["share"], 1) for t in body["tasks"]] == [90.9, 9.1] + assert "OR" not in table.group_by.await_args.kwargs["where"] + assert "take" not in table.group_by.await_args.kwargs + + +def test_model_insight_tasks_leader_follows_the_selected_metric() -> None: + by_spend = _call(MagicMock(group_by=AsyncMock(return_value=_task_rows())), "metric=spend", "/model-insights/tasks") + by_requests = _call( + MagicMock(group_by=AsyncMock(return_value=_task_rows())), "metric=requests", "/model-insights/tasks" + ) + + assert by_spend.json()["tasks"][0]["leader"] == "big" + assert by_requests.json()["tasks"][0]["leader"] == "busy" + + +def test_model_insight_tasks_unknown_task_shows_as_uncategorized() -> None: + row = _grouped_row(task_type="uncategorized", model_group="a", model="a", custom_llm_provider="openai") + body = _call(MagicMock(group_by=AsyncMock(return_value=[row])), "metric=spend", "/model-insights/tasks").json() + + assert [(t["label"], t["category"]) for t in body["tasks"]] == [("Uncategorized", "General")] + + +def test_model_insight_tasks_require_an_admin() -> None: + app = FastAPI() + app.include_router(router) + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth( + api_key="sk-test", user_id="u", user_role=LitellmUserRoles.INTERNAL_USER + ) + with patch("litellm.proxy.proxy_server.prisma_client", MagicMock()): + assert TestClient(app).get("/model-insights/tasks").status_code == 403 + + +def test_model_insights_rejects_unknown_metric() -> None: + assert _call(MagicMock(group_by=AsyncMock()), "metric=bogus").status_code == 422 + + +class _InMemoryUsageTable: + def __init__(self) -> None: + self.rows: dict[tuple[str, ...], dict[str, float]] = {} + + async def upsert(self, where: dict, data: dict) -> None: + key_fields = where["date_model_group_model_custom_llm_provider_task_type"] + key = tuple(key_fields.values()) + if key not in self.rows: + self.rows[key] = {**key_fields, **{k: v for k, v in data["create"].items() if k not in key_fields}} + return + for field, change in data["update"].items(): + self.rows[key][field] += change["increment"] + + async def group_by(self, by: list[str], sum: dict, where: dict, **_: object) -> list[dict]: + grouped: dict[tuple, dict] = {} + for row in self.rows.values(): + if not where["date"]["gte"] <= row["date"] <= where["date"]["lte"]: + continue + if where.get("OR") and not any(all(row[k] == v for k, v in option.items()) for option in where["OR"]): + continue + bucket = grouped.setdefault(tuple(row[k] for k in by), {**{k: row[k] for k in by}, "_sum": {}}) + for field in sum: + bucket["_sum"][field] = bucket["_sum"].get(field, 0) + row[field] + return list(grouped.values()) + + +@pytest.mark.asyncio +async def test_model_insights_reads_back_what_the_rollup_wrote() -> None: + table = _InMemoryUsageTable() + prisma = MagicMock() + prisma.db.litellm_dailymodelusage = table + payload = { + "call_type": "acompletion", + "spend": 0.5, + "prompt_tokens": 10, + "completion_tokens": 20, + "startTime": datetime(2026, 9, 28, tzinfo=timezone.utc), + "model": "gpt-5", + "model_group": "gpt-5", + "metadata": "{}", + "request_tags": '["task:debugging"]', + "custom_llm_provider": "openai", + "status": "success", + } + + await increment_daily_model_usage(prisma, payload) + await increment_daily_model_usage(prisma, {**payload, "request_tags": "[]"}) + + body = _call(table, "metric=requests").json() + + assert [(m["model_group"], m["requests"], m["prompt_tokens"]) for m in body["top_models"]] == [("gpt-5", 2, 20)] + tasks = _call(table, "metric=requests", path="/model-insights/tasks").json()["tasks"] + assert sorted((t["task_type"], t["value"]) for t in tasks) == [("debugging", 1), ("uncategorized", 1)] + assert [(d["date"], d["requests"]) for d in body["daily"]] == [("2026-09-28", 2)] diff --git a/ui/litellm-dashboard/src/app/(dashboard)/legacyPageRoutes.ts b/ui/litellm-dashboard/src/app/(dashboard)/legacyPageRoutes.ts index cf943b331b9..bfc1b1ba4a8 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/legacyPageRoutes.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/legacyPageRoutes.ts @@ -35,6 +35,7 @@ const LEGACY_PAGE_ROUTES: ReadonlyMap = new Map( new_usage: "usage", usage: "old-usage", "cost-optimization": "cost-optimization", + "model-insights": "model-insights", agents: "agents", "router-settings": "router-settings", users: "users", diff --git a/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.test.tsx new file mode 100644 index 00000000000..67e57a794a1 --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.test.tsx @@ -0,0 +1,148 @@ +import { render, screen, waitFor } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import type React from "react"; +import { beforeEach, describe, expect, it, vi } from "vitest"; + +import ModelInsightsView from "./ModelInsightsView"; +import { apiClient } from "@/components/networking"; + +vi.mock("@/components/networking", () => ({ apiClient: { get: vi.fn() } })); +vi.mock("@/components/ui/chart", () => ({ + ChartContainer: ({ children }: { children: React.ReactNode }) =>
{children}
, + ChartTooltip: () => null, + ChartTooltipContent: () => null, +})); +vi.mock("recharts", () => ({ + Bar: () => null, + BarChart: ({ children }: { children: React.ReactNode }) =>
{children}
, + CartesianGrid: () => null, + Treemap: () => null, + XAxis: () => null, + YAxis: () => null, +})); + +const metrics = { + model_group: "fast-chat", + model: "openai/gpt-5.4-mini", + provider: "openai", + spend: 2.5, + prompt_tokens: 1000, + completion_tokens: 2000, + requests: 12, + successful_requests: 12, + failed_requests: 0, +}; + +const response = { + start_date: "2025-09-29", + end_date: "2026-09-28", + top_models: [metrics], + daily: [{ ...metrics, date: "2026-09-28" }], +}; + +const taskResponse = { + start_date: "2025-09-29", + end_date: "2026-09-28", + tasks: [ + { + task_type: "code_generation", + label: "Code Generation", + category: "Code", + value: 2.5, + share: 100, + leader: "fast-chat", + provider: "openai", + }, + ], +}; + +const mockApi = (tasks: unknown = taskResponse) => + vi + .mocked(apiClient.get) + .mockImplementation((path: string) => + path === "/model-insights/tasks" ? (tasks as Promise) : Promise.resolve(response), + ); + +describe("ModelInsightsView", () => { + beforeEach(() => { + vi.mocked(apiClient.get).mockReset(); + mockApi(Promise.resolve(taskResponse)); + }); + + it("shows the ranking with share and the task legend from the API response", async () => { + render(); + + expect(await screen.findByText("fast-chat")).toBeInTheDocument(); + expect(screen.getByText("by openai")).toBeInTheDocument(); + expect(await screen.findByText("Code")).toBeInTheDocument(); + expect(screen.getAllByText("100.0%")).toHaveLength(2); + expect(screen.getByRole("tab", { name: "tokens" })).toHaveAttribute("aria-selected", "true"); + expect(screen.getByRole("tab", { name: "log" })).toBeInTheDocument(); + expect(apiClient.get).toHaveBeenCalledWith("/model-insights", { + accessToken: "token", + query: { metric: "tokens" }, + }); + }); + + it("refetches with the selected metric so top models are ranked by it", async () => { + render(); + await screen.findByText("fast-chat"); + + await userEvent.click(screen.getByRole("tab", { name: "requests" })); + + await waitFor(() => + expect(apiClient.get).toHaveBeenCalledWith("/model-insights", { + accessToken: "token", + query: { metric: "requests" }, + }), + ); + }); + + it("does not refetch the task breakdown when the chart metric changes", async () => { + render(); + await screen.findByText("Code"); + const taskCalls = () => + vi.mocked(apiClient.get).mock.calls.filter(([path]) => path === "/model-insights/tasks").length; + const before = taskCalls(); + + await userEvent.click(screen.getByRole("tab", { name: "requests" })); + await waitFor(() => + expect(apiClient.get).toHaveBeenCalledWith("/model-insights", { + accessToken: "token", + query: { metric: "requests" }, + }), + ); + + expect(taskCalls()).toBe(before); + }); + + it("shows the API error instead of loading forever", async () => { + vi.mocked(apiClient.get).mockRejectedValue(new Error("Only proxy admins can view deployment-wide model insights")); + render(); + + expect(await screen.findByText("Could not load model insights")).toBeInTheDocument(); + expect(screen.getByText("Only proxy admins can view deployment-wide model insights")).toBeInTheDocument(); + }); + + it("keeps the previous ranking, dimmed, until the new metric's data arrives", async () => { + render(); + await screen.findByText("fast-chat"); + let resolve: (value: typeof response) => void = () => {}; + vi.mocked(apiClient.get).mockImplementation((path: string) => + path === "/model-insights/tasks" + ? Promise.resolve(taskResponse) + : new Promise((done) => (resolve = done as typeof resolve)), + ); + + await userEvent.click(screen.getByRole("tab", { name: "spend" })); + + expect( + screen.getByText("Share of tokens, with the change between the first and second half of the period"), + ).toBeInTheDocument(); + + resolve(response); + expect( + await screen.findByText("Share of spend, with the change between the first and second half of the period"), + ).toBeInTheDocument(); + }); +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.tsx b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.tsx new file mode 100644 index 00000000000..5f195c9383b --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/ModelInsightsView.tsx @@ -0,0 +1,352 @@ +"use client"; + +import React from "react"; +import { Bar, BarChart, CartesianGrid, Treemap, XAxis, YAxis } from "recharts"; +import { ArrowDownRight, ArrowUpRight, BarChart3, Layers, Minus } from "lucide-react"; + +import { apiClient } from "@/components/networking"; +import { extractErrorMessage } from "@/utils/errorUtils"; +import { ProviderLogo } from "@/components/molecules/models/ProviderLogo"; +import { PageHeader } from "@/components/shared/PageHeader"; +import { Alert, AlertDescription, AlertTitle } from "@/components/ui/alert"; +import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card"; +import { ChartConfig, ChartContainer, ChartTooltip, ChartTooltipContent } from "@/components/ui/chart"; +import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select"; +import { Skeleton } from "@/components/ui/skeleton"; +import { Tabs, TabsList, TabsTrigger } from "@/components/ui/tabs"; +import { + buildWeeklySeries, + formatMetric, + Metric, + ModelInsightsResponse, + ModelInsightTasksResponse, + TaskSummary, + modelOrder, + rankModels, + RankedModel, +} from "./modelInsightsData"; + +const PALETTE = [ + "#ec4899", + "#a855f7", + "#f59e0b", + "#3b82f6", + "#10b981", + "#ef4444", + "#14b8a6", + "#84cc16", + "#6366f1", + "#f97316", +]; +const FALLBACK_COLOR = "#64748b"; +const CATEGORY_COLORS: Record = { + General: "#ee8650", + Agent: "#7666e4", + Code: "#5fb074", + Data: "#3b82f6", +}; +const SCALES = ["linear", "log"] as const; +const METRIC_LABELS: Record = { requests: "requests", spend: "spend", tokens: "tokens" }; +const RANKING_ROWS = 5; + +type Scale = (typeof SCALES)[number]; + +const formatDelta = (value: number) => `${value > 0 ? "+" : ""}${value.toFixed(1)}`; + +const DeltaBadge = ({ value }: { value: number }) => { + if (Math.abs(value) < 0.05) { + return ( + + 0.0 + + ); + } + const up = value > 0; + const Icon = up ? ArrowUpRight : ArrowDownRight; + return ( + + {formatDelta(value)} + + ); +}; + +const RankingRow = ({ model, rank }: { model: RankedModel; rank: number }) => ( +
  • + {rank} + +
    +

    {model.model_group}

    +

    by {model.provider}

    +
    +
    +

    {model.share.toFixed(1)}%

    + +
    +
  • +); + +type TileProps = TaskSummary & { x: number; y: number; width: number; height: number; index: number }; + +const TaskTileContent = ({ x, y, width, height, category, label, leader }: TileProps) => { + if (width <= 0 || height <= 0) return null; + const color = CATEGORY_COLORS[category] ?? FALLBACK_COLOR; + const fits = width > 90 && height > 44; + return ( + + + {fits && ( + <> + + {label} + + + {leader} + + + )} + + ); +}; + +export default function ModelInsightsView({ accessToken }: { accessToken: string | null }) { + const [loaded, setLoaded] = React.useState<{ metric: Metric; response: ModelInsightsResponse } | null>(null); + const [metric, setMetric] = React.useState("tokens"); + const [scale, setScale] = React.useState("linear"); + const [taskMetric, setTaskMetric] = React.useState("spend"); + const [taskData, setTaskData] = React.useState(null); + const [taskError, setTaskError] = React.useState(null); + const [error, setError] = React.useState(null); + + React.useEffect(() => { + if (!accessToken) return; + let cancelled = false; + apiClient + .get("/model-insights", { accessToken, query: { metric } }) + .then((response) => { + if (cancelled) return; + setError(null); + setLoaded({ metric, response }); + }) + .catch((err: unknown) => { + if (!cancelled) setError(extractErrorMessage(err)); + }); + return () => { + cancelled = true; + }; + }, [accessToken, metric]); + + React.useEffect(() => { + if (!accessToken) return; + let cancelled = false; + apiClient + .get("/model-insights/tasks", { accessToken, query: { metric: taskMetric } }) + .then((response) => { + if (cancelled) return; + setTaskError(null); + setTaskData(response); + }) + .catch((err: unknown) => { + if (!cancelled) setTaskError(extractErrorMessage(err)); + }); + return () => { + cancelled = true; + }; + }, [accessToken, taskMetric]); + + const data = loaded?.response ?? null; + const shown = loaded?.metric ?? metric; + const isStale = loaded !== null && loaded.metric !== metric; + const range = React.useMemo(() => ({ start: data?.start_date ?? "", end: data?.end_date ?? "" }), [data]); + const models = React.useMemo(() => (data ? modelOrder(data.daily, shown) : []), [data, shown]); + const series = React.useMemo( + () => (data ? buildWeeklySeries(data.daily, models, shown, range) : []), + [data, models, shown, range], + ); + const ranking = React.useMemo( + () => (data ? rankModels(data.top_models, data.daily, shown, range) : []), + [data, shown, range], + ); + const tiles = React.useMemo(() => taskData?.tasks ?? [], [taskData]); + const categoryShares = React.useMemo( + () => + [...new Set(tiles.map((tile) => tile.category))].map((category) => ({ + category, + share: tiles.filter((tile) => tile.category === category).reduce((sum, tile) => sum + tile.share, 0), + })), + [tiles], + ); + + if (error) { + return ( +
    + + Could not load model insights + {error} + +
    + ); + } + + if (!data) { + return ( +
    + + +
    + ); + } + + const chartConfig = Object.fromEntries( + models.map((model, index) => [model, { label: model, color: PALETTE[index % PALETTE.length] }]), + ) satisfies ChartConfig; + + return ( +
    + } + title="Model Leaderboard" + subtitle={`See which models your gateway used from ${data.start_date} through ${data.end_date}`} + /> + + + +
    + Top models + Weekly {METRIC_LABELS[shown]} across your gateway +
    +
    + setMetric(value as Metric)}> + + {(["requests", "spend", "tokens"] as const).map((value) => ( + + {value} + + ))} + + + setScale(value as Scale)}> + + {SCALES.map((value) => ( + + {value} + + ))} + + +
    +
    + + + + + + formatMetric(Number(value), shown)} + /> + } /> + {models.map((model, index) => ( + + ))} + + + +
    + + + + Leaderboard + + Share of {METRIC_LABELS[shown]}, with the change between the first and second half of the period + + + +
      + {ranking.slice(0, RANKING_ROWS).map((model, index) => ( + + ))} +
    +
      + {ranking.slice(RANKING_ROWS).map((model, index) => ( + + ))} +
    +
    +
    + + + +
    + + Top models by task + + + Each task's share of {METRIC_LABELS[taskMetric]}, labelled with its leading model + +
    + +
    + + {taskError && ( + + Could not load tasks + {taskError} + + )} + + ({ ...tile, name: tile.task_type }))} + dataKey="value" + isAnimationActive={false} + content={} + /> + +
      + {categoryShares.map(({ category, share }) => ( +
    • + + {category} + {share.toFixed(1)}% +
    • + ))} +
    +
    +
    + + + + Cost per session + Session cost is not estimated from request counts + + +

    + Add a stable session_id to requests to unlock accurate session-level model comparisons in a future bounded + session rollup +

    +
    +
    +
    + ); +} diff --git a/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.test.ts b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.test.ts new file mode 100644 index 00000000000..e23196bad68 --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.test.ts @@ -0,0 +1,92 @@ +import { describe, expect, it } from "vitest"; + +import { buildWeeklySeries, DailyMetric, formatMetric, modelOrder, rankModels } from "./modelInsightsData"; + +const row = (over: Partial): DailyMetric => ({ + model_group: "a", + model: "a", + provider: "openai", + date: "2026-01-01", + spend: 0, + prompt_tokens: 0, + completion_tokens: 0, + requests: 0, + successful_requests: 0, + failed_requests: 0, + ...over, +}); + +describe("buildWeeklySeries", () => { + const range = { start: "2026-01-01", end: "2026-01-15" }; + + it("sums days into 7-day buckets per model", () => { + const rows = [ + row({ date: "2026-01-01", requests: 1 }), + row({ date: "2026-01-07", requests: 2 }), + row({ date: "2026-01-08", requests: 4 }), + row({ date: "2026-01-02", model_group: "b", requests: 8 }), + ]; + expect(buildWeeklySeries(rows, ["a", "b"], "requests", range)).toEqual([ + { date: "2026-01-01", a: 3, b: 8 }, + { date: "2026-01-08", a: 4, b: 0 }, + { date: "2026-01-15", a: 0, b: 0 }, + ]); + }); + + it("keeps weeks with no usage as zero instead of dropping them", () => { + const rows = [row({ date: "2026-01-01", requests: 1 }), row({ date: "2026-01-15", requests: 2 })]; + expect(buildWeeklySeries(rows, ["a"], "requests", range).map((week) => [week.date, week.a])).toEqual([ + ["2026-01-01", 1], + ["2026-01-08", 0], + ["2026-01-15", 2], + ]); + }); +}); + +describe("modelOrder", () => { + it("orders models by the selected metric, largest first", () => { + const rows = [row({ model_group: "a", spend: 1, requests: 9 }), row({ model_group: "b", spend: 5, requests: 1 })]; + expect(modelOrder(rows, "spend")).toEqual(["b", "a"]); + expect(modelOrder(rows, "requests")).toEqual(["a", "b"]); + }); +}); + +describe("rankModels", () => { + const range = { start: "2026-01-01", end: "2026-01-10" }; + const totals = [row({ model_group: "a", requests: 40 }), row({ model_group: "b", requests: 40 })]; + + it("computes share and the change in share between the first and second half of the range", () => { + const daily = [ + row({ date: "2026-01-01", model_group: "a", requests: 30 }), + row({ date: "2026-01-01", model_group: "b", requests: 10 }), + row({ date: "2026-01-10", model_group: "a", requests: 10 }), + row({ date: "2026-01-10", model_group: "b", requests: 30 }), + ]; + const ranked = rankModels(totals, daily, "requests", range); + expect(ranked.find((m) => m.model_group === "a")).toMatchObject({ share: 50, delta: -50 }); + expect(ranked.find((m) => m.model_group === "b")).toMatchObject({ share: 50, delta: 50 }); + }); + + it("splits at the middle of the range, not the middle of the days that had usage", () => { + const daily = [ + row({ date: "2026-01-01", model_group: "a", requests: 10 }), + row({ date: "2026-01-02", model_group: "b", requests: 10 }), + row({ date: "2026-01-03", model_group: "b", requests: 10 }), + ]; + const ranked = rankModels(totals, daily, "requests", range); + expect(ranked.find((m) => m.model_group === "a")?.delta).toBe(0); + }); + + it("shows no change when one half of the range has no usage to compare against", () => { + const daily = [row({ date: "2026-01-10", model_group: "a", requests: 10 })]; + const ranked = rankModels(totals, daily, "requests", range); + expect(ranked.map((m) => m.delta)).toEqual([0, 0]); + }); +}); + +describe("formatMetric", () => { + it("formats spend as currency and counts compactly", () => { + expect(formatMetric(12.5, "spend")).toBe("$12.50"); + expect(formatMetric(1_500_000, "tokens")).toBe("1.5M"); + }); +}); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.ts b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.ts new file mode 100644 index 00000000000..3cadf0dc95c --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/_components/modelInsightsData.ts @@ -0,0 +1,132 @@ +export type Metric = "requests" | "spend" | "tokens"; + +export type ModelMetric = { + model_group: string; + model: string; + provider: string; + spend: number; + prompt_tokens: number; + completion_tokens: number; + requests: number; + successful_requests: number; + failed_requests: number; +}; +export type DailyMetric = ModelMetric & { date: string }; +export type ModelInsightsResponse = { + start_date: string; + end_date: string; + daily: DailyMetric[]; + top_models: ModelMetric[]; +}; +export type TaskSummary = { + task_type: string; + label: string; + category: string; + value: number; + share: number; + leader: string; + provider: string; +}; +export type ModelInsightTasksResponse = { start_date: string; end_date: string; tasks: TaskSummary[] }; + +export type RankedModel = { model_group: string; provider: string; share: number; delta: number }; +const DAY_MS = 86_400_000; +const WEEK_DAYS = 7; + +export const metricValue = (row: ModelMetric, metric: Metric) => { + if (metric === "requests") return row.requests; + if (metric === "spend") return row.spend; + return row.prompt_tokens + row.completion_tokens; +}; + +const COMPACT_SPEND_FROM = 10_000; + +export const formatMetric = (value: number, metric: Metric) => { + if (metric === "spend") { + const compact = value >= COMPACT_SPEND_FROM; + const options: Intl.NumberFormatOptions = { + style: "currency", + currency: "USD", + notation: compact ? "compact" : "standard", + maximumFractionDigits: compact ? 1 : 2, + }; + return new Intl.NumberFormat("en-US", options).format(value); + } + return new Intl.NumberFormat("en-US", { notation: "compact", maximumFractionDigits: 1 }).format(value); +}; + +const toDay = (date: string) => Date.parse(`${date}T00:00:00Z`); +const isoDay = (ms: number) => new Date(ms).toISOString().slice(0, 10); + +export type DateRange = { start: string; end: string }; + +export const modelOrder = (rows: DailyMetric[], metric: Metric) => { + const totals = new Map(); + for (const row of rows) totals.set(row.model_group, (totals.get(row.model_group) ?? 0) + metricValue(row, metric)); + return [...totals.entries()].sort((a, b) => b[1] - a[1]).map(([model]) => model); +}; + +export const buildWeeklySeries = (rows: DailyMetric[], models: string[], metric: Metric, range: DateRange) => { + const weekMs = WEEK_DAYS * DAY_MS; + const origin = toDay(range.start); + const weekCount = Math.floor((toDay(range.end) - origin) / weekMs) + 1; + const buckets = Array.from({ length: weekCount }, (_, week) => ({ + date: isoDay(origin + week * weekMs), + ...Object.fromEntries(models.map((model) => [model, 0])), + })) as Record[]; + for (const row of rows) { + const bucket = buckets[Math.floor((toDay(row.date) - origin) / weekMs)]; + if (bucket) bucket[row.model_group] = Number(bucket[row.model_group] ?? 0) + metricValue(row, metric); + } + return buckets; +}; + +const shareByModel = (rows: { model_group: string; provider: string }[], values: number[]) => { + const totals = new Map(); + rows.forEach((row, index) => { + const current = totals.get(row.model_group) ?? { provider: row.provider, value: 0 }; + totals.set(row.model_group, { provider: row.provider, value: current.value + values[index] }); + }); + const grand = [...totals.values()].reduce((sum, entry) => sum + entry.value, 0); + return { totals, grand }; +}; + +const halfShares = (daily: DailyMetric[], metric: Metric, range: DateRange) => { + const midpoint = isoDay(toDay(range.start) + Math.floor((toDay(range.end) - toDay(range.start)) / 2 + DAY_MS / 2)); + const share = (rows: DailyMetric[]) => { + const { totals, grand } = shareByModel( + rows, + rows.map((row) => metricValue(row, metric)), + ); + return { + hasUsage: grand > 0, + of: (model: string) => (grand === 0 ? 0 : ((totals.get(model)?.value ?? 0) / grand) * 100), + }; + }; + return { + earlier: share(daily.filter((row) => row.date < midpoint)), + later: share(daily.filter((row) => row.date >= midpoint)), + }; +}; + +export const rankModels = ( + rows: ModelMetric[], + daily: DailyMetric[], + metric: Metric, + range: DateRange, +): RankedModel[] => { + const { totals, grand } = shareByModel( + rows, + rows.map((row) => metricValue(row, metric)), + ); + const { earlier, later } = halfShares(daily, metric, range); + const comparable = earlier.hasUsage && later.hasUsage; + return [...totals.entries()] + .sort((a, b) => b[1].value - a[1].value) + .map(([model_group, entry]) => ({ + model_group, + provider: entry.provider, + share: grand === 0 ? 0 : (entry.value / grand) * 100, + delta: comparable ? later.of(model_group) - earlier.of(model_group) : 0, + })); +}; diff --git a/ui/litellm-dashboard/src/app/(dashboard)/model-insights/page.tsx b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/page.tsx new file mode 100644 index 00000000000..01a673ce10f --- /dev/null +++ b/ui/litellm-dashboard/src/app/(dashboard)/model-insights/page.tsx @@ -0,0 +1,9 @@ +"use client"; + +import useAuthorized from "@/app/(dashboard)/hooks/useAuthorized"; +import ModelInsightsView from "./_components/ModelInsightsView"; + +export default function ModelInsightsPage() { + const { accessToken } = useAuthorized(); + return ; +} diff --git a/ui/litellm-dashboard/src/components/leftnav.tsx b/ui/litellm-dashboard/src/components/leftnav.tsx index 9d772f45153..824e8e14a1c 100644 --- a/ui/litellm-dashboard/src/components/leftnav.tsx +++ b/ui/litellm-dashboard/src/components/leftnav.tsx @@ -204,6 +204,17 @@ const menuGroups: MenuGroup[] = [ roles: [...all_admin_roles, ...internalUserRoles], label: "Usage", }, + { + key: "model-insights", + page: "model-insights", + icon: , + roles: all_admin_roles, + label: ( + + Model Leaderboard + + ), + }, { key: "cost-optimization", page: "cost-optimization", diff --git a/ui/litellm-dashboard/src/components/page_metadata.ts b/ui/litellm-dashboard/src/components/page_metadata.ts index 0f2ff639bb3..6f76be5fb07 100644 --- a/ui/litellm-dashboard/src/components/page_metadata.ts +++ b/ui/litellm-dashboard/src/components/page_metadata.ts @@ -20,6 +20,7 @@ export const pageDescriptions: Record = { "vector-stores": "Manage vector databases for embeddings", new_usage: "View usage analytics and metrics", "cost-optimization": "Track and configure cost-saving features: prompt compression, caching, and auto routing", + "model-insights": "Model Leaderboard: compare usage, spend, tokens, and task mix across this gateway", logs: "Access request and response logs", "guardrails-monitor": "Monitor guardrail performance and view logs", users: "Manage internal user accounts and permissions", diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 6b4087a4664..93f4718a586 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -9189,6 +9189,40 @@ export interface paths { patch: operations["mistral_proxy_route_mistral__endpoint__patch"]; trace?: never; }; + "/model-insights": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** Get Model Insights */ + get: operations["get_model_insights_model_insights_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/model-insights/tasks": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** Get Model Insight Tasks */ + get: operations["get_model_insight_tasks_model_insights_tasks_get"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/model/block": { parameters: { query?: never; @@ -35974,6 +36008,87 @@ export interface components { /** Id */ id: string; }; + /** ModelInsightDailyMetric */ + ModelInsightDailyMetric: { + /** Completion Tokens */ + completion_tokens: number; + /** Date */ + date: string; + /** Failed Requests */ + failed_requests: number; + /** Model */ + model: string; + /** Model Group */ + model_group: string; + /** Prompt Tokens */ + prompt_tokens: number; + /** Provider */ + provider: string; + /** Requests */ + requests: number; + /** Spend */ + spend: number; + /** Successful Requests */ + successful_requests: number; + }; + /** ModelInsightMetric */ + ModelInsightMetric: { + /** Completion Tokens */ + completion_tokens: number; + /** Failed Requests */ + failed_requests: number; + /** Model */ + model: string; + /** Model Group */ + model_group: string; + /** Prompt Tokens */ + prompt_tokens: number; + /** Provider */ + provider: string; + /** Requests */ + requests: number; + /** Spend */ + spend: number; + /** Successful Requests */ + successful_requests: number; + }; + /** ModelInsightTaskSummary */ + ModelInsightTaskSummary: { + /** Category */ + category: string; + /** Label */ + label: string; + /** Leader */ + leader: string; + /** Provider */ + provider: string; + /** Share */ + share: number; + /** Task Type */ + task_type: string; + /** Value */ + value: number; + }; + /** ModelInsightTasksResponse */ + ModelInsightTasksResponse: { + /** End Date */ + end_date: string; + /** Start Date */ + start_date: string; + /** Tasks */ + tasks: components["schemas"]["ModelInsightTaskSummary"][]; + }; + /** ModelInsightsResponse */ + ModelInsightsResponse: { + /** Daily */ + daily: components["schemas"]["ModelInsightDailyMetric"][]; + /** End Date */ + end_date: string; + /** Start Date */ + start_date: string; + /** Top Models */ + top_models: components["schemas"]["ModelInsightMetric"][]; + }; /** ModelParams */ ModelParams: { /** Litellm Params */ @@ -59332,6 +59447,78 @@ export interface operations { }; }; }; + get_model_insights_model_insights_get: { + parameters: { + query?: { + /** @description YYYY-MM-DD, defaults to 365 days ago */ + start_date?: string | null; + /** @description YYYY-MM-DD, defaults to today */ + end_date?: string | null; + /** @description Metric the top models are ranked by */ + metric?: "requests" | "spend" | "tokens"; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["ModelInsightsResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; + get_model_insight_tasks_model_insights_tasks_get: { + parameters: { + query?: { + /** @description YYYY-MM-DD, defaults to 365 days ago */ + start_date?: string | null; + /** @description YYYY-MM-DD, defaults to today */ + end_date?: string | null; + /** @description Metric task shares are computed from */ + metric?: "requests" | "spend" | "tokens"; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Successful Response */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["ModelInsightTasksResponse"]; + }; + }; + /** @description Validation Error */ + 422: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["HTTPValidationError"]; + }; + }; + }; + }; block_model_model_block_post: { parameters: { query?: never;