diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 9eaf6c7e9ed..147174c73d0 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -4016,6 +4016,13 @@ }, "metrics": { "$ref": "#/components/schemas/SpendMetrics" + }, + "provider_breakdown": { + "additionalProperties": { + "$ref": "#/components/schemas/ProviderThroughputMetrics" + }, + "title": "Provider Breakdown", + "type": "object" } }, "required": [ @@ -4343,6 +4350,38 @@ "title": "PatchAgentRequest", "type": "object" }, + "ProviderThroughputMetrics": { + "properties": { + "completion_tokens": { + "default": 0, + "title": "Completion Tokens", + "type": "integer" + }, + "output_tokens_per_second": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Output Tokens Per Second" + }, + "timed_requests": { + "default": 0, + "title": "Timed Requests", + "type": "integer" + }, + "total_response_time_ms": { + "default": 0, + "title": "Total Response Time Ms", + "type": "integer" + } + }, + "title": "ProviderThroughputMetrics", + "type": "object" + }, "SpendAnalyticsPaginatedResponse": { "properties": { "metadata": { diff --git a/litellm/proxy/management_endpoints/common_daily_activity.py b/litellm/proxy/management_endpoints/common_daily_activity.py index 28dfbb09eab..36ce7d0da5e 100644 --- a/litellm/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/proxy/management_endpoints/common_daily_activity.py @@ -31,6 +31,7 @@ from litellm.types.proxy.management_endpoints.common_daily_activity import ( KeyMetadata, KeyMetricWithMetadata, MetricWithMetadata, + ProviderThroughputMetrics, SpendAnalyticsPaginatedResponse, SpendMetrics, ) @@ -236,6 +237,35 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) -> return existing_metrics +def _provider_throughput( + completion_tokens: int, + total_response_time_ms: int, + timed_requests: int, +) -> ProviderThroughputMetrics: + output_tokens_per_second: Final = ( + completion_tokens * 1000 / total_response_time_ms if timed_requests > 0 and total_response_time_ms > 0 else None + ) + return ProviderThroughputMetrics( + completion_tokens=completion_tokens, + total_response_time_ms=total_response_time_ms, + timed_requests=timed_requests, + output_tokens_per_second=output_tokens_per_second, + ) + + +def _update_provider_throughput( + target: MetricWithMetadata, + provider: str, + record: DailySpendRecord, +) -> None: + existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics()) + target.provider_breakdown[provider] = _provider_throughput( + completion_tokens=existing.completion_tokens + (record.completion_tokens or 0), + total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0), + timed_requests=existing.timed_requests + (record.timed_requests or 0), + ) + + def _is_user_agent_tag(tag: str | None) -> bool: """Determine whether a tag should be treated as a User-Agent tag.""" if not tag: @@ -312,6 +342,12 @@ def update_breakdown_metrics( breakdown.models[model_key].metrics = update_metrics(breakdown.models[model_key].metrics, record) if not is_ptu_sentinel: + _update_provider_throughput( + breakdown.models[model_key], + record.custom_llm_provider or "unknown", + record, + ) + # Update API key breakdown for this model if record.api_key not in breakdown.models[model_key].api_key_breakdown: breakdown.models[model_key].api_key_breakdown[record.api_key] = KeyMetricWithMetadata( @@ -336,6 +372,12 @@ def update_breakdown_metrics( ) if not is_ptu_sentinel: + _update_provider_throughput( + breakdown.model_groups[model_group_key], + record.custom_llm_provider or "unknown", + record, + ) + # Update API key breakdown for this model if record.api_key not in breakdown.model_groups[model_group_key].api_key_breakdown: breakdown.model_groups[model_group_key].api_key_breakdown[record.api_key] = KeyMetricWithMetadata( @@ -697,8 +739,10 @@ _API_KEY_ROLLED_UP_BIT: Final = 32 # 0b0100000 _GROUP_DATE_API_KEY: Final = 31 # 0b0011111 _GROUP_DATE_MODEL: Final = 47 # 0b0101111 _GROUP_DATE_MODEL_API_KEY: Final = 15 # 0b0001111 +_GROUP_DATE_MODEL_PROVIDER: Final = 43 # 0b0101011 _GROUP_DATE_MODEL_GROUP: Final = 55 # 0b0110111 _GROUP_DATE_MODEL_GROUP_API_KEY: Final = 23 # 0b0010111 +_GROUP_DATE_MODEL_GROUP_PROVIDER: Final = 51 # 0b0110011 _GROUP_DATE_PROVIDER: Final = 59 # 0b0111011 _GROUP_DATE_PROVIDER_API_KEY: Final = 27 # 0b0011011 _GROUP_DATE_MCP: Final = 61 # 0b0111101 @@ -779,6 +823,32 @@ def _aggregate_grouping_sets_records_sync( metrics=metrics, metadata=_key_metadata(api_key_metadata, api_key) ) + def assign_provider_breakdown( + target: dict[str, MetricWithMetadata], + parent_key: str, + provider: str, + metrics: SpendMetrics, + ) -> None: + parent: Final = target.get(parent_key) + if parent is None: + target[parent_key] = MetricWithMetadata( + metrics=SpendMetrics(), + metadata={}, + provider_breakdown={ + provider: _provider_throughput( + metrics.completion_tokens, + metrics.total_response_time_ms, + metrics.timed_requests, + ) + }, + ) + return + parent.provider_breakdown[provider] = _provider_throughput( + metrics.completion_tokens, + metrics.total_response_time_ms, + metrics.timed_requests, + ) + for record in records: level = record.group_level metrics = _record_to_spend_metrics(record) @@ -806,6 +876,14 @@ def _aggregate_grouping_sets_records_sync( elif level == _GROUP_DATE_MODEL_API_KEY: if record.model and record.api_key and not is_ptu_sentinel: assign_api_key_breakdown(breakdown.models, record.model, record.api_key, metrics) + elif level == _GROUP_DATE_MODEL_PROVIDER: + if record.model: + assign_provider_breakdown( + breakdown.models, + record.model, + record.custom_llm_provider or "unknown", + metrics, + ) elif level == _GROUP_DATE_MODEL_GROUP: if record.model_group: assign_metric_with_metadata(breakdown.model_groups, record.model_group, metrics) @@ -817,6 +895,14 @@ def _aggregate_grouping_sets_records_sync( record.api_key, metrics, ) + elif level == _GROUP_DATE_MODEL_GROUP_PROVIDER: + if record.model_group: + assign_provider_breakdown( + breakdown.model_groups, + record.model_group, + record.custom_llm_provider or "unknown", + metrics, + ) elif level == _GROUP_DATE_PROVIDER: # Only PTU sentinel rows carry ptu_flat_cost and they have no provider, so at # this level the sentinel's cost would land under "unknown". Withholding the diff --git a/litellm/repositories/daily_activity_sql.py b/litellm/repositories/daily_activity_sql.py index f12dc1be5ae..7495bb6abbf 100644 --- a/litellm/repositories/daily_activity_sql.py +++ b/litellm/repositories/daily_activity_sql.py @@ -177,7 +177,9 @@ def build_aggregated_sql(scope: DailyActivityScope, *, api_key_limit: int) -> Sq GROUP BY GROUPING SETS ( (date), (date, model), + (date, model, custom_llm_provider), (date, {_MODEL_GROUP_EXPR}), + (date, {_MODEL_GROUP_EXPR}, custom_llm_provider), (date, custom_llm_provider), (date, mcp_namespaced_tool_name), (date, endpoint), diff --git a/litellm/types/proxy/management_endpoints/common_daily_activity.py b/litellm/types/proxy/management_endpoints/common_daily_activity.py index 37804032569..4fc27a03a99 100644 --- a/litellm/types/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/types/proxy/management_endpoints/common_daily_activity.py @@ -56,10 +56,18 @@ class KeyMetricWithMetadata(MetricBase): metadata: KeyMetadata = Field(default_factory=KeyMetadata) +class ProviderThroughputMetrics(BaseModel): + completion_tokens: int = Field(default=0) + total_response_time_ms: int = Field(default=0) + timed_requests: int = Field(default=0) + output_tokens_per_second: float | None = Field(default=None) + + class MetricWithMetadata(MetricBase): metadata: dict[str, Any] = Field(default_factory=dict) # API key breakdown for this metric (e.g., which API keys are using this MCP server) api_key_breakdown: dict[str, KeyMetricWithMetadata] = Field(default_factory=dict) # api_key -> {metrics, metadata} + provider_breakdown: dict[str, ProviderThroughputMetrics] = Field(default_factory=dict) class BreakdownMetrics(BaseModel): diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index e6a6680d3e4..c14e6e098d6 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -1675,6 +1675,9 @@ def _grouping_row( endpoint=None, spend=0.0, ptu_flat_cost=0.0, + completion_tokens=0, + total_response_time_ms=0, + timed_requests=0, ): return GroupingSetsRow( date="2024-01-01", @@ -1689,7 +1692,7 @@ def _grouping_row( spend=spend, ptu_flat_cost=ptu_flat_cost, prompt_tokens=0, - completion_tokens=0, + completion_tokens=completion_tokens, cache_read_input_tokens=0, cache_creation_input_tokens=0, compression_saved_tokens=0, @@ -1697,8 +1700,8 @@ def _grouping_row( prompt_caching_savings_spend=0.0, gateway_injected_caching_savings_spend=0.0, autorouter_savings_spend=0.0, - total_response_time_ms=0, - timed_requests=0, + total_response_time_ms=total_response_time_ms, + timed_requests=timed_requests, api_requests=0, successful_requests=0, failed_requests=0, @@ -1794,6 +1797,73 @@ def test_grouping_sets_dispatcher_populates_every_breakdown_level(ptu_cost_attri assert "real-key" in day.breakdown.endpoints["/v1/chat/completions"].api_key_breakdown +def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_model_groups(): + from litellm.proxy.management_endpoints.common_daily_activity import ( + _GROUP_DATE_MODEL_GROUP_PROVIDER, + _GROUP_DATE_MODEL_PROVIDER, + _aggregate_grouping_sets_records_sync, + ) + + records = [ + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + custom_llm_provider="openai", + completion_tokens=900, + total_response_time_ms=3000, + timed_requests=3, + ), + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + custom_llm_provider="azure", + completion_tokens=400, + total_response_time_ms=2000, + timed_requests=2, + ), + _grouping_row( + _GROUP_DATE_MODEL_GROUP_PROVIDER, + model_group="public-gpt-4o", + custom_llm_provider="openai", + completion_tokens=900, + total_response_time_ms=3000, + timed_requests=3, + ), + ] + + day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] + + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].model_dump() == { + "completion_tokens": 900, + "total_response_time_ms": 3000, + "timed_requests": 3, + "output_tokens_per_second": 300.0, + } + assert day.breakdown.models["gpt-4o"].provider_breakdown["azure"].output_tokens_per_second == 200.0 + assert day.breakdown.model_groups["public-gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 300.0 + + +def test_grouping_sets_dispatcher_returns_no_throughput_without_positive_duration(): + from litellm.proxy.management_endpoints.common_daily_activity import ( + _GROUP_DATE_MODEL_PROVIDER, + _aggregate_grouping_sets_records_sync, + ) + + records = [ + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + completion_tokens=900, + total_response_time_ms=0, + timed_requests=1, + ) + ] + + day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] + + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None + + def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown(): """Sentinel rows carry no provider, so their flat cost must not surface under the "unknown" provider - the per-row path skips them for exactly the same reason.""" @@ -1851,7 +1921,7 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib endpoint="/v1/chat/completions", spend=5.0, prompt_tokens=0, - completion_tokens=0, + completion_tokens=600, cache_read_input_tokens=0, cache_creation_input_tokens=0, compression_saved_tokens=0, @@ -1859,8 +1929,8 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib prompt_caching_savings_spend=0, gateway_injected_caching_savings_spend=0, autorouter_savings_spend=0, - total_response_time_ms=0, - timed_requests=0, + total_response_time_ms=2000, + timed_requests=2, total_tokens=0, api_requests=0, successful_requests=0, @@ -1874,6 +1944,8 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib assert "real-key" in breakdown.mcp_servers["srv/tool"].api_key_breakdown assert "/v1/chat/completions" in breakdown.endpoints assert "azure" in breakdown.providers + assert breakdown.models["gpt-4o-mini-ptu"].provider_breakdown["azure"].output_tokens_per_second == 300.0 + assert breakdown.model_groups["grp"].provider_breakdown["azure"].output_tokens_per_second == 300.0 assert "team-1" in breakdown.entities assert "real-key" in breakdown.entities["team-1"].api_key_breakdown diff --git a/tests/unit/repositories/test_daily_activity_sql.py b/tests/unit/repositories/test_daily_activity_sql.py index c775dfd1f88..d413615a0a1 100644 --- a/tests/unit/repositories/test_daily_activity_sql.py +++ b/tests/unit/repositories/test_daily_activity_sql.py @@ -243,6 +243,13 @@ def test_aggregate_query_sums_all_savings_drivers_and_response_time() -> None: assert all(f"SUM({field})" in query.sql for field in fields) +def test_aggregated_query_groups_models_and_model_groups_by_provider() -> None: + query = build_aggregated_sql(_scope(), api_key_limit=constants.USAGE_TOP_API_KEYS_DEFAULT) + + assert "(date, model, custom_llm_provider)" in query.sql + assert "(date, COALESCE(NULLIF(model_group, ''), model), custom_llm_provider)" in query.sql + + def test_aggregated_query_binds_sentinel_and_api_key_limit_after_scope_values() -> None: scope = _scope(entity_ids=None, api_keys=("key-1",)) diff --git a/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts b/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts index 36c2837d9ac..703f6353363 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts @@ -76,6 +76,7 @@ const toMetric = (entry: SchemaMetricWithMetadata): MetricWithMetadata => ({ api_key_breakdown: Object.fromEntries( Object.entries(entry.api_key_breakdown ?? {}).map(([key, value]) => [key, toKeyMetric(value)]), ), + provider_breakdown: entry.provider_breakdown ?? {}, }); const toMetricMap = ( diff --git a/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts b/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts index b35ea72896a..54145e3c9c6 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts @@ -56,6 +56,14 @@ const aggregatedResponse: DailyActivityAggregatedResponse = { metrics: completeMetrics, metadata: {}, api_key_breakdown: { "key-hash": apiKeyActivity }, + provider_breakdown: { + openai: { + completion_tokens: 900, + output_tokens_per_second: 300, + timed_requests: 3, + total_response_time_ms: 3000, + }, + }, }, }, }, @@ -105,6 +113,14 @@ describe("key activity data", () => { }, }, ]); + expect(toDailyData(aggregatedResponse)[0].breakdown.models["gpt-4o-mini"].provider_breakdown).toEqual({ + openai: { + completion_tokens: 900, + output_tokens_per_second: 300, + timed_requests: 3, + total_response_time_ms: 3000, + }, + }); }); it("appends pages without duplicate keys and compares the server offset to the total", () => { diff --git a/ui/litellm-dashboard/src/components/UsagePage/types.ts b/ui/litellm-dashboard/src/components/UsagePage/types.ts index 06f1856a5a6..41a09874050 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/types.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/types.ts @@ -38,6 +38,14 @@ export interface MetricWithMetadata { metrics: SpendMetrics; metadata: object; api_key_breakdown: { [key: string]: KeyMetricWithMetadata }; + provider_breakdown?: { [key: string]: ProviderThroughputMetrics }; +} + +export interface ProviderThroughputMetrics { + completion_tokens: number; + total_response_time_ms: number; + timed_requests: number; + output_tokens_per_second?: number | null; } export interface KeyMetricWithMetadata { @@ -92,6 +100,7 @@ export interface ModelActivityData { cache_creation_input_tokens: number; avg_response_time_ms?: number | null; }; + provider_throughput?: Record; }[]; } diff --git a/ui/litellm-dashboard/src/components/activity_metrics.test.tsx b/ui/litellm-dashboard/src/components/activity_metrics.test.tsx index 8ce18884718..f9fc2870aa3 100644 --- a/ui/litellm-dashboard/src/components/activity_metrics.test.tsx +++ b/ui/litellm-dashboard/src/components/activity_metrics.test.tsx @@ -1,7 +1,13 @@ import { fireEvent, render, screen, waitFor } from "@testing-library/react"; import React from "react"; import { beforeAll, describe, expect, it, vi } from "vitest"; -import { ActivityMetrics, formatKeyLabel, processActivityData, ResponseTimeTooltip } from "./activity_metrics"; +import { + ActivityMetrics, + formatKeyLabel, + processActivityData, + providerThroughputChartData, + ResponseTimeTooltip, +} from "./activity_metrics"; import type { ChartTooltipProps } from "@/components/shared/charts"; import { Team } from "./key_team_helpers/key_list"; import { DailyData, KeyMetricWithMetadata, ModelActivityData } from "./UsagePage/types"; @@ -1402,6 +1408,38 @@ describe("processActivityData", () => { expect(result["gpt-5.5"].total_timed_requests).toBe(0); expect(result["gpt-5.5"].daily_data[0].metrics.avg_response_time_ms).toBeNull(); }); + + it("preserves daily provider throughput for model and model-group views", () => { + const metric = { + metrics: EMPTY_SPEND_METRICS, + metadata: {}, + api_key_breakdown: {}, + provider_breakdown: { + openai: { + completion_tokens: 900, + total_response_time_ms: 3000, + timed_requests: 3, + output_tokens_per_second: 300, + }, + }, + }; + const activity: { results: DailyData[] } = { + results: [ + createMockDailyData("2025-01-01", EMPTY_SPEND_METRICS, { + ...EMPTY_BREAKDOWN, + models: { "gpt-4o": metric }, + model_groups: { "public-gpt-4o": metric }, + }), + ], + }; + + expect(processActivityData(activity, "models")["gpt-4o"].daily_data[0].provider_throughput).toEqual({ + openai: 300, + }); + expect(processActivityData(activity, "model_groups")["public-gpt-4o"].daily_data[0].provider_throughput).toEqual({ + openai: 300, + }); + }); }); describe("ActivityMetrics response time", () => { @@ -1482,6 +1520,58 @@ describe("ActivityMetrics response time", () => { }); }); +describe("ActivityMetrics provider throughput", () => { + const model = createMockModelActivityData("GPT-4o", { + daily_data: [ + { + ...createMockModelActivityData("GPT-4o").daily_data[0], + date: "2025-01-01", + provider_throughput: { openai: 300, azure: 200 }, + }, + { + ...createMockModelActivityData("GPT-4o").daily_data[0], + date: "2025-01-02", + provider_throughput: { openai: 250 }, + }, + ], + }); + + it("renders one daily line per provider", () => { + render(); + + expect(screen.getByText("Output tokens per second of response time")).toBeInTheDocument(); + expect(screen.getByText("Openai")).toBeInTheDocument(); + expect(screen.getByText("Azure")).toBeInTheDocument(); + expect(screen.getAllByText("2025-01-01").length).toBeGreaterThan(0); + expect(screen.getAllByText("2025-01-02").length).toBeGreaterThan(0); + }); + + it("keeps missing provider dates as gaps", () => { + expect(providerThroughputChartData(model.daily_data)).toEqual({ + providers: ["azure", "openai"], + data: [ + { date: "2025-01-01", azure: 200, openai: 300 }, + { date: "2025-01-02", azure: null, openai: 250 }, + ], + }); + }); + + it("hides the chart when every throughput value is unavailable", () => { + const unavailable = createMockModelActivityData("GPT-4o", { + daily_data: [ + { + ...createMockModelActivityData("GPT-4o").daily_data[0], + provider_throughput: { openai: null }, + }, + ], + }); + + render(); + + expect(screen.queryByText("Output tokens per second of response time")).not.toBeInTheDocument(); + }); +}); + describe("formatKeyLabel", () => { it("should return key_alias when no team_id is present", () => { const modelData = createMockKeyMetricWithMetadata({ diff --git a/ui/litellm-dashboard/src/components/activity_metrics.tsx b/ui/litellm-dashboard/src/components/activity_metrics.tsx index e7712a99c71..86091052eb8 100644 --- a/ui/litellm-dashboard/src/components/activity_metrics.tsx +++ b/ui/litellm-dashboard/src/components/activity_metrics.tsx @@ -4,6 +4,7 @@ import { type ChartTooltipProps, CustomLegend, CustomTooltip, + DEFAULT_COLOR_CYCLE, formatCategoryName, LineChart, ValueTooltip, @@ -18,7 +19,13 @@ import { Team } from "./key_team_helpers/key_list"; import KeyModelUsageView from "./UsagePage/components/KeyModelUsageView"; import { keyActivityLabel } from "./UsagePage/keyActivityLabel"; import type { ModelTopKeysResponse } from "./UsagePage/dailyActivityApi"; -import { DailyData, KeyMetricWithMetadata, ModelActivityData, TopModelData } from "./UsagePage/types"; +import { + DailyData, + KeyMetricWithMetadata, + MetricWithMetadata, + ModelActivityData, + TopModelData, +} from "./UsagePage/types"; import { averageResponseTimeMs, formatResponseTime, valueFormatter } from "./UsagePage/utils/value_formatters"; interface ActivityMetricsProps { @@ -41,6 +48,30 @@ export const ResponseTimeTooltip = ({ active, payload, label }: ChartTooltipProp /> ); +const formatTokensPerSecond = (value: number): string => + `${value.toLocaleString(undefined, { maximumFractionDigits: 2 })} tokens/s`; + +export const providerThroughputChartData = (dailyData: ModelActivityData["daily_data"]) => { + const providers = Array.from(new Set(dailyData.flatMap((day) => Object.keys(day.provider_throughput ?? {})))) + .filter((provider) => + dailyData.some((day) => { + const value = day.provider_throughput?.[provider]; + return typeof value === "number" && Number.isFinite(value); + }), + ) + .sort(); + const data = dailyData.map((day) => ({ + date: day.date, + ...Object.fromEntries( + providers.map((provider) => { + const value = day.provider_throughput?.[provider]; + return [provider, typeof value === "number" && Number.isFinite(value) ? value : null]; + }), + ), + })); + return { providers, data }; +}; + const ModelTopKeys = ({ modelName, fetchTopApiKeys, @@ -178,6 +209,8 @@ export const ModelSection = ({ hidePromptCachingMetrics?: boolean; fetchTopApiKeys?: (model: string) => Promise; }) => { + const throughputChart = providerThroughputChartData(metrics.daily_data); + return (
{/* Summary Cards */} @@ -316,6 +349,27 @@ export const ModelSection = ({ )} + {throughputChart.providers.length > 0 && ( + + +
+

Output tokens per second of response time

+ +
+ +
+
+ )} +
@@ -683,6 +737,13 @@ export const processActivityData = ( modelMetrics[model].total_response_time_ms = (modelMetrics[model].total_response_time_ms ?? 0) + dayResponseTimeMs; modelMetrics[model].total_timed_requests = (modelMetrics[model].total_timed_requests ?? 0) + dayTimedRequests; + const providerBreakdown = (modelData as MetricWithMetadata).provider_breakdown ?? {}; + const providerThroughput = Object.fromEntries( + Object.entries(providerBreakdown).map(([provider, providerMetrics]) => [ + provider, + providerMetrics.output_tokens_per_second ?? null, + ]), + ); // Add daily data modelMetrics[model].daily_data.push({ @@ -699,6 +760,7 @@ export const processActivityData = ( cache_creation_input_tokens: modelData.metrics.cache_creation_input_tokens || 0, avg_response_time_ms: averageResponseTimeMs(dayResponseTimeMs, dayTimedRequests), }, + ...(Object.keys(providerThroughput).length > 0 ? { provider_throughput: providerThroughput } : {}), }); }); }); diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 3f0629f05d6..f75f772e08c 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -37805,6 +37805,10 @@ export interface components { [key: string]: unknown; }; metrics: components["schemas"]["SpendMetrics"]; + /** Provider Breakdown */ + provider_breakdown?: { + [key: string]: components["schemas"]["ProviderThroughputMetrics"]; + }; }; /** Mode */ Mode: { @@ -41071,6 +41075,26 @@ export interface components { /** Tooltip */ tooltip?: string | null; }; + /** ProviderThroughputMetrics */ + ProviderThroughputMetrics: { + /** + * Completion Tokens + * @default 0 + */ + completion_tokens: number; + /** Output Tokens Per Second */ + output_tokens_per_second?: number | null; + /** + * Timed Requests + * @default 0 + */ + timed_requests: number; + /** + * Total Response Time Ms + * @default 0 + */ + total_response_time_ms: number; + }; /** * ProxyChatCompletionRequest * @description Pydantic model for chat completion requests that includes both OpenAI standard fields