mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
feat(usage): graph provider throughput by model
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
09313bcf1b
commit
5a9f1045ad
12 changed files with 424 additions and 8 deletions
|
|
@ -4016,6 +4016,13 @@
|
|||
},
|
||||
"metrics": {
|
||||
"$ref": "#/components/schemas/SpendMetrics"
|
||||
},
|
||||
"provider_breakdown": {
|
||||
"additionalProperties": {
|
||||
"$ref": "#/components/schemas/ProviderThroughputMetrics"
|
||||
},
|
||||
"title": "Provider Breakdown",
|
||||
"type": "object"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
|
|
@ -4343,6 +4350,38 @@
|
|||
"title": "PatchAgentRequest",
|
||||
"type": "object"
|
||||
},
|
||||
"ProviderThroughputMetrics": {
|
||||
"properties": {
|
||||
"completion_tokens": {
|
||||
"default": 0,
|
||||
"title": "Completion Tokens",
|
||||
"type": "integer"
|
||||
},
|
||||
"output_tokens_per_second": {
|
||||
"anyOf": [
|
||||
{
|
||||
"type": "number"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
],
|
||||
"title": "Output Tokens Per Second"
|
||||
},
|
||||
"timed_requests": {
|
||||
"default": 0,
|
||||
"title": "Timed Requests",
|
||||
"type": "integer"
|
||||
},
|
||||
"total_response_time_ms": {
|
||||
"default": 0,
|
||||
"title": "Total Response Time Ms",
|
||||
"type": "integer"
|
||||
}
|
||||
},
|
||||
"title": "ProviderThroughputMetrics",
|
||||
"type": "object"
|
||||
},
|
||||
"SpendAnalyticsPaginatedResponse": {
|
||||
"properties": {
|
||||
"metadata": {
|
||||
|
|
|
|||
|
|
@ -31,6 +31,7 @@ from litellm.types.proxy.management_endpoints.common_daily_activity import (
|
|||
KeyMetadata,
|
||||
KeyMetricWithMetadata,
|
||||
MetricWithMetadata,
|
||||
ProviderThroughputMetrics,
|
||||
SpendAnalyticsPaginatedResponse,
|
||||
SpendMetrics,
|
||||
)
|
||||
|
|
@ -236,6 +237,35 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) ->
|
|||
return existing_metrics
|
||||
|
||||
|
||||
def _provider_throughput(
|
||||
completion_tokens: int,
|
||||
total_response_time_ms: int,
|
||||
timed_requests: int,
|
||||
) -> ProviderThroughputMetrics:
|
||||
output_tokens_per_second: Final = (
|
||||
completion_tokens * 1000 / total_response_time_ms if timed_requests > 0 and total_response_time_ms > 0 else None
|
||||
)
|
||||
return ProviderThroughputMetrics(
|
||||
completion_tokens=completion_tokens,
|
||||
total_response_time_ms=total_response_time_ms,
|
||||
timed_requests=timed_requests,
|
||||
output_tokens_per_second=output_tokens_per_second,
|
||||
)
|
||||
|
||||
|
||||
def _update_provider_throughput(
|
||||
target: MetricWithMetadata,
|
||||
provider: str,
|
||||
record: DailySpendRecord,
|
||||
) -> None:
|
||||
existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics())
|
||||
target.provider_breakdown[provider] = _provider_throughput(
|
||||
completion_tokens=existing.completion_tokens + (record.completion_tokens or 0),
|
||||
total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0),
|
||||
timed_requests=existing.timed_requests + (record.timed_requests or 0),
|
||||
)
|
||||
|
||||
|
||||
def _is_user_agent_tag(tag: str | None) -> bool:
|
||||
"""Determine whether a tag should be treated as a User-Agent tag."""
|
||||
if not tag:
|
||||
|
|
@ -312,6 +342,12 @@ def update_breakdown_metrics(
|
|||
breakdown.models[model_key].metrics = update_metrics(breakdown.models[model_key].metrics, record)
|
||||
|
||||
if not is_ptu_sentinel:
|
||||
_update_provider_throughput(
|
||||
breakdown.models[model_key],
|
||||
record.custom_llm_provider or "unknown",
|
||||
record,
|
||||
)
|
||||
|
||||
# Update API key breakdown for this model
|
||||
if record.api_key not in breakdown.models[model_key].api_key_breakdown:
|
||||
breakdown.models[model_key].api_key_breakdown[record.api_key] = KeyMetricWithMetadata(
|
||||
|
|
@ -336,6 +372,12 @@ def update_breakdown_metrics(
|
|||
)
|
||||
|
||||
if not is_ptu_sentinel:
|
||||
_update_provider_throughput(
|
||||
breakdown.model_groups[model_group_key],
|
||||
record.custom_llm_provider or "unknown",
|
||||
record,
|
||||
)
|
||||
|
||||
# Update API key breakdown for this model
|
||||
if record.api_key not in breakdown.model_groups[model_group_key].api_key_breakdown:
|
||||
breakdown.model_groups[model_group_key].api_key_breakdown[record.api_key] = KeyMetricWithMetadata(
|
||||
|
|
@ -697,8 +739,10 @@ _API_KEY_ROLLED_UP_BIT: Final = 32 # 0b0100000
|
|||
_GROUP_DATE_API_KEY: Final = 31 # 0b0011111
|
||||
_GROUP_DATE_MODEL: Final = 47 # 0b0101111
|
||||
_GROUP_DATE_MODEL_API_KEY: Final = 15 # 0b0001111
|
||||
_GROUP_DATE_MODEL_PROVIDER: Final = 43 # 0b0101011
|
||||
_GROUP_DATE_MODEL_GROUP: Final = 55 # 0b0110111
|
||||
_GROUP_DATE_MODEL_GROUP_API_KEY: Final = 23 # 0b0010111
|
||||
_GROUP_DATE_MODEL_GROUP_PROVIDER: Final = 51 # 0b0110011
|
||||
_GROUP_DATE_PROVIDER: Final = 59 # 0b0111011
|
||||
_GROUP_DATE_PROVIDER_API_KEY: Final = 27 # 0b0011011
|
||||
_GROUP_DATE_MCP: Final = 61 # 0b0111101
|
||||
|
|
@ -779,6 +823,32 @@ def _aggregate_grouping_sets_records_sync(
|
|||
metrics=metrics, metadata=_key_metadata(api_key_metadata, api_key)
|
||||
)
|
||||
|
||||
def assign_provider_breakdown(
|
||||
target: dict[str, MetricWithMetadata],
|
||||
parent_key: str,
|
||||
provider: str,
|
||||
metrics: SpendMetrics,
|
||||
) -> None:
|
||||
parent: Final = target.get(parent_key)
|
||||
if parent is None:
|
||||
target[parent_key] = MetricWithMetadata(
|
||||
metrics=SpendMetrics(),
|
||||
metadata={},
|
||||
provider_breakdown={
|
||||
provider: _provider_throughput(
|
||||
metrics.completion_tokens,
|
||||
metrics.total_response_time_ms,
|
||||
metrics.timed_requests,
|
||||
)
|
||||
},
|
||||
)
|
||||
return
|
||||
parent.provider_breakdown[provider] = _provider_throughput(
|
||||
metrics.completion_tokens,
|
||||
metrics.total_response_time_ms,
|
||||
metrics.timed_requests,
|
||||
)
|
||||
|
||||
for record in records:
|
||||
level = record.group_level
|
||||
metrics = _record_to_spend_metrics(record)
|
||||
|
|
@ -806,6 +876,14 @@ def _aggregate_grouping_sets_records_sync(
|
|||
elif level == _GROUP_DATE_MODEL_API_KEY:
|
||||
if record.model and record.api_key and not is_ptu_sentinel:
|
||||
assign_api_key_breakdown(breakdown.models, record.model, record.api_key, metrics)
|
||||
elif level == _GROUP_DATE_MODEL_PROVIDER:
|
||||
if record.model:
|
||||
assign_provider_breakdown(
|
||||
breakdown.models,
|
||||
record.model,
|
||||
record.custom_llm_provider or "unknown",
|
||||
metrics,
|
||||
)
|
||||
elif level == _GROUP_DATE_MODEL_GROUP:
|
||||
if record.model_group:
|
||||
assign_metric_with_metadata(breakdown.model_groups, record.model_group, metrics)
|
||||
|
|
@ -817,6 +895,14 @@ def _aggregate_grouping_sets_records_sync(
|
|||
record.api_key,
|
||||
metrics,
|
||||
)
|
||||
elif level == _GROUP_DATE_MODEL_GROUP_PROVIDER:
|
||||
if record.model_group:
|
||||
assign_provider_breakdown(
|
||||
breakdown.model_groups,
|
||||
record.model_group,
|
||||
record.custom_llm_provider or "unknown",
|
||||
metrics,
|
||||
)
|
||||
elif level == _GROUP_DATE_PROVIDER:
|
||||
# Only PTU sentinel rows carry ptu_flat_cost and they have no provider, so at
|
||||
# this level the sentinel's cost would land under "unknown". Withholding the
|
||||
|
|
|
|||
|
|
@ -177,7 +177,9 @@ def build_aggregated_sql(scope: DailyActivityScope, *, api_key_limit: int) -> Sq
|
|||
GROUP BY GROUPING SETS (
|
||||
(date),
|
||||
(date, model),
|
||||
(date, model, custom_llm_provider),
|
||||
(date, {_MODEL_GROUP_EXPR}),
|
||||
(date, {_MODEL_GROUP_EXPR}, custom_llm_provider),
|
||||
(date, custom_llm_provider),
|
||||
(date, mcp_namespaced_tool_name),
|
||||
(date, endpoint),
|
||||
|
|
|
|||
|
|
@ -56,10 +56,18 @@ class KeyMetricWithMetadata(MetricBase):
|
|||
metadata: KeyMetadata = Field(default_factory=KeyMetadata)
|
||||
|
||||
|
||||
class ProviderThroughputMetrics(BaseModel):
|
||||
completion_tokens: int = Field(default=0)
|
||||
total_response_time_ms: int = Field(default=0)
|
||||
timed_requests: int = Field(default=0)
|
||||
output_tokens_per_second: float | None = Field(default=None)
|
||||
|
||||
|
||||
class MetricWithMetadata(MetricBase):
|
||||
metadata: dict[str, Any] = Field(default_factory=dict)
|
||||
# API key breakdown for this metric (e.g., which API keys are using this MCP server)
|
||||
api_key_breakdown: dict[str, KeyMetricWithMetadata] = Field(default_factory=dict) # api_key -> {metrics, metadata}
|
||||
provider_breakdown: dict[str, ProviderThroughputMetrics] = Field(default_factory=dict)
|
||||
|
||||
|
||||
class BreakdownMetrics(BaseModel):
|
||||
|
|
|
|||
|
|
@ -1675,6 +1675,9 @@ def _grouping_row(
|
|||
endpoint=None,
|
||||
spend=0.0,
|
||||
ptu_flat_cost=0.0,
|
||||
completion_tokens=0,
|
||||
total_response_time_ms=0,
|
||||
timed_requests=0,
|
||||
):
|
||||
return GroupingSetsRow(
|
||||
date="2024-01-01",
|
||||
|
|
@ -1689,7 +1692,7 @@ def _grouping_row(
|
|||
spend=spend,
|
||||
ptu_flat_cost=ptu_flat_cost,
|
||||
prompt_tokens=0,
|
||||
completion_tokens=0,
|
||||
completion_tokens=completion_tokens,
|
||||
cache_read_input_tokens=0,
|
||||
cache_creation_input_tokens=0,
|
||||
compression_saved_tokens=0,
|
||||
|
|
@ -1697,8 +1700,8 @@ def _grouping_row(
|
|||
prompt_caching_savings_spend=0.0,
|
||||
gateway_injected_caching_savings_spend=0.0,
|
||||
autorouter_savings_spend=0.0,
|
||||
total_response_time_ms=0,
|
||||
timed_requests=0,
|
||||
total_response_time_ms=total_response_time_ms,
|
||||
timed_requests=timed_requests,
|
||||
api_requests=0,
|
||||
successful_requests=0,
|
||||
failed_requests=0,
|
||||
|
|
@ -1794,6 +1797,73 @@ def test_grouping_sets_dispatcher_populates_every_breakdown_level(ptu_cost_attri
|
|||
assert "real-key" in day.breakdown.endpoints["/v1/chat/completions"].api_key_breakdown
|
||||
|
||||
|
||||
def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_model_groups():
|
||||
from litellm.proxy.management_endpoints.common_daily_activity import (
|
||||
_GROUP_DATE_MODEL_GROUP_PROVIDER,
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
_aggregate_grouping_sets_records_sync,
|
||||
)
|
||||
|
||||
records = [
|
||||
_grouping_row(
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
model="gpt-4o",
|
||||
custom_llm_provider="openai",
|
||||
completion_tokens=900,
|
||||
total_response_time_ms=3000,
|
||||
timed_requests=3,
|
||||
),
|
||||
_grouping_row(
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
model="gpt-4o",
|
||||
custom_llm_provider="azure",
|
||||
completion_tokens=400,
|
||||
total_response_time_ms=2000,
|
||||
timed_requests=2,
|
||||
),
|
||||
_grouping_row(
|
||||
_GROUP_DATE_MODEL_GROUP_PROVIDER,
|
||||
model_group="public-gpt-4o",
|
||||
custom_llm_provider="openai",
|
||||
completion_tokens=900,
|
||||
total_response_time_ms=3000,
|
||||
timed_requests=3,
|
||||
),
|
||||
]
|
||||
|
||||
day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0]
|
||||
|
||||
assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].model_dump() == {
|
||||
"completion_tokens": 900,
|
||||
"total_response_time_ms": 3000,
|
||||
"timed_requests": 3,
|
||||
"output_tokens_per_second": 300.0,
|
||||
}
|
||||
assert day.breakdown.models["gpt-4o"].provider_breakdown["azure"].output_tokens_per_second == 200.0
|
||||
assert day.breakdown.model_groups["public-gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 300.0
|
||||
|
||||
|
||||
def test_grouping_sets_dispatcher_returns_no_throughput_without_positive_duration():
|
||||
from litellm.proxy.management_endpoints.common_daily_activity import (
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
_aggregate_grouping_sets_records_sync,
|
||||
)
|
||||
|
||||
records = [
|
||||
_grouping_row(
|
||||
_GROUP_DATE_MODEL_PROVIDER,
|
||||
model="gpt-4o",
|
||||
completion_tokens=900,
|
||||
total_response_time_ms=0,
|
||||
timed_requests=1,
|
||||
)
|
||||
]
|
||||
|
||||
day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0]
|
||||
|
||||
assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None
|
||||
|
||||
|
||||
def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown():
|
||||
"""Sentinel rows carry no provider, so their flat cost must not surface under the
|
||||
"unknown" provider - the per-row path skips them for exactly the same reason."""
|
||||
|
|
@ -1851,7 +1921,7 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib
|
|||
endpoint="/v1/chat/completions",
|
||||
spend=5.0,
|
||||
prompt_tokens=0,
|
||||
completion_tokens=0,
|
||||
completion_tokens=600,
|
||||
cache_read_input_tokens=0,
|
||||
cache_creation_input_tokens=0,
|
||||
compression_saved_tokens=0,
|
||||
|
|
@ -1859,8 +1929,8 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib
|
|||
prompt_caching_savings_spend=0,
|
||||
gateway_injected_caching_savings_spend=0,
|
||||
autorouter_savings_spend=0,
|
||||
total_response_time_ms=0,
|
||||
timed_requests=0,
|
||||
total_response_time_ms=2000,
|
||||
timed_requests=2,
|
||||
total_tokens=0,
|
||||
api_requests=0,
|
||||
successful_requests=0,
|
||||
|
|
@ -1874,6 +1944,8 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib
|
|||
assert "real-key" in breakdown.mcp_servers["srv/tool"].api_key_breakdown
|
||||
assert "/v1/chat/completions" in breakdown.endpoints
|
||||
assert "azure" in breakdown.providers
|
||||
assert breakdown.models["gpt-4o-mini-ptu"].provider_breakdown["azure"].output_tokens_per_second == 300.0
|
||||
assert breakdown.model_groups["grp"].provider_breakdown["azure"].output_tokens_per_second == 300.0
|
||||
assert "team-1" in breakdown.entities
|
||||
assert "real-key" in breakdown.entities["team-1"].api_key_breakdown
|
||||
|
||||
|
|
|
|||
|
|
@ -243,6 +243,13 @@ def test_aggregate_query_sums_all_savings_drivers_and_response_time() -> None:
|
|||
assert all(f"SUM({field})" in query.sql for field in fields)
|
||||
|
||||
|
||||
def test_aggregated_query_groups_models_and_model_groups_by_provider() -> None:
|
||||
query = build_aggregated_sql(_scope(), api_key_limit=constants.USAGE_TOP_API_KEYS_DEFAULT)
|
||||
|
||||
assert "(date, model, custom_llm_provider)" in query.sql
|
||||
assert "(date, COALESCE(NULLIF(model_group, ''), model), custom_llm_provider)" in query.sql
|
||||
|
||||
|
||||
def test_aggregated_query_binds_sentinel_and_api_key_limit_after_scope_values() -> None:
|
||||
scope = _scope(entity_ids=None, api_keys=("key-1",))
|
||||
|
||||
|
|
|
|||
|
|
@ -76,6 +76,7 @@ const toMetric = (entry: SchemaMetricWithMetadata): MetricWithMetadata => ({
|
|||
api_key_breakdown: Object.fromEntries(
|
||||
Object.entries(entry.api_key_breakdown ?? {}).map(([key, value]) => [key, toKeyMetric(value)]),
|
||||
),
|
||||
provider_breakdown: entry.provider_breakdown ?? {},
|
||||
});
|
||||
|
||||
const toMetricMap = (
|
||||
|
|
|
|||
|
|
@ -56,6 +56,14 @@ const aggregatedResponse: DailyActivityAggregatedResponse = {
|
|||
metrics: completeMetrics,
|
||||
metadata: {},
|
||||
api_key_breakdown: { "key-hash": apiKeyActivity },
|
||||
provider_breakdown: {
|
||||
openai: {
|
||||
completion_tokens: 900,
|
||||
output_tokens_per_second: 300,
|
||||
timed_requests: 3,
|
||||
total_response_time_ms: 3000,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
|
|
@ -105,6 +113,14 @@ describe("key activity data", () => {
|
|||
},
|
||||
},
|
||||
]);
|
||||
expect(toDailyData(aggregatedResponse)[0].breakdown.models["gpt-4o-mini"].provider_breakdown).toEqual({
|
||||
openai: {
|
||||
completion_tokens: 900,
|
||||
output_tokens_per_second: 300,
|
||||
timed_requests: 3,
|
||||
total_response_time_ms: 3000,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("appends pages without duplicate keys and compares the server offset to the total", () => {
|
||||
|
|
|
|||
|
|
@ -38,6 +38,14 @@ export interface MetricWithMetadata {
|
|||
metrics: SpendMetrics;
|
||||
metadata: object;
|
||||
api_key_breakdown: { [key: string]: KeyMetricWithMetadata };
|
||||
provider_breakdown?: { [key: string]: ProviderThroughputMetrics };
|
||||
}
|
||||
|
||||
export interface ProviderThroughputMetrics {
|
||||
completion_tokens: number;
|
||||
total_response_time_ms: number;
|
||||
timed_requests: number;
|
||||
output_tokens_per_second?: number | null;
|
||||
}
|
||||
|
||||
export interface KeyMetricWithMetadata {
|
||||
|
|
@ -92,6 +100,7 @@ export interface ModelActivityData {
|
|||
cache_creation_input_tokens: number;
|
||||
avg_response_time_ms?: number | null;
|
||||
};
|
||||
provider_throughput?: Record<string, number | null>;
|
||||
}[];
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,13 @@
|
|||
import { fireEvent, render, screen, waitFor } from "@testing-library/react";
|
||||
import React from "react";
|
||||
import { beforeAll, describe, expect, it, vi } from "vitest";
|
||||
import { ActivityMetrics, formatKeyLabel, processActivityData, ResponseTimeTooltip } from "./activity_metrics";
|
||||
import {
|
||||
ActivityMetrics,
|
||||
formatKeyLabel,
|
||||
processActivityData,
|
||||
providerThroughputChartData,
|
||||
ResponseTimeTooltip,
|
||||
} from "./activity_metrics";
|
||||
import type { ChartTooltipProps } from "@/components/shared/charts";
|
||||
import { Team } from "./key_team_helpers/key_list";
|
||||
import { DailyData, KeyMetricWithMetadata, ModelActivityData } from "./UsagePage/types";
|
||||
|
|
@ -1402,6 +1408,38 @@ describe("processActivityData", () => {
|
|||
expect(result["gpt-5.5"].total_timed_requests).toBe(0);
|
||||
expect(result["gpt-5.5"].daily_data[0].metrics.avg_response_time_ms).toBeNull();
|
||||
});
|
||||
|
||||
it("preserves daily provider throughput for model and model-group views", () => {
|
||||
const metric = {
|
||||
metrics: EMPTY_SPEND_METRICS,
|
||||
metadata: {},
|
||||
api_key_breakdown: {},
|
||||
provider_breakdown: {
|
||||
openai: {
|
||||
completion_tokens: 900,
|
||||
total_response_time_ms: 3000,
|
||||
timed_requests: 3,
|
||||
output_tokens_per_second: 300,
|
||||
},
|
||||
},
|
||||
};
|
||||
const activity: { results: DailyData[] } = {
|
||||
results: [
|
||||
createMockDailyData("2025-01-01", EMPTY_SPEND_METRICS, {
|
||||
...EMPTY_BREAKDOWN,
|
||||
models: { "gpt-4o": metric },
|
||||
model_groups: { "public-gpt-4o": metric },
|
||||
}),
|
||||
],
|
||||
};
|
||||
|
||||
expect(processActivityData(activity, "models")["gpt-4o"].daily_data[0].provider_throughput).toEqual({
|
||||
openai: 300,
|
||||
});
|
||||
expect(processActivityData(activity, "model_groups")["public-gpt-4o"].daily_data[0].provider_throughput).toEqual({
|
||||
openai: 300,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("ActivityMetrics response time", () => {
|
||||
|
|
@ -1482,6 +1520,58 @@ describe("ActivityMetrics response time", () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe("ActivityMetrics provider throughput", () => {
|
||||
const model = createMockModelActivityData("GPT-4o", {
|
||||
daily_data: [
|
||||
{
|
||||
...createMockModelActivityData("GPT-4o").daily_data[0],
|
||||
date: "2025-01-01",
|
||||
provider_throughput: { openai: 300, azure: 200 },
|
||||
},
|
||||
{
|
||||
...createMockModelActivityData("GPT-4o").daily_data[0],
|
||||
date: "2025-01-02",
|
||||
provider_throughput: { openai: 250 },
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
it("renders one daily line per provider", () => {
|
||||
render(<ActivityMetrics modelMetrics={{ "gpt-4o": model }} />);
|
||||
|
||||
expect(screen.getByText("Output tokens per second of response time")).toBeInTheDocument();
|
||||
expect(screen.getByText("Openai")).toBeInTheDocument();
|
||||
expect(screen.getByText("Azure")).toBeInTheDocument();
|
||||
expect(screen.getAllByText("2025-01-01").length).toBeGreaterThan(0);
|
||||
expect(screen.getAllByText("2025-01-02").length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it("keeps missing provider dates as gaps", () => {
|
||||
expect(providerThroughputChartData(model.daily_data)).toEqual({
|
||||
providers: ["azure", "openai"],
|
||||
data: [
|
||||
{ date: "2025-01-01", azure: 200, openai: 300 },
|
||||
{ date: "2025-01-02", azure: null, openai: 250 },
|
||||
],
|
||||
});
|
||||
});
|
||||
|
||||
it("hides the chart when every throughput value is unavailable", () => {
|
||||
const unavailable = createMockModelActivityData("GPT-4o", {
|
||||
daily_data: [
|
||||
{
|
||||
...createMockModelActivityData("GPT-4o").daily_data[0],
|
||||
provider_throughput: { openai: null },
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
render(<ActivityMetrics modelMetrics={{ "gpt-4o": unavailable }} />);
|
||||
|
||||
expect(screen.queryByText("Output tokens per second of response time")).not.toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
||||
describe("formatKeyLabel", () => {
|
||||
it("should return key_alias when no team_id is present", () => {
|
||||
const modelData = createMockKeyMetricWithMetadata({
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import {
|
|||
type ChartTooltipProps,
|
||||
CustomLegend,
|
||||
CustomTooltip,
|
||||
DEFAULT_COLOR_CYCLE,
|
||||
formatCategoryName,
|
||||
LineChart,
|
||||
ValueTooltip,
|
||||
|
|
@ -18,7 +19,13 @@ import { Team } from "./key_team_helpers/key_list";
|
|||
import KeyModelUsageView from "./UsagePage/components/KeyModelUsageView";
|
||||
import { keyActivityLabel } from "./UsagePage/keyActivityLabel";
|
||||
import type { ModelTopKeysResponse } from "./UsagePage/dailyActivityApi";
|
||||
import { DailyData, KeyMetricWithMetadata, ModelActivityData, TopModelData } from "./UsagePage/types";
|
||||
import {
|
||||
DailyData,
|
||||
KeyMetricWithMetadata,
|
||||
MetricWithMetadata,
|
||||
ModelActivityData,
|
||||
TopModelData,
|
||||
} from "./UsagePage/types";
|
||||
import { averageResponseTimeMs, formatResponseTime, valueFormatter } from "./UsagePage/utils/value_formatters";
|
||||
|
||||
interface ActivityMetricsProps {
|
||||
|
|
@ -41,6 +48,30 @@ export const ResponseTimeTooltip = ({ active, payload, label }: ChartTooltipProp
|
|||
/>
|
||||
);
|
||||
|
||||
const formatTokensPerSecond = (value: number): string =>
|
||||
`${value.toLocaleString(undefined, { maximumFractionDigits: 2 })} tokens/s`;
|
||||
|
||||
export const providerThroughputChartData = (dailyData: ModelActivityData["daily_data"]) => {
|
||||
const providers = Array.from(new Set(dailyData.flatMap((day) => Object.keys(day.provider_throughput ?? {}))))
|
||||
.filter((provider) =>
|
||||
dailyData.some((day) => {
|
||||
const value = day.provider_throughput?.[provider];
|
||||
return typeof value === "number" && Number.isFinite(value);
|
||||
}),
|
||||
)
|
||||
.sort();
|
||||
const data = dailyData.map((day) => ({
|
||||
date: day.date,
|
||||
...Object.fromEntries(
|
||||
providers.map((provider) => {
|
||||
const value = day.provider_throughput?.[provider];
|
||||
return [provider, typeof value === "number" && Number.isFinite(value) ? value : null];
|
||||
}),
|
||||
),
|
||||
}));
|
||||
return { providers, data };
|
||||
};
|
||||
|
||||
const ModelTopKeys = ({
|
||||
modelName,
|
||||
fetchTopApiKeys,
|
||||
|
|
@ -178,6 +209,8 @@ export const ModelSection = ({
|
|||
hidePromptCachingMetrics?: boolean;
|
||||
fetchTopApiKeys?: (model: string) => Promise<ModelTopKeysResponse>;
|
||||
}) => {
|
||||
const throughputChart = providerThroughputChartData(metrics.daily_data);
|
||||
|
||||
return (
|
||||
<div className="space-y-2">
|
||||
{/* Summary Cards */}
|
||||
|
|
@ -316,6 +349,27 @@ export const ModelSection = ({
|
|||
</Card>
|
||||
)}
|
||||
|
||||
{throughputChart.providers.length > 0 && (
|
||||
<Card>
|
||||
<CardContent>
|
||||
<div className="flex justify-between items-center">
|
||||
<h3 className="text-lg font-medium text-foreground">Output tokens per second of response time</h3>
|
||||
<CustomLegend categories={throughputChart.providers} colors={DEFAULT_COLOR_CYCLE} />
|
||||
</div>
|
||||
<LineChart
|
||||
className="mt-4"
|
||||
data={throughputChart.data}
|
||||
index="date"
|
||||
categories={throughputChart.providers}
|
||||
colors={DEFAULT_COLOR_CYCLE}
|
||||
valueFormatter={formatTokensPerSecond}
|
||||
connectNulls={false}
|
||||
showLegend={false}
|
||||
/>
|
||||
</CardContent>
|
||||
</Card>
|
||||
)}
|
||||
|
||||
<Card>
|
||||
<CardContent>
|
||||
<div className="flex justify-between items-center">
|
||||
|
|
@ -683,6 +737,13 @@ export const processActivityData = (
|
|||
modelMetrics[model].total_response_time_ms =
|
||||
(modelMetrics[model].total_response_time_ms ?? 0) + dayResponseTimeMs;
|
||||
modelMetrics[model].total_timed_requests = (modelMetrics[model].total_timed_requests ?? 0) + dayTimedRequests;
|
||||
const providerBreakdown = (modelData as MetricWithMetadata).provider_breakdown ?? {};
|
||||
const providerThroughput = Object.fromEntries(
|
||||
Object.entries(providerBreakdown).map(([provider, providerMetrics]) => [
|
||||
provider,
|
||||
providerMetrics.output_tokens_per_second ?? null,
|
||||
]),
|
||||
);
|
||||
|
||||
// Add daily data
|
||||
modelMetrics[model].daily_data.push({
|
||||
|
|
@ -699,6 +760,7 @@ export const processActivityData = (
|
|||
cache_creation_input_tokens: modelData.metrics.cache_creation_input_tokens || 0,
|
||||
avg_response_time_ms: averageResponseTimeMs(dayResponseTimeMs, dayTimedRequests),
|
||||
},
|
||||
...(Object.keys(providerThroughput).length > 0 ? { provider_throughput: providerThroughput } : {}),
|
||||
});
|
||||
});
|
||||
});
|
||||
|
|
|
|||
24
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
24
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -37805,6 +37805,10 @@ export interface components {
|
|||
[key: string]: unknown;
|
||||
};
|
||||
metrics: components["schemas"]["SpendMetrics"];
|
||||
/** Provider Breakdown */
|
||||
provider_breakdown?: {
|
||||
[key: string]: components["schemas"]["ProviderThroughputMetrics"];
|
||||
};
|
||||
};
|
||||
/** Mode */
|
||||
Mode: {
|
||||
|
|
@ -41071,6 +41075,26 @@ export interface components {
|
|||
/** Tooltip */
|
||||
tooltip?: string | null;
|
||||
};
|
||||
/** ProviderThroughputMetrics */
|
||||
ProviderThroughputMetrics: {
|
||||
/**
|
||||
* Completion Tokens
|
||||
* @default 0
|
||||
*/
|
||||
completion_tokens: number;
|
||||
/** Output Tokens Per Second */
|
||||
output_tokens_per_second?: number | null;
|
||||
/**
|
||||
* Timed Requests
|
||||
* @default 0
|
||||
*/
|
||||
timed_requests: number;
|
||||
/**
|
||||
* Total Response Time Ms
|
||||
* @default 0
|
||||
*/
|
||||
total_response_time_ms: number;
|
||||
};
|
||||
/**
|
||||
* ProxyChatCompletionRequest
|
||||
* @description Pydantic model for chat completion requests that includes both OpenAI standard fields
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue