From 5a9f1045ad5f0add739427bfa0db6d172d9bf203 Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 15:48:32 +0530 Subject: [PATCH 1/8] feat(usage): graph provider throughput by model Co-authored-by: Cursor --- litellm/proxy/_lazy_openapi_snapshot.json | 39 ++++++++ .../common_daily_activity.py | 86 +++++++++++++++++ litellm/repositories/daily_activity_sql.py | 2 + .../common_daily_activity.py | 8 ++ .../test_common_daily_activity.py | 84 +++++++++++++++-- .../repositories/test_daily_activity_sql.py | 7 ++ .../components/UsagePage/dailyActivityApi.ts | 1 + .../UsagePage/keyActivityData.test.ts | 16 ++++ .../src/components/UsagePage/types.ts | 9 ++ .../src/components/activity_metrics.test.tsx | 92 ++++++++++++++++++- .../src/components/activity_metrics.tsx | 64 ++++++++++++- ui/litellm-dashboard/src/lib/http/schema.d.ts | 24 +++++ 12 files changed, 424 insertions(+), 8 deletions(-) diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 9eaf6c7e9ed..147174c73d0 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -4016,6 +4016,13 @@ }, "metrics": { "$ref": "#/components/schemas/SpendMetrics" + }, + "provider_breakdown": { + "additionalProperties": { + "$ref": "#/components/schemas/ProviderThroughputMetrics" + }, + "title": "Provider Breakdown", + "type": "object" } }, "required": [ @@ -4343,6 +4350,38 @@ "title": "PatchAgentRequest", "type": "object" }, + "ProviderThroughputMetrics": { + "properties": { + "completion_tokens": { + "default": 0, + "title": "Completion Tokens", + "type": "integer" + }, + "output_tokens_per_second": { + "anyOf": [ + { + "type": "number" + }, + { + "type": "null" + } + ], + "title": "Output Tokens Per Second" + }, + "timed_requests": { + "default": 0, + "title": "Timed Requests", + "type": "integer" + }, + "total_response_time_ms": { + "default": 0, + "title": "Total Response Time Ms", + "type": "integer" + } + }, + "title": "ProviderThroughputMetrics", + "type": "object" + }, "SpendAnalyticsPaginatedResponse": { "properties": { "metadata": { diff --git a/litellm/proxy/management_endpoints/common_daily_activity.py b/litellm/proxy/management_endpoints/common_daily_activity.py index 28dfbb09eab..36ce7d0da5e 100644 --- a/litellm/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/proxy/management_endpoints/common_daily_activity.py @@ -31,6 +31,7 @@ from litellm.types.proxy.management_endpoints.common_daily_activity import ( KeyMetadata, KeyMetricWithMetadata, MetricWithMetadata, + ProviderThroughputMetrics, SpendAnalyticsPaginatedResponse, SpendMetrics, ) @@ -236,6 +237,35 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) -> return existing_metrics +def _provider_throughput( + completion_tokens: int, + total_response_time_ms: int, + timed_requests: int, +) -> ProviderThroughputMetrics: + output_tokens_per_second: Final = ( + completion_tokens * 1000 / total_response_time_ms if timed_requests > 0 and total_response_time_ms > 0 else None + ) + return ProviderThroughputMetrics( + completion_tokens=completion_tokens, + total_response_time_ms=total_response_time_ms, + timed_requests=timed_requests, + output_tokens_per_second=output_tokens_per_second, + ) + + +def _update_provider_throughput( + target: MetricWithMetadata, + provider: str, + record: DailySpendRecord, +) -> None: + existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics()) + target.provider_breakdown[provider] = _provider_throughput( + completion_tokens=existing.completion_tokens + (record.completion_tokens or 0), + total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0), + timed_requests=existing.timed_requests + (record.timed_requests or 0), + ) + + def _is_user_agent_tag(tag: str | None) -> bool: """Determine whether a tag should be treated as a User-Agent tag.""" if not tag: @@ -312,6 +342,12 @@ def update_breakdown_metrics( breakdown.models[model_key].metrics = update_metrics(breakdown.models[model_key].metrics, record) if not is_ptu_sentinel: + _update_provider_throughput( + breakdown.models[model_key], + record.custom_llm_provider or "unknown", + record, + ) + # Update API key breakdown for this model if record.api_key not in breakdown.models[model_key].api_key_breakdown: breakdown.models[model_key].api_key_breakdown[record.api_key] = KeyMetricWithMetadata( @@ -336,6 +372,12 @@ def update_breakdown_metrics( ) if not is_ptu_sentinel: + _update_provider_throughput( + breakdown.model_groups[model_group_key], + record.custom_llm_provider or "unknown", + record, + ) + # Update API key breakdown for this model if record.api_key not in breakdown.model_groups[model_group_key].api_key_breakdown: breakdown.model_groups[model_group_key].api_key_breakdown[record.api_key] = KeyMetricWithMetadata( @@ -697,8 +739,10 @@ _API_KEY_ROLLED_UP_BIT: Final = 32 # 0b0100000 _GROUP_DATE_API_KEY: Final = 31 # 0b0011111 _GROUP_DATE_MODEL: Final = 47 # 0b0101111 _GROUP_DATE_MODEL_API_KEY: Final = 15 # 0b0001111 +_GROUP_DATE_MODEL_PROVIDER: Final = 43 # 0b0101011 _GROUP_DATE_MODEL_GROUP: Final = 55 # 0b0110111 _GROUP_DATE_MODEL_GROUP_API_KEY: Final = 23 # 0b0010111 +_GROUP_DATE_MODEL_GROUP_PROVIDER: Final = 51 # 0b0110011 _GROUP_DATE_PROVIDER: Final = 59 # 0b0111011 _GROUP_DATE_PROVIDER_API_KEY: Final = 27 # 0b0011011 _GROUP_DATE_MCP: Final = 61 # 0b0111101 @@ -779,6 +823,32 @@ def _aggregate_grouping_sets_records_sync( metrics=metrics, metadata=_key_metadata(api_key_metadata, api_key) ) + def assign_provider_breakdown( + target: dict[str, MetricWithMetadata], + parent_key: str, + provider: str, + metrics: SpendMetrics, + ) -> None: + parent: Final = target.get(parent_key) + if parent is None: + target[parent_key] = MetricWithMetadata( + metrics=SpendMetrics(), + metadata={}, + provider_breakdown={ + provider: _provider_throughput( + metrics.completion_tokens, + metrics.total_response_time_ms, + metrics.timed_requests, + ) + }, + ) + return + parent.provider_breakdown[provider] = _provider_throughput( + metrics.completion_tokens, + metrics.total_response_time_ms, + metrics.timed_requests, + ) + for record in records: level = record.group_level metrics = _record_to_spend_metrics(record) @@ -806,6 +876,14 @@ def _aggregate_grouping_sets_records_sync( elif level == _GROUP_DATE_MODEL_API_KEY: if record.model and record.api_key and not is_ptu_sentinel: assign_api_key_breakdown(breakdown.models, record.model, record.api_key, metrics) + elif level == _GROUP_DATE_MODEL_PROVIDER: + if record.model: + assign_provider_breakdown( + breakdown.models, + record.model, + record.custom_llm_provider or "unknown", + metrics, + ) elif level == _GROUP_DATE_MODEL_GROUP: if record.model_group: assign_metric_with_metadata(breakdown.model_groups, record.model_group, metrics) @@ -817,6 +895,14 @@ def _aggregate_grouping_sets_records_sync( record.api_key, metrics, ) + elif level == _GROUP_DATE_MODEL_GROUP_PROVIDER: + if record.model_group: + assign_provider_breakdown( + breakdown.model_groups, + record.model_group, + record.custom_llm_provider or "unknown", + metrics, + ) elif level == _GROUP_DATE_PROVIDER: # Only PTU sentinel rows carry ptu_flat_cost and they have no provider, so at # this level the sentinel's cost would land under "unknown". Withholding the diff --git a/litellm/repositories/daily_activity_sql.py b/litellm/repositories/daily_activity_sql.py index f12dc1be5ae..7495bb6abbf 100644 --- a/litellm/repositories/daily_activity_sql.py +++ b/litellm/repositories/daily_activity_sql.py @@ -177,7 +177,9 @@ def build_aggregated_sql(scope: DailyActivityScope, *, api_key_limit: int) -> Sq GROUP BY GROUPING SETS ( (date), (date, model), + (date, model, custom_llm_provider), (date, {_MODEL_GROUP_EXPR}), + (date, {_MODEL_GROUP_EXPR}, custom_llm_provider), (date, custom_llm_provider), (date, mcp_namespaced_tool_name), (date, endpoint), diff --git a/litellm/types/proxy/management_endpoints/common_daily_activity.py b/litellm/types/proxy/management_endpoints/common_daily_activity.py index 37804032569..4fc27a03a99 100644 --- a/litellm/types/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/types/proxy/management_endpoints/common_daily_activity.py @@ -56,10 +56,18 @@ class KeyMetricWithMetadata(MetricBase): metadata: KeyMetadata = Field(default_factory=KeyMetadata) +class ProviderThroughputMetrics(BaseModel): + completion_tokens: int = Field(default=0) + total_response_time_ms: int = Field(default=0) + timed_requests: int = Field(default=0) + output_tokens_per_second: float | None = Field(default=None) + + class MetricWithMetadata(MetricBase): metadata: dict[str, Any] = Field(default_factory=dict) # API key breakdown for this metric (e.g., which API keys are using this MCP server) api_key_breakdown: dict[str, KeyMetricWithMetadata] = Field(default_factory=dict) # api_key -> {metrics, metadata} + provider_breakdown: dict[str, ProviderThroughputMetrics] = Field(default_factory=dict) class BreakdownMetrics(BaseModel): diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index e6a6680d3e4..c14e6e098d6 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -1675,6 +1675,9 @@ def _grouping_row( endpoint=None, spend=0.0, ptu_flat_cost=0.0, + completion_tokens=0, + total_response_time_ms=0, + timed_requests=0, ): return GroupingSetsRow( date="2024-01-01", @@ -1689,7 +1692,7 @@ def _grouping_row( spend=spend, ptu_flat_cost=ptu_flat_cost, prompt_tokens=0, - completion_tokens=0, + completion_tokens=completion_tokens, cache_read_input_tokens=0, cache_creation_input_tokens=0, compression_saved_tokens=0, @@ -1697,8 +1700,8 @@ def _grouping_row( prompt_caching_savings_spend=0.0, gateway_injected_caching_savings_spend=0.0, autorouter_savings_spend=0.0, - total_response_time_ms=0, - timed_requests=0, + total_response_time_ms=total_response_time_ms, + timed_requests=timed_requests, api_requests=0, successful_requests=0, failed_requests=0, @@ -1794,6 +1797,73 @@ def test_grouping_sets_dispatcher_populates_every_breakdown_level(ptu_cost_attri assert "real-key" in day.breakdown.endpoints["/v1/chat/completions"].api_key_breakdown +def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_model_groups(): + from litellm.proxy.management_endpoints.common_daily_activity import ( + _GROUP_DATE_MODEL_GROUP_PROVIDER, + _GROUP_DATE_MODEL_PROVIDER, + _aggregate_grouping_sets_records_sync, + ) + + records = [ + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + custom_llm_provider="openai", + completion_tokens=900, + total_response_time_ms=3000, + timed_requests=3, + ), + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + custom_llm_provider="azure", + completion_tokens=400, + total_response_time_ms=2000, + timed_requests=2, + ), + _grouping_row( + _GROUP_DATE_MODEL_GROUP_PROVIDER, + model_group="public-gpt-4o", + custom_llm_provider="openai", + completion_tokens=900, + total_response_time_ms=3000, + timed_requests=3, + ), + ] + + day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] + + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].model_dump() == { + "completion_tokens": 900, + "total_response_time_ms": 3000, + "timed_requests": 3, + "output_tokens_per_second": 300.0, + } + assert day.breakdown.models["gpt-4o"].provider_breakdown["azure"].output_tokens_per_second == 200.0 + assert day.breakdown.model_groups["public-gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 300.0 + + +def test_grouping_sets_dispatcher_returns_no_throughput_without_positive_duration(): + from litellm.proxy.management_endpoints.common_daily_activity import ( + _GROUP_DATE_MODEL_PROVIDER, + _aggregate_grouping_sets_records_sync, + ) + + records = [ + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + completion_tokens=900, + total_response_time_ms=0, + timed_requests=1, + ) + ] + + day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] + + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None + + def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown(): """Sentinel rows carry no provider, so their flat cost must not surface under the "unknown" provider - the per-row path skips them for exactly the same reason.""" @@ -1851,7 +1921,7 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib endpoint="/v1/chat/completions", spend=5.0, prompt_tokens=0, - completion_tokens=0, + completion_tokens=600, cache_read_input_tokens=0, cache_creation_input_tokens=0, compression_saved_tokens=0, @@ -1859,8 +1929,8 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib prompt_caching_savings_spend=0, gateway_injected_caching_savings_spend=0, autorouter_savings_spend=0, - total_response_time_ms=0, - timed_requests=0, + total_response_time_ms=2000, + timed_requests=2, total_tokens=0, api_requests=0, successful_requests=0, @@ -1874,6 +1944,8 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib assert "real-key" in breakdown.mcp_servers["srv/tool"].api_key_breakdown assert "/v1/chat/completions" in breakdown.endpoints assert "azure" in breakdown.providers + assert breakdown.models["gpt-4o-mini-ptu"].provider_breakdown["azure"].output_tokens_per_second == 300.0 + assert breakdown.model_groups["grp"].provider_breakdown["azure"].output_tokens_per_second == 300.0 assert "team-1" in breakdown.entities assert "real-key" in breakdown.entities["team-1"].api_key_breakdown diff --git a/tests/unit/repositories/test_daily_activity_sql.py b/tests/unit/repositories/test_daily_activity_sql.py index c775dfd1f88..d413615a0a1 100644 --- a/tests/unit/repositories/test_daily_activity_sql.py +++ b/tests/unit/repositories/test_daily_activity_sql.py @@ -243,6 +243,13 @@ def test_aggregate_query_sums_all_savings_drivers_and_response_time() -> None: assert all(f"SUM({field})" in query.sql for field in fields) +def test_aggregated_query_groups_models_and_model_groups_by_provider() -> None: + query = build_aggregated_sql(_scope(), api_key_limit=constants.USAGE_TOP_API_KEYS_DEFAULT) + + assert "(date, model, custom_llm_provider)" in query.sql + assert "(date, COALESCE(NULLIF(model_group, ''), model), custom_llm_provider)" in query.sql + + def test_aggregated_query_binds_sentinel_and_api_key_limit_after_scope_values() -> None: scope = _scope(entity_ids=None, api_keys=("key-1",)) diff --git a/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts b/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts index 36c2837d9ac..703f6353363 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/dailyActivityApi.ts @@ -76,6 +76,7 @@ const toMetric = (entry: SchemaMetricWithMetadata): MetricWithMetadata => ({ api_key_breakdown: Object.fromEntries( Object.entries(entry.api_key_breakdown ?? {}).map(([key, value]) => [key, toKeyMetric(value)]), ), + provider_breakdown: entry.provider_breakdown ?? {}, }); const toMetricMap = ( diff --git a/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts b/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts index b35ea72896a..54145e3c9c6 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts @@ -56,6 +56,14 @@ const aggregatedResponse: DailyActivityAggregatedResponse = { metrics: completeMetrics, metadata: {}, api_key_breakdown: { "key-hash": apiKeyActivity }, + provider_breakdown: { + openai: { + completion_tokens: 900, + output_tokens_per_second: 300, + timed_requests: 3, + total_response_time_ms: 3000, + }, + }, }, }, }, @@ -105,6 +113,14 @@ describe("key activity data", () => { }, }, ]); + expect(toDailyData(aggregatedResponse)[0].breakdown.models["gpt-4o-mini"].provider_breakdown).toEqual({ + openai: { + completion_tokens: 900, + output_tokens_per_second: 300, + timed_requests: 3, + total_response_time_ms: 3000, + }, + }); }); it("appends pages without duplicate keys and compares the server offset to the total", () => { diff --git a/ui/litellm-dashboard/src/components/UsagePage/types.ts b/ui/litellm-dashboard/src/components/UsagePage/types.ts index 06f1856a5a6..41a09874050 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/types.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/types.ts @@ -38,6 +38,14 @@ export interface MetricWithMetadata { metrics: SpendMetrics; metadata: object; api_key_breakdown: { [key: string]: KeyMetricWithMetadata }; + provider_breakdown?: { [key: string]: ProviderThroughputMetrics }; +} + +export interface ProviderThroughputMetrics { + completion_tokens: number; + total_response_time_ms: number; + timed_requests: number; + output_tokens_per_second?: number | null; } export interface KeyMetricWithMetadata { @@ -92,6 +100,7 @@ export interface ModelActivityData { cache_creation_input_tokens: number; avg_response_time_ms?: number | null; }; + provider_throughput?: Record; }[]; } diff --git a/ui/litellm-dashboard/src/components/activity_metrics.test.tsx b/ui/litellm-dashboard/src/components/activity_metrics.test.tsx index 8ce18884718..f9fc2870aa3 100644 --- a/ui/litellm-dashboard/src/components/activity_metrics.test.tsx +++ b/ui/litellm-dashboard/src/components/activity_metrics.test.tsx @@ -1,7 +1,13 @@ import { fireEvent, render, screen, waitFor } from "@testing-library/react"; import React from "react"; import { beforeAll, describe, expect, it, vi } from "vitest"; -import { ActivityMetrics, formatKeyLabel, processActivityData, ResponseTimeTooltip } from "./activity_metrics"; +import { + ActivityMetrics, + formatKeyLabel, + processActivityData, + providerThroughputChartData, + ResponseTimeTooltip, +} from "./activity_metrics"; import type { ChartTooltipProps } from "@/components/shared/charts"; import { Team } from "./key_team_helpers/key_list"; import { DailyData, KeyMetricWithMetadata, ModelActivityData } from "./UsagePage/types"; @@ -1402,6 +1408,38 @@ describe("processActivityData", () => { expect(result["gpt-5.5"].total_timed_requests).toBe(0); expect(result["gpt-5.5"].daily_data[0].metrics.avg_response_time_ms).toBeNull(); }); + + it("preserves daily provider throughput for model and model-group views", () => { + const metric = { + metrics: EMPTY_SPEND_METRICS, + metadata: {}, + api_key_breakdown: {}, + provider_breakdown: { + openai: { + completion_tokens: 900, + total_response_time_ms: 3000, + timed_requests: 3, + output_tokens_per_second: 300, + }, + }, + }; + const activity: { results: DailyData[] } = { + results: [ + createMockDailyData("2025-01-01", EMPTY_SPEND_METRICS, { + ...EMPTY_BREAKDOWN, + models: { "gpt-4o": metric }, + model_groups: { "public-gpt-4o": metric }, + }), + ], + }; + + expect(processActivityData(activity, "models")["gpt-4o"].daily_data[0].provider_throughput).toEqual({ + openai: 300, + }); + expect(processActivityData(activity, "model_groups")["public-gpt-4o"].daily_data[0].provider_throughput).toEqual({ + openai: 300, + }); + }); }); describe("ActivityMetrics response time", () => { @@ -1482,6 +1520,58 @@ describe("ActivityMetrics response time", () => { }); }); +describe("ActivityMetrics provider throughput", () => { + const model = createMockModelActivityData("GPT-4o", { + daily_data: [ + { + ...createMockModelActivityData("GPT-4o").daily_data[0], + date: "2025-01-01", + provider_throughput: { openai: 300, azure: 200 }, + }, + { + ...createMockModelActivityData("GPT-4o").daily_data[0], + date: "2025-01-02", + provider_throughput: { openai: 250 }, + }, + ], + }); + + it("renders one daily line per provider", () => { + render(); + + expect(screen.getByText("Output tokens per second of response time")).toBeInTheDocument(); + expect(screen.getByText("Openai")).toBeInTheDocument(); + expect(screen.getByText("Azure")).toBeInTheDocument(); + expect(screen.getAllByText("2025-01-01").length).toBeGreaterThan(0); + expect(screen.getAllByText("2025-01-02").length).toBeGreaterThan(0); + }); + + it("keeps missing provider dates as gaps", () => { + expect(providerThroughputChartData(model.daily_data)).toEqual({ + providers: ["azure", "openai"], + data: [ + { date: "2025-01-01", azure: 200, openai: 300 }, + { date: "2025-01-02", azure: null, openai: 250 }, + ], + }); + }); + + it("hides the chart when every throughput value is unavailable", () => { + const unavailable = createMockModelActivityData("GPT-4o", { + daily_data: [ + { + ...createMockModelActivityData("GPT-4o").daily_data[0], + provider_throughput: { openai: null }, + }, + ], + }); + + render(); + + expect(screen.queryByText("Output tokens per second of response time")).not.toBeInTheDocument(); + }); +}); + describe("formatKeyLabel", () => { it("should return key_alias when no team_id is present", () => { const modelData = createMockKeyMetricWithMetadata({ diff --git a/ui/litellm-dashboard/src/components/activity_metrics.tsx b/ui/litellm-dashboard/src/components/activity_metrics.tsx index e7712a99c71..86091052eb8 100644 --- a/ui/litellm-dashboard/src/components/activity_metrics.tsx +++ b/ui/litellm-dashboard/src/components/activity_metrics.tsx @@ -4,6 +4,7 @@ import { type ChartTooltipProps, CustomLegend, CustomTooltip, + DEFAULT_COLOR_CYCLE, formatCategoryName, LineChart, ValueTooltip, @@ -18,7 +19,13 @@ import { Team } from "./key_team_helpers/key_list"; import KeyModelUsageView from "./UsagePage/components/KeyModelUsageView"; import { keyActivityLabel } from "./UsagePage/keyActivityLabel"; import type { ModelTopKeysResponse } from "./UsagePage/dailyActivityApi"; -import { DailyData, KeyMetricWithMetadata, ModelActivityData, TopModelData } from "./UsagePage/types"; +import { + DailyData, + KeyMetricWithMetadata, + MetricWithMetadata, + ModelActivityData, + TopModelData, +} from "./UsagePage/types"; import { averageResponseTimeMs, formatResponseTime, valueFormatter } from "./UsagePage/utils/value_formatters"; interface ActivityMetricsProps { @@ -41,6 +48,30 @@ export const ResponseTimeTooltip = ({ active, payload, label }: ChartTooltipProp /> ); +const formatTokensPerSecond = (value: number): string => + `${value.toLocaleString(undefined, { maximumFractionDigits: 2 })} tokens/s`; + +export const providerThroughputChartData = (dailyData: ModelActivityData["daily_data"]) => { + const providers = Array.from(new Set(dailyData.flatMap((day) => Object.keys(day.provider_throughput ?? {})))) + .filter((provider) => + dailyData.some((day) => { + const value = day.provider_throughput?.[provider]; + return typeof value === "number" && Number.isFinite(value); + }), + ) + .sort(); + const data = dailyData.map((day) => ({ + date: day.date, + ...Object.fromEntries( + providers.map((provider) => { + const value = day.provider_throughput?.[provider]; + return [provider, typeof value === "number" && Number.isFinite(value) ? value : null]; + }), + ), + })); + return { providers, data }; +}; + const ModelTopKeys = ({ modelName, fetchTopApiKeys, @@ -178,6 +209,8 @@ export const ModelSection = ({ hidePromptCachingMetrics?: boolean; fetchTopApiKeys?: (model: string) => Promise; }) => { + const throughputChart = providerThroughputChartData(metrics.daily_data); + return (
{/* Summary Cards */} @@ -316,6 +349,27 @@ export const ModelSection = ({ )} + {throughputChart.providers.length > 0 && ( + + +
+

Output tokens per second of response time

+ +
+ +
+
+ )} +
@@ -683,6 +737,13 @@ export const processActivityData = ( modelMetrics[model].total_response_time_ms = (modelMetrics[model].total_response_time_ms ?? 0) + dayResponseTimeMs; modelMetrics[model].total_timed_requests = (modelMetrics[model].total_timed_requests ?? 0) + dayTimedRequests; + const providerBreakdown = (modelData as MetricWithMetadata).provider_breakdown ?? {}; + const providerThroughput = Object.fromEntries( + Object.entries(providerBreakdown).map(([provider, providerMetrics]) => [ + provider, + providerMetrics.output_tokens_per_second ?? null, + ]), + ); // Add daily data modelMetrics[model].daily_data.push({ @@ -699,6 +760,7 @@ export const processActivityData = ( cache_creation_input_tokens: modelData.metrics.cache_creation_input_tokens || 0, avg_response_time_ms: averageResponseTimeMs(dayResponseTimeMs, dayTimedRequests), }, + ...(Object.keys(providerThroughput).length > 0 ? { provider_throughput: providerThroughput } : {}), }); }); }); diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 3f0629f05d6..f75f772e08c 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -37805,6 +37805,10 @@ export interface components { [key: string]: unknown; }; metrics: components["schemas"]["SpendMetrics"]; + /** Provider Breakdown */ + provider_breakdown?: { + [key: string]: components["schemas"]["ProviderThroughputMetrics"]; + }; }; /** Mode */ Mode: { @@ -41071,6 +41075,26 @@ export interface components { /** Tooltip */ tooltip?: string | null; }; + /** ProviderThroughputMetrics */ + ProviderThroughputMetrics: { + /** + * Completion Tokens + * @default 0 + */ + completion_tokens: number; + /** Output Tokens Per Second */ + output_tokens_per_second?: number | null; + /** + * Timed Requests + * @default 0 + */ + timed_requests: number; + /** + * Total Response Time Ms + * @default 0 + */ + total_response_time_ms: number; + }; /** * ProxyChatCompletionRequest * @description Pydantic model for chat completion requests that includes both OpenAI standard fields From 45f3cf4d1bbbc27374922699d7649f8b50a4180f Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 20:16:53 +0530 Subject: [PATCH 2/8] fix(usage): measure throughput from timed tokens Co-authored-by: Cursor --- .../migration.sql | 6 +++ litellm/proxy/_lazy_openapi_snapshot.json | 10 ++-- litellm/proxy/_types.py | 1 + litellm/proxy/db/daily_spend_bulk_upsert.py | 1 + litellm/proxy/db/db_spend_update_writer.py | 1 + .../daily_spend_update_queue.py | 4 ++ .../common_daily_activity.py | 49 ++++++++++++------- litellm/proxy/schema.prisma | 6 +++ litellm/repositories/daily_activity_sql.py | 3 +- .../common_daily_activity.py | 2 +- litellm/types/repositories/daily_activity.py | 2 + schema.prisma | 6 +++ .../test_daily_spend_update_queue.py | 8 +++ .../proxy/db/test_daily_spend_bulk_upsert.py | 8 +-- .../proxy/db/test_db_spend_update_writer.py | 2 + .../test_common_daily_activity.py | 26 ++++++++-- .../test_daily_activity_routes.py | 4 ++ .../repositories/test_daily_activity_sql.py | 8 +-- .../UsagePage/keyActivityData.test.ts | 4 +- .../src/components/UsagePage/types.ts | 2 +- .../src/components/activity_metrics.test.tsx | 2 +- ui/litellm-dashboard/src/lib/http/schema.d.ts | 10 ++-- 22 files changed, 116 insertions(+), 49 deletions(-) create mode 100644 litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql new file mode 100644 index 00000000000..6d08b2092aa --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql @@ -0,0 +1,6 @@ +ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 147174c73d0..06a696a6b6c 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -4352,11 +4352,6 @@ }, "ProviderThroughputMetrics": { "properties": { - "completion_tokens": { - "default": 0, - "title": "Completion Tokens", - "type": "integer" - }, "output_tokens_per_second": { "anyOf": [ { @@ -4368,6 +4363,11 @@ ], "title": "Output Tokens Per Second" }, + "timed_completion_tokens": { + "default": 0, + "title": "Timed Completion Tokens", + "type": "integer" + }, "timed_requests": { "default": 0, "title": "Timed Requests", diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index 0abec51cc49..f1c43b3f1ca 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -5703,6 +5703,7 @@ class BaseDailySpendTransaction(TypedDict): failed_requests: int total_response_time_ms: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place timed_requests: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place + timed_completion_tokens: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place class DailyTeamSpendTransaction(BaseDailySpendTransaction): diff --git a/litellm/proxy/db/daily_spend_bulk_upsert.py b/litellm/proxy/db/daily_spend_bulk_upsert.py index eb130a5196f..e4f8b62e6ae 100644 --- a/litellm/proxy/db/daily_spend_bulk_upsert.py +++ b/litellm/proxy/db/daily_spend_bulk_upsert.py @@ -91,6 +91,7 @@ _COUNTER_COLUMNS: Final = ( "compression_saved_tokens", "total_response_time_ms", "timed_requests", + "timed_completion_tokens", ) _SPEND_COLUMNS: Final = ( "spend", diff --git a/litellm/proxy/db/db_spend_update_writer.py b/litellm/proxy/db/db_spend_update_writer.py index 26a21069c83..1897a6700ba 100644 --- a/litellm/proxy/db/db_spend_update_writer.py +++ b/litellm/proxy/db/db_spend_update_writer.py @@ -2798,6 +2798,7 @@ class DBSpendUpdateWriter: autorouter_savings_spend=0.0 if is_internal_call else savings_spend.autorouter, total_response_time_ms=timed_duration_ms or 0, timed_requests=0 if timed_duration_ms is None else 1, + timed_completion_tokens=0 if timed_duration_ms is None else cast(int, payload["completion_tokens"]), ) return daily_transaction except Exception as e: diff --git a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py index 288c85c3513..cc17bceba71 100644 --- a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py +++ b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py @@ -162,6 +162,10 @@ class DailySpendUpdateQueue(BaseUpdateQueue): payload.get("timed_requests", 0) or 0 ) + daily_transaction.get("timed_requests", 0) + daily_transaction["timed_completion_tokens"] = ( + payload.get("timed_completion_tokens", 0) or 0 + ) + daily_transaction.get("timed_completion_tokens", 0) + else: aggregated_daily_spend_update_transactions[_key] = deepcopy(payload) return aggregated_daily_spend_update_transactions diff --git a/litellm/proxy/management_endpoints/common_daily_activity.py b/litellm/proxy/management_endpoints/common_daily_activity.py index 36ce7d0da5e..f663dcf120b 100644 --- a/litellm/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/proxy/management_endpoints/common_daily_activity.py @@ -161,6 +161,9 @@ class DailySpendRecord(Protocol): @property def timed_requests(self) -> int: ... + @property + def timed_completion_tokens(self) -> int: ... + class _KeyMetadataDict(TypedDict, total=False): key_alias: ReadOnly[str | None] @@ -238,15 +241,17 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) -> def _provider_throughput( - completion_tokens: int, + timed_completion_tokens: int, total_response_time_ms: int, timed_requests: int, ) -> ProviderThroughputMetrics: output_tokens_per_second: Final = ( - completion_tokens * 1000 / total_response_time_ms if timed_requests > 0 and total_response_time_ms > 0 else None + timed_completion_tokens * 1000 / total_response_time_ms + if timed_completion_tokens > 0 and timed_requests > 0 and total_response_time_ms > 0 + else None ) return ProviderThroughputMetrics( - completion_tokens=completion_tokens, + timed_completion_tokens=timed_completion_tokens, total_response_time_ms=total_response_time_ms, timed_requests=timed_requests, output_tokens_per_second=output_tokens_per_second, @@ -259,11 +264,14 @@ def _update_provider_throughput( record: DailySpendRecord, ) -> None: existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics()) - target.provider_breakdown[provider] = _provider_throughput( - completion_tokens=existing.completion_tokens + (record.completion_tokens or 0), - total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0), - timed_requests=existing.timed_requests + (record.timed_requests or 0), - ) + target.provider_breakdown = { + **target.provider_breakdown, + provider: _provider_throughput( + timed_completion_tokens=existing.timed_completion_tokens + (record.timed_completion_tokens or 0), + total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0), + timed_requests=existing.timed_requests + (record.timed_requests or 0), + ), + } def _is_user_agent_tag(tag: str | None) -> bool: @@ -827,7 +835,7 @@ def _aggregate_grouping_sets_records_sync( target: dict[str, MetricWithMetadata], parent_key: str, provider: str, - metrics: SpendMetrics, + record: GroupingSetsRow, ) -> None: parent: Final = target.get(parent_key) if parent is None: @@ -836,18 +844,21 @@ def _aggregate_grouping_sets_records_sync( metadata={}, provider_breakdown={ provider: _provider_throughput( - metrics.completion_tokens, - metrics.total_response_time_ms, - metrics.timed_requests, + record.timed_completion_tokens or 0, + record.total_response_time_ms or 0, + record.timed_requests or 0, ) }, ) return - parent.provider_breakdown[provider] = _provider_throughput( - metrics.completion_tokens, - metrics.total_response_time_ms, - metrics.timed_requests, - ) + parent.provider_breakdown = { + **parent.provider_breakdown, + provider: _provider_throughput( + record.timed_completion_tokens or 0, + record.total_response_time_ms or 0, + record.timed_requests or 0, + ), + } for record in records: level = record.group_level @@ -882,7 +893,7 @@ def _aggregate_grouping_sets_records_sync( breakdown.models, record.model, record.custom_llm_provider or "unknown", - metrics, + record, ) elif level == _GROUP_DATE_MODEL_GROUP: if record.model_group: @@ -901,7 +912,7 @@ def _aggregate_grouping_sets_records_sync( breakdown.model_groups, record.model_group, record.custom_llm_provider or "unknown", - metrics, + record, ) elif level == _GROUP_DATE_PROVIDER: # Only PTU sentinel rows carry ptu_flat_cost and they have no provider, so at diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index cf76b764350..83b70ef6481 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -870,6 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -939,6 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -977,6 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1014,6 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1051,6 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1091,6 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/litellm/repositories/daily_activity_sql.py b/litellm/repositories/daily_activity_sql.py index 7495bb6abbf..945136a2e8b 100644 --- a/litellm/repositories/daily_activity_sql.py +++ b/litellm/repositories/daily_activity_sql.py @@ -129,7 +129,8 @@ def _rollup_metric_select(table: DailyActivityTable) -> str: SUM(successful_requests)::bigint AS successful_requests, SUM(failed_requests)::bigint AS failed_requests, SUM(total_response_time_ms)::bigint AS total_response_time_ms, - SUM(timed_requests)::bigint AS timed_requests""" + SUM(timed_requests)::bigint AS timed_requests, + SUM(timed_completion_tokens)::bigint AS timed_completion_tokens""" def _validate_api_key_limit(api_key_limit: int) -> None: diff --git a/litellm/types/proxy/management_endpoints/common_daily_activity.py b/litellm/types/proxy/management_endpoints/common_daily_activity.py index 4fc27a03a99..ee0d599a720 100644 --- a/litellm/types/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/types/proxy/management_endpoints/common_daily_activity.py @@ -57,7 +57,7 @@ class KeyMetricWithMetadata(MetricBase): class ProviderThroughputMetrics(BaseModel): - completion_tokens: int = Field(default=0) + timed_completion_tokens: int = Field(default=0) total_response_time_ms: int = Field(default=0) timed_requests: int = Field(default=0) output_tokens_per_second: float | None = Field(default=None) diff --git a/litellm/types/repositories/daily_activity.py b/litellm/types/repositories/daily_activity.py index df302234398..b4dae9cd9e1 100644 --- a/litellm/types/repositories/daily_activity.py +++ b/litellm/types/repositories/daily_activity.py @@ -124,6 +124,7 @@ class RollupMetricsRow: failed_requests: int | None total_response_time_ms: int | None timed_requests: int | None + timed_completion_tokens: int | None @dataclass(frozen=True, slots=True) @@ -184,6 +185,7 @@ class DailyActivityRow(Protocol): failed_requests: int total_response_time_ms: int timed_requests: int + timed_completion_tokens: int @dataclass(frozen=True, slots=True) diff --git a/schema.prisma b/schema.prisma index cf76b764350..83b70ef6481 100644 --- a/schema.prisma +++ b/schema.prisma @@ -870,6 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -939,6 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -977,6 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1014,6 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1051,6 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1091,6 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py index c17ba75db03..f12f6d076f5 100644 --- a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py +++ b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py @@ -40,6 +40,7 @@ async def test_add_single_update(daily_spend_update_queue): "spend": 10.0, "prompt_tokens": 100, "completion_tokens": 50, + "timed_completion_tokens": 50, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -73,6 +74,7 @@ async def test_add_multiple_updates(daily_spend_update_queue): "spend": 5.0, "prompt_tokens": 200, "completion_tokens": 30, + "timed_completion_tokens": 0, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -181,6 +183,7 @@ async def test_get_aggregated_daily_spend_update_transactions_same_key(): "spend": 10.0, "prompt_tokens": 100, "completion_tokens": 50, + "timed_completion_tokens": 50, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -190,6 +193,7 @@ async def test_get_aggregated_daily_spend_update_transactions_same_key(): "spend": 5.0, "prompt_tokens": 200, "completion_tokens": 30, + "timed_completion_tokens": 0, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -199,6 +203,7 @@ async def test_get_aggregated_daily_spend_update_transactions_same_key(): "spend": 15.0, # 10 + 5 "prompt_tokens": 300, # 100 + 200 "completion_tokens": 80, # 50 + 30 + "timed_completion_tokens": 50, "api_requests": 2, # 1 + 1 "successful_requests": 2, # 1 + 1 "failed_requests": 0, # 0 + 0 @@ -235,6 +240,7 @@ async def test_flush_and_get_aggregated_daily_spend_update_transactions( "spend": 10.0, "prompt_tokens": 100, "completion_tokens": 50, + "timed_completion_tokens": 50, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -244,6 +250,7 @@ async def test_flush_and_get_aggregated_daily_spend_update_transactions( "spend": 5.0, "prompt_tokens": 200, "completion_tokens": 30, + "timed_completion_tokens": 0, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -253,6 +260,7 @@ async def test_flush_and_get_aggregated_daily_spend_update_transactions( "spend": 15.0, # 10 + 5 "prompt_tokens": 300, # 100 + 200 "completion_tokens": 80, # 50 + 30 + "timed_completion_tokens": 50, "api_requests": 2, # 1 + 1 "successful_requests": 2, # 1 + 1 "failed_requests": 0, # 0 + 0 diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index 7893fb82281..4941ae0f827 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -34,6 +34,7 @@ def tag_txn(**overrides): "endpoint": "/chat/completions", "prompt_tokens": 10, "completion_tokens": 20, + "timed_completion_tokens": 20, "spend": 0.25, "api_requests": 1, "successful_requests": 1, @@ -87,10 +88,10 @@ def test_one_statement_carries_every_row_in_the_batch(): assert sql.count("INSERT INTO") == 1 assert len(re.findall(r"ON CONFLICT", sql)) == 1 - # 25 bound columns per row plus the inlined updated_at, so the row count is what + # 26 bound columns per row plus the inlined updated_at, so the row count is what # separates one multi-row statement from a hundred single-row ones. - assert len(params) == 100 * 25 - assert "$2500::text" in sql + assert len(params) == 100 * 26 + assert "$2600::text" in sql assert sql.count("(NOW() AT TIME ZONE 'UTC')") == 100 + 1 @@ -115,6 +116,7 @@ def test_conflict_target_is_the_full_unique_constraint(): "failed_requests", "total_response_time_ms", "timed_requests", + "timed_completion_tokens", ], ) def test_counters_increment_rather_than_overwrite(column): diff --git a/tests/unit/proxy/db/test_db_spend_update_writer.py b/tests/unit/proxy/db/test_db_spend_update_writer.py index 4de90d5d7f7..c70affce0df 100644 --- a/tests/unit/proxy/db/test_db_spend_update_writer.py +++ b/tests/unit/proxy/db/test_db_spend_update_writer.py @@ -3714,6 +3714,7 @@ async def test_daily_transaction_rolls_up_response_time_for_successful_requests( assert transaction is not None assert transaction["total_response_time_ms"] == request_duration_ms assert transaction["timed_requests"] == 1 + assert transaction["timed_completion_tokens"] == 5 @pytest.mark.asyncio @@ -3746,6 +3747,7 @@ async def test_daily_transaction_excludes_untimed_requests_from_response_time( assert transaction is not None assert transaction["total_response_time_ms"] == 0 assert transaction["timed_requests"] == 0 + assert transaction["timed_completion_tokens"] == 0 def _deadlock_error(): diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index c14e6e098d6..fcb7ce547dc 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -366,6 +366,7 @@ async def test_get_daily_activity_aggregated_with_endpoint_breakdown(): "autorouter_savings_spend": 0.0, "total_response_time_ms": 0, "timed_requests": 0, + "timed_completion_tokens": 0, "failed_requests": 0, } mock_rows = [ @@ -950,6 +951,7 @@ def test_update_breakdown_metrics_includes_user_email(): autorouter_savings_spend=0, total_response_time_ms=0, timed_requests=0, + timed_completion_tokens=0, total_tokens=2, api_requests=1, successful_requests=1, @@ -1031,6 +1033,7 @@ async def test_tag_daily_activity_metadata_totals_not_zero(): mock_record_1.autorouter_savings_spend = 0.0 mock_record_1.total_response_time_ms = 18_000 mock_record_1.timed_requests = 9 + mock_record_1.timed_completion_tokens = 200 mock_record_1.api_requests = 10 mock_record_1.successful_requests = 9 mock_record_1.failed_requests = 1 @@ -1057,6 +1060,7 @@ async def test_tag_daily_activity_metadata_totals_not_zero(): mock_record_2.autorouter_savings_spend = 0.0 mock_record_2.total_response_time_ms = 2_500 mock_record_2.timed_requests = 5 + mock_record_2.timed_completion_tokens = 100 mock_record_2.api_requests = 5 mock_record_2.successful_requests = 5 mock_record_2.failed_requests = 0 @@ -1129,6 +1133,7 @@ async def test_aggregated_activity_preserves_metadata_for_deleted_keys(): "autorouter_savings_spend": 0.0, "total_response_time_ms": 0, "timed_requests": 0, + "timed_completion_tokens": 0, "failed_requests": 0, } mock_rows = [ @@ -1225,6 +1230,7 @@ async def test_aggregated_activity_flags_only_keys_that_key_info_can_still_resol "autorouter_savings_spend": 0.0, "total_response_time_ms": 0, "timed_requests": 0, + "timed_completion_tokens": 0, "api_requests": 1, "successful_requests": 1, "failed_requests": 0, @@ -1295,6 +1301,7 @@ def _daily_user_spend_record(*, user_id, api_key, spend, model="gpt-4", model_gr autorouter_savings_spend=0.0, total_response_time_ms=0, timed_requests=0, + timed_completion_tokens=0, api_requests=1, successful_requests=1, failed_requests=0, @@ -1444,6 +1451,7 @@ async def test_get_daily_activity_aggregated_empty_result_set(): "autorouter_savings_spend": None, "total_response_time_ms": None, "timed_requests": None, + "timed_completion_tokens": None, "api_requests": None, "successful_requests": None, "failed_requests": None, @@ -1492,6 +1500,7 @@ def _no_spend_record(): autorouter_savings_spend=None, total_response_time_ms=None, timed_requests=None, + timed_completion_tokens=None, api_requests=None, successful_requests=None, failed_requests=None, @@ -1631,6 +1640,7 @@ def _spend_record(api_key, *, model="gpt-4o-mini-ptu", spend=0.0, ptu_flat_cost= autorouter_savings_spend=0, total_response_time_ms=0, timed_requests=0, + timed_completion_tokens=0, total_tokens=0, api_requests=0, successful_requests=0, @@ -1676,6 +1686,7 @@ def _grouping_row( spend=0.0, ptu_flat_cost=0.0, completion_tokens=0, + timed_completion_tokens=0, total_response_time_ms=0, timed_requests=0, ): @@ -1693,6 +1704,7 @@ def _grouping_row( ptu_flat_cost=ptu_flat_cost, prompt_tokens=0, completion_tokens=completion_tokens, + timed_completion_tokens=timed_completion_tokens, cache_read_input_tokens=0, cache_creation_input_tokens=0, compression_saved_tokens=0, @@ -1809,7 +1821,8 @@ def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_mod _GROUP_DATE_MODEL_PROVIDER, model="gpt-4o", custom_llm_provider="openai", - completion_tokens=900, + completion_tokens=1900, + timed_completion_tokens=900, total_response_time_ms=3000, timed_requests=3, ), @@ -1818,6 +1831,7 @@ def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_mod model="gpt-4o", custom_llm_provider="azure", completion_tokens=400, + timed_completion_tokens=400, total_response_time_ms=2000, timed_requests=2, ), @@ -1826,6 +1840,7 @@ def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_mod model_group="public-gpt-4o", custom_llm_provider="openai", completion_tokens=900, + timed_completion_tokens=900, total_response_time_ms=3000, timed_requests=3, ), @@ -1834,7 +1849,7 @@ def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_mod day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].model_dump() == { - "completion_tokens": 900, + "timed_completion_tokens": 900, "total_response_time_ms": 3000, "timed_requests": 3, "output_tokens_per_second": 300.0, @@ -1843,7 +1858,7 @@ def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_mod assert day.breakdown.model_groups["public-gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 300.0 -def test_grouping_sets_dispatcher_returns_no_throughput_without_positive_duration(): +def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_timed_tokens(): from litellm.proxy.management_endpoints.common_daily_activity import ( _GROUP_DATE_MODEL_PROVIDER, _aggregate_grouping_sets_records_sync, @@ -1854,7 +1869,8 @@ def test_grouping_sets_dispatcher_returns_no_throughput_without_positive_duratio _GROUP_DATE_MODEL_PROVIDER, model="gpt-4o", completion_tokens=900, - total_response_time_ms=0, + timed_completion_tokens=0, + total_response_time_ms=3000, timed_requests=1, ) ] @@ -1931,6 +1947,7 @@ def test_update_breakdown_metrics_covers_mcp_endpoint_and_entity(ptu_cost_attrib autorouter_savings_spend=0, total_response_time_ms=2000, timed_requests=2, + timed_completion_tokens=600, total_tokens=0, api_requests=0, successful_requests=0, @@ -2288,6 +2305,7 @@ async def test_get_daily_activity_aggregated_with_entity_breakdown(): "autorouter_savings_spend": 0.0, "total_response_time_ms": 0, "timed_requests": 0, + "timed_completion_tokens": 0, "failed_requests": 0, "prompt_tokens": 0, "completion_tokens": 0, diff --git a/tests/unit/proxy/management_endpoints/test_daily_activity_routes.py b/tests/unit/proxy/management_endpoints/test_daily_activity_routes.py index c3f4fdfef54..7b542d83c97 100644 --- a/tests/unit/proxy/management_endpoints/test_daily_activity_routes.py +++ b/tests/unit/proxy/management_endpoints/test_daily_activity_routes.py @@ -59,6 +59,7 @@ class _Activity: failed_requests: int total_response_time_ms: int timed_requests: int + timed_completion_tokens: int _ENTITY_CASES: Final[tuple[tuple[str, str, str], ...]] = ( @@ -101,6 +102,7 @@ def _activity_for_entity( failed_requests=0, total_response_time_ms=100, timed_requests=1, + timed_completion_tokens=5, ) for api_key, date, model, spend, cache_read in key_rows ) @@ -157,6 +159,7 @@ def _seeded_activity() -> tuple[_Activity, ...]: failed_requests=0, total_response_time_ms=100, timed_requests=1, + timed_completion_tokens=5, ) for table in entity_ids ) @@ -180,6 +183,7 @@ def _metrics(rows: Sequence[_Activity]) -> Mapping[str, int | float]: "failed_requests": sum(row.failed_requests for row in rows), "total_response_time_ms": sum(row.total_response_time_ms for row in rows), "timed_requests": sum(row.timed_requests for row in rows), + "timed_completion_tokens": sum(row.timed_completion_tokens for row in rows), } diff --git a/tests/unit/repositories/test_daily_activity_sql.py b/tests/unit/repositories/test_daily_activity_sql.py index d413615a0a1..761e24ac47c 100644 --- a/tests/unit/repositories/test_daily_activity_sql.py +++ b/tests/unit/repositories/test_daily_activity_sql.py @@ -237,19 +237,13 @@ def test_aggregate_query_sums_all_savings_drivers_and_response_time() -> None: fields: Final = tuple(field for field in SpendMetrics.model_fields if field.endswith("_savings_spend")) + ( "total_response_time_ms", "timed_requests", + "timed_completion_tokens", ) assert fields assert all(f"SUM({field})" in query.sql for field in fields) -def test_aggregated_query_groups_models_and_model_groups_by_provider() -> None: - query = build_aggregated_sql(_scope(), api_key_limit=constants.USAGE_TOP_API_KEYS_DEFAULT) - - assert "(date, model, custom_llm_provider)" in query.sql - assert "(date, COALESCE(NULLIF(model_group, ''), model), custom_llm_provider)" in query.sql - - def test_aggregated_query_binds_sentinel_and_api_key_limit_after_scope_values() -> None: scope = _scope(entity_ids=None, api_keys=("key-1",)) diff --git a/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts b/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts index 54145e3c9c6..33f6ba84038 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/keyActivityData.test.ts @@ -58,7 +58,7 @@ const aggregatedResponse: DailyActivityAggregatedResponse = { api_key_breakdown: { "key-hash": apiKeyActivity }, provider_breakdown: { openai: { - completion_tokens: 900, + timed_completion_tokens: 900, output_tokens_per_second: 300, timed_requests: 3, total_response_time_ms: 3000, @@ -115,7 +115,7 @@ describe("key activity data", () => { ]); expect(toDailyData(aggregatedResponse)[0].breakdown.models["gpt-4o-mini"].provider_breakdown).toEqual({ openai: { - completion_tokens: 900, + timed_completion_tokens: 900, output_tokens_per_second: 300, timed_requests: 3, total_response_time_ms: 3000, diff --git a/ui/litellm-dashboard/src/components/UsagePage/types.ts b/ui/litellm-dashboard/src/components/UsagePage/types.ts index 41a09874050..fa8fb6c3a4e 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/types.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/types.ts @@ -42,7 +42,7 @@ export interface MetricWithMetadata { } export interface ProviderThroughputMetrics { - completion_tokens: number; + timed_completion_tokens: number; total_response_time_ms: number; timed_requests: number; output_tokens_per_second?: number | null; diff --git a/ui/litellm-dashboard/src/components/activity_metrics.test.tsx b/ui/litellm-dashboard/src/components/activity_metrics.test.tsx index f9fc2870aa3..869f679300c 100644 --- a/ui/litellm-dashboard/src/components/activity_metrics.test.tsx +++ b/ui/litellm-dashboard/src/components/activity_metrics.test.tsx @@ -1416,7 +1416,7 @@ describe("processActivityData", () => { api_key_breakdown: {}, provider_breakdown: { openai: { - completion_tokens: 900, + timed_completion_tokens: 900, total_response_time_ms: 3000, timed_requests: 3, output_tokens_per_second: 300, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index f75f772e08c..1f437c6d38b 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -41077,13 +41077,13 @@ export interface components { }; /** ProviderThroughputMetrics */ ProviderThroughputMetrics: { - /** - * Completion Tokens - * @default 0 - */ - completion_tokens: number; /** Output Tokens Per Second */ output_tokens_per_second?: number | null; + /** + * Timed Completion Tokens + * @default 0 + */ + timed_completion_tokens: number; /** * Timed Requests * @default 0 From 3c6dfb463f442086f7d0e398dd24517f5b740bef Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 21:55:00 +0530 Subject: [PATCH 3/8] fix(usage): address provider throughput review Co-authored-by: Cursor --- .../litellm_proxy_extras/schema.prisma | 6 + .../daily_spend_update_queue.py | 109 +++++++++--------- .../common_daily_activity.py | 2 +- .../test_daily_spend_update_queue.py | 37 ++---- .../proxy/db/test_daily_spend_bulk_upsert.py | 4 +- .../proxy/db/test_db_spend_update_writer.py | 67 +++++++---- .../test_common_daily_activity.py | 4 +- 7 files changed, 114 insertions(+), 115 deletions(-) diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index cf76b764350..83b70ef6481 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -870,6 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -939,6 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -977,6 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1014,6 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1051,6 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1091,6 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) + timed_completion_tokens BigInt @default(0) created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py index cc17bceba71..87c2915f7fd 100644 --- a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py +++ b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py @@ -1,6 +1,8 @@ import asyncio -from collections.abc import Coroutine +from collections.abc import Coroutine, Iterator from copy import deepcopy +from functools import reduce +from itertools import groupby from typing import Final from litellm._logging import verbose_proxy_logger @@ -13,6 +15,47 @@ from litellm.proxy.db.db_transaction_queue.base_update_queue import ( from litellm.types.services import ServiceTypes +def _daily_spend_updates( + updates: list[dict[str, BaseDailySpendTransaction]], +) -> Iterator[tuple[str, BaseDailySpendTransaction]]: + for update in updates: + yield from update.items() + + +def _merge_daily_spend_transactions( + existing: BaseDailySpendTransaction, + payload: BaseDailySpendTransaction, +) -> BaseDailySpendTransaction: + return { + **existing, + "spend": existing["spend"] + payload["spend"], + "prompt_tokens": existing["prompt_tokens"] + payload["prompt_tokens"], + "completion_tokens": existing["completion_tokens"] + payload["completion_tokens"], + "api_requests": existing["api_requests"] + payload["api_requests"], + "successful_requests": existing["successful_requests"] + payload["successful_requests"], + "failed_requests": existing["failed_requests"] + payload["failed_requests"], + "cache_read_input_tokens": (existing.get("cache_read_input_tokens", 0) or 0) + + (payload.get("cache_read_input_tokens", 0) or 0), + "cache_creation_input_tokens": (existing.get("cache_creation_input_tokens", 0) or 0) + + (payload.get("cache_creation_input_tokens", 0) or 0), + "compression_saved_tokens": (existing.get("compression_saved_tokens", 0) or 0) + + (payload.get("compression_saved_tokens", 0) or 0), + "compression_savings_spend": (existing.get("compression_savings_spend", 0) or 0) + + (payload.get("compression_savings_spend", 0) or 0), + "prompt_caching_savings_spend": (existing.get("prompt_caching_savings_spend", 0) or 0) + + (payload.get("prompt_caching_savings_spend", 0) or 0), + "gateway_injected_caching_savings_spend": (existing.get("gateway_injected_caching_savings_spend", 0) or 0) + + (payload.get("gateway_injected_caching_savings_spend", 0) or 0), + "autorouter_savings_spend": (existing.get("autorouter_savings_spend", 0) or 0) + + (payload.get("autorouter_savings_spend", 0) or 0), + "total_response_time_ms": (existing.get("total_response_time_ms", 0) or 0) + + (payload.get("total_response_time_ms", 0) or 0), + "timed_requests": (existing.get("timed_requests", 0) or 0) + (payload.get("timed_requests", 0) or 0), + "timed_completion_tokens": (existing.get("timed_completion_tokens", 0) or 0) + + (payload.get("timed_completion_tokens", 0) or 0), + } + + class DailySpendUpdateQueue(BaseUpdateQueue): """ In memory buffer for daily spend updates that should be committed to the database @@ -113,62 +156,14 @@ class DailySpendUpdateQueue(BaseUpdateQueue): updates: list[dict[str, BaseDailySpendTransaction]], ) -> dict[str, BaseDailySpendTransaction]: """Aggregate updates by daily_transaction_key.""" - aggregated_daily_spend_update_transactions: Final[dict[str, BaseDailySpendTransaction]] = {} - for _update in updates: - for _key, payload in _update.items(): - if _key in aggregated_daily_spend_update_transactions: - daily_transaction = aggregated_daily_spend_update_transactions[_key] - daily_transaction["spend"] += payload["spend"] - daily_transaction["prompt_tokens"] += payload["prompt_tokens"] - daily_transaction["completion_tokens"] += payload["completion_tokens"] - daily_transaction["api_requests"] += payload["api_requests"] - daily_transaction["successful_requests"] += payload["successful_requests"] - daily_transaction["failed_requests"] += payload["failed_requests"] - - # Add optional metrics cache_read_input_tokens and cache_creation_input_tokens - daily_transaction["cache_read_input_tokens"] = ( - payload.get("cache_read_input_tokens", 0) or 0 - ) + daily_transaction.get("cache_read_input_tokens", 0) - - daily_transaction["cache_creation_input_tokens"] = ( - payload.get("cache_creation_input_tokens", 0) or 0 - ) + daily_transaction.get("cache_creation_input_tokens", 0) - - daily_transaction["compression_saved_tokens"] = ( - payload.get("compression_saved_tokens", 0) or 0 - ) + daily_transaction.get("compression_saved_tokens", 0) - - daily_transaction["compression_savings_spend"] = ( - payload.get("compression_savings_spend", 0) or 0 - ) + daily_transaction.get("compression_savings_spend", 0) - - daily_transaction["prompt_caching_savings_spend"] = ( - payload.get("prompt_caching_savings_spend", 0) or 0 - ) + daily_transaction.get("prompt_caching_savings_spend", 0) - - daily_transaction["gateway_injected_caching_savings_spend"] = ( - payload.get("gateway_injected_caching_savings_spend", 0) or 0 - ) + daily_transaction.get("gateway_injected_caching_savings_spend", 0) - - daily_transaction["autorouter_savings_spend"] = ( - payload.get("autorouter_savings_spend", 0) or 0 - ) + daily_transaction.get("autorouter_savings_spend", 0) - - daily_transaction["total_response_time_ms"] = ( - payload.get("total_response_time_ms", 0) or 0 - ) + daily_transaction.get("total_response_time_ms", 0) - - daily_transaction["timed_requests"] = ( - payload.get("timed_requests", 0) or 0 - ) + daily_transaction.get("timed_requests", 0) - - daily_transaction["timed_completion_tokens"] = ( - payload.get("timed_completion_tokens", 0) or 0 - ) + daily_transaction.get("timed_completion_tokens", 0) - - else: - aggregated_daily_spend_update_transactions[_key] = deepcopy(payload) - return aggregated_daily_spend_update_transactions + ordered_updates: Final = sorted(_daily_spend_updates(updates), key=lambda update: update[0]) + return { + key: reduce( + _merge_daily_spend_transactions, + (deepcopy(payload) for _, payload in grouped_updates), + ) + for key, grouped_updates in groupby(ordered_updates, key=lambda update: update[0]) + } async def _emit_new_item_added_to_queue_event( self, diff --git a/litellm/proxy/management_endpoints/common_daily_activity.py b/litellm/proxy/management_endpoints/common_daily_activity.py index f663dcf120b..02649a16cb9 100644 --- a/litellm/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/proxy/management_endpoints/common_daily_activity.py @@ -247,7 +247,7 @@ def _provider_throughput( ) -> ProviderThroughputMetrics: output_tokens_per_second: Final = ( timed_completion_tokens * 1000 / total_response_time_ms - if timed_completion_tokens > 0 and timed_requests > 0 and total_response_time_ms > 0 + if timed_requests > 0 and total_response_time_ms > 0 else None ) return ProviderThroughputMetrics( diff --git a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py index f12f6d076f5..ee9dff12019 100644 --- a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py +++ b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py @@ -107,9 +107,7 @@ async def test_add_multiple_updates(daily_spend_update_queue): @pytest.mark.asyncio async def test_aggregated_daily_spend_update_empty(daily_spend_update_queue): """Test aggregating updates from an empty queue""" - result = ( - await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() - ) + result = await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() assert result == {} @@ -129,9 +127,7 @@ async def test_get_aggregated_daily_spend_update_transactions_single_key(): updates = [{test_key: test_transaction}] # Test aggregation - result = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions( - updates - ) + result = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions(updates) assert len(result) == 1 assert test_key in result @@ -164,9 +160,7 @@ async def test_get_aggregated_daily_spend_update_transactions_multiple_keys(): updates = [{test_key1: test_transaction1}, {test_key2: test_transaction2}] # Test aggregation - result = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions( - updates - ) + result = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions(updates) assert len(result) == 2 assert test_key1 in result @@ -221,9 +215,7 @@ async def test_get_aggregated_daily_spend_update_transactions_same_key(): updates = [{test_key: test_transaction1}, {test_key: test_transaction2}] # Test aggregation - result = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions( - updates - ) + result = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions(updates) assert len(result) == 1 assert test_key in result @@ -280,9 +272,7 @@ async def test_flush_and_get_aggregated_daily_spend_update_transactions( await daily_spend_update_queue.add_update({test_key: test_transaction2}) # Flush and get aggregated transactions - result = ( - await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() - ) + result = await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() assert len(result) == 1 assert test_key in result @@ -290,9 +280,7 @@ async def test_flush_and_get_aggregated_daily_spend_update_transactions( @pytest.mark.asyncio -async def test_queue_max_size_triggers_aggregation( - monkeypatch, daily_spend_update_queue -): +async def test_queue_max_size_triggers_aggregation(monkeypatch, daily_spend_update_queue): """Test that reaching MAX_SIZE_IN_MEMORY_QUEUE triggers aggregation""" # Override MAX_SIZE_IN_MEMORY_QUEUE for testing litellm._turn_on_debug() @@ -316,9 +304,7 @@ async def test_queue_max_size_triggers_aggregation( assert daily_spend_update_queue.update_queue.qsize() == 1 # Verify the aggregated values - result = ( - await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() - ) + result = await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() assert result[test_key]["spend"] == 6.0 assert result[test_key]["prompt_tokens"] == 600 assert result[test_key]["completion_tokens"] == 300 @@ -435,9 +421,7 @@ async def test_cache_token_fields_aggregation(daily_spend_update_queue): @pytest.mark.asyncio -async def test_queue_size_reduction_with_large_volume( - monkeypatch, daily_spend_update_queue -): +async def test_queue_size_reduction_with_large_volume(monkeypatch, daily_spend_update_queue): """Test that queue size is actually reduced when dealing with many items""" # Set a smaller MAX_SIZE for testing monkeypatch.setattr(daily_spend_update_queue, "MAX_SIZE_IN_MEMORY_QUEUE", 10) @@ -478,9 +462,7 @@ async def test_queue_size_reduction_with_large_volume( assert daily_spend_update_queue.update_queue.qsize() <= 10 # Verify total costs are correct - result = ( - await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() - ) + result = await daily_spend_update_queue.flush_and_get_aggregated_daily_spend_update_transactions() print("RESULT", json.dumps(result, indent=4)) assert result[user1_key]["spend"] == 200 * 0.5 # 10.0 @@ -553,6 +535,7 @@ async def test_every_optional_daily_metric_aggregates(daily_spend_update_queue): paths, so the driver reads as zero on the dashboard however much it saved. """ test_key = "user1_2023-01-01_key123_claude-haiku-4-5_anthropic" + def _numeric(annotation): # additive metrics may be declared NotRequired[float] for rows queued by a pod # running the previous release, so unwrap before matching diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index 4941ae0f827..0e5d2eff81b 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -127,9 +127,7 @@ def test_counters_increment_rather_than_overwrite(column): def test_request_id_is_preserved_when_a_later_batch_carries_none(): - sql, params = build_bulk_upsert( - TAG_TABLE, merge_by_conflict_key(TAG_TABLE, (tag_txn(request_id=None),)) - ) + sql, params = build_bulk_upsert(TAG_TABLE, merge_by_conflict_key(TAG_TABLE, (tag_txn(request_id=None),))) assert '"request_id" = COALESCE(EXCLUDED."request_id", "LiteLLM_DailyTagSpend"."request_id")' in sql assert None in params diff --git a/tests/unit/proxy/db/test_db_spend_update_writer.py b/tests/unit/proxy/db/test_db_spend_update_writer.py index c70affce0df..96d095c0595 100644 --- a/tests/unit/proxy/db/test_db_spend_update_writer.py +++ b/tests/unit/proxy/db/test_db_spend_update_writer.py @@ -103,11 +103,21 @@ async def test_update_database_attributes_router_rejected_failure_to_model_group ) with ( - patch("litellm.proxy.proxy_server.disable_spend_logs", True), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam - patch("litellm.proxy.proxy_server.prisma_client", MagicMock()), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam - patch("litellm.proxy.proxy_server.user_api_key_cache", MagicMock()), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam - patch("litellm.proxy.proxy_server.litellm_proxy_budget_name", "test-budget"), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam - patch("litellm.proxy.proxy_server.llm_router", llm_router), # test-quality-ok: get_llm_router reads this proxy_server module global at call time; no injection seam + patch( + "litellm.proxy.proxy_server.disable_spend_logs", True + ), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam + patch( + "litellm.proxy.proxy_server.prisma_client", MagicMock() + ), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam + patch( + "litellm.proxy.proxy_server.user_api_key_cache", MagicMock() + ), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam + patch( + "litellm.proxy.proxy_server.litellm_proxy_budget_name", "test-budget" + ), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam + patch( + "litellm.proxy.proxy_server.llm_router", llm_router + ), # test-quality-ok: get_llm_router reads this proxy_server module global at call time; no injection seam ): await db_writer.update_database( token="test-token", @@ -320,7 +330,9 @@ async def test_a_routed_request_reaches_the_auto_router_rollup_whether_or_not_sp } with ( - patch("litellm.proxy.proxy_server.disable_spend_logs", disable_spend_logs), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam + patch( + "litellm.proxy.proxy_server.disable_spend_logs", disable_spend_logs + ), # test-quality-ok: update_database reads this proxy_server module global at call time; no injection seam patch("litellm.proxy.proxy_server.prisma_client", prisma), patch("litellm.proxy.proxy_server.litellm_proxy_budget_name", "test-budget"), patch( @@ -2868,17 +2880,22 @@ async def test_daily_transaction_carries_compression_saved_tokens(): @pytest.mark.asyncio -@pytest.mark.parametrize("estimate, recorded_savings, expected", [ - pytest.param(None, None, -0.005, id="plain-classifier-cost"), - pytest.param({"version": 1, "status": "unknown"}, None, 0.0, id="unknown"), - pytest.param({"version": 2, "status": "unknown"}, None, 0.0, id="unknown-v2"), - pytest.param({"version": 1, "status": "unknown"}, -0.003, 0.0, id="unknown-stale-value"), - pytest.param({"version": 0, "status": "estimated"}, -0.003, 0.0, id="unsupported-version"), - pytest.param({"version": 1, "status": "estimated"}, -0.003, -0.003, id="estimated"), - pytest.param(None, -0.003, -0.003, id="legacy"), -]) +@pytest.mark.parametrize( + "estimate, recorded_savings, expected", + [ + pytest.param(None, None, -0.005, id="plain-classifier-cost"), + pytest.param({"version": 1, "status": "unknown"}, None, 0.0, id="unknown"), + pytest.param({"version": 2, "status": "unknown"}, None, 0.0, id="unknown-v2"), + pytest.param({"version": 1, "status": "unknown"}, -0.003, 0.0, id="unknown-stale-value"), + pytest.param({"version": 0, "status": "estimated"}, -0.003, 0.0, id="unsupported-version"), + pytest.param({"version": 1, "status": "estimated"}, -0.003, -0.003, id="estimated"), + pytest.param(None, -0.003, -0.003, id="legacy"), + ], +) async def test_daily_transaction_compression_saved_tokens_zero_when_absent( - estimate: dict[str, object] | None, recorded_savings: float | None, expected: float, + estimate: dict[str, object] | None, + recorded_savings: float | None, + expected: float, ) -> None: """Requests without any compression metadata produce a zero count.""" writer = DBSpendUpdateWriter() @@ -2897,12 +2914,14 @@ async def test_daily_transaction_compression_saved_tokens_zero_when_absent( "prompt_tokens": 100, "completion_tokens": 10, "spend": 0.01, - "metadata": json.dumps({ - "usage_object": {"prompt_tokens": 100, "completion_tokens": 10}, - "routing_decision": {"savings_baseline_model": "anthropic/claude-sonnet-5", "classifier_cost": 0.005}, - "autorouter_savings": recorded_savings, - "autorouter_savings_estimate": estimate, - }), + "metadata": json.dumps( + { + "usage_object": {"prompt_tokens": 100, "completion_tokens": 10}, + "routing_decision": {"savings_baseline_model": "anthropic/claude-sonnet-5", "classifier_cost": 0.005}, + "autorouter_savings": recorded_savings, + "autorouter_savings_estimate": estimate, + } + ), } transaction = await writer._common_add_spend_log_transaction_to_daily_transaction( @@ -3395,9 +3414,7 @@ async def test_failed_per_entity_increment_from_redis_restores_only_what_may_sti ) mock_redis_update_buffer.restore_transactions_to_redis.assert_awaited_once() - restored = mock_redis_update_buffer.restore_transactions_to_redis.call_args.kwargs[ - "db_spend_update_transactions" - ] + restored = mock_redis_update_buffer.restore_transactions_to_redis.call_args.kwargs["db_spend_update_transactions"] assert restored["user_list_transactions"] is None assert restored["team_list_transactions"] == {"team-1": 1.5} assert restored["key_list_transactions"] == ({"key-1": 1.5} if safe_to_resend else None) diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index fcb7ce547dc..636b83fefea 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -1858,7 +1858,7 @@ def test_grouping_sets_dispatcher_returns_provider_throughput_for_models_and_mod assert day.breakdown.model_groups["public-gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 300.0 -def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_timed_tokens(): +def test_grouping_sets_dispatcher_returns_zero_for_timed_requests_without_completion_tokens(): from litellm.proxy.management_endpoints.common_daily_activity import ( _GROUP_DATE_MODEL_PROVIDER, _aggregate_grouping_sets_records_sync, @@ -1877,7 +1877,7 @@ def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_ day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] - assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 0.0 def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown(): From ed862773d11b10342deb97fd16bc0b206aa0435d Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 22:09:09 +0530 Subject: [PATCH 4/8] fix(usage): preserve unknown historical throughput Co-authored-by: Cursor --- .../migration.sql | 12 +++---- .../litellm_proxy_extras/schema.prisma | 12 +++---- litellm/proxy/_lazy_openapi_snapshot.json | 12 +++++-- litellm/proxy/_types.py | 2 +- litellm/proxy/db/daily_spend_bulk_upsert.py | 18 ++++++++-- .../daily_spend_update_queue.py | 10 ++++-- .../common_daily_activity.py | 34 ++++++++++++++----- litellm/proxy/schema.prisma | 12 +++---- litellm/repositories/daily_activity_sql.py | 5 ++- .../common_daily_activity.py | 2 +- litellm/types/repositories/daily_activity.py | 2 +- schema.prisma | 12 +++---- .../test_daily_spend_update_queue.py | 11 +++++- .../proxy/db/test_daily_spend_bulk_upsert.py | 9 +++++ .../test_common_daily_activity.py | 22 ++++++++++++ .../src/components/UsagePage/types.ts | 2 +- ui/litellm-dashboard/src/lib/http/schema.d.ts | 7 ++-- 17 files changed, 133 insertions(+), 51 deletions(-) diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql index 6d08b2092aa..05a3a559d74 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql @@ -1,6 +1,6 @@ -ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 83b70ef6481..a584048f11a 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 06a696a6b6c..edbdb130218 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -4364,9 +4364,15 @@ "title": "Output Tokens Per Second" }, "timed_completion_tokens": { - "default": 0, - "title": "Timed Completion Tokens", - "type": "integer" + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Timed Completion Tokens" }, "timed_requests": { "default": 0, diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index f1c43b3f1ca..ab7ae03c789 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -5703,7 +5703,7 @@ class BaseDailySpendTransaction(TypedDict): failed_requests: int total_response_time_ms: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place timed_requests: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place - timed_completion_tokens: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place + timed_completion_tokens: NotRequired[int | None] class DailyTeamSpendTransaction(BaseDailySpendTransaction): diff --git a/litellm/proxy/db/daily_spend_bulk_upsert.py b/litellm/proxy/db/daily_spend_bulk_upsert.py index e4f8b62e6ae..b2bfb68813a 100644 --- a/litellm/proxy/db/daily_spend_bulk_upsert.py +++ b/litellm/proxy/db/daily_spend_bulk_upsert.py @@ -125,6 +125,20 @@ def _as_float(value: object) -> float: return float(value) if isinstance(value, (int, float)) else 0.0 +def _counter_total(column: str, group: Sequence[SpendRow]) -> int | None: + values: Final = tuple(row.get(column) for row in group) + if column == "timed_completion_tokens" and any(value is None for value in values): + return None + return sum(_as_int(value) for value in values) + + +def _counter_value(column: str, transaction: SpendRow) -> int | None: + value: Final = transaction.get(column) + if column == "timed_completion_tokens" and value is None: + return None + return _as_int(value) + + def conflict_key(table: DailySpendTable, transaction: SpendRow) -> tuple[str, ...]: """The tuple the database arbitrates the upsert on, normalized free of NULLs.""" return tuple(_as_text(transaction.get(column)) for column in (table.entity_id_column, *_KEY_COLUMNS)) @@ -135,7 +149,7 @@ def _merge(group: Sequence[SpendRow]) -> SpendRow: return group[0] return { **group[0], - **{column: sum(_as_int(row.get(column)) for row in group) for column in _COUNTER_COLUMNS}, + **{column: _counter_total(column, group) for column in _COUNTER_COLUMNS}, **{column: sum(_as_float(row.get(column)) for row in group) for column in _SPEND_COLUMNS}, } @@ -166,7 +180,7 @@ def _row_params( str(uuid.uuid4()), *key, None if transaction.get("model_group") is None else _as_text(transaction.get("model_group")), - *(_as_int(transaction.get(column)) for column in _COUNTER_COLUMNS), + *(_counter_value(column, transaction) for column in _COUNTER_COLUMNS), *(_as_float(transaction.get(column)) for column in _SPEND_COLUMNS), *((None if request_id is None else _as_text(request_id),) if table.carries_request_id else ()), ) diff --git a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py index 87c2915f7fd..6bfaab45ad1 100644 --- a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py +++ b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py @@ -22,6 +22,10 @@ def _daily_spend_updates( yield from update.items() +def _sum_timed_completion_tokens(left: int | None, right: int | None) -> int | None: + return left + right if left is not None and right is not None else None + + def _merge_daily_spend_transactions( existing: BaseDailySpendTransaction, payload: BaseDailySpendTransaction, @@ -51,8 +55,10 @@ def _merge_daily_spend_transactions( "total_response_time_ms": (existing.get("total_response_time_ms", 0) or 0) + (payload.get("total_response_time_ms", 0) or 0), "timed_requests": (existing.get("timed_requests", 0) or 0) + (payload.get("timed_requests", 0) or 0), - "timed_completion_tokens": (existing.get("timed_completion_tokens", 0) or 0) - + (payload.get("timed_completion_tokens", 0) or 0), + "timed_completion_tokens": _sum_timed_completion_tokens( + existing.get("timed_completion_tokens"), + payload.get("timed_completion_tokens"), + ), } diff --git a/litellm/proxy/management_endpoints/common_daily_activity.py b/litellm/proxy/management_endpoints/common_daily_activity.py index 02649a16cb9..b5aaeff549d 100644 --- a/litellm/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/proxy/management_endpoints/common_daily_activity.py @@ -162,7 +162,7 @@ class DailySpendRecord(Protocol): def timed_requests(self) -> int: ... @property - def timed_completion_tokens(self) -> int: ... + def timed_completion_tokens(self) -> int | None: ... class _KeyMetadataDict(TypedDict, total=False): @@ -241,13 +241,13 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) -> def _provider_throughput( - timed_completion_tokens: int, + timed_completion_tokens: int | None, total_response_time_ms: int, timed_requests: int, ) -> ProviderThroughputMetrics: output_tokens_per_second: Final = ( timed_completion_tokens * 1000 / total_response_time_ms - if timed_requests > 0 and total_response_time_ms > 0 + if timed_completion_tokens is not None and timed_requests > 0 and total_response_time_ms > 0 else None ) return ProviderThroughputMetrics( @@ -258,18 +258,34 @@ def _provider_throughput( ) +def _combined_timed_completion_tokens( + existing: ProviderThroughputMetrics | None, + current: int | None, +) -> int | None: + if existing is None: + return current + if existing.timed_completion_tokens is None or current is None: + return None + return existing.timed_completion_tokens + current + + def _update_provider_throughput( target: MetricWithMetadata, provider: str, record: DailySpendRecord, ) -> None: - existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics()) + existing: Final = target.provider_breakdown.get(provider) + timed_completion_tokens: Final = _combined_timed_completion_tokens( + existing, + record.timed_completion_tokens, + ) target.provider_breakdown = { **target.provider_breakdown, provider: _provider_throughput( - timed_completion_tokens=existing.timed_completion_tokens + (record.timed_completion_tokens or 0), - total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0), - timed_requests=existing.timed_requests + (record.timed_requests or 0), + timed_completion_tokens=timed_completion_tokens, + total_response_time_ms=(existing.total_response_time_ms if existing is not None else 0) + + (record.total_response_time_ms or 0), + timed_requests=(existing.timed_requests if existing is not None else 0) + (record.timed_requests or 0), ), } @@ -844,7 +860,7 @@ def _aggregate_grouping_sets_records_sync( metadata={}, provider_breakdown={ provider: _provider_throughput( - record.timed_completion_tokens or 0, + record.timed_completion_tokens, record.total_response_time_ms or 0, record.timed_requests or 0, ) @@ -854,7 +870,7 @@ def _aggregate_grouping_sets_records_sync( parent.provider_breakdown = { **parent.provider_breakdown, provider: _provider_throughput( - record.timed_completion_tokens or 0, + record.timed_completion_tokens, record.total_response_time_ms or 0, record.timed_requests or 0, ), diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 83b70ef6481..a584048f11a 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/litellm/repositories/daily_activity_sql.py b/litellm/repositories/daily_activity_sql.py index 945136a2e8b..3a253c5de3a 100644 --- a/litellm/repositories/daily_activity_sql.py +++ b/litellm/repositories/daily_activity_sql.py @@ -130,7 +130,10 @@ def _rollup_metric_select(table: DailyActivityTable) -> str: SUM(failed_requests)::bigint AS failed_requests, SUM(total_response_time_ms)::bigint AS total_response_time_ms, SUM(timed_requests)::bigint AS timed_requests, - SUM(timed_completion_tokens)::bigint AS timed_completion_tokens""" + CASE + WHEN COUNT(timed_completion_tokens) = COUNT(*) THEN SUM(timed_completion_tokens)::bigint + ELSE NULL::bigint + END AS timed_completion_tokens""" def _validate_api_key_limit(api_key_limit: int) -> None: diff --git a/litellm/types/proxy/management_endpoints/common_daily_activity.py b/litellm/types/proxy/management_endpoints/common_daily_activity.py index ee0d599a720..9f7436c68a7 100644 --- a/litellm/types/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/types/proxy/management_endpoints/common_daily_activity.py @@ -57,7 +57,7 @@ class KeyMetricWithMetadata(MetricBase): class ProviderThroughputMetrics(BaseModel): - timed_completion_tokens: int = Field(default=0) + timed_completion_tokens: int | None = Field(default=None) total_response_time_ms: int = Field(default=0) timed_requests: int = Field(default=0) output_tokens_per_second: float | None = Field(default=None) diff --git a/litellm/types/repositories/daily_activity.py b/litellm/types/repositories/daily_activity.py index b4dae9cd9e1..f15eeba7208 100644 --- a/litellm/types/repositories/daily_activity.py +++ b/litellm/types/repositories/daily_activity.py @@ -185,7 +185,7 @@ class DailyActivityRow(Protocol): failed_requests: int total_response_time_ms: int timed_requests: int - timed_completion_tokens: int + timed_completion_tokens: int | None @dataclass(frozen=True, slots=True) diff --git a/schema.prisma b/schema.prisma index 83b70ef6481..a584048f11a 100644 --- a/schema.prisma +++ b/schema.prisma @@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py index ee9dff12019..5212c29a6c7 100644 --- a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py +++ b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py @@ -575,7 +575,15 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates( await daily_spend_update_queue.add_update({test_key: dict(base)}) await daily_spend_update_queue.add_update( - {test_key: {**base, "autorouter_savings_spend": 0.25, "total_response_time_ms": 900, "timed_requests": 1}} + { + test_key: { + **base, + "autorouter_savings_spend": 0.25, + "total_response_time_ms": 900, + "timed_requests": 1, + "timed_completion_tokens": 5, + } + } ) await daily_spend_update_queue.aggregate_queue_updates() updates = await daily_spend_update_queue.flush_all_updates_from_in_memory_queue() @@ -583,3 +591,4 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates( assert updates[0][test_key]["autorouter_savings_spend"] == pytest.approx(0.25) assert updates[0][test_key]["total_response_time_ms"] == 900 assert updates[0][test_key]["timed_requests"] == 1 + assert updates[0][test_key]["timed_completion_tokens"] is None diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index 0e5d2eff81b..f63ef1acb64 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -72,6 +72,15 @@ def test_null_and_empty_provider_merge_into_one_row(order): assert folded["api_requests"] == 4 +def test_unknown_timed_tokens_remain_unknown_when_rows_merge(): + legacy = tag_txn() + del legacy["timed_completion_tokens"] + + merged = merge_by_conflict_key(TAG_TABLE, (legacy, tag_txn(timed_completion_tokens=7))) + + assert merged[0][1]["timed_completion_tokens"] is None + + def test_distinct_keys_are_not_merged_and_are_ordered_deterministically(): unordered = (tag_txn(tag="z-team"), tag_txn(tag="a-team"), tag_txn(tag="m-team")) diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index 636b83fefea..7e3a80f75da 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -1880,6 +1880,28 @@ def test_grouping_sets_dispatcher_returns_zero_for_timed_requests_without_comple assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 0.0 +def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_timed_tokens(): + from litellm.proxy.management_endpoints.common_daily_activity import ( + _GROUP_DATE_MODEL_PROVIDER, + _aggregate_grouping_sets_records_sync, + ) + + records = [ + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + completion_tokens=900, + timed_completion_tokens=None, + total_response_time_ms=3000, + timed_requests=1, + ) + ] + + day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] + + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None + + def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown(): """Sentinel rows carry no provider, so their flat cost must not surface under the "unknown" provider - the per-row path skips them for exactly the same reason.""" diff --git a/ui/litellm-dashboard/src/components/UsagePage/types.ts b/ui/litellm-dashboard/src/components/UsagePage/types.ts index fa8fb6c3a4e..bf6769046b2 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/types.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/types.ts @@ -42,7 +42,7 @@ export interface MetricWithMetadata { } export interface ProviderThroughputMetrics { - timed_completion_tokens: number; + timed_completion_tokens?: number | null; total_response_time_ms: number; timed_requests: number; output_tokens_per_second?: number | null; diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index ba4158a8b5d..521223b587c 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -41076,11 +41076,8 @@ export interface components { ProviderThroughputMetrics: { /** Output Tokens Per Second */ output_tokens_per_second?: number | null; - /** - * Timed Completion Tokens - * @default 0 - */ - timed_completion_tokens: number; + /** Timed Completion Tokens */ + timed_completion_tokens?: number | null; /** * Timed Requests * @default 0 From deec7f26d152887df989a0d1ed68b609e769682b Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 22:25:48 +0530 Subject: [PATCH 5/8] fix(usage): retain new throughput after upgrades Co-authored-by: Cursor --- litellm/proxy/db/daily_spend_bulk_upsert.py | 24 ++++++++++++++----- .../daily_spend_update_queue.py | 13 ++++++++-- litellm/repositories/daily_activity_sql.py | 6 +++-- .../test_daily_spend_update_queue.py | 20 +++++++++++++++- .../proxy/db/test_daily_spend_bulk_upsert.py | 21 ++++++++++++++-- 5 files changed, 71 insertions(+), 13 deletions(-) diff --git a/litellm/proxy/db/daily_spend_bulk_upsert.py b/litellm/proxy/db/daily_spend_bulk_upsert.py index b2bfb68813a..bfc49de3dd6 100644 --- a/litellm/proxy/db/daily_spend_bulk_upsert.py +++ b/litellm/proxy/db/daily_spend_bulk_upsert.py @@ -127,14 +127,17 @@ def _as_float(value: object) -> float: def _counter_total(column: str, group: Sequence[SpendRow]) -> int | None: values: Final = tuple(row.get(column) for row in group) - if column == "timed_completion_tokens" and any(value is None for value in values): + has_unknown_timed_tokens: Final = column == "timed_completion_tokens" and any( + row.get("timed_completion_tokens") is None and _as_int(row.get("timed_requests")) > 0 for row in group + ) + if has_unknown_timed_tokens: return None return sum(_as_int(value) for value in values) def _counter_value(column: str, transaction: SpendRow) -> int | None: value: Final = transaction.get(column) - if column == "timed_completion_tokens" and value is None: + if column == "timed_completion_tokens" and value is None and _as_int(transaction.get("timed_requests")) > 0: return None return _as_int(value) @@ -214,9 +217,18 @@ def build_bulk_upsert( + ", (NOW() AT TIME ZONE 'UTC'))" for row_index in range(len(batch)) ) - increments: Final = ", ".join( - f'"{column}" = {quoted_table}."{column}" + EXCLUDED."{column}"' - for column in (*_COUNTER_COLUMNS, *_SPEND_COLUMNS) + regular_increment_columns: Final = tuple( + column for column in (*_COUNTER_COLUMNS, *_SPEND_COLUMNS) if column != "timed_completion_tokens" + ) + regular_increments: Final = ", ".join( + f'"{column}" = {quoted_table}."{column}" + EXCLUDED."{column}"' for column in regular_increment_columns + ) + timed_completion_tokens_increment: Final = ( + f'"timed_completion_tokens" = CASE ' + f'WHEN ({quoted_table}."timed_requests" > 0 AND {quoted_table}."timed_completion_tokens" IS NULL) ' + 'OR (EXCLUDED."timed_requests" > 0 AND EXCLUDED."timed_completion_tokens" IS NULL) THEN NULL ' + f'ELSE COALESCE({quoted_table}."timed_completion_tokens", 0) ' + '+ COALESCE(EXCLUDED."timed_completion_tokens", 0) END' ) # request_id names one arbitrary contributing request, so an entry carrying none must # not blank out the one already recorded. @@ -229,7 +241,7 @@ def build_bulk_upsert( f'INSERT INTO {quoted_table} ({_quoted(columns)}, "updated_at")\n' f"VALUES {rows}\n" f"ON CONFLICT ({_quoted((table.entity_id_column, *_KEY_COLUMNS))}) DO UPDATE SET\n" - f" {increments}{request_id_update},\n" + f" {regular_increments}, {timed_completion_tokens_increment}{request_id_update},\n" f" \"updated_at\" = (NOW() AT TIME ZONE 'UTC')" ) return sql, tuple(value for key, transaction in batch for value in _row_params(table, key, transaction)) diff --git a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py index 6bfaab45ad1..7dc3d42b818 100644 --- a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py +++ b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py @@ -22,8 +22,15 @@ def _daily_spend_updates( yield from update.items() -def _sum_timed_completion_tokens(left: int | None, right: int | None) -> int | None: - return left + right if left is not None and right is not None else None +def _sum_timed_completion_tokens( + left_tokens: int | None, + left_requests: int, + right_tokens: int | None, + right_requests: int, +) -> int | None: + if (left_tokens is None and left_requests > 0) or (right_tokens is None and right_requests > 0): + return None + return (left_tokens or 0) + (right_tokens or 0) def _merge_daily_spend_transactions( @@ -57,7 +64,9 @@ def _merge_daily_spend_transactions( "timed_requests": (existing.get("timed_requests", 0) or 0) + (payload.get("timed_requests", 0) or 0), "timed_completion_tokens": _sum_timed_completion_tokens( existing.get("timed_completion_tokens"), + existing.get("timed_requests", 0) or 0, payload.get("timed_completion_tokens"), + payload.get("timed_requests", 0) or 0, ), } diff --git a/litellm/repositories/daily_activity_sql.py b/litellm/repositories/daily_activity_sql.py index 3a253c5de3a..005372a598e 100644 --- a/litellm/repositories/daily_activity_sql.py +++ b/litellm/repositories/daily_activity_sql.py @@ -131,8 +131,10 @@ def _rollup_metric_select(table: DailyActivityTable) -> str: SUM(total_response_time_ms)::bigint AS total_response_time_ms, SUM(timed_requests)::bigint AS timed_requests, CASE - WHEN COUNT(timed_completion_tokens) = COUNT(*) THEN SUM(timed_completion_tokens)::bigint - ELSE NULL::bigint + WHEN COUNT(*) FILTER ( + WHERE timed_requests > 0 AND timed_completion_tokens IS NULL + ) > 0 THEN NULL::bigint + ELSE COALESCE(SUM(timed_completion_tokens), 0)::bigint END AS timed_completion_tokens""" diff --git a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py index 5212c29a6c7..0b203b9703d 100644 --- a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py +++ b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py @@ -591,4 +591,22 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates( assert updates[0][test_key]["autorouter_savings_spend"] == pytest.approx(0.25) assert updates[0][test_key]["total_response_time_ms"] == 900 assert updates[0][test_key]["timed_requests"] == 1 - assert updates[0][test_key]["timed_completion_tokens"] is None + assert updates[0][test_key]["timed_completion_tokens"] == 5 + + +def test_legacy_timed_request_keeps_timed_tokens_unknown(): + key = "user1_2023-01-01_key123_gpt-4o_openai" + base = { + "spend": 1.0, + "prompt_tokens": 10, + "completion_tokens": 5, + "api_requests": 1, + "successful_requests": 1, + "failed_requests": 0, + "timed_requests": 1, + } + updates = [{key: base}, {key: {**base, "timed_completion_tokens": 5}}] + + aggregated = DailySpendUpdateQueue.get_aggregated_daily_spend_update_transactions(updates) + + assert aggregated[key]["timed_completion_tokens"] is None diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index f63ef1acb64..81fb28ffe58 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -72,12 +72,21 @@ def test_null_and_empty_provider_merge_into_one_row(order): assert folded["api_requests"] == 4 -def test_unknown_timed_tokens_remain_unknown_when_rows_merge(): +def test_untimed_legacy_rows_do_not_hide_known_timed_tokens_when_rows_merge(): legacy = tag_txn() del legacy["timed_completion_tokens"] merged = merge_by_conflict_key(TAG_TABLE, (legacy, tag_txn(timed_completion_tokens=7))) + assert merged[0][1]["timed_completion_tokens"] == 7 + + +def test_legacy_timed_requests_keep_timed_tokens_unknown_when_rows_merge(): + legacy = tag_txn(timed_requests=1) + del legacy["timed_completion_tokens"] + + merged = merge_by_conflict_key(TAG_TABLE, (legacy, tag_txn(timed_completion_tokens=7))) + assert merged[0][1]["timed_completion_tokens"] is None @@ -125,7 +134,6 @@ def test_conflict_target_is_the_full_unique_constraint(): "failed_requests", "total_response_time_ms", "timed_requests", - "timed_completion_tokens", ], ) def test_counters_increment_rather_than_overwrite(column): @@ -135,6 +143,15 @@ def test_counters_increment_rather_than_overwrite(column): assert f'"{column}" = "LiteLLM_DailyTagSpend"."{column}" + EXCLUDED."{column}"' in sql +def test_timed_token_upsert_preserves_unknown_legacy_measurements(): + sql, _ = build_bulk_upsert(TAG_TABLE, merge_by_conflict_key(TAG_TABLE, (tag_txn(),))) + + assert '"timed_requests" > 0 AND "LiteLLM_DailyTagSpend"."timed_completion_tokens" IS NULL' in sql + assert 'EXCLUDED."timed_requests" > 0 AND EXCLUDED."timed_completion_tokens" IS NULL' in sql + assert 'COALESCE("LiteLLM_DailyTagSpend"."timed_completion_tokens", 0)' in sql + assert '+ COALESCE(EXCLUDED."timed_completion_tokens", 0)' in sql + + def test_request_id_is_preserved_when_a_later_batch_carries_none(): sql, params = build_bulk_upsert(TAG_TABLE, merge_by_conflict_key(TAG_TABLE, (tag_txn(request_id=None),))) From 311494b59d3556cd7e78d931fb62ce62ffc8cd73 Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 22:38:35 +0530 Subject: [PATCH 6/8] test(usage): keep throughput assertions behavioral Co-authored-by: Cursor --- tests/unit/proxy/db/test_daily_spend_bulk_upsert.py | 9 --------- 1 file changed, 9 deletions(-) diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index 81fb28ffe58..2101d5c2a4e 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -143,15 +143,6 @@ def test_counters_increment_rather_than_overwrite(column): assert f'"{column}" = "LiteLLM_DailyTagSpend"."{column}" + EXCLUDED."{column}"' in sql -def test_timed_token_upsert_preserves_unknown_legacy_measurements(): - sql, _ = build_bulk_upsert(TAG_TABLE, merge_by_conflict_key(TAG_TABLE, (tag_txn(),))) - - assert '"timed_requests" > 0 AND "LiteLLM_DailyTagSpend"."timed_completion_tokens" IS NULL' in sql - assert 'EXCLUDED."timed_requests" > 0 AND EXCLUDED."timed_completion_tokens" IS NULL' in sql - assert 'COALESCE("LiteLLM_DailyTagSpend"."timed_completion_tokens", 0)' in sql - assert '+ COALESCE(EXCLUDED."timed_completion_tokens", 0)' in sql - - def test_request_id_is_preserved_when_a_later_batch_carries_none(): sql, params = build_bulk_upsert(TAG_TABLE, merge_by_conflict_key(TAG_TABLE, (tag_txn(request_id=None),))) From 5e2381f217fd1d7c9bc845e0fd48c809febac63b Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 22:54:02 +0530 Subject: [PATCH 7/8] test(usage): cover unknown throughput branches Co-authored-by: Cursor --- tests/unit/proxy/db/test_daily_spend_bulk_upsert.py | 8 ++++++++ .../management_endpoints/test_common_daily_activity.py | 9 +++++++++ 2 files changed, 17 insertions(+) diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index 2101d5c2a4e..ff5b43e5b39 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -8,6 +8,7 @@ import pytest from litellm.proxy.db.daily_spend_bulk_upsert import ( DAILY_SPEND_TABLES, + _counter_value, build_bulk_upsert, conflict_key, merge_by_conflict_key, @@ -90,6 +91,13 @@ def test_legacy_timed_requests_keep_timed_tokens_unknown_when_rows_merge(): assert merged[0][1]["timed_completion_tokens"] is None +def test_legacy_timed_request_serializes_unknown_token_total(): + legacy = tag_txn(timed_requests=1) + del legacy["timed_completion_tokens"] + + assert _counter_value("timed_completion_tokens", legacy) is None + + def test_distinct_keys_are_not_merged_and_are_ordered_deterministically(): unordered = (tag_txn(tag="z-team"), tag_txn(tag="a-team"), tag_txn(tag="m-team")) diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index 7e3a80f75da..74b3b2ae463 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -1902,6 +1902,15 @@ def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_ assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None +def test_combined_timed_tokens_propagates_unknown_measurements(): + from litellm.proxy.management_endpoints.common_daily_activity import _combined_timed_completion_tokens + from litellm.types.proxy.management_endpoints.common_daily_activity import ProviderThroughputMetrics + + existing = ProviderThroughputMetrics(timed_completion_tokens=10) + + assert _combined_timed_completion_tokens(existing, None) is None + + def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown(): """Sentinel rows carry no provider, so their flat cost must not surface under the "unknown" provider - the per-row path skips them for exactly the same reason.""" From 01fafb270c65e3fcb4e63f6ac4dec7026662b0ab Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 22:55:47 +0530 Subject: [PATCH 8/8] fix(observability): register background settlement model Co-authored-by: Cursor --- litellm/integrations/otel/model/spans.py | 1 + 1 file changed, 1 insertion(+) diff --git a/litellm/integrations/otel/model/spans.py b/litellm/integrations/otel/model/spans.py index 2cc8e035ebd..f04a629278c 100644 --- a/litellm/integrations/otel/model/spans.py +++ b/litellm/integrations/otel/model/spans.py @@ -347,6 +347,7 @@ _PRISMA_MODELS: Final[frozenset[str]] = frozenset( "LiteLLM_WorkflowRun", "LiteLLM_WorkflowEvent", "LiteLLM_WorkflowMessage", + "LiteLLM_BackgroundInteractionSettlement", "LiteLLM_Lens", "LiteLLM_LensRun", "LiteLLM_LensWorker",