From ed862773d11b10342deb97fd16bc0b206aa0435d Mon Sep 17 00:00:00 2001 From: atul naik Date: Sat, 3 Oct 2026 22:09:09 +0530 Subject: [PATCH] fix(usage): preserve unknown historical throughput Co-authored-by: Cursor --- .../migration.sql | 12 +++---- .../litellm_proxy_extras/schema.prisma | 12 +++---- litellm/proxy/_lazy_openapi_snapshot.json | 12 +++++-- litellm/proxy/_types.py | 2 +- litellm/proxy/db/daily_spend_bulk_upsert.py | 18 ++++++++-- .../daily_spend_update_queue.py | 10 ++++-- .../common_daily_activity.py | 34 ++++++++++++++----- litellm/proxy/schema.prisma | 12 +++---- litellm/repositories/daily_activity_sql.py | 5 ++- .../common_daily_activity.py | 2 +- litellm/types/repositories/daily_activity.py | 2 +- schema.prisma | 12 +++---- .../test_daily_spend_update_queue.py | 11 +++++- .../proxy/db/test_daily_spend_bulk_upsert.py | 9 +++++ .../test_common_daily_activity.py | 22 ++++++++++++ .../src/components/UsagePage/types.ts | 2 +- ui/litellm-dashboard/src/lib/http/schema.d.ts | 7 ++-- 17 files changed, 133 insertions(+), 51 deletions(-) diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql index 6d08b2092aa..05a3a559d74 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261003140000_add_timed_completion_tokens/migration.sql @@ -1,6 +1,6 @@ -ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; -ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT NOT NULL DEFAULT 0; +ALTER TABLE "LiteLLM_DailyUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyOrganizationSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyEndUserSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyAgentSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyTeamSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; +ALTER TABLE "LiteLLM_DailyTagSpend" ADD COLUMN IF NOT EXISTS "timed_completion_tokens" BIGINT; diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 83b70ef6481..a584048f11a 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/litellm/proxy/_lazy_openapi_snapshot.json b/litellm/proxy/_lazy_openapi_snapshot.json index 06a696a6b6c..edbdb130218 100644 --- a/litellm/proxy/_lazy_openapi_snapshot.json +++ b/litellm/proxy/_lazy_openapi_snapshot.json @@ -4364,9 +4364,15 @@ "title": "Output Tokens Per Second" }, "timed_completion_tokens": { - "default": 0, - "title": "Timed Completion Tokens", - "type": "integer" + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "title": "Timed Completion Tokens" }, "timed_requests": { "default": 0, diff --git a/litellm/proxy/_types.py b/litellm/proxy/_types.py index f1c43b3f1ca..ab7ae03c789 100644 --- a/litellm/proxy/_types.py +++ b/litellm/proxy/_types.py @@ -5703,7 +5703,7 @@ class BaseDailySpendTransaction(TypedDict): failed_requests: int total_response_time_ms: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place timed_requests: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place - timed_completion_tokens: NotRequired[int] # writable-ok: the rollup queue accumulates into this key in place + timed_completion_tokens: NotRequired[int | None] class DailyTeamSpendTransaction(BaseDailySpendTransaction): diff --git a/litellm/proxy/db/daily_spend_bulk_upsert.py b/litellm/proxy/db/daily_spend_bulk_upsert.py index e4f8b62e6ae..b2bfb68813a 100644 --- a/litellm/proxy/db/daily_spend_bulk_upsert.py +++ b/litellm/proxy/db/daily_spend_bulk_upsert.py @@ -125,6 +125,20 @@ def _as_float(value: object) -> float: return float(value) if isinstance(value, (int, float)) else 0.0 +def _counter_total(column: str, group: Sequence[SpendRow]) -> int | None: + values: Final = tuple(row.get(column) for row in group) + if column == "timed_completion_tokens" and any(value is None for value in values): + return None + return sum(_as_int(value) for value in values) + + +def _counter_value(column: str, transaction: SpendRow) -> int | None: + value: Final = transaction.get(column) + if column == "timed_completion_tokens" and value is None: + return None + return _as_int(value) + + def conflict_key(table: DailySpendTable, transaction: SpendRow) -> tuple[str, ...]: """The tuple the database arbitrates the upsert on, normalized free of NULLs.""" return tuple(_as_text(transaction.get(column)) for column in (table.entity_id_column, *_KEY_COLUMNS)) @@ -135,7 +149,7 @@ def _merge(group: Sequence[SpendRow]) -> SpendRow: return group[0] return { **group[0], - **{column: sum(_as_int(row.get(column)) for row in group) for column in _COUNTER_COLUMNS}, + **{column: _counter_total(column, group) for column in _COUNTER_COLUMNS}, **{column: sum(_as_float(row.get(column)) for row in group) for column in _SPEND_COLUMNS}, } @@ -166,7 +180,7 @@ def _row_params( str(uuid.uuid4()), *key, None if transaction.get("model_group") is None else _as_text(transaction.get("model_group")), - *(_as_int(transaction.get(column)) for column in _COUNTER_COLUMNS), + *(_counter_value(column, transaction) for column in _COUNTER_COLUMNS), *(_as_float(transaction.get(column)) for column in _SPEND_COLUMNS), *((None if request_id is None else _as_text(request_id),) if table.carries_request_id else ()), ) diff --git a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py index 87c2915f7fd..6bfaab45ad1 100644 --- a/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py +++ b/litellm/proxy/db/db_transaction_queue/daily_spend_update_queue.py @@ -22,6 +22,10 @@ def _daily_spend_updates( yield from update.items() +def _sum_timed_completion_tokens(left: int | None, right: int | None) -> int | None: + return left + right if left is not None and right is not None else None + + def _merge_daily_spend_transactions( existing: BaseDailySpendTransaction, payload: BaseDailySpendTransaction, @@ -51,8 +55,10 @@ def _merge_daily_spend_transactions( "total_response_time_ms": (existing.get("total_response_time_ms", 0) or 0) + (payload.get("total_response_time_ms", 0) or 0), "timed_requests": (existing.get("timed_requests", 0) or 0) + (payload.get("timed_requests", 0) or 0), - "timed_completion_tokens": (existing.get("timed_completion_tokens", 0) or 0) - + (payload.get("timed_completion_tokens", 0) or 0), + "timed_completion_tokens": _sum_timed_completion_tokens( + existing.get("timed_completion_tokens"), + payload.get("timed_completion_tokens"), + ), } diff --git a/litellm/proxy/management_endpoints/common_daily_activity.py b/litellm/proxy/management_endpoints/common_daily_activity.py index 02649a16cb9..b5aaeff549d 100644 --- a/litellm/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/proxy/management_endpoints/common_daily_activity.py @@ -162,7 +162,7 @@ class DailySpendRecord(Protocol): def timed_requests(self) -> int: ... @property - def timed_completion_tokens(self) -> int: ... + def timed_completion_tokens(self) -> int | None: ... class _KeyMetadataDict(TypedDict, total=False): @@ -241,13 +241,13 @@ def update_metrics(existing_metrics: SpendMetrics, record: DailySpendRecord) -> def _provider_throughput( - timed_completion_tokens: int, + timed_completion_tokens: int | None, total_response_time_ms: int, timed_requests: int, ) -> ProviderThroughputMetrics: output_tokens_per_second: Final = ( timed_completion_tokens * 1000 / total_response_time_ms - if timed_requests > 0 and total_response_time_ms > 0 + if timed_completion_tokens is not None and timed_requests > 0 and total_response_time_ms > 0 else None ) return ProviderThroughputMetrics( @@ -258,18 +258,34 @@ def _provider_throughput( ) +def _combined_timed_completion_tokens( + existing: ProviderThroughputMetrics | None, + current: int | None, +) -> int | None: + if existing is None: + return current + if existing.timed_completion_tokens is None or current is None: + return None + return existing.timed_completion_tokens + current + + def _update_provider_throughput( target: MetricWithMetadata, provider: str, record: DailySpendRecord, ) -> None: - existing: Final = target.provider_breakdown.get(provider, ProviderThroughputMetrics()) + existing: Final = target.provider_breakdown.get(provider) + timed_completion_tokens: Final = _combined_timed_completion_tokens( + existing, + record.timed_completion_tokens, + ) target.provider_breakdown = { **target.provider_breakdown, provider: _provider_throughput( - timed_completion_tokens=existing.timed_completion_tokens + (record.timed_completion_tokens or 0), - total_response_time_ms=existing.total_response_time_ms + (record.total_response_time_ms or 0), - timed_requests=existing.timed_requests + (record.timed_requests or 0), + timed_completion_tokens=timed_completion_tokens, + total_response_time_ms=(existing.total_response_time_ms if existing is not None else 0) + + (record.total_response_time_ms or 0), + timed_requests=(existing.timed_requests if existing is not None else 0) + (record.timed_requests or 0), ), } @@ -844,7 +860,7 @@ def _aggregate_grouping_sets_records_sync( metadata={}, provider_breakdown={ provider: _provider_throughput( - record.timed_completion_tokens or 0, + record.timed_completion_tokens, record.total_response_time_ms or 0, record.timed_requests or 0, ) @@ -854,7 +870,7 @@ def _aggregate_grouping_sets_records_sync( parent.provider_breakdown = { **parent.provider_breakdown, provider: _provider_throughput( - record.timed_completion_tokens or 0, + record.timed_completion_tokens, record.total_response_time_ms or 0, record.timed_requests or 0, ), diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 83b70ef6481..a584048f11a 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/litellm/repositories/daily_activity_sql.py b/litellm/repositories/daily_activity_sql.py index 945136a2e8b..3a253c5de3a 100644 --- a/litellm/repositories/daily_activity_sql.py +++ b/litellm/repositories/daily_activity_sql.py @@ -130,7 +130,10 @@ def _rollup_metric_select(table: DailyActivityTable) -> str: SUM(failed_requests)::bigint AS failed_requests, SUM(total_response_time_ms)::bigint AS total_response_time_ms, SUM(timed_requests)::bigint AS timed_requests, - SUM(timed_completion_tokens)::bigint AS timed_completion_tokens""" + CASE + WHEN COUNT(timed_completion_tokens) = COUNT(*) THEN SUM(timed_completion_tokens)::bigint + ELSE NULL::bigint + END AS timed_completion_tokens""" def _validate_api_key_limit(api_key_limit: int) -> None: diff --git a/litellm/types/proxy/management_endpoints/common_daily_activity.py b/litellm/types/proxy/management_endpoints/common_daily_activity.py index ee0d599a720..9f7436c68a7 100644 --- a/litellm/types/proxy/management_endpoints/common_daily_activity.py +++ b/litellm/types/proxy/management_endpoints/common_daily_activity.py @@ -57,7 +57,7 @@ class KeyMetricWithMetadata(MetricBase): class ProviderThroughputMetrics(BaseModel): - timed_completion_tokens: int = Field(default=0) + timed_completion_tokens: int | None = Field(default=None) total_response_time_ms: int = Field(default=0) timed_requests: int = Field(default=0) output_tokens_per_second: float | None = Field(default=None) diff --git a/litellm/types/repositories/daily_activity.py b/litellm/types/repositories/daily_activity.py index b4dae9cd9e1..f15eeba7208 100644 --- a/litellm/types/repositories/daily_activity.py +++ b/litellm/types/repositories/daily_activity.py @@ -185,7 +185,7 @@ class DailyActivityRow(Protocol): failed_requests: int total_response_time_ms: int timed_requests: int - timed_completion_tokens: int + timed_completion_tokens: int | None @dataclass(frozen=True, slots=True) diff --git a/schema.prisma b/schema.prisma index 83b70ef6481..a584048f11a 100644 --- a/schema.prisma +++ b/schema.prisma @@ -870,7 +870,7 @@ model LiteLLM_DailyUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -940,7 +940,7 @@ model LiteLLM_DailyOrganizationSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -979,7 +979,7 @@ model LiteLLM_DailyEndUserSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([end_user_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1017,7 +1017,7 @@ model LiteLLM_DailyAgentSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt @@unique([agent_id, date, api_key, model, custom_llm_provider, mcp_namespaced_tool_name, endpoint]) @@ -1055,7 +1055,7 @@ model LiteLLM_DailyTeamSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? ptu_flat_cost Float @default(0.0) created_at DateTime @default(now()) updated_at DateTime @updatedAt @@ -1096,7 +1096,7 @@ model LiteLLM_DailyTagSpend { failed_requests BigInt @default(0) total_response_time_ms BigInt @default(0) timed_requests BigInt @default(0) - timed_completion_tokens BigInt @default(0) + timed_completion_tokens BigInt? created_at DateTime @default(now()) updated_at DateTime @updatedAt diff --git a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py index ee9dff12019..5212c29a6c7 100644 --- a/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py +++ b/tests/unit/proxy/db/db_transaction_queue/test_daily_spend_update_queue.py @@ -575,7 +575,15 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates( await daily_spend_update_queue.add_update({test_key: dict(base)}) await daily_spend_update_queue.add_update( - {test_key: {**base, "autorouter_savings_spend": 0.25, "total_response_time_ms": 900, "timed_requests": 1}} + { + test_key: { + **base, + "autorouter_savings_spend": 0.25, + "total_response_time_ms": 900, + "timed_requests": 1, + "timed_completion_tokens": 5, + } + } ) await daily_spend_update_queue.aggregate_queue_updates() updates = await daily_spend_update_queue.flush_all_updates_from_in_memory_queue() @@ -583,3 +591,4 @@ async def test_optional_metric_missing_from_an_older_payload_still_aggregates( assert updates[0][test_key]["autorouter_savings_spend"] == pytest.approx(0.25) assert updates[0][test_key]["total_response_time_ms"] == 900 assert updates[0][test_key]["timed_requests"] == 1 + assert updates[0][test_key]["timed_completion_tokens"] is None diff --git a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py index 0e5d2eff81b..f63ef1acb64 100644 --- a/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py +++ b/tests/unit/proxy/db/test_daily_spend_bulk_upsert.py @@ -72,6 +72,15 @@ def test_null_and_empty_provider_merge_into_one_row(order): assert folded["api_requests"] == 4 +def test_unknown_timed_tokens_remain_unknown_when_rows_merge(): + legacy = tag_txn() + del legacy["timed_completion_tokens"] + + merged = merge_by_conflict_key(TAG_TABLE, (legacy, tag_txn(timed_completion_tokens=7))) + + assert merged[0][1]["timed_completion_tokens"] is None + + def test_distinct_keys_are_not_merged_and_are_ordered_deterministically(): unordered = (tag_txn(tag="z-team"), tag_txn(tag="a-team"), tag_txn(tag="m-team")) diff --git a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py index 636b83fefea..7e3a80f75da 100644 --- a/tests/unit/proxy/management_endpoints/test_common_daily_activity.py +++ b/tests/unit/proxy/management_endpoints/test_common_daily_activity.py @@ -1880,6 +1880,28 @@ def test_grouping_sets_dispatcher_returns_zero_for_timed_requests_without_comple assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second == 0.0 +def test_grouping_sets_dispatcher_returns_no_throughput_for_legacy_rows_without_timed_tokens(): + from litellm.proxy.management_endpoints.common_daily_activity import ( + _GROUP_DATE_MODEL_PROVIDER, + _aggregate_grouping_sets_records_sync, + ) + + records = [ + _grouping_row( + _GROUP_DATE_MODEL_PROVIDER, + model="gpt-4o", + completion_tokens=900, + timed_completion_tokens=None, + total_response_time_ms=3000, + timed_requests=1, + ) + ] + + day = _aggregate_grouping_sets_records_sync(records=records, api_key_metadata={})["results"][0] + + assert day.breakdown.models["gpt-4o"].provider_breakdown["openai"].output_tokens_per_second is None + + def test_grouping_sets_dispatcher_keeps_ptu_flat_cost_out_of_the_provider_breakdown(): """Sentinel rows carry no provider, so their flat cost must not surface under the "unknown" provider - the per-row path skips them for exactly the same reason.""" diff --git a/ui/litellm-dashboard/src/components/UsagePage/types.ts b/ui/litellm-dashboard/src/components/UsagePage/types.ts index fa8fb6c3a4e..bf6769046b2 100644 --- a/ui/litellm-dashboard/src/components/UsagePage/types.ts +++ b/ui/litellm-dashboard/src/components/UsagePage/types.ts @@ -42,7 +42,7 @@ export interface MetricWithMetadata { } export interface ProviderThroughputMetrics { - timed_completion_tokens: number; + timed_completion_tokens?: number | null; total_response_time_ms: number; timed_requests: number; output_tokens_per_second?: number | null; diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index ba4158a8b5d..521223b587c 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -41076,11 +41076,8 @@ export interface components { ProviderThroughputMetrics: { /** Output Tokens Per Second */ output_tokens_per_second?: number | null; - /** - * Timed Completion Tokens - * @default 0 - */ - timed_completion_tokens: number; + /** Timed Completion Tokens */ + timed_completion_tokens?: number | null; /** * Timed Requests * @default 0