diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261001200000_add_autorouter_daily_spend/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261001200000_add_autorouter_daily_spend/migration.sql new file mode 100644 index 00000000000..ce166b4df45 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261001200000_add_autorouter_daily_spend/migration.sql @@ -0,0 +1,17 @@ +CREATE TABLE IF NOT EXISTS "LiteLLM_AutoRouterDailySpend" ( + "date" TEXT NOT NULL, + "api_key" TEXT NOT NULL, + "user_id" TEXT NOT NULL, + "router_name" TEXT NOT NULL, + "router_type" TEXT NOT NULL, + "turns" INTEGER NOT NULL DEFAULT 0, + "spend" DOUBLE PRECISION NOT NULL DEFAULT 0, + "saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0, + "savings_estimated_turns" INTEGER NOT NULL DEFAULT 0, + "savings_estimated_actual_spend" DOUBLE PRECISION NOT NULL DEFAULT 0, + "savings_estimated_saved_spend" DOUBLE PRECISION NOT NULL DEFAULT 0, + "classifier_cost" DOUBLE PRECISION NOT NULL DEFAULT 0, + "classifier_cost_recorded_turns" INTEGER NOT NULL DEFAULT 0, + + CONSTRAINT "LiteLLM_AutoRouterDailySpend_pkey" PRIMARY KEY ("date", "api_key", "user_id", "router_name", "router_type") +); diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 6f285e9dc39..aba89526cf6 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -1744,6 +1744,27 @@ model LiteLLM_AutoRouterUserSession { @@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn") } +// Auto-routed requests per UTC request day and router: the selected-day money behind the +// auto-router usage view. Written in the same statement as the session rollup, so a day row +// and its session row never disagree; corrected in the same transaction as late baselines. +model LiteLLM_AutoRouterDailySpend { + date String + api_key String + user_id String + router_name String + router_type String + turns Int @default(0) + spend Float @default(0) + saved_spend Float @default(0) + savings_estimated_turns Int @default(0) + savings_estimated_actual_spend Float @default(0) + savings_estimated_saved_spend Float @default(0) + classifier_cost Float @default(0) + classifier_cost_recorded_turns Int @default(0) + + @@id([date, api_key, user_id, router_name, router_type]) +} + // Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in // either direction. forward duplicates the requests the keys did not route through the // router through it, answering whether they should adopt it; reverse duplicates the diff --git a/litellm/proxy/db/autorouter_session_rollup.py b/litellm/proxy/db/autorouter_session_rollup.py index b762a40f344..cdc949a6dca 100644 --- a/litellm/proxy/db/autorouter_session_rollup.py +++ b/litellm/proxy/db/autorouter_session_rollup.py @@ -4,11 +4,12 @@ Per-session auto-router benchmarks rollup. At request time the spend writer builds one AutoRouterTurnTransaction per successful auto-routed request (a request whose metadata carries a routing_decision) and queues it on the prisma client. The spend-log flush job drains the queue into -key and user session rollups with one atomic statement per turn: each upsert classifies +key and user session rollups, plus the per-day router rollup, with one atomic statement +per turn: each upsert classifies the turn (same model, first visit, return to a model the session already used, out of order) against the row's own columns, so nothing is read before the write and concurrent -pods compose. The benchmarks endpoint aggregates these rows and can recover matching historical -costs from retained spend logs when estimate coverage predates these columns. +pods compose. The benchmarks endpoint reads session shape from the session rows and money from the +day rows, so spend and savings count only requests on the selected UTC days. """ from __future__ import annotations @@ -71,45 +72,82 @@ tier_maps AS ( GROUP BY router_name, router_type, kv.key ) per_tier GROUP BY router_name, router_type +), +sessions AS ( + SELECT + router_name, + router_type, + COUNT(*)::int AS sessions, + SUM(turns)::int AS session_turns, + SUM(unordered_turns)::int AS unordered_turns, + SUM(covered_turns)::int AS covered_turns, + SUM(cache_hits)::int AS cache_hits, + SUM(same_model_turns)::int AS same_model_turns, + SUM(same_model_hits)::int AS same_model_hits, + SUM(first_visit_turns)::int AS first_visit_turns, + SUM(first_visit_hits)::int AS first_visit_hits, + SUM(return_turns)::int AS return_turns, + SUM(return_hits)::int AS return_hits, + SUM(return_expired_misses)::int AS return_expired_misses, + SUM(return_within_ttl_misses)::int AS return_within_ttl_misses, + SUM(ttl_5m_turns)::int AS ttl_5m_turns, + SUM(ttl_1h_turns)::int AS ttl_1h_turns, + SUM(total_tokens)::bigint AS total_tokens, + SUM(EXTRACT(EPOCH FROM (last_turn_at - first_turn_at)))::float8 AS session_seconds + FROM windowed + GROUP BY router_name, router_type +), +days AS ( + SELECT + router_name, + router_type, + SUM(turns)::int AS turns, + SUM(spend)::float8 AS spend, + SUM(saved_spend)::float8 AS saved_spend, + SUM(savings_estimated_turns)::int AS savings_estimated_turns, + SUM(savings_estimated_actual_spend)::float8 AS savings_estimated_actual_spend, + SUM(savings_estimated_saved_spend)::float8 AS savings_estimated_saved_spend, + SUM(classifier_cost)::float8 AS classifier_cost, + SUM(classifier_cost_recorded_turns)::int AS classifier_cost_recorded_turns + FROM "LiteLLM_AutoRouterDailySpend" + WHERE date >= $5 AND date <= $6 + AND ($3::text IS NULL OR api_key = $3::text) + AND ($4::text IS NULL OR user_id = $4::text) + GROUP BY router_name, router_type ) -SELECT - agg.*, - COALESCE(tier_maps.tier_turns, '{{}}'::jsonb) AS tier_turns -FROM ( SELECT router_name, router_type, - COUNT(*)::int AS sessions, - COALESCE(SUM(turns), 0)::int AS turns, - COALESCE(SUM(unordered_turns), 0)::int AS unordered_turns, - COALESCE(SUM(covered_turns), 0)::int AS covered_turns, - COALESCE(SUM(cache_hits), 0)::int AS cache_hits, - COALESCE(SUM(same_model_turns), 0)::int AS same_model_turns, - COALESCE(SUM(same_model_hits), 0)::int AS same_model_hits, - COALESCE(SUM(first_visit_turns), 0)::int AS first_visit_turns, - COALESCE(SUM(first_visit_hits), 0)::int AS first_visit_hits, - COALESCE(SUM(return_turns), 0)::int AS return_turns, - COALESCE(SUM(return_hits), 0)::int AS return_hits, - COALESCE(SUM(return_expired_misses), 0)::int AS return_expired_misses, - COALESCE(SUM(return_within_ttl_misses), 0)::int AS return_within_ttl_misses, - COALESCE(SUM(ttl_5m_turns), 0)::int AS ttl_5m_turns, - COALESCE(SUM(ttl_1h_turns), 0)::int AS ttl_1h_turns, - COALESCE(SUM(total_tokens), 0)::bigint AS total_tokens, - COALESCE(SUM(spend), 0)::float8 AS spend, - COALESCE(SUM(saved_spend), 0)::float8 AS saved_spend, - COALESCE(SUM(savings_estimated_turns), 0)::int AS savings_estimated_turns, - COALESCE(SUM(savings_estimated_actual_spend), 0)::float8 AS savings_estimated_actual_spend, - CASE WHEN BOOL_AND(savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns) - THEN SUM(classifier_cost)::float8 END AS savings_estimated_classifier_cost, - COALESCE(SUM(savings_estimated_saved_spend), 0)::float8 AS savings_estimated_saved_spend, - COALESCE(SUM(classifier_cost), 0)::float8 AS classifier_cost, - COALESCE(SUM(classifier_cost_recorded_turns), 0)::int AS classifier_cost_recorded_turns, - COALESCE(SUM(EXTRACT(EPOCH FROM (last_turn_at - first_turn_at))), 0)::float8 AS session_seconds -FROM windowed -GROUP BY router_name, router_type -) agg + COALESCE(tier_maps.tier_turns, '{{}}'::jsonb) AS tier_turns, + COALESCE(sessions.sessions, 0) AS sessions, + COALESCE(sessions.session_turns, 0) AS session_turns, + COALESCE(sessions.unordered_turns, 0) AS unordered_turns, + COALESCE(sessions.covered_turns, 0) AS covered_turns, + COALESCE(sessions.cache_hits, 0) AS cache_hits, + COALESCE(sessions.same_model_turns, 0) AS same_model_turns, + COALESCE(sessions.same_model_hits, 0) AS same_model_hits, + COALESCE(sessions.first_visit_turns, 0) AS first_visit_turns, + COALESCE(sessions.first_visit_hits, 0) AS first_visit_hits, + COALESCE(sessions.return_turns, 0) AS return_turns, + COALESCE(sessions.return_hits, 0) AS return_hits, + COALESCE(sessions.return_expired_misses, 0) AS return_expired_misses, + COALESCE(sessions.return_within_ttl_misses, 0) AS return_within_ttl_misses, + COALESCE(sessions.ttl_5m_turns, 0) AS ttl_5m_turns, + COALESCE(sessions.ttl_1h_turns, 0) AS ttl_1h_turns, + COALESCE(sessions.total_tokens, 0) AS total_tokens, + COALESCE(sessions.session_seconds, 0) AS session_seconds, + COALESCE(days.turns, 0) AS turns, + COALESCE(days.spend, 0) AS spend, + COALESCE(days.saved_spend, 0) AS saved_spend, + COALESCE(days.savings_estimated_turns, 0) AS savings_estimated_turns, + COALESCE(days.savings_estimated_actual_spend, 0) AS savings_estimated_actual_spend, + COALESCE(days.savings_estimated_saved_spend, 0) AS savings_estimated_saved_spend, + COALESCE(days.classifier_cost, 0) AS classifier_cost, + COALESCE(days.classifier_cost_recorded_turns, 0) AS classifier_cost_recorded_turns +FROM sessions +FULL OUTER JOIN days USING (router_name, router_type) LEFT JOIN tier_maps USING (router_name, router_type) -ORDER BY agg.spend DESC +ORDER BY spend DESC, router_name, router_type """ @@ -391,15 +429,43 @@ ON CONFLICT ({user_column}api_key, session_id, router_name) DO UPDATE SET """ +_DAY_UPSERT_SQL: Final = f""" +day_rollup AS ( + INSERT INTO "LiteLLM_AutoRouterDailySpend" AS d ( + date, api_key, user_id, router_name, router_type, turns, spend, saved_spend, savings_estimated_turns, + savings_estimated_actual_spend, savings_estimated_saved_spend, classifier_cost, classifier_cost_recorded_turns + ) + VALUES ( + ({_TURN_AT}::timestamp)::date::text, {_p("api_key")}::text, {_p("user_id")}::text, {_p("router_name")}, + {_p("router_type")}, 1, {_p("spend")}::float8, {_p("saved_spend")}::float8, {_p("savings_estimated_turns")}::int, + {_p("savings_estimated_actual_spend")}::float8, {_p("savings_estimated_saved_spend")}::float8, + {_p("classifier_cost")}::float8, 1 + ) + ON CONFLICT (date, api_key, user_id, router_name, router_type) DO UPDATE SET + turns = d.turns + 1, + spend = d.spend + EXCLUDED.spend, + saved_spend = d.saved_spend + EXCLUDED.saved_spend, + savings_estimated_turns = d.savings_estimated_turns + EXCLUDED.savings_estimated_turns, + savings_estimated_actual_spend = d.savings_estimated_actual_spend + EXCLUDED.savings_estimated_actual_spend, + savings_estimated_saved_spend = d.savings_estimated_saved_spend + EXCLUDED.savings_estimated_saved_spend, + classifier_cost = d.classifier_cost + EXCLUDED.classifier_cost, + classifier_cost_recorded_turns = d.classifier_cost_recorded_turns + 1 + RETURNING 1 +) +""" + UPSERT_AUTOROUTER_SESSION_SQL: Final = f""" WITH key_rollup AS ( {_session_upsert_sql(user_scoped=False)} RETURNING 1 -) +), {_DAY_UPSERT_SQL} {_session_upsert_sql(user_scoped=True)} """ -UPSERT_AUTOROUTER_USER_SESSION_SQL: Final = _session_upsert_sql(user_scoped=True) +UPSERT_AUTOROUTER_USER_SESSION_SQL: Final = f""" +WITH {_DAY_UPSERT_SQL} +{_session_upsert_sql(user_scoped=True)} +""" def _as_sql_param(value: str | float | bool | datetime | None) -> str | float | None: diff --git a/litellm/proxy/db/baseline_accounting.py b/litellm/proxy/db/baseline_accounting.py index 05a4a989152..9536f8d740a 100644 --- a/litellm/proxy/db/baseline_accounting.py +++ b/litellm/proxy/db/baseline_accounting.py @@ -179,6 +179,8 @@ class _Change(BaseModel): actual_delta: float savings_delta: float daily: DailyBaselineAttribution | None + date: str | None = None + router_type: str | None = None class _TransactionManager(Protocol): @@ -303,6 +305,26 @@ WHERE {user_match}session.api_key = totals.api_key AND session.session_id = tota _UPDATE_SESSIONS: Final = _session_correction_sql(user_scoped=False) _UPDATE_USER_SESSIONS: Final = _session_correction_sql(user_scoped=True) +_UPDATE_DAYS: Final = """ +WITH totals AS ( + SELECT date, api_key, user_id, router_name, router_type, SUM(covered_delta)::int AS covered_delta, + SUM(actual_delta) AS actual_delta, SUM(savings_delta) AS savings_delta + FROM jsonb_to_recordset($1::jsonb) AS x( + date text, api_key text, user_id text, router_name text, router_type text, + covered_delta int, actual_delta float8, savings_delta float8 + ) + WHERE date IS NOT NULL + GROUP BY date, api_key, user_id, router_name, router_type +) +UPDATE "LiteLLM_AutoRouterDailySpend" AS day +SET saved_spend = day.saved_spend + totals.savings_delta, + savings_estimated_turns = day.savings_estimated_turns + totals.covered_delta, + savings_estimated_actual_spend = day.savings_estimated_actual_spend + totals.actual_delta, + savings_estimated_saved_spend = day.savings_estimated_saved_spend + totals.savings_delta +FROM totals +WHERE day.date = totals.date AND day.api_key = totals.api_key AND day.user_id = totals.user_id + AND day.router_name = totals.router_name AND day.router_type = totals.router_type +""" def _primary_transaction(client: PrismaClient) -> _TransactionManager: @@ -331,6 +353,8 @@ def _change(record: BaselineAccountingRecord, old: BaselinePublication | None, n savings_delta=(current.savings if current is not None else 0.0) - (previous.savings if previous is not None else 0.0), daily=record.daily, + date=record.turn.turn_at.date().isoformat() if record.turn is not None else None, + router_type=record.turn.router_type if record.turn is not None else None, ) @@ -373,6 +397,7 @@ async def _publish(db: SupportsRawQueries, changes: Sequence[_Change]) -> None: await db.execute_raw(_UPDATE_SESSIONS, serialized) if any(change.user_id for change in changes): await db.execute_raw(_UPDATE_USER_SESSIONS, serialized) + await db.execute_raw(_UPDATE_DAYS, serialized) for entity, table in DAILY_SPEND_TABLES.items(): if adjustments := tuple( change.daily.adjustment(target, change.savings_delta, change.request_id) diff --git a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py index 06e4d06fca4..679e286cedd 100644 --- a/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py +++ b/litellm/proxy/db/db_transaction_queue/spend_log_cleanup.py @@ -540,6 +540,18 @@ class SpendLogCleanup: deadline=deadline, ) + async def _delete_old_autorouter_daily_rows( + self, prisma_client: PrismaClient, cutoff_day: str, deadline: float + ) -> TableCleanupResult: + return await self._delete_old_rows_batched( + prisma_client, + cutoff_day, + table_name="LiteLLM_AutoRouterDailySpend", + key_columns=("date", "api_key", "user_id", "router_name", "router_type"), + time_column="date", + deadline=deadline, + ) + async def _delete_old_health_check_rows( self, prisma_client: PrismaClient, cutoff_date: datetime, deadline: float ) -> TableCleanupResult: @@ -623,16 +635,20 @@ class SpendLogCleanup: except Exception: # noqa: BLE001 # retained observations are retried by the next cleanup job verbose_proxy_logger.warning("Auto-router baseline retention remains pending") sessions_result: Final = await self._delete_old_autorouter_session_rows( - prisma_client, session_cutoff, self._group_deadline(deadline, 2) + prisma_client, session_cutoff, self._group_deadline(deadline, 3) ) verbose_proxy_logger.info("Deleted %s expired auto-router session rollup rows", sessions_result.rows_deleted) user_sessions_result: Final = await self._delete_old_autorouter_user_session_rows( - prisma_client, session_cutoff, deadline + prisma_client, session_cutoff, self._group_deadline(deadline, 2) ) verbose_proxy_logger.info( "Deleted %s expired auto-router user session rollup rows", user_sessions_result.rows_deleted ) - return (sessions_result, user_sessions_result) + days_result: Final = await self._delete_old_autorouter_daily_rows( + prisma_client, session_cutoff.date().isoformat(), deadline + ) + verbose_proxy_logger.info("Deleted %s expired auto-router daily rollup rows", days_result.rows_deleted) + return (sessions_result, user_sessions_result, days_result) async def _clean_health_checks( self, prisma_client: PrismaClient, retention_seconds: int, deadline: float diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index 50145ed3923..35ff9186f5e 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -5,6 +5,8 @@ POST /auto_router/test_routing - Route one request through an unsaved complexity POST /auto_router/validate_complexity_router_config - Dry-run the complexity-router write gate without saving """ +import asyncio +import math from collections.abc import Mapping, Sequence from datetime import datetime, timedelta, timezone from itertools import chain, groupby @@ -40,6 +42,7 @@ from litellm.proxy.litellm_pre_call_utils import ( refresh_proxy_server_request_body_snapshot, ) from litellm.proxy.management.teams.access import is_team_admin +from litellm.proxy.management_endpoints.common_daily_activity import daily_activity_scope from litellm.proxy.management_helpers.auto_router_permissions import ( authorize_member_auto_router_dependencies, authorize_member_auto_router_team, @@ -47,6 +50,7 @@ from litellm.proxy.management_helpers.auto_router_permissions import ( ) from litellm.repositories.autorouter_session_repository import AutoRouterSessionRepository from litellm.repositories.base_repository import SupportsModelDump +from litellm.repositories.daily_activity_sql import build_where_clause from litellm.repositories.team_repository import TeamRepository from litellm.router_strategy.complexity_router import ComplexityRouter from litellm.router_utils.auto_router_model_naming import ( @@ -616,34 +620,37 @@ async def preview_auto_router_routing( class _SessionAggRow(BaseModel): + """One router's window: session shape from overlapping sessions, money from the selected days.""" + router_name: str router_type: str - tier_turns: Mapping[str, int] - sessions: int - turns: int - unordered_turns: int - covered_turns: int - cache_hits: int - same_model_turns: int - same_model_hits: int - first_visit_turns: int - first_visit_hits: int - return_turns: int - return_hits: int - return_expired_misses: int - return_within_ttl_misses: int - ttl_5m_turns: int - ttl_1h_turns: int - total_tokens: int - spend: float - saved_spend: float + tier_turns: Mapping[str, int] = MappingProxyType({}) + sessions: int = 0 + session_turns: int = 0 + unordered_turns: int = 0 + covered_turns: int = 0 + cache_hits: int = 0 + same_model_turns: int = 0 + same_model_hits: int = 0 + first_visit_turns: int = 0 + first_visit_hits: int = 0 + return_turns: int = 0 + return_hits: int = 0 + return_expired_misses: int = 0 + return_within_ttl_misses: int = 0 + ttl_5m_turns: int = 0 + ttl_1h_turns: int = 0 + total_tokens: int = 0 + session_seconds: float = 0.0 + turns: int = 0 + spend: float = 0.0 + saved_spend: float = 0.0 savings_estimated_turns: int = 0 savings_estimated_actual_spend: float = 0.0 savings_estimated_classifier_cost: float | None = None savings_estimated_saved_spend: float = 0.0 - classifier_cost: float - classifier_cost_recorded_turns: int - session_seconds: float + classifier_cost: float = 0.0 + classifier_cost_recorded_turns: int = 0 _SESSION_AGG_ROWS: Final = TypeAdapter(list[_SessionAggRow]) @@ -692,6 +699,13 @@ def _compared_row(row: _SessionAggRow) -> _SessionAggRow: ) +def _per_session(row: _SessionAggRow, total: float) -> float | None: + """Unknown, not zero, when routed requests have no session rows of their own to average over.""" + if row.sessions: + return total / row.sessions + return None if row.turns else 0.0 + + def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: return_misses: Final = row.return_turns - row.return_hits saved_spend, baseline_spend = _savings_cohort( @@ -701,9 +715,9 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: return AutoRouterBenchmarkTotals( sessions=sessions, turns=row.turns, - avg_turns_per_session=row.turns / sessions if sessions else 0.0, - avg_session_seconds=row.session_seconds / sessions if sessions else 0.0, - avg_tokens_per_session=row.total_tokens / sessions if sessions else 0.0, + avg_turns_per_session=_per_session(row, row.session_turns), + avg_session_seconds=_per_session(row, row.session_seconds), + avg_tokens_per_session=_per_session(row, row.total_tokens), spend=row.spend, savings_estimated_turns=row.savings_estimated_turns, savings_estimated_actual_spend=row.savings_estimated_actual_spend, @@ -712,9 +726,8 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: classifier_cost=row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None, baseline_spend=baseline_spend, saved_pct=_pct(saved_spend, baseline_spend) if saved_spend is not None and baseline_spend is not None else None, - saved_per_session=(saved_spend / sessions if sessions else 0.0) if saved_spend is not None else None, cache=AutoRouterCacheStats( - coverage_pct=_pct(row.covered_turns, row.turns), + coverage_pct=_pct(row.covered_turns, row.session_turns), hit_rate_pct=_pct(row.cache_hits, row.covered_turns), same_model=_cache_bucket(row.same_model_turns, row.same_model_hits), first_visit=_cache_bucket(row.first_visit_turns, row.first_visit_hits), @@ -748,7 +761,6 @@ def _benchmark_group(row: _SessionAggRow) -> AutoRouterBenchmarkGroup: classifier_cost=totals.classifier_cost, baseline_spend=totals.baseline_spend, saved_pct=totals.saved_pct, - saved_per_session=totals.saved_per_session, cache=totals.cache, ) @@ -759,6 +771,7 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow: router_type="", tier_turns=MappingProxyType({}), sessions=sum(row.sessions for row in rows), + session_turns=sum(row.session_turns for row in rows), turns=sum(row.turns for row in rows), unordered_turns=sum(row.unordered_turns for row in rows), covered_turns=sum(row.covered_turns for row in rows), @@ -790,6 +803,49 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow: ) +async def _recorded_autorouter_savings( + prisma_client: "PrismaClient", start_day: str, end_day: str, api_key: str | None, user_id: str | None +) -> float: + """The selected days' auto-router savings exactly as the Overall view sums them: same table, same filters.""" + where, params = build_where_clause( + daily_activity_scope( + table="litellm_dailyuserspend", + entity_id_field="user_id", + entity_id=user_id, + exclude_entity_ids=None, + api_key=api_key, + start_date=start_day, + end_date=end_day, + model=None, + timezone_offset_minutes=None, + ) + ) + rows: Final = await _query_raw( + prisma_client, + f'SELECT COALESCE(SUM(autorouter_savings_spend), 0)::float8 AS saved FROM "LiteLLM_DailyUserSpend" WHERE {where}', + *params, + ) + return float(rows[0]["saved"]) if rows else 0.0 + + +def _with_recorded_savings( + totals: AutoRouterBenchmarkTotals, rows: Sequence[_SessionAggRow], recorded: float +) -> AutoRouterBenchmarkTotals: + """The headline is the recorded total. Savings outside the compared routers void the cost comparison, + and the part no router's day rows account for is reported as unattributed.""" + if math.isclose(recorded, totals.saved_spend or 0.0, abs_tol=1e-9): + return totals + unattributed: Final = recorded - sum(row.saved_spend for row in rows) + return totals.model_copy( + update={ + "saved_spend": recorded, + "unattributed_saved_spend": None if math.isclose(unattributed, 0.0, abs_tol=1e-9) else unattributed, + "baseline_spend": None, + "saved_pct": None, + } + ) + + def _strategy_router_key(deployment: object) -> tuple[str, str] | None: """``(model_name, kind)`` for a deployment whose routing the session rollup records. @@ -849,7 +905,7 @@ async def get_auto_router_benchmarks( str | None, Query(description="YYYY-MM-DD UTC, inclusive (defaults to 30 days before end_date)") ] = None, end_date: Annotated[str | None, Query(description="YYYY-MM-DD UTC, inclusive (defaults to today)")] = None, - api_key: Annotated[str | None, Query(description="Filter to one virtual key token hash")] = None, + api_key: Annotated[str | None, Query(min_length=1, description="Filter to one virtual key token hash")] = None, user_id: Annotated[ str | None, Query(min_length=1, description="Filter to one canonical internal user recorded on each turn") ] = None, @@ -860,9 +916,10 @@ async def get_auto_router_benchmarks( Reads session rollups folded once per request at spend-write time, so this endpoint never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that - internal user when written; older key-only history remains outside user views. A session - is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before - end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is + internal user when written; older key-only history remains outside user views. Money counts + only requests on the selected UTC days, and the all-router savings headline is the same daily + total the Overall view reads. Session shape and caching cover every session that overlaps the + window, whole. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is over that bucket's turns. The rollup supplies the measures, never the list. Which routers appear comes from the @@ -885,24 +942,35 @@ async def get_auto_router_benchmarks( if end_day < start_day: raise HTTPException(status_code=400, detail="end_date must not be earlier than start_date") - raw_rows: Final = await _query_raw( - prisma_client, - AUTOROUTER_BENCHMARKS_SQL, - start_day.isoformat(), - (end_day + timedelta(days=1)).isoformat(), - api_key, - user_id, + first_day: Final = start_day.strftime("%Y-%m-%d") + last_day: Final = end_day.strftime("%Y-%m-%d") + raw_rows, recorded = await asyncio.gather( + _query_raw( + prisma_client, + AUTOROUTER_BENCHMARKS_SQL, + start_day.isoformat(), + (end_day + timedelta(days=1)).isoformat(), + api_key, + user_id, + first_day, + last_day, + ), + _recorded_autorouter_savings(prisma_client, first_day, last_day, api_key, user_id), ) rows: Final = tuple(_compared_row(row) for row in _SESSION_AGG_ROWS.validate_python(raw_rows or ())) + totals: Final = _with_recorded_savings(_benchmark_totals(_summed_agg_row(rows)), rows, recorded) + unattributed: Final = MappingProxyType( + {"baseline_spend": None, "saved_pct": None} if totals.unattributed_saved_spend is not None else {} + ) groups: Final = ( - *(_benchmark_group(row) for row in rows), + *(_benchmark_group(row).model_copy(update=unattributed) for row in rows), *_idle_router_groups(llm_router, frozenset((row.router_name, row.router_type) for row in rows)), ) return AutoRouterBenchmarksResponse( - start_date=start_day.strftime("%Y-%m-%d"), - end_date=end_day.strftime("%Y-%m-%d"), + start_date=first_day, + end_date=last_day, routers_in_scope=len(groups), - totals=_benchmark_totals(_summed_agg_row(rows)), + totals=totals, groups=groups, ) diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 6f285e9dc39..aba89526cf6 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -1744,6 +1744,27 @@ model LiteLLM_AutoRouterUserSession { @@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn") } +// Auto-routed requests per UTC request day and router: the selected-day money behind the +// auto-router usage view. Written in the same statement as the session rollup, so a day row +// and its session row never disagree; corrected in the same transaction as late baselines. +model LiteLLM_AutoRouterDailySpend { + date String + api_key String + user_id String + router_name String + router_type String + turns Int @default(0) + spend Float @default(0) + saved_spend Float @default(0) + savings_estimated_turns Int @default(0) + savings_estimated_actual_spend Float @default(0) + savings_estimated_saved_spend Float @default(0) + classifier_cost Float @default(0) + classifier_cost_recorded_turns Int @default(0) + + @@id([date, api_key, user_id, router_name, router_type]) +} + // Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in // either direction. forward duplicates the requests the keys did not route through the // router through it, answering whether they should adopt it; reverse duplicates the diff --git a/litellm/types/management_endpoints/auto_router_endpoints.py b/litellm/types/management_endpoints/auto_router_endpoints.py index 00083e01f54..ded971f6705 100644 --- a/litellm/types/management_endpoints/auto_router_endpoints.py +++ b/litellm/types/management_endpoints/auto_router_endpoints.py @@ -205,14 +205,18 @@ class AutoRouterCacheStats(BaseModel): class AutoRouterBenchmarkTotals(BaseModel): - """Session-shape and savings aggregates over auto-routed traffic in the window.""" + """Auto-routed traffic in the window. Turns, spend and savings count requests on the selected UTC days; + the session averages and cache stats describe every session overlapping the window, whole.""" - sessions: int - turns: int - avg_turns_per_session: float - avg_session_seconds: float - avg_tokens_per_session: float - spend: float = Field(description="What the routed traffic actually cost") + sessions: int = Field(description="Sessions overlapping the window, counted whole") + turns: int = Field(description="Auto-routed requests on the selected UTC days") + avg_turns_per_session: float | None = Field( + description="Lifetime turns per overlapping session; null when the window has routed requests but no session " + "rows for this router type, such as an alias whose router type changed mid-session" + ) + avg_session_seconds: float | None = Field(description="Lifetime seconds per overlapping session; null as above") + avg_tokens_per_session: float | None = Field(description="Lifetime tokens per overlapping session; null as above") + spend: float = Field(description="What the selected days' routed traffic actually cost") classifier_cost: float | None = Field( description="Recorded LLM classifier cost already included in spend; null when any session turns predate " "subtotal recording, and zero for an empty window" @@ -229,14 +233,19 @@ class AutoRouterBenchmarkTotals(BaseModel): "null when classification costs for those requests are unavailable", ) saved_spend: float | None = Field( - description="Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates" + description="Recorded savings on the selected UTC days; null when traffic has no recorded savings estimates. " + "On totals this is the same daily figure the Overall savings view reports" + ) + unattributed_saved_spend: float | None = Field( + default=None, + description="Part of saved_spend no router's daily rows account for, such as history recorded before " + "per-router daily tracking; when set, baseline_spend and saved_pct are null", ) baseline_spend: float | None = Field( description="Estimated single-model cost: compared actual spend plus recorded savings; " "null when traffic has no recorded savings" ) saved_pct: float | None = Field(description="Recorded savings over baseline_spend, as a percentage") - saved_per_session: float | None = Field(description="Recorded savings per session, including historical estimates") cache: AutoRouterCacheStats @@ -291,7 +300,7 @@ class AutoRouterSessionResponse(BaseModel): class AutoRouterBenchmarksResponse(BaseModel): - """Benchmarks for the auto-router dashboard, aggregated from the per-session rollup.""" + """Benchmarks for the auto-router dashboard, aggregated from the per-session and per-day rollups.""" start_date: str = Field(description="Window start day, YYYY-MM-DD UTC, inclusive") end_date: str = Field(description="Window end day, YYYY-MM-DD UTC, inclusive") diff --git a/schema.prisma b/schema.prisma index 6f285e9dc39..aba89526cf6 100644 --- a/schema.prisma +++ b/schema.prisma @@ -1744,6 +1744,27 @@ model LiteLLM_AutoRouterUserSession { @@index([user_id, last_turn_at], map: "idx_autorouter_user_session_user_last_turn") } +// Auto-routed requests per UTC request day and router: the selected-day money behind the +// auto-router usage view. Written in the same statement as the session rollup, so a day row +// and its session row never disagree; corrected in the same transaction as late baselines. +model LiteLLM_AutoRouterDailySpend { + date String + api_key String + user_id String + router_name String + router_type String + turns Int @default(0) + spend Float @default(0) + saved_spend Float @default(0) + savings_estimated_turns Int @default(0) + savings_estimated_actual_spend Float @default(0) + savings_estimated_saved_spend Float @default(0) + classifier_cost Float @default(0) + classifier_cost_recorded_turns Int @default(0) + + @@id([date, api_key, user_id, router_name, router_type]) +} + // Shadow eval: evaluation of an auto-router against one or more keys' live traffic, in // either direction. forward duplicates the requests the keys did not route through the // router through it, answering whether they should adopt it; reverse duplicates the diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py index 2c648f309f6..a5c6f5962a2 100644 --- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py +++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py @@ -80,6 +80,25 @@ async def _turn( ) +async def _benchmark_rows( + db, start: datetime, end: datetime, key: str | None = None, user_id: str | None = None +) -> list[dict]: + return await db.query_raw( + AUTOROUTER_BENCHMARKS_SQL, + start.isoformat(), + end.isoformat(), + key, + user_id, + start.date().isoformat(), + (end - timedelta(days=1)).date().isoformat(), + ) + + +async def _days(db, key: str | None = None, user_id: str | None = None, router: str | None = None) -> list[dict]: + rows = await _benchmark_rows(db, T0 - timedelta(days=1), T0 + timedelta(days=2), key, user_id) + return [row for row in rows if row["turns"] and (router is None or row["router_name"] == router)] + + async def _row(db, key: str, session_id: str = "s1", router: str = "auto-1") -> dict: rows = await db.query_raw( 'SELECT * FROM "LiteLLM_AutoRouterSession" WHERE api_key = $1 AND session_id = $2 AND router_name = $3', @@ -225,18 +244,15 @@ async def test_subtotal_coverage_survives_legacy_and_rolling_writers(db, writers assert row["savings_estimated_turns"] == sum(writers) assert row["savings_estimated_actual_spend"] == pytest.approx(0.01 * sum(writers)) assert row["savings_estimated_saved_spend"] == pytest.approx(0.02 * sum(writers)) - groups: Final = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, T0.isoformat(), (T0 + timedelta(days=1)).isoformat(), key, None - ) - assert len(groups) == 1 - assert groups[0]["classifier_cost"] == row["classifier_cost"] - assert groups[0]["classifier_cost_recorded_turns"] == sum(writers) - assert groups[0]["turns"] == len(writers) - assert groups[0]["spend"] == row["spend"] - assert groups[0]["saved_spend"] == row["saved_spend"] - assert groups[0]["savings_estimated_turns"] == sum(writers) - assert groups[0]["savings_estimated_actual_spend"] == row["savings_estimated_actual_spend"] - assert groups[0]["savings_estimated_saved_spend"] == row["savings_estimated_saved_spend"] + days: Final = await _days(db, key) + assert len(days) == int(any(writers)) + for day in days: + assert day["classifier_cost"] == row["classifier_cost"] + assert day["classifier_cost_recorded_turns"] == day["turns"] == sum(writers) + assert day["spend"] == pytest.approx(0.01 * sum(writers)) + assert day["saved_spend"] == pytest.approx(0.02 * sum(writers)) + assert day["savings_estimated_actual_spend"] == row["savings_estimated_actual_spend"] + assert day["savings_estimated_saved_spend"] == row["savings_estimated_saved_spend"] async def test_unknown_and_legacy_turns_preserve_actual_spend_without_entering_the_estimated_cohort(db: Prisma) -> None: @@ -250,13 +266,10 @@ async def test_unknown_and_legacy_turns_preserve_actual_spend_without_entering_t row: Final = await _row(db, key) assert row["saved_spend"] == pytest.approx(-0.03) assert row["savings_estimated_baseline_models"] == {"opus": 1} - groups: Final = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, T0.isoformat(), (T0 + timedelta(days=1)).isoformat(), key, None - ) - assert len(groups) == 1 - for actual in (row, groups[0]): - assert actual["turns"] == 3 - assert actual["spend"] == pytest.approx(0.96) + (day,) = await _days(db, key) + assert (row["turns"], day["turns"]) == (3, 2) + assert (row["spend"], day["spend"]) == (pytest.approx(0.96), pytest.approx(0.95)) + for actual in (row, day): assert actual["savings_estimated_turns"] == 1 assert actual["savings_estimated_actual_spend"] == pytest.approx(0.25) assert actual["savings_estimated_saved_spend"] == pytest.approx(-0.05) @@ -281,25 +294,20 @@ async def test_the_benchmarks_aggregate_reads_only_overlapping_sessions(db): await _turn(db, key, "A", T0, session_id=in_window, router=router, saved=0.5, spend=0.25, classifier_cost=0.02) await _turn(db, key, "A", T0 - timedelta(days=40), session_id=out_of_window, router=router, classifier_cost=9.0) - rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - None, - None, - ) + rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), None, None) matching = [row for row in rows if row["router_name"] == router] assert len(matching) == 1 grouped = matching[0] assert grouped["router_type"] == "complexity" assert grouped["sessions"] == 1 - assert grouped["turns"] == 2 - assert grouped["spend"] == pytest.approx(0.5) - assert grouped["saved_spend"] == pytest.approx(1.0) - assert grouped["classifier_cost"] == pytest.approx(0.03) - assert grouped["classifier_cost_recorded_turns"] == 2 + assert grouped["session_turns"] == 2 assert grouped["unordered_turns"] == 1 assert grouped["session_seconds"] == pytest.approx(60.0) + (day,) = await _days(db, router=router) + assert (day["turns"], day["classifier_cost_recorded_turns"]) == (2, 2) + assert day["spend"] == pytest.approx(0.5) + assert day["saved_spend"] == pytest.approx(1.0) + assert day["classifier_cost"] == pytest.approx(0.03) async def test_the_benchmarks_aggregate_can_filter_to_one_key(db): @@ -309,32 +317,22 @@ async def test_the_benchmarks_aggregate_can_filter_to_one_key(db): await _turn(db, first_key, "A", T0, router=router, saved=0.5, classifier_cost=0.01) await _turn(db, second_key, "A", T0, router=router, saved=9.0, classifier_cost=0.09) - rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - first_key, - None, - ) + rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), first_key, None) matching = [row for row in rows if row["router_name"] == router] assert len(matching) == 1 assert matching[0]["sessions"] == 1 - assert matching[0]["saved_spend"] == pytest.approx(0.5) - assert matching[0]["classifier_cost"] == pytest.approx(0.01) - assert matching[0]["classifier_cost_recorded_turns"] == 1 + (day,) = await _days(db, first_key, router=router) + assert day["saved_spend"] == pytest.approx(0.5) + assert day["classifier_cost"] == pytest.approx(0.01) + assert day["classifier_cost_recorded_turns"] == 1 - unknown_key_rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - f"k-{uuid.uuid4()}", - None, - ) + unknown_key_rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), f"k-{uuid.uuid4()}", None) assert [row for row in unknown_key_rows if row["router_name"] == router] == [] class _BenchmarkRow(TypedDict): sessions: ReadOnly[int] + session_turns: ReadOnly[int] turns: ReadOnly[int] same_model_turns: ReadOnly[int] first_visit_turns: ReadOnly[int] @@ -350,14 +348,11 @@ class _BenchmarkRow(TypedDict): async def _scoped_benchmarks( db: Prisma, router: str, user_id: str | None = None, key: str | None = None ) -> tuple[_BenchmarkRow, ...]: - rows: Final = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - key, - user_id, + rows: Final = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), key, user_id) + days: Final = await _days(db, key, user_id, router) + return tuple( + cast(_BenchmarkRow, {**row, **next(iter(days), {})}) for row in rows if row["router_name"] == router ) - return tuple(cast(_BenchmarkRow, row) for row in rows if row["router_name"] == router) async def test_users_keep_written_identity_across_shared_keys_and_keyless_sessions(db: Prisma) -> None: @@ -384,28 +379,33 @@ async def test_users_keep_written_identity_across_shared_keys_and_keyless_sessio intersection: Final = await _scoped_benchmarks(db, router, user_id=alice, key=first_key) assert len(alice_rows) == len(bob_rows) == len(global_rows) == len(key_rows) == len(intersection) == 1 assert (alice_rows[0]["sessions"], alice_rows[0]["turns"], alice_rows[0]["same_model_turns"]) == (3, 4, 1) + assert (alice_rows[0]["session_turns"], bob_rows[0]["session_turns"]) == (4, 2) assert (bob_rows[0]["sessions"], bob_rows[0]["turns"], bob_rows[0]["first_visit_turns"]) == (2, 2, 2) assert alice_rows[0]["spend"] == pytest.approx(0.05) assert bob_rows[0]["spend"] == pytest.approx(0.07) assert alice_rows[0]["tier_turns"] == {"simple": 1} assert bob_rows[0]["tier_turns"] == {"complex": 1} assert (alice_rows[0]["cache_hits"], bob_rows[0]["cache_hits"]) == (1, 0) - assert (global_rows[0]["sessions"], global_rows[0]["turns"]) == (4, 7) + assert (global_rows[0]["sessions"], global_rows[0]["session_turns"], global_rows[0]["turns"]) == (4, 7, 6) assert (alice_rows[0]["savings_estimated_turns"], bob_rows[0]["savings_estimated_turns"]) == (4, 2) assert global_rows[0]["savings_estimated_turns"] == 6 for scoped in (alice_rows[0], bob_rows[0]): assert scoped["savings_estimated_actual_spend"] == pytest.approx(scoped["spend"]) assert scoped["savings_estimated_saved_spend"] == pytest.approx(scoped["saved_spend"]) - assert global_rows[0]["spend"] == pytest.approx(alice_rows[0]["spend"] + bob_rows[0]["spend"] + 0.01) - assert global_rows[0]["saved_spend"] == pytest.approx(alice_rows[0]["saved_spend"] + bob_rows[0]["saved_spend"] + 0.02) + assert global_rows[0]["spend"] == pytest.approx(alice_rows[0]["spend"] + bob_rows[0]["spend"]) + assert global_rows[0]["saved_spend"] == pytest.approx(alice_rows[0]["saved_spend"] + bob_rows[0]["saved_spend"]) assert global_rows[0]["tier_turns"] == {"simple": 1, "complex": 1} - assert (key_rows[0]["sessions"], key_rows[0]["turns"]) == (1, 3) - assert key_rows[0]["spend"] == pytest.approx(0.05) + assert (key_rows[0]["sessions"], key_rows[0]["session_turns"], key_rows[0]["turns"]) == (1, 3, 2) + assert key_rows[0]["spend"] == pytest.approx(0.04) assert (intersection[0]["sessions"], intersection[0]["turns"]) == (1, 1) assert intersection[0]["spend"] == pytest.approx(0.01) assert await _scoped_benchmarks(db, router, user_id=bob, key=second_key) == () assert await _scoped_benchmarks(db, router, user_id=f"u-{uuid.uuid4()}") == () - assert await _scoped_benchmarks(db, router, user_id="") == () + assert [ + row + for row in await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), None, "") + if row["router_name"] == router + ] == [] async def test_a_failed_user_projection_rolls_back_the_keys_increment(db: Prisma) -> None: @@ -419,6 +419,7 @@ async def test_a_failed_user_projection_rolls_back_the_keys_increment(db: Prisma assert await _row(db, key) == before assert await db.query_raw('SELECT user_id FROM "LiteLLM_AutoRouterUserSession" WHERE user_id = $1', user_id) == [] + assert [day["turns"] for day in await _days(db, key)] == [1] first_user: Final = f"u-{uuid.uuid4()}" second_user: Final = f"u-{uuid.uuid4()}" @@ -463,6 +464,14 @@ async def test_a_failed_user_projection_rolls_back_the_keys_increment(db: Prisma assert (row["turns"], row["same_model_turns"], row["unordered_turns"], row["last_model"]) == (count, 1, 0, model) assert row["spend"] == pytest.approx(count * 0.01) assert row["saved_spend"] == pytest.approx(count * 0.02) + days: Final = await db.query_raw( + 'SELECT user_id, turns, saved_spend FROM "LiteLLM_AutoRouterDailySpend" WHERE api_key = $1', key + ) + assert {day["user_id"]: (day["turns"], day["saved_spend"]) for day in days} == { + "": (1, pytest.approx(0.02)), + first_user: (3, pytest.approx(0.06)), + second_user: (2, pytest.approx(0.04)), + } async def test_user_session_cleanup_keeps_another_users_recent_keyless_session(db: Prisma) -> None: @@ -490,13 +499,7 @@ async def test_a_reconfigured_alias_reports_each_router_type_as_its_own_group(db db, key, "A", T0 + timedelta(seconds=10), session_id=f"s-{uuid.uuid4()}", router=router, router_type="quality" ) - rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - None, - None, - ) + rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), None, None) matching = sorted( (row for row in rows if row["router_name"] == router), key=lambda row: row["router_type"], @@ -575,16 +578,10 @@ async def test_the_benchmarks_aggregate_sums_tier_turns_across_sessions(db): await _turn(db, key, "B", T0 + timedelta(seconds=20), session_id=f"s-{uuid.uuid4()}", router=router, tier="complex") await _turn(db, key, "C", T0 + timedelta(seconds=30), session_id=f"s-{uuid.uuid4()}", router=router, tier=None) - rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - None, - None, - ) + rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), None, None) grouped = next(row for row in rows if row["router_name"] == router) assert grouped["tier_turns"] == {"simple": 2, "complex": 1} - assert grouped["turns"] == 4 + assert grouped["session_turns"] == 4 async def test_tier_maps_stay_separate_per_router_type_on_a_reconfigured_alias(db): @@ -604,13 +601,7 @@ async def test_tier_maps_stay_separate_per_router_type_on_a_reconfigured_alias(d tier="2", ) - rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - None, - None, - ) + rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), None, None) by_type = {row["router_type"]: row["tier_turns"] for row in rows if row["router_name"] == router} assert by_type == {"complexity": {"medium": 1}, "quality": {"2": 1}} @@ -620,13 +611,7 @@ async def test_a_window_with_no_tiered_turns_aggregates_to_an_empty_map(db): router = f"r-{uuid.uuid4()}" await _turn(db, key, "A", T0, session_id=f"s-{uuid.uuid4()}", router=router, tier=None) - rows = await db.query_raw( - AUTOROUTER_BENCHMARKS_SQL, - (T0 - timedelta(days=1)).isoformat(), - (T0 + timedelta(days=1)).isoformat(), - None, - None, - ) + rows = await _benchmark_rows(db, (T0 - timedelta(days=1)), (T0 + timedelta(days=1)), None, None) grouped = next(row for row in rows if row["router_name"] == router) assert grouped["tier_turns"] == {} @@ -653,3 +638,49 @@ async def test_an_out_of_order_hit_still_counts_toward_the_overall_hit_rate(db): assert row["unordered_turns"] == 1 assert row["cache_hits"] == 1 assert row["same_model_hits"] + row["first_visit_hits"] + row["return_hits"] == 0 + + +async def test_a_cross_midnight_session_splits_its_money_by_request_day(db): + key = f"k-{uuid.uuid4()}" + router = f"auto-{uuid.uuid4()}" + midnight = datetime(2026, 9, 2) + await _turn(db, key, "A", midnight - timedelta(minutes=10), router=router, spend=1.0, saved=7.0, user_id="u1") + await _turn(db, key, "A", midnight + timedelta(minutes=10), router=router, spend=1.0, saved=3.0, user_id="u1") + await _turn(db, key, "B", midnight + timedelta(days=1), router=router, spend=1.0, saved=11.0, user_id="u1") + + assert (await _row(db, key, router=router))["saved_spend"] == 21.0 + days = await db.query_raw( + 'SELECT date, turns, saved_spend FROM "LiteLLM_AutoRouterDailySpend" WHERE api_key = $1 ORDER BY date', key + ) + assert [(d["date"], d["turns"], d["saved_spend"]) for d in days] == [ + ("2026-09-01", 1, 7.0), + ("2026-09-02", 1, 3.0), + ("2026-09-03", 1, 11.0), + ] + for user_id in (None, "u1"): + (selected,) = await _benchmark_rows(db, midnight, midnight + timedelta(days=1), key, user_id) + assert (selected["sessions"], selected["session_turns"]) == (1, 3) + assert (selected["turns"], selected["spend"], selected["saved_spend"]) == (1, 1.0, 3.0) + + +async def test_a_router_type_change_within_a_day_keeps_each_types_money_apart(db): + key = f"k-{uuid.uuid4()}" + router = f"auto-{uuid.uuid4()}" + await _turn(db, key, "A", T0, router=router, router_type="complexity", spend=1.0, saved=4.0) + await _turn(db, key, "A", T0 + timedelta(hours=1), router=router, router_type="quality", spend=2.0, saved=0.0) + + days = {day["router_type"]: (day["turns"], day["spend"], day["saved_spend"]) for day in await _days(db, key)} + assert days == {"complexity": (1, 1.0, 4.0), "quality": (1, 2.0, 0.0)} + + +async def test_a_router_type_change_mid_session_keeps_session_shape_with_the_sessions_type(db): + key = f"k-{uuid.uuid4()}" + router = f"auto-{uuid.uuid4()}" + await _turn(db, key, "A", T0, router=router, router_type="complexity", spend=1.0, saved=4.0) + await _turn(db, key, "B", T0 + timedelta(hours=1), router=router, router_type="quality", spend=2.0, saved=0.0) + + rows = {row["router_type"]: row for row in await _benchmark_rows(db, T0, T0 + timedelta(days=1), key)} + assert set(rows) == {"complexity", "quality"} + assert (rows["complexity"]["sessions"], rows["complexity"]["session_turns"], rows["complexity"]["turns"]) == (1, 2, 1) + assert (rows["quality"]["sessions"], rows["quality"]["session_turns"], rows["quality"]["turns"]) == (0, 0, 1) + assert rows["quality"]["spend"] == 2.0 diff --git a/tests/proxy_behavior/spend/test_baseline_accounting.py b/tests/proxy_behavior/spend/test_baseline_accounting.py index 3504751d132..8fb82d0c80e 100644 --- a/tests/proxy_behavior/spend/test_baseline_accounting.py +++ b/tests/proxy_behavior/spend/test_baseline_accounting.py @@ -151,6 +151,13 @@ async def test_late_replay_updates_all_projections_without_rebilling(db: Prisma, ): assert after_users["late-user"][field] == after[field] assert after_users["late-user"]["turns"] == 1 and after_users["late-user"]["spend"] == 0.17 + days: Final = await db.query_raw( + 'SELECT * FROM "LiteLLM_AutoRouterDailySpend" WHERE api_key=$1 ORDER BY user_id', late.api_key + ) + assert [(day["date"], day["user_id"]) for day in days] == [("1970-01-01", "early-user"), ("1970-01-01", "late-user")] + assert days[0]["saved_spend"] == days[0]["savings_estimated_turns"] == 0 + for field in ("saved_spend", "savings_estimated_turns", "savings_estimated_actual_spend", "savings_estimated_saved_spend"): + assert days[1][field] == after[field] for table in ("DailyUserSpend", "DailyTeamSpend", "DailyOrganizationSpend", "DailyEndUserSpend", "DailyAgentSpend", "DailyTagSpend"): rows: Final = await db.query_raw(f'SELECT spend,api_requests,autorouter_savings_spend FROM "LiteLLM_{table}" WHERE api_key=$1', late.api_key) assert rows[0]["spend"] == rows[0]["api_requests"] == 0 diff --git a/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py index 2cbba9da8b3..01e41e8b03f 100644 --- a/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py +++ b/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py @@ -619,6 +619,18 @@ def test_classifier_plugin_is_not_settable_over_http(): _request("what is 2+2", classifier_type="custom", classifier_plugin="my_module.instance") +def _benchmark_db(rows: Sequence[Mapping[str, object]], recorded: float | None = None) -> SimpleNamespace: + """The joined benchmark statement returns the rows as given; any other statement is the Overall total.""" + from litellm.proxy.db.autorouter_session_rollup import AUTOROUTER_BENCHMARKS_SQL + + total: Final = recorded if recorded is not None else sum(float(row.get("saved_spend") or 0.0) for row in rows) + + async def query_raw(sql: str, *params: object) -> Sequence[Mapping[str, object]]: + return rows if sql == AUTOROUTER_BENCHMARKS_SQL else ({"saved": total},) + + return SimpleNamespace(db=SimpleNamespace(query_raw=AsyncMock(side_effect=query_raw))) + + class TestAutoRouterBenchmarks: from litellm.proxy.management_endpoints.auto_router_endpoints import _SessionAggRow @@ -635,15 +647,12 @@ class TestAutoRouterBenchmarks: rows: Sequence[Mapping[str, object]], model_list: Sequence[object], api_key: str | None = None, + recorded: float | None = None, ) -> AutoRouterBenchmarksResponse: from litellm.proxy import proxy_server from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks - class _DB: - async def query_raw(self, sql: str, *params: object): - return rows - - monkeypatch.setattr(proxy_server, "prisma_client", type("P", (), {"db": _DB()})()) + monkeypatch.setattr(proxy_server, "prisma_client", _benchmark_db(rows, recorded)) monkeypatch.setattr(proxy_server, "llm_router", type("R", (), {"model_list": model_list})()) return await get_auto_router_benchmarks( user_api_key_dict=ADMIN, @@ -657,6 +666,7 @@ class TestAutoRouterBenchmarks: router_type="complexity", tier_turns={}, sessions=4, + session_turns=40, turns=40, unordered_turns=1, covered_turns=38, @@ -703,7 +713,6 @@ class TestAutoRouterBenchmarks: assert totals.baseline_spend == 40.0 assert totals.saved_pct == 75.0 assert totals.savings_estimated_classifier_cost == 0.4 - assert totals.saved_per_session == 7.5 assert totals.cache.coverage_pct == 95.0 assert totals.cache.hit_rate_pct == pytest.approx(73.7) assert totals.cache.same_model.hit_rate_pct == 95.0 @@ -742,7 +751,6 @@ class TestAutoRouterBenchmarks: assert (totals.spend, totals.saved_spend, totals.baseline_spend, totals.saved_pct) == (10.0, 30.0, 40.0, 75.0) assert (totals.savings_estimated_turns, totals.savings_estimated_actual_spend) == (40, 10.0) assert totals.savings_estimated_classifier_cost == 0.4 - assert totals.saved_per_session == 7.5 @pytest.mark.asyncio @pytest.mark.parametrize("router_type, saved", [("adaptive", 0.0), ("quality", 0.0), ("quality", 2.0)]) @@ -772,9 +780,50 @@ class TestAutoRouterBenchmarks: totals: Final = response.totals assert (totals.turns, totals.spend) == (50, 13.0) assert (totals.savings_estimated_turns, totals.savings_estimated_actual_spend) == (40, 10.0) - assert (totals.saved_spend, totals.baseline_spend, totals.saved_pct) == (30.0, 40.0, 75.0) + assert totals.unattributed_saved_spend is None + assert (totals.saved_spend, totals.baseline_spend, totals.saved_pct) == ( + (30.0, 40.0, 75.0) if saved == 0.0 else (32.0, None, None) + ) assert totals.savings_estimated_classifier_cost == 0.4 + @pytest.mark.asyncio + @pytest.mark.parametrize("recorded, unattributed", [(30.0, None), (33.0, 3.0), (27.0, -3.0)]) + async def test_the_headline_is_the_overall_daily_total_and_untracked_savings_void_the_baseline( + self, recorded: float, unattributed: float | None, monkeypatch: pytest.MonkeyPatch + ) -> None: + response: Final = await self._benchmarks( + monkeypatch, rows=[self.ROW.model_dump()], model_list=[], recorded=recorded + ) + totals: Final = response.totals + assert (totals.saved_spend, totals.unattributed_saved_spend) == (recorded, unattributed) + assert (totals.baseline_spend, totals.saved_pct) == ((40.0, 75.0) if unattributed is None else (None, None)) + group: Final = response.groups[0] + assert group.saved_spend == 30.0 + assert (group.baseline_spend, group.saved_pct) == ((40.0, 75.0) if unattributed is None else (None, None)) + + @pytest.mark.asyncio + async def test_a_window_holding_only_untracked_history_shows_no_router_baseline( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + history_only: Final = self.ROW.model_dump( + exclude={ + "turns", + "spend", + "saved_spend", + "savings_estimated_turns", + "savings_estimated_actual_spend", + "savings_estimated_classifier_cost", + "savings_estimated_saved_spend", + "classifier_cost", + "classifier_cost_recorded_turns", + } + ) + response: Final = await self._benchmarks(monkeypatch, rows=[history_only], model_list=[], recorded=3.0) + assert (response.totals.saved_spend, response.totals.unattributed_saved_spend) == (3.0, 3.0) + group: Final = response.groups[0] + assert (group.sessions, group.turns, group.saved_spend) == (4, 0, 0.0) + assert (group.baseline_spend, group.saved_pct) == (None, None) + def test_an_empty_window_folds_to_zeros(self): from litellm.proxy.management_endpoints.auto_router_endpoints import ( _benchmark_totals, @@ -805,7 +854,7 @@ class TestAutoRouterBenchmarks: "savings_estimated_classifier_cost": 0.0, } ) - summed = _summed_agg_row([self.ROW, other]) + summed = _summed_agg_row([self.ROW, other.model_copy(update={"session_turns": 10})]) totals = _benchmark_totals(summed) assert summed.sessions == 5 assert summed.turns == 50 @@ -868,6 +917,28 @@ class TestAutoRouterBenchmarks: assert response.status_code == 422 query.assert_not_awaited() + @pytest.mark.asyncio + async def test_an_empty_key_filter_is_rejected_before_querying_deployment_data( + self, monkeypatch: pytest.MonkeyPatch + ) -> None: + import httpx + from fastapi import FastAPI + + from litellm.proxy import proxy_server + from litellm.proxy.auth.user_api_key_auth import user_api_key_auth + from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks + + query: Final = AsyncMock(return_value=[]) + monkeypatch.setattr(proxy_server, "prisma_client", SimpleNamespace(db=SimpleNamespace(query_raw=query))) + app: Final = FastAPI() + app.get("/auto_router/benchmarks")(get_auto_router_benchmarks) + app.dependency_overrides[user_api_key_auth] = lambda: ADMIN + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=app), base_url="http://test") as client: + response: Final = await client.get("/auto_router/benchmarks", params={"api_key": ""}) + + assert response.status_code == 422 + query.assert_not_awaited() + @pytest.mark.asyncio async def test_a_reversed_window_is_rejected(self, monkeypatch: pytest.MonkeyPatch): from litellm.proxy import proxy_server @@ -891,15 +962,8 @@ class TestAutoRouterBenchmarks: from litellm.proxy import proxy_server from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks - captured: dict = {} - - class _DB: - async def query_raw(self, sql: str, *params: object): - captured["sql"] = sql - captured["params"] = params - return [TestAutoRouterBenchmarks.ROW.model_dump()] - - monkeypatch.setattr(proxy_server, "prisma_client", type("P", (), {"db": _DB()})()) + prisma_client: Final = _benchmark_db([TestAutoRouterBenchmarks.ROW.model_dump()]) + monkeypatch.setattr(proxy_server, "prisma_client", prisma_client) response = await get_auto_router_benchmarks( user_api_key_dict=UserAPIKeyAuth(user_role=role, api_key="sk-admin", user_id="viewer"), @@ -908,7 +972,11 @@ class TestAutoRouterBenchmarks: api_key="key-hash", user_id=user_id, ) - assert captured["params"] == ("2026-07-01T00:00:00", "2026-08-02T00:00:00", "key-hash", user_id) + params: Final = tuple(call.args[1:] for call in prisma_client.db.query_raw.await_args_list) + assert params == ( + ("2026-07-01T00:00:00", "2026-08-02T00:00:00", "key-hash", user_id, "2026-07-01", "2026-08-01"), + ("2026-07-01", "2026-08-01", *(([user_id],) if user_id else ()), ["key-hash"]), + ) assert response.routers_in_scope == 1 assert response.groups[0].router_name == "live-auto" assert response.groups[0].saved_pct == response.totals.saved_pct == 75.0 @@ -946,7 +1014,6 @@ class TestAutoRouterBenchmarks: assert response.totals.saved_spend == 29.5 assert response.totals.baseline_spend == 41.5 assert response.totals.saved_pct == 71.1 - assert response.totals.saved_per_session == 5.9 @pytest.mark.asyncio @pytest.mark.parametrize( @@ -958,11 +1025,9 @@ class TestAutoRouterBenchmarks: from litellm.proxy import proxy_server from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_benchmarks - class _DB: - async def query_raw(self, sql: str, *params: object): - return [{**TestAutoRouterBenchmarks.ROW.model_dump(), "tier_turns": wire_value}] - - monkeypatch.setattr(proxy_server, "prisma_client", type("P", (), {"db": _DB()})()) + monkeypatch.setattr( + proxy_server, "prisma_client", _benchmark_db([{**TestAutoRouterBenchmarks.ROW.model_dump(), "tier_turns": wire_value}]) + ) response = await get_auto_router_benchmarks( user_api_key_dict=ADMIN, @@ -1006,7 +1071,7 @@ class TestAutoRouterBenchmarks: 0.0, 0.0, ) - assert (idle.saved_pct, idle.saved_per_session, idle.avg_turns_per_session) == (0.0, 0.0, 0.0) + assert (idle.saved_pct, idle.avg_turns_per_session) == (0.0, 0.0) assert (idle.cache.hit_rate_pct, idle.cache.coverage_pct) == (0.0, 0.0) assert idle.cache.same_model.turns == idle.cache.return_to_tier.hits == 0 assert idle.tier_turns == {} @@ -3724,3 +3789,18 @@ async def test_availability_waits_for_the_first_complete_catalog(monkeypatch): with pytest.raises(HTTPException) as error: await auto_router_endpoints.get_auto_router_availability(AutoRouterAvailabilityRequest(), ADMIN) assert error.value.status_code == 503 + + +class TestPerSessionAverages: + @pytest.mark.parametrize( + "sessions, turns, expected", + [(4, 40, (10.0, 100.0, 1000.0)), (0, 0, (0.0, 0.0, 0.0)), (0, 3, (None, None, None))], + ) + def test_requests_without_session_rows_have_unknown_averages_not_zero( + self, sessions: int, turns: int, expected: tuple[float | None, ...] + ) -> None: + from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals + + row: Final = TestAutoRouterBenchmarks.ROW.model_copy(update={"sessions": sessions, "turns": turns}) + totals: Final = _benchmark_totals(row) + assert (totals.avg_turns_per_session, totals.avg_session_seconds, totals.avg_tokens_per_session) == expected diff --git a/tests/unit/proxy/test_spend_log_cleanup.py b/tests/unit/proxy/test_spend_log_cleanup.py index 46ac1234615..399c76d97c1 100644 --- a/tests/unit/proxy/test_spend_log_cleanup.py +++ b/tests/unit/proxy/test_spend_log_cleanup.py @@ -796,19 +796,23 @@ async def test_spend_logs_retention_alone_does_not_touch_the_session_rollup(): assert any('"LiteLLM_SpendLogs"' in sql for sql in tables) assert not any('"LiteLLM_AutoRouterSession"' in sql for sql in tables) assert not any('"LiteLLM_AutoRouterUserSession"' in sql for sql in tables) + assert not any('"LiteLLM_AutoRouterDailySpend"' in sql for sql in tables) assert not any('"LiteLLM_HealthCheckTable"' in sql for sql in tables) @pytest.mark.asyncio -async def test_session_retention_alone_cleans_both_session_rollups(): - client = _mock_prisma_for_retention([0, 0]) +async def test_session_retention_alone_cleans_both_session_rollups_and_the_daily_rollup(): + client = _mock_prisma_for_retention([0, 0, 0]) cleaner = SpendLogCleanup(general_settings={"maximum_autorouter_session_retention_period": "365d"}) cleaner.pod_lock_manager = None await cleaner.cleanup_old_spend_logs(client) - tables = [call[0][0] for call in client.db.execute_raw.call_args_list] - assert len(tables) == 2 + calls = client.db.execute_raw.call_args_list + tables = [call[0][0] for call in calls] + assert len(tables) == 3 assert '"LiteLLM_AutoRouterSession"' in tables[0] assert '"LiteLLM_AutoRouterUserSession"' in tables[1] + assert '"LiteLLM_AutoRouterDailySpend"' in tables[2] + assert calls[2][0][1] == calls[0][0][1].date().isoformat() @pytest.mark.asyncio @@ -852,7 +856,7 @@ async def test_spend_logs_retention_alone_keeps_daily_tag_spend_forever(): @pytest.mark.asyncio async def test_each_retention_key_cuts_off_at_its_own_horizon(): - client = _mock_prisma_for_retention([0, 0, 0, 0, 0]) + client = _mock_prisma_for_retention([0, 0, 0, 0, 0, 0]) cleaner = SpendLogCleanup( general_settings={ "maximum_spend_logs_retention_period": "7d", @@ -868,6 +872,8 @@ async def test_each_retention_key_cuts_off_at_its_own_horizon(): if '"LiteLLM_AutoRouterSession"' in call[0][0] else "LiteLLM_AutoRouterUserSession" if '"LiteLLM_AutoRouterUserSession"' in call[0][0] + else "LiteLLM_AutoRouterDailySpend" + if '"LiteLLM_AutoRouterDailySpend"' in call[0][0] else "LiteLLM_HealthCheckTable" if '"LiteLLM_HealthCheckTable"' in call[0][0] else "logs" @@ -878,6 +884,7 @@ async def test_each_retention_key_cuts_off_at_its_own_horizon(): assert (now - cutoffs["logs"]).days == 7 assert (now - cutoffs["LiteLLM_AutoRouterSession"]).days == 365 assert cutoffs["LiteLLM_AutoRouterUserSession"] == cutoffs["LiteLLM_AutoRouterSession"] + assert cutoffs["LiteLLM_AutoRouterDailySpend"] == cutoffs["LiteLLM_AutoRouterSession"].date().isoformat() assert (now - cutoffs["LiteLLM_HealthCheckTable"]).days == 30 diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx index b4b34dbfaf3..081eb7f6e09 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx @@ -75,7 +75,6 @@ const totals = (overrides: Partial = {}): Totals => ({ saved_spend: 2174.59, baseline_spend: 2534.45, saved_pct: 85.8, - saved_per_session: 23.13, cache: cache(), ...overrides, }); @@ -110,7 +109,6 @@ const zeroTotals: Totals = { saved_spend: 0, baseline_spend: 0, saved_pct: 0, - saved_per_session: 0, cache: zeroCache, }; @@ -173,7 +171,6 @@ describe("AutoRouterBenchmarksTab", () => { saved_spend: saved, baseline_spend: estimatedTurns ? actual + (saved ?? 0) : null, saved_pct: pct, - saved_per_session: null, }; mockHook({ data: response([], totals(comparison)), @@ -204,18 +201,15 @@ describe("AutoRouterBenchmarksTab", () => { } }); - it("leads with total estimated savings, before the four session-shape metrics", () => { + it("leads with total estimated savings, before the three session-shape metrics", () => { mockHook({ data: response([group(), group({ router_name: "gpt-auto" })]) }); renderTab(); const labels = screen - .getAllByText( - /Total estimated savings|Avg saved per session|Avg turns per session|Avg session length|Avg tokens per session/, - ) + .getAllByText(/Total estimated savings|Avg turns per session|Avg session length|Avg tokens per session/) .map((node) => node.textContent); expect(labels).toEqual([ "Total estimated savings", - "Avg saved per session", "Avg turns per session", "Avg session length", "Avg tokens per session", @@ -271,15 +265,40 @@ describe("AutoRouterBenchmarksTab", () => { }, ); - it("pairs the savings with the session count it was earned over, in its own tile", () => { + it("labels selected-day money apart from whole-session metrics, with no savings-per-session tile", () => { mockHook({ data: response([group(), group({ router_name: "gpt-auto" })]) }); renderTab(); - const tile = screen.getByText("Avg saved per session").closest('[data-slot="card"]'); - if (!tile) throw new Error("expected avg saved per session to render as a metric tile"); - - expect(within(tile).getByText("$23.13")).toBeInTheDocument(); + const tile = screen.getByText("Avg turns per session").closest('[data-slot="card"]'); + if (!tile) throw new Error("expected avg turns per session to render as a metric tile"); expect(within(tile).getByText("· 94 sessions")).toBeInTheDocument(); + expect(screen.queryByText("Avg saved per session")).not.toBeInTheDocument(); + expect(screen.getByText(/Savings and spend count requests on the selected UTC days/)).toBeInTheDocument(); + expect(screen.getByText(/Session metrics cover every session that overlaps the range/)).toBeInTheDocument(); + }); + + it("shows session averages as unavailable, not zero, when routed requests have no session rows", () => { + const noSessions = { + sessions: 0, + avg_turns_per_session: null, + avg_session_seconds: null, + avg_tokens_per_session: null, + }; + mockHook({ data: response([], totals(noSessions)) }); + renderTab(); + + expect(screen.getAllByText("Unavailable")).toHaveLength(3); + expect(screen.queryByText("0.0")).not.toBeInTheDocument(); + }); + + it.each([3, -3])("explains a %s gap between router records and recorded savings instead of comparing", (gap) => { + const residual = { saved_spend: 5, unattributed_saved_spend: gap, baseline_spend: null, saved_pct: null }; + mockHook({ data: response([], totals(residual)) }); + renderTab(); + + expect(screen.getByText("$5.00")).toBeInTheDocument(); + expect(screen.getByText(/Per-router records differ from recorded savings by \$3\.00/)).toBeInTheDocument(); + expect(screen.getByText("Estimated baseline spend").nextSibling?.textContent).toBe("Unavailable"); }); it("exposes each spend row as a term and its value, not as loose text", () => { @@ -431,7 +450,7 @@ describe("AutoRouterBenchmarksTab", () => { renderTab(); expect(screen.getByText("Total estimated savings")).toBeInTheDocument(); - expect(screen.getAllByText("$0.00")).toHaveLength(6); + expect(screen.getAllByText("$0.00")).toHaveLength(5); expect(screen.getByText("· 0 sessions")).toBeInTheDocument(); expect(screen.getByText("0s")).toBeInTheDocument(); expect(screen.getByText(/turns measured/)).toBeInTheDocument(); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx index 24a97587e32..27e8df6db87 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx @@ -105,6 +105,12 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => { adaptive and quality routers are excluded

)} + {stats.unattributed_saved_spend != null && ( +

+ Per-router records differ from recorded savings by {usd(Math.abs(stats.unattributed_saved_spend))}, for + example history from before per-router tracking, so the baseline comparison is unavailable +

+ )}
@@ -297,22 +303,34 @@ const BenchmarksBody: React.FC = ({ isPending, error, data, -
- - - - -
+

+ Savings and spend count requests on the selected UTC days. Actual spend covers every request on complexity + routers, including LLM classification cost. Baseline is actual spend plus recorded savings, so savings can be + zero or negative. +

- Actual spend covers every request on complexity routers, including LLM classification cost. Baseline is actual - spend plus recorded savings, so savings can be zero or negative. The range counts whole sessions that overlap - it, so totals can differ from savings views that group usage by UTC day. + Session metrics cover every session that overlaps the range, including its turns outside the range.

+
+ + + +
diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/TierTurnsChart.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/TierTurnsChart.test.tsx index e4417d77463..42444fd8f06 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/TierTurnsChart.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/TierTurnsChart.test.tsx @@ -12,21 +12,21 @@ vi.mock("@/components/shared/charts", () => ({ })); import TierTurnsChart, { tierDisplayLabel } from "./TierTurnsChart"; -import type { AutoRouterBenchmarkGroup, BenchmarkView } from "./autoRouterBenchmarks"; +import type { AutoRouterBenchmarkGroup, AutoRouterBenchmarkTotals, BenchmarkView } from "./autoRouterBenchmarks"; -const totalsOnly = { +const totalsOnly: AutoRouterBenchmarkTotals = { sessions: 3, turns: 9, avg_turns_per_session: 3, avg_session_seconds: 60, avg_tokens_per_session: 100, spend: 1, + classifier_cost: 0, savings_estimated_turns: 9, savings_estimated_actual_spend: 1, saved_spend: 1, baseline_spend: 2, saved_pct: 50, - saved_per_session: 0.33, cache: { coverage_pct: 0, hit_rate_pct: 0, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/autoRouterBenchmarks.test.ts b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/autoRouterBenchmarks.test.ts index 0586163e77e..62d0c0e4e5e 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/autoRouterBenchmarks.test.ts +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/autoRouterBenchmarks.test.ts @@ -43,7 +43,6 @@ const totals = (overrides: Partial = {}) => ({ saved_spend: 2174.59, baseline_spend: 2534.45, saved_pct: 85.8, - saved_per_session: 23.13, cache: cache(), ...overrides, }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx index 676bc7d29eb..4195a3c9913 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/users/_components/view_users/user_info_view.integration.test.tsx @@ -347,7 +347,6 @@ const routerUsageResponse = (saved: number): AutoRouterBenchmarksResponse => ({ saved_spend: saved, baseline_spend: 10 + saved, saved_pct: (100 * saved) / (10 + saved), - saved_per_session: saved / 2, cache: { coverage_pct: 100, hit_rate_pct: 0, diff --git a/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx b/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx index 5c7b9b28789..2ce585e895e 100644 --- a/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx @@ -39,7 +39,6 @@ const stats = { saved_spend: 8.75, baseline_spend: 10, saved_pct: 87.5, - saved_per_session: 4.375, cache, }; diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index 1d50f37e573..fc1a9c67946 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -1350,9 +1350,10 @@ export interface paths { * * Reads session rollups folded once per request at spend-write time, so this endpoint * never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that - * internal user when written; older key-only history remains outside user views. A session - * is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before - * end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is + * internal user when written; older key-only history remains outside user views. Money counts + * only requests on the selected UTC days, and the all-router savings headline is the same daily + * total the Overall view reads. Session shape and caching cover every session that overlaps the + * window, whole. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is * over that bucket's turns. * * The rollup supplies the measures, never the list. Which routers appear comes from the @@ -26169,12 +26170,21 @@ export interface components { * @description One auto-router's slice of the benchmarks. */ AutoRouterBenchmarkGroup: { - /** Avg Session Seconds */ - avg_session_seconds: number; - /** Avg Tokens Per Session */ - avg_tokens_per_session: number; - /** Avg Turns Per Session */ - avg_turns_per_session: number; + /** + * Avg Session Seconds + * @description Lifetime seconds per overlapping session; null as above + */ + avg_session_seconds: number | null; + /** + * Avg Tokens Per Session + * @description Lifetime tokens per overlapping session; null as above + */ + avg_tokens_per_session: number | null; + /** + * Avg Turns Per Session + * @description Lifetime turns per overlapping session; null when the window has routed requests but no session rows for this router type, such as an alias whose router type changed mid-session + */ + avg_turns_per_session: number | null; /** * Baseline Spend * @description Estimated single-model cost: compared actual spend plus recorded savings; null when traffic has no recorded savings @@ -26201,14 +26211,9 @@ export interface components { * @description Recorded savings over baseline_spend, as a percentage */ saved_pct: number | null; - /** - * Saved Per Session - * @description Recorded savings per session, including historical estimates - */ - saved_per_session: number | null; /** * Saved Spend - * @description Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates + * @description Recorded savings on the selected UTC days; null when traffic has no recorded savings estimates. On totals this is the same daily figure the Overall savings view reports */ saved_spend: number | null; /** @@ -26226,11 +26231,14 @@ export interface components { * @description Requests compared against the baseline: every request on complexity routers that recorded savings */ savings_estimated_turns: number; - /** Sessions */ + /** + * Sessions + * @description Sessions overlapping the window, counted whole + */ sessions: number; /** * Spend - * @description What the routed traffic actually cost + * @description What the selected days' routed traffic actually cost */ spend: number; /** @@ -26240,20 +26248,38 @@ export interface components { tier_turns?: { [key: string]: number; }; - /** Turns */ + /** + * Turns + * @description Auto-routed requests on the selected UTC days + */ turns: number; + /** + * Unattributed Saved Spend + * @description Part of saved_spend no router's daily rows account for, such as history recorded before per-router daily tracking; when set, baseline_spend and saved_pct are null + */ + unattributed_saved_spend?: number | null; }; /** * AutoRouterBenchmarkTotals - * @description Session-shape and savings aggregates over auto-routed traffic in the window. + * @description Auto-routed traffic in the window. Turns, spend and savings count requests on the selected UTC days; + * the session averages and cache stats describe every session overlapping the window, whole. */ AutoRouterBenchmarkTotals: { - /** Avg Session Seconds */ - avg_session_seconds: number; - /** Avg Tokens Per Session */ - avg_tokens_per_session: number; - /** Avg Turns Per Session */ - avg_turns_per_session: number; + /** + * Avg Session Seconds + * @description Lifetime seconds per overlapping session; null as above + */ + avg_session_seconds: number | null; + /** + * Avg Tokens Per Session + * @description Lifetime tokens per overlapping session; null as above + */ + avg_tokens_per_session: number | null; + /** + * Avg Turns Per Session + * @description Lifetime turns per overlapping session; null when the window has routed requests but no session rows for this router type, such as an alias whose router type changed mid-session + */ + avg_turns_per_session: number | null; /** * Baseline Spend * @description Estimated single-model cost: compared actual spend plus recorded savings; null when traffic has no recorded savings @@ -26270,14 +26296,9 @@ export interface components { * @description Recorded savings over baseline_spend, as a percentage */ saved_pct: number | null; - /** - * Saved Per Session - * @description Recorded savings per session, including historical estimates - */ - saved_per_session: number | null; /** * Saved Spend - * @description Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates + * @description Recorded savings on the selected UTC days; null when traffic has no recorded savings estimates. On totals this is the same daily figure the Overall savings view reports */ saved_spend: number | null; /** @@ -26295,19 +26316,30 @@ export interface components { * @description Requests compared against the baseline: every request on complexity routers that recorded savings */ savings_estimated_turns: number; - /** Sessions */ + /** + * Sessions + * @description Sessions overlapping the window, counted whole + */ sessions: number; /** * Spend - * @description What the routed traffic actually cost + * @description What the selected days' routed traffic actually cost */ spend: number; - /** Turns */ + /** + * Turns + * @description Auto-routed requests on the selected UTC days + */ turns: number; + /** + * Unattributed Saved Spend + * @description Part of saved_spend no router's daily rows account for, such as history recorded before per-router daily tracking; when set, baseline_spend and saved_pct are null + */ + unattributed_saved_spend?: number | null; }; /** * AutoRouterBenchmarksResponse - * @description Benchmarks for the auto-router dashboard, aggregated from the per-session rollup. + * @description Benchmarks for the auto-router dashboard, aggregated from the per-session and per-day rollups. */ AutoRouterBenchmarksResponse: { /**