diff --git a/litellm/models/autorouter_session.py b/litellm/models/autorouter_session.py index ddce2b5ef81..9df9af18dff 100644 --- a/litellm/models/autorouter_session.py +++ b/litellm/models/autorouter_session.py @@ -34,15 +34,12 @@ class LiteLLM_AutoRouterSession(LiteLLMPydanticObjectBase): @property def baseline_model(self) -> str | None: - """The baseline most covered turns were priced against, or None when none were estimated. - - A router reconfigured mid-session leaves turns priced against two baselines; the row keeps both - counts, and the label is the one that priced the most money-carrying turns rather than whatever the - router is configured with now. - """ - if not self.savings_estimated_baseline_models: + """A recorded baseline label when excluded turns cannot change the selected model.""" + if not self.baseline_models: + return None + if self.savings_estimated_turns < self.turns and len(self.baseline_models) > 1: return None return max( - self.savings_estimated_baseline_models, - key=lambda model: (self.savings_estimated_baseline_models[model], model), + self.baseline_models, + key=lambda model: (self.baseline_models[model], model), ) diff --git a/litellm/proxy/db/autorouter_savings_comparison.py b/litellm/proxy/db/autorouter_savings_comparison.py new file mode 100644 index 00000000000..041496d63f2 --- /dev/null +++ b/litellm/proxy/db/autorouter_savings_comparison.py @@ -0,0 +1,147 @@ +from collections.abc import Mapping +from contextlib import AbstractAsyncContextManager +from datetime import timedelta +from math import isclose +from types import MappingProxyType +from typing import TYPE_CHECKING, Final, Protocol, cast + +from pydantic import BaseModel, ConfigDict, TypeAdapter + +from litellm._logging import verbose_proxy_logger +from litellm.constants import MAX_SPENDLOG_ROWS_TO_QUERY +from litellm.proxy.db.autorouter_session_rollup import AUTOROUTER_SESSION_WINDOW_SQL +from litellm.proxy.db.create_views import SupportsRawQueries + +if TYPE_CHECKING: + from litellm.proxy.utils import PrismaClient + + +class SessionSavingsComparison(BaseModel): + model_config = ConfigDict(frozen=True, allow_inf_nan=False) + + router_name: str + router_type: str + turns: int + estimated_turns: int + actual_spend: float + classifier_cost: float | None + saved_spend: float + complete: bool + + def coverage_fields(self, recorded_savings: float, recorded_turns: int) -> Mapping[str, float | int]: + if self.turns != recorded_turns or not self.complete: + return MappingProxyType({}) + if not isclose(self.saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9): + return MappingProxyType({}) + return MappingProxyType( + { + "savings_estimated_turns": self.estimated_turns, + "savings_estimated_actual_spend": self.actual_spend, + "savings_estimated_saved_spend": self.saved_spend, + } + ) + + +class _ReadTransactions(Protocol): + def tx(self, *, timeout: timedelta, max_wait: timedelta) -> AbstractAsyncContextManager[SupportsRawQueries]: ... + + +_COMPARISONS: Final = TypeAdapter(tuple[SessionSavingsComparison, ...]) + + +async def historical_session_comparisons( + prisma_client: "PrismaClient", + start_date: str, + end_date: str, + api_key: str | None, + user_id: str | None, + session_id: str | None = None, +) -> Mapping[tuple[str, str], SessionSavingsComparison]: + try: + reader: Final = cast(_ReadTransactions, prisma_client.read_db) # cast-ok: untyped Prisma transaction delegate + async with reader.tx(timeout=timedelta(seconds=3), max_wait=timedelta(seconds=1)) as transaction: + await transaction.execute_raw("SET TRANSACTION READ ONLY") + await transaction.execute_raw("SET LOCAL statement_timeout = 2000") + rows: Final = await transaction.query_raw( + HISTORICAL_SESSION_COMPARISONS_SQL, + start_date, + end_date, + api_key, + user_id, + session_id, + ) + comparisons: Final = _COMPARISONS.validate_python(rows or ()) + return MappingProxyType({(row.router_name, row.router_type): row for row in comparisons}) + except Exception: # noqa: BLE001 # missing retained logs must not discard recorded dollar savings + verbose_proxy_logger.warning("Historical auto-router cost comparison unavailable; preserving recorded savings") + return MappingProxyType({}) + + +HISTORICAL_SESSION_COMPARISONS_SQL: Final = f""" +WITH {AUTOROUTER_SESSION_WINDOW_SQL}, scoped AS MATERIALIZED ( + SELECT * FROM windowed WHERE $5::text IS NULL OR session_id = $5::text +), limited_logs AS MATERIALIZED ( + SELECT session.api_key, session.session_id, session.router_name, session.router_type, session.comparison_user_id, + session.classifier_cost_recorded_turns = session.turns AS classifier_cost_tracked, + logs.spend, logs.prompt_tokens + logs.completion_tokens AS tokens, + logs.metadata::jsonb -> 'routing_decision' AS decision, + logs.metadata::jsonb -> 'autorouter_savings' AS savings, + logs.metadata::jsonb -> 'autorouter_savings_estimate' AS estimate + FROM scoped AS session JOIN "LiteLLM_SpendLogs" AS logs + ON logs.api_key = session.api_key + AND CASE WHEN char_length(logs.session_id) > 256 + THEN 'sha256:' || encode(sha256(convert_to(logs.session_id, 'UTF8')), 'hex') + ELSE logs.session_id END = session.session_id + AND (session.comparison_user_id IS NULL OR logs."user" = session.comparison_user_id) + AND logs."startTime" BETWEEN session.first_turn_at AND session.last_turn_at + AND COALESCE(logs.metadata::jsonb #>> '{{routing_decision,router_model_name}}', logs.model_group) + = session.router_name + WHERE session.savings_estimated_turns < session.turns + AND logs.status = 'success' AND COALESCE(logs.metadata::jsonb ->> 'internal_call_origin', '') = '' + LIMIT {MAX_SPENDLOG_ROWS_TO_QUERY + 1} +), facts AS ( + SELECT *, + CASE WHEN jsonb_typeof(decision -> 'classifier_cost') = 'number' + THEN (decision ->> 'classifier_cost')::float8 + WHEN classifier_cost_tracked THEN 0 END AS classifier, + CASE WHEN jsonb_typeof(savings) = 'number' AND ( + estimate IS NULL OR estimate = 'null'::jsonb OR ( + jsonb_typeof(estimate -> 'version') = 'number' AND estimate ->> 'version' IN ('1', '2', '3') + AND estimate ->> 'status' = 'estimated' + ) + ) THEN savings::text::float8 END AS saved + FROM limited_logs +), compared AS ( + SELECT api_key, session_id, router_name, router_type, comparison_user_id, + COUNT(*) AS turns, SUM(spend + COALESCE(classifier, 0)) AS spend, SUM(tokens) AS total_tokens, + COUNT(saved) AS estimated_turns, + COALESCE(SUM(spend + COALESCE(classifier, 0)) FILTER (WHERE saved IS NOT NULL), 0)::float8 AS actual_spend, + CASE WHEN COUNT(saved) = COUNT(classifier) FILTER (WHERE saved IS NOT NULL) + THEN COALESCE(SUM(classifier) FILTER (WHERE saved IS NOT NULL), 0)::float8 + END AS estimated_classifier_cost, + COALESCE(SUM(saved), 0)::float8 AS saved_spend + FROM facts GROUP BY 1, 2, 3, 4, 5 +), reconciled AS ( + SELECT session.*, logs.estimated_turns, logs.actual_spend, logs.estimated_classifier_cost, + COALESCE((SELECT COUNT(*) FROM limited_logs) <= {MAX_SPENDLOG_ROWS_TO_QUERY} + AND logs.turns = session.turns AND logs.total_tokens = session.total_tokens + AND ABS(logs.spend - session.spend) <= GREATEST(1e-9, ABS(session.spend) * 1e-9) + AND ABS(logs.saved_spend - session.saved_spend) <= GREATEST(1e-9, ABS(session.saved_spend) * 1e-9), FALSE + ) AS recovered + FROM scoped AS session LEFT JOIN compared AS logs + ON logs.api_key = session.api_key AND logs.session_id = session.session_id + AND logs.router_name = session.router_name AND logs.router_type = session.router_type + AND logs.comparison_user_id IS NOT DISTINCT FROM session.comparison_user_id +) +SELECT router_name, router_type, + SUM(turns)::bigint AS turns, + SUM(CASE WHEN recovered THEN estimated_turns ELSE savings_estimated_turns END)::bigint AS estimated_turns, + SUM(CASE WHEN recovered THEN actual_spend ELSE savings_estimated_actual_spend END)::float8 AS actual_spend, + CASE WHEN BOOL_AND(CASE WHEN recovered THEN estimated_classifier_cost IS NOT NULL + ELSE savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns END) + THEN SUM(CASE WHEN recovered THEN estimated_classifier_cost ELSE classifier_cost END)::float8 + END AS classifier_cost, + SUM(saved_spend)::float8 AS saved_spend, + BOOL_AND(recovered OR savings_estimated_turns = turns) AS complete +FROM reconciled GROUP BY router_name, router_type +""" diff --git a/litellm/proxy/db/autorouter_session_rollup.py b/litellm/proxy/db/autorouter_session_rollup.py index dd08cfd1bef..b762a40f344 100644 --- a/litellm/proxy/db/autorouter_session_rollup.py +++ b/litellm/proxy/db/autorouter_session_rollup.py @@ -7,8 +7,8 @@ on the prisma client. The spend-log flush job drains the queue into key and user session rollups with one atomic statement per turn: each upsert classifies the turn (same model, first visit, return to a model the session already used, out of order) against the row's own columns, so nothing is read before the write and concurrent -pods compose. The benchmarks endpoint aggregates these rows and never touches -LiteLLM_SpendLogs. +pods compose. The benchmarks endpoint aggregates these rows and can recover matching historical +costs from retained spend logs when estimate coverage predates these columns. """ from __future__ import annotations @@ -45,20 +45,24 @@ _SESSION_COLUMNS: Final = """ savings_estimated_baseline_models """ -AUTOROUTER_BENCHMARKS_SQL: Final = f""" -WITH windowed AS ( - SELECT {_SESSION_COLUMNS} FROM "LiteLLM_AutoRouterSession" +AUTOROUTER_SESSION_WINDOW_SQL: Final = f""" +windowed AS ( + SELECT {_SESSION_COLUMNS}, NULL::text AS comparison_user_id FROM "LiteLLM_AutoRouterSession" WHERE $4::text IS NULL AND last_turn_at >= $1::timestamp AND first_turn_at < $2::timestamp AND ($3::text IS NULL OR api_key = $3::text) UNION ALL - SELECT {_SESSION_COLUMNS} FROM "LiteLLM_AutoRouterUserSession" + SELECT {_SESSION_COLUMNS}, user_id AS comparison_user_id FROM "LiteLLM_AutoRouterUserSession" WHERE (($4::text IS NOT NULL AND user_id = $4::text) OR ($4::text IS NULL AND api_key = '')) AND last_turn_at >= $1::timestamp AND first_turn_at < $2::timestamp AND ($3::text IS NULL OR api_key = $3::text) -), +) +""" + +AUTOROUTER_BENCHMARKS_SQL: Final = f""" +WITH {AUTOROUTER_SESSION_WINDOW_SQL}, tier_maps AS ( SELECT router_name, router_type, jsonb_object_agg(tier, tier_turns) AS tier_turns FROM ( @@ -95,6 +99,8 @@ SELECT COALESCE(SUM(saved_spend), 0)::float8 AS saved_spend, COALESCE(SUM(savings_estimated_turns), 0)::int AS savings_estimated_turns, COALESCE(SUM(savings_estimated_actual_spend), 0)::float8 AS savings_estimated_actual_spend, + CASE WHEN BOOL_AND(savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns) + THEN SUM(classifier_cost)::float8 END AS savings_estimated_classifier_cost, COALESCE(SUM(savings_estimated_saved_spend), 0)::float8 AS savings_estimated_saved_spend, COALESCE(SUM(classifier_cost), 0)::float8 AS classifier_cost, COALESCE(SUM(classifier_cost_recorded_turns), 0)::int AS classifier_cost_recorded_turns, diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index 9708161397a..e3bb2b0b6cc 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -8,6 +8,7 @@ POST /auto_router/validate_complexity_router_config - Dry-run the complexity-rou from collections.abc import Mapping, Sequence from datetime import datetime, timedelta, timezone from itertools import chain, groupby +from math import isclose from types import MappingProxyType from typing import TYPE_CHECKING, Annotated, Final, Protocol from uuid import uuid4 @@ -31,6 +32,7 @@ from litellm.proxy.auth.auth_checks import ( can_key_call_resolved_model, ) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth +from litellm.proxy.db.autorouter_savings_comparison import historical_session_comparisons from litellm.proxy.db.autorouter_session_rollup import ( AUTOROUTER_BENCHMARKS_SQL, bounded_session_id, @@ -651,7 +653,9 @@ class _SessionAggRow(BaseModel): saved_spend: float savings_estimated_turns: int = 0 savings_estimated_actual_spend: float = 0.0 + savings_estimated_classifier_cost: float | None = None savings_estimated_saved_spend: float = 0.0 + savings_comparison_complete: bool = True classifier_cost: float classifier_cost_recorded_turns: int session_seconds: float @@ -679,18 +683,25 @@ def _cache_bucket(turns: int, hits: int) -> AutoRouterCacheBucket: def _savings_cohort( - turns: int, estimated_turns: int, actual_spend: float, saved_spend: float + turns: int, estimated_turns: int, actual_spend: float, saved_spend: float, recorded_savings: float ) -> tuple[float | None, float | None]: - if turns > 0 and estimated_turns == 0: + if turns > 0 and estimated_turns == 0 and recorded_savings == 0: return None, None - return saved_spend, actual_spend + saved_spend + if not isclose(saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9): + return recorded_savings, None + return recorded_savings, actual_spend + recorded_savings def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: return_misses: Final = row.return_turns - row.return_hits - saved_spend, baseline_spend = _savings_cohort( - row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend + saved_spend, compared_baseline = _savings_cohort( + row.turns, + row.savings_estimated_turns, + row.savings_estimated_actual_spend, + row.savings_estimated_saved_spend, + row.saved_spend, ) + baseline_spend: Final = compared_baseline if row.savings_comparison_complete else None sessions: Final = row.sessions return AutoRouterBenchmarkTotals( sessions=sessions, @@ -701,13 +712,12 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: spend=row.spend, savings_estimated_turns=row.savings_estimated_turns, savings_estimated_actual_spend=row.savings_estimated_actual_spend, + savings_estimated_classifier_cost=row.savings_estimated_classifier_cost if baseline_spend is not None else None, saved_spend=saved_spend, classifier_cost=row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None, baseline_spend=baseline_spend, saved_pct=_pct(saved_spend, baseline_spend) if saved_spend is not None and baseline_spend is not None else None, - saved_per_session=(row.savings_estimated_saved_spend / sessions if sessions else 0.0) - if row.savings_estimated_turns == row.turns - else None, + saved_per_session=(saved_spend / sessions if sessions else 0.0) if saved_spend is not None else None, cache=AutoRouterCacheStats( coverage_pct=_pct(row.covered_turns, row.turns), hit_rate_pct=_pct(row.cache_hits, row.covered_turns), @@ -739,6 +749,7 @@ def _benchmark_group(row: _SessionAggRow) -> AutoRouterBenchmarkGroup: saved_spend=totals.saved_spend, savings_estimated_turns=totals.savings_estimated_turns, savings_estimated_actual_spend=totals.savings_estimated_actual_spend, + savings_estimated_classifier_cost=totals.savings_estimated_classifier_cost, classifier_cost=totals.classifier_cost, baseline_spend=totals.baseline_spend, saved_pct=totals.saved_pct, @@ -772,7 +783,13 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow: saved_spend=sum(row.saved_spend for row in rows), savings_estimated_turns=sum(row.savings_estimated_turns for row in rows), savings_estimated_actual_spend=sum(row.savings_estimated_actual_spend for row in rows), + savings_estimated_classifier_cost=( + sum(row.savings_estimated_classifier_cost or 0.0 for row in rows) + if all(row.savings_estimated_classifier_cost is not None for row in rows) + else None + ), savings_estimated_saved_spend=sum(row.savings_estimated_saved_spend for row in rows), + savings_comparison_complete=all(row.savings_comparison_complete for row in rows), classifier_cost=sum(row.classifier_cost for row in rows), classifier_cost_recorded_turns=sum(row.classifier_cost_recorded_turns for row in rows), session_seconds=sum(row.session_seconds for row in rows), @@ -847,8 +864,8 @@ async def get_auto_router_benchmarks( Benchmarks for the auto-router dashboard: session shape, savings against the configured baseline, and prompt-caching behaviour bucketed by what the router did. - Reads session rollups folded once per request at spend-write time, so this endpoint - never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that + Reads session rollups folded once per request at spend-write time, with bounded + retained-log recovery for historical comparisons. A user filter selects only turns attributed to that internal user when written; older key-only history remains outside user views. A session is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is @@ -882,7 +899,44 @@ async def get_auto_router_benchmarks( api_key, user_id, ) - rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ()) + recorded_rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ()) + comparisons: Final = ( + await historical_session_comparisons( + prisma_client, + start_day.isoformat(), + (end_day + timedelta(days=1)).isoformat(), + api_key, + user_id, + ) + if any(row.savings_estimated_turns < row.turns for row in recorded_rows) + else MappingProxyType({}) + ) + covered_rows: Final = tuple( + row.model_copy( + update={ + **comparison.coverage_fields(row.saved_spend, row.turns), + "savings_estimated_classifier_cost": comparison.classifier_cost, + "savings_comparison_complete": comparison.complete and comparison.turns == row.turns, + } + ) + if (comparison := comparisons.get((row.router_name, row.router_type))) + else row.model_copy(update={"savings_comparison_complete": row.savings_estimated_turns == row.turns}) + for row in recorded_rows + ) + rows: Final = tuple( + row.model_copy( + update={ + "savings_comparison_complete": row.savings_comparison_complete + and isclose( + row.saved_spend, + row.savings_estimated_saved_spend, + rel_tol=1e-9, + abs_tol=1e-9, + ), + } + ) + for row in covered_rows + ) groups: Final = ( *(_benchmark_group(row) for row in rows), *_idle_router_groups(llm_router, frozenset((row.router_name, row.router_type) for row in rows)), @@ -920,15 +974,43 @@ async def get_auto_router_session( if prisma_client is None: raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value) - row: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key( + recorded: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key( user_api_key_dict.api_key, bounded_session_id(session_id) ) - if row is None: + if recorded is None: raise HTTPException( status_code=404, detail=f"No auto-routed turns recorded for session {session_id!r} under this key" ) - saved_spend, baseline_spend = _savings_cohort( - row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend + comparisons: Final = ( + await historical_session_comparisons( + prisma_client, + recorded.first_turn_at.isoformat(), + (recorded.last_turn_at + timedelta(microseconds=1)).isoformat(), + user_api_key_dict.api_key, + None, + bounded_session_id(session_id), + ) + if recorded.savings_estimated_turns < recorded.turns + else MappingProxyType({}) + ) + comparison: Final = comparisons.get((recorded.router_name, recorded.router_type)) + row: Final = ( + recorded.model_copy(update=comparison.coverage_fields(recorded.saved_spend, recorded.turns)) + if comparison + else recorded + ) + saved_spend, compared_baseline = _savings_cohort( + row.turns, + row.savings_estimated_turns, + row.savings_estimated_actual_spend, + row.savings_estimated_saved_spend, + row.saved_spend, + ) + baseline_spend: Final = ( + compared_baseline + if row.savings_estimated_turns == row.turns + or (comparison and comparison.complete and comparison.turns == row.turns) + else None ) return AutoRouterSessionResponse( session_id=session_id, @@ -943,7 +1025,7 @@ async def get_auto_router_session( baseline_spend=baseline_spend if row.savings_estimated_turns == row.turns else None, savings_estimated_baseline_spend=baseline_spend, baseline_model=row.baseline_model, - baseline_models=row.savings_estimated_baseline_models, + baseline_models=row.baseline_models, ) diff --git a/litellm/types/management_endpoints/auto_router_endpoints.py b/litellm/types/management_endpoints/auto_router_endpoints.py index e191470ec6e..75a80beac5c 100644 --- a/litellm/types/management_endpoints/auto_router_endpoints.py +++ b/litellm/types/management_endpoints/auto_router_endpoints.py @@ -224,19 +224,24 @@ class AutoRouterBenchmarkTotals(BaseModel): "subtotal recording, and zero for an empty window" ) savings_estimated_turns: int = Field( - description="Turns covered by the current savings estimator; legacy estimates are excluded" + description="Requests with a matching savings comparison, including historical recorded estimates" ) savings_estimated_actual_spend: float = Field( description="Actual spend, including classifier cost, for covered turns only" ) + savings_estimated_classifier_cost: float | None = Field( + default=None, + description="Classifier cost included in the matching historical and newer savings comparison; " + "null when classification costs for those requests are unavailable", + ) saved_spend: float | None = Field( - description="Signed savings for covered turns only; null when traffic has no current estimates" + description="Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates" ) baseline_spend: float | None = Field(description="Estimated single-model cost for covered turns only") - saved_pct: float | None = Field(description="Covered savings over covered baseline spend, as a percentage") - saved_per_session: float | None = Field( - description="Average session savings; unavailable unless every turn is covered" + saved_pct: float | None = Field( + description="Total recorded savings over the matching historical and current baseline; null when costs are unavailable" ) + saved_per_session: float | None = Field(description="Recorded savings per session, including historical estimates") cache: AutoRouterCacheStats @@ -268,12 +273,14 @@ class AutoRouterSessionResponse(BaseModel): last_model: str = Field(description="The deployment model the most recent turn was routed to") spend: float = Field(description="What the session's routed traffic actually cost, classifier calls included") savings_estimated_turns: int = Field( - description="Turns covered by the current savings estimator; legacy estimates are excluded" + description="Requests with a matching savings comparison, including historical recorded estimates" ) savings_estimated_actual_spend: float = Field( description="Actual spend, including classifier cost, for covered turns only" ) - saved_spend: float | None = Field(description="Estimated savings for covered turns only, net of classifier cost") + saved_spend: float | None = Field( + description="Recorded historical savings plus newer estimates, net of classifier cost" + ) baseline_spend: float | None = Field( description="Estimated single-model cost; unavailable unless every turn is covered" ) @@ -281,14 +288,14 @@ class AutoRouterSessionResponse(BaseModel): description="Estimated single-model cost for covered turns only" ) baseline_model: str | None = Field( - description="The savings baseline most covered turns were priced against, recorded turn by " + description="The savings baseline recorded by most session turns, including historical turns, recorded turn by " "turn, so it still names the counterfactual after the router is reconfigured or removed. None when no " "turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, " "which derive no baseline and so report no savings" ) baseline_models: Mapping[str, int] = Field( - description="Covered turns priced against each baseline model; more than one entry means the router's " - "baseline changed mid-session and baseline_spend mixes both" + description="Session turns recording each baseline model; more than one entry means the router's " + "baseline changed mid-session; these counts do not imply savings coverage" ) diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py index 77549b527d8..9ef42f5dc7a 100644 --- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py +++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py @@ -6,6 +6,7 @@ tests/test_litellm/proxy/db/test_autorouter_session_rollup.py. """ import asyncio +import json import time import uuid from datetime import datetime, timedelta, timezone @@ -24,6 +25,10 @@ from litellm.proxy.db.autorouter_session_rollup import ( flush_autorouter_turn_transactions, ) from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import SpendLogCleanup +from litellm.proxy.db.autorouter_savings_comparison import ( + HISTORICAL_SESSION_COMPARISONS_SQL, + SessionSavingsComparison, +) pytestmark = pytest.mark.asyncio(loop_scope="session") @@ -91,6 +96,66 @@ async def _row(db, key: str, session_id: str = "s1", router: str = "auto-1") -> return rows[0] +@pytest.mark.parametrize("historical_saved, damaged, user_id, split_sessions, current_classifier", [ + (29.5, None, None, False, 0.2), (29.5, None, "owner", False, 0.2), (0.0, None, None, False, 0.2), + (-3.0, None, None, False, 0.2), (29.5, "missing", None, False, 0.2), (29.5, "cost", None, False, 0.2), + (0.0, "missing", None, False, 0.2), (29.5, None, None, True, 0.2), (29.5, None, None, False, 0.0), +]) +async def test_historical_and_new_savings_compare_matching_costs_and_exclude_unknown_requests( + db: Prisma, historical_saved: float, damaged: str | None, user_id: str | None, split_sessions: bool, + current_classifier: float, +) -> None: + async with db.tx() as tx: + for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession", "LiteLLM_SpendLogs"): + await tx.execute_raw(f'CREATE TEMP TABLE "{table}" (LIKE public."{table}" INCLUDING ALL) ON COMMIT DROP') + for name, spend, saved, classifier, estimated in ( + ("historical", 9.0, historical_saved, 0.1, False), + ("current", 1.0, 0.5, current_classifier, True), + ("unknown", 99.0, 0.0, 3.0, False), + ): + session_id: Final = "s2" if split_sessions and name == "current" else "s1" + await _turn(tx, "key", "model", T0, spend=spend, saved=saved, classifier_cost=classifier, + estimated=estimated, session_id=session_id) + metadata: Final = { + "routing_decision": {"router_model_name": "auto-1", **({"classifier_cost": classifier} if classifier else {})}, + "autorouter_savings": saved if name != "unknown" else None, + **({"autorouter_savings_estimate": { + "version": 3, "status": "estimated" if estimated else "unknown", + }} if name != "historical" else {}), + } + await tx.execute_raw('''INSERT INTO "LiteLLM_SpendLogs" + (request_id,api_key,session_id,model,"user","startTime","endTime",call_type, + spend,prompt_tokens,completion_tokens,status,metadata) + VALUES ($1,'key',$5,'model','owner',$2::timestamp,$2::timestamp,'acompletion', + $3::float8,100,0,'success',$4::jsonb) + ''', name, T0.isoformat(), spend - classifier, json.dumps(metadata), session_id) + await tx.execute_raw('''INSERT INTO "LiteLLM_AutoRouterUserSession" + (user_id,api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model, + turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend, + savings_estimated_saved_spend) + SELECT 'owner',api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model, + turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend, + savings_estimated_saved_spend FROM "LiteLLM_AutoRouterSession" + ''') + if damaged == "missing": + await tx.execute_raw('DELETE FROM "LiteLLM_SpendLogs" WHERE request_id = \'historical\'') + elif damaged == "cost": + await tx.execute_raw('UPDATE "LiteLLM_SpendLogs" SET spend = 1 WHERE request_id = \'historical\'') + rows: Final = await tx.query_raw( + HISTORICAL_SESSION_COMPARISONS_SQL, "2026-08-01", "2026-08-02", "key", user_id, None, + ) + comparison: Final = SessionSavingsComparison.model_validate(rows[0]) + assert comparison.saved_spend == historical_saved + 0.5 + assert comparison.complete is (damaged is None) + assert comparison.classifier_cost == (pytest.approx(0.1 + current_classifier) if damaged is None else None) + assert comparison.coverage_fields(historical_saved + 0.5, 4) == {} + assert comparison.coverage_fields(historical_saved + 0.5, 3) == ({ + "savings_estimated_turns": 2, + "savings_estimated_actual_spend": 10.0, + "savings_estimated_saved_spend": historical_saved + 0.5, + } if damaged is None else {}) + + async def test_every_turn_lands_in_exactly_one_bucket(db): key = f"k-{uuid.uuid4()}" await _turn(db, key, "A", T0, ttl=300) diff --git a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py index ff3d19e8637..385b2b1cc5b 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py @@ -676,6 +676,7 @@ class TestAutoRouterBenchmarks: saved_spend=30.0, savings_estimated_turns=40, savings_estimated_actual_spend=10.0, + savings_estimated_classifier_cost=0.4, savings_estimated_saved_spend=30.0, classifier_cost=0.4, classifier_cost_recorded_turns=40, @@ -701,6 +702,7 @@ class TestAutoRouterBenchmarks: assert totals.avg_tokens_per_session == 1000.0 assert totals.baseline_spend == 40.0 assert totals.saved_pct == 75.0 + assert totals.savings_estimated_classifier_cost == 0.4 assert totals.saved_per_session == 7.5 assert totals.cache.coverage_pct == 95.0 assert totals.cache.hit_rate_pct == pytest.approx(73.7) @@ -720,7 +722,7 @@ class TestAutoRouterBenchmarks: assert totals.classifier_cost == 0.4 @pytest.mark.parametrize("estimated_turns", [0, 4]) - def test_savings_compare_only_the_current_estimated_cohort(self, estimated_turns: int) -> None: + def test_recorded_savings_survive_when_historical_comparison_costs_are_missing(self, estimated_turns: int) -> None: from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals row: Final = self.ROW.model_copy( @@ -733,10 +735,11 @@ class TestAutoRouterBenchmarks: totals: Final = _benchmark_totals(row) assert totals.spend == 10.0 assert totals.savings_estimated_turns == estimated_turns - assert totals.saved_spend == (-0.5 if estimated_turns else None) - assert totals.baseline_spend == (1.5 if estimated_turns else None) - assert totals.saved_pct == (pytest.approx(-33.3) if estimated_turns else None) - assert totals.saved_per_session is None + assert totals.saved_spend == 30.0 + assert totals.baseline_spend is None + assert totals.savings_estimated_classifier_cost is None + assert totals.saved_pct is None + assert totals.saved_per_session == 7.5 def test_an_empty_window_folds_to_zeros(self): from litellm.proxy.management_endpoints.auto_router_endpoints import ( @@ -765,6 +768,7 @@ class TestAutoRouterBenchmarks: "spend": 0.0, "savings_estimated_turns": 10, "savings_estimated_actual_spend": 0.0, + "savings_estimated_classifier_cost": 0.0, } ) summed = _summed_agg_row([self.ROW, other]) @@ -773,6 +777,9 @@ class TestAutoRouterBenchmarks: assert summed.turns == 50 assert totals.avg_turns_per_session == 10.0 assert totals.spend == 10.0 + assert totals.savings_estimated_classifier_cost == 0.4 + unknown_cost = other.model_copy(update={"savings_estimated_classifier_cost": None}) + assert _benchmark_totals(_summed_agg_row([self.ROW, unknown_cost])).savings_estimated_classifier_cost is None def test_tier_names_stay_scoped_to_the_router_type_that_recorded_them(self): quality = self.ROW.model_copy( @@ -1128,13 +1135,13 @@ class TestAutoRouterSession: "turns": turns, "last_model": "anthropic/claude-sonnet-5", "spend": spend, - "saved_spend": (0.24 if turns == 3 else -0.04) if estimated else None, + "saved_spend": 0.24, "savings_estimated_turns": 3 if estimated else 0, "savings_estimated_actual_spend": 0.14 if estimated else 0.0, "baseline_spend": pytest.approx(0.38) if turns == 3 else None, - "savings_estimated_baseline_spend": pytest.approx(0.38 if turns == 3 else 0.1) if estimated else None, - "baseline_model": "anthropic/claude-opus-5" if estimated else None, - "baseline_models": {"anthropic/claude-opus-5": 3} if estimated else {}, + "savings_estimated_baseline_spend": pytest.approx(0.38) if turns == 3 else None, + "baseline_model": "anthropic/claude-opus-5", + "baseline_models": {"anthropic/claude-opus-5": 3}, } @pytest.mark.asyncio @@ -1168,11 +1175,10 @@ class TestAutoRouterSession: assert response.router_name == "new-auto" @pytest.mark.asyncio - async def test_a_reconfigured_router_keeps_the_label_the_money_was_priced_against( - self, monkeypatch: pytest.MonkeyPatch + @pytest.mark.parametrize("mixed", [False, True]) + async def test_session_preserves_historical_baseline_labels( + self, monkeypatch: pytest.MonkeyPatch, mixed: bool ): - # The proxy's router now prices against a different baseline, but the row's money was priced - # against opus for two of three turns, and the label says so; the full split is on the response. from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session priced = {"anthropic/claude-opus-5": 2, "anthropic/claude-sonnet-5": 1} @@ -1183,14 +1189,15 @@ class TestAutoRouterSession: **self.ROW, "api_key": ADMIN.api_key, "session_id": "s", - "baseline_models": {"old-baseline": 100}, + "baseline_models": {"old-baseline": 100, **({"unknown-baseline": 200} if mixed else {})}, + "savings_estimated_turns": 1, "savings_estimated_baseline_models": priced, } ], ) response = await get_auto_router_session(user_api_key_dict=ADMIN, session_id="s") - assert response.baseline_model == "anthropic/claude-opus-5" - assert response.baseline_models == priced + assert response.baseline_model == (None if mixed else "old-baseline") + assert response.baseline_models == {"old-baseline": 100, **({"unknown-baseline": 200} if mixed else {})} @pytest.mark.asyncio async def test_an_oversized_client_session_id_is_bounded_like_the_writer_bounded_it( diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx index a144630cdd0..5e8533c8b82 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx @@ -158,35 +158,49 @@ describe("AutoRouterBenchmarksTab", () => { }); it.each([ - { estimatedTurns: 0, saved: null, pct: null }, - { estimatedTurns: 10, saved: -0.5, pct: -33.3 }, - { estimatedTurns: 10, saved: 0, pct: 0 }, - ])("preserves costs for $estimatedTurns estimated turns with savings $saved", ({ estimatedTurns, saved, pct }) => { - const cohort = { + { estimatedTurns: 0, actual: 0, saved: null, pct: null }, + { estimatedTurns: 0, actual: 0, saved: 30, pct: null }, + { estimatedTurns: 10, actual: 2, saved: -0.5, pct: -33.3 }, + { estimatedTurns: 10, actual: 2, saved: 0, pct: 0 }, + { estimatedTurns: 40, actual: 10, saved: 30, pct: 75 }, + ])("compares matching old and new requests with savings $saved", ({ estimatedTurns, actual, saved, pct }) => { + const comparison = { + spend: actual + 99, savings_estimated_turns: estimatedTurns, - savings_estimated_actual_spend: estimatedTurns ? 2 : 0, + savings_estimated_actual_spend: actual, + savings_estimated_classifier_cost: 0.1, saved_spend: saved, - baseline_spend: estimatedTurns ? 2 + (saved ?? 0) : null, + baseline_spend: estimatedTurns ? actual + (saved ?? 0) : null, saved_pct: pct, saved_per_session: null, }; - const partial = totals(cohort); - mockHook({ data: response([], partial) }); + mockHook({ + data: response([], totals(comparison)), + }); renderTab(); - expect(screen.getByText("Estimated savings on covered turns")).toBeInTheDocument(); - expect(screen.getByText(`${estimatedTurns} of 3,073 turns estimated`)).toBeInTheDocument(); - expect(screen.getByText("$359.86")).toBeInTheDocument(); - expect(screen.getByText("Actual spend on covered turns")).toBeInTheDocument(); - expect(screen.getByText("Estimated baseline spend on covered turns")).toBeInTheDocument(); - expect(screen.getAllByText("Unavailable")).toHaveLength(estimatedTurns ? 1 : 3); - if (saved === 0) { - expect(screen.getByText("0%")).toBeInTheDocument(); - expect(screen.getAllByText("$2.00")).toHaveLength(2); - } else if (estimatedTurns) { - expect(screen.getByText("-$0.5000")).toBeInTheDocument(); - expect(screen.getByText("+33%")).toBeInTheDocument(); - } else { - expect(screen.queryByText("+0%")).not.toBeInTheDocument(); + expect(screen.getByText("Total estimated savings")).toBeInTheDocument(); + expect(screen.getAllByRole("definition").map((row) => row.textContent)).toEqual( + estimatedTurns + ? [ + `$${actual.toFixed(2)}`, + `$${(actual - 0.1).toFixed(2)}`, + "$0.1000", + `$${(actual + (saved ?? 0)).toFixed(2)}`, + ] + : ["Unavailable", "Unavailable", "Unavailable", "Unavailable"], + ); + expect(screen.queryByText("Actual spend on covered turns")).not.toBeInTheDocument(); + expect(screen.getByLabelText("question-circle")).toBeInTheDocument(); + if (estimatedTurns) { + expect(screen.getByText(`Savings based on ${estimatedTurns} of 3,073 requests`)).toBeInTheDocument(); + const sign = pct && pct > 0 ? "-" : "+"; + const badge = pct === 0 ? "0%" : `${sign}${Math.abs(pct ?? 0).toFixed(0)}%`; + expect(screen.getByText(badge)).toBeInTheDocument(); + } else if (saved != null) { + expect(screen.getByText("$30.00")).toBeInTheDocument(); + expect( + screen.getByText("Historical savings are included. Matching cost details are unavailable."), + ).toBeInTheDocument(); } }); @@ -216,7 +230,7 @@ describe("AutoRouterBenchmarksTab", () => { expect(screen.getByText("-86%")).toBeInTheDocument(); expect(screen.getByText("Actual auto-router spend")).toBeInTheDocument(); expect(screen.getByText("$359.86")).toBeInTheDocument(); - expect(screen.getByText("Estimated spend at highest-tier model")).toBeInTheDocument(); + expect(screen.getByText("Estimated baseline spend")).toBeInTheDocument(); expect(screen.getByText("$2,534.45")).toBeInTheDocument(); expect(screen.getByText("32.7")).toBeInTheDocument(); expect(screen.getByText("2.1h")).toBeInTheDocument(); @@ -242,17 +256,20 @@ describe("AutoRouterBenchmarksTab", () => { expect(screen.getAllByText("$10,126.28").length).toBeGreaterThan(0); }); - it.each([null, undefined])("keeps totals when the classification breakdown is %s", (classifier_cost) => { - const stats = totals({ classifier_cost }); - mockHook({ data: response([group(stats)], stats) }); - renderTab(); + it.each([null, undefined])( + "keeps eligible totals when the classification breakdown is %s", + (savings_estimated_classifier_cost) => { + const stats = totals({ savings_estimated_turns: 30, savings_estimated_classifier_cost }); + mockHook({ data: response([group(stats)], stats) }); + renderTab(); - expect(screen.getAllByText("Unavailable")).toHaveLength(2); - expect(screen.queryByText(/\/ 1K turns/)).not.toBeInTheDocument(); - expect(screen.getByText("$359.86")).toBeInTheDocument(); - expect(screen.getByText("$2,174.59")).toBeInTheDocument(); - expect(screen.getByText(/some usage predates classification-cost tracking/)).toBeInTheDocument(); - }); + expect(screen.getAllByText("Unavailable")).toHaveLength(2); + expect(screen.queryByText(/\/ 1K turns/)).not.toBeInTheDocument(); + expect(screen.getByText("$359.86")).toBeInTheDocument(); + expect(screen.getByText("$2,174.59")).toBeInTheDocument(); + expect(screen.getByText(/some usage predates classification-cost tracking/)).toBeInTheDocument(); + }, + ); it("pairs the savings with the session count it was earned over, in its own tile", () => { mockHook({ data: response([group(), group({ router_name: "gpt-auto" })]) }); @@ -275,7 +292,7 @@ describe("AutoRouterBenchmarksTab", () => { "Actual auto-router spend", "LLM spend", "Classification cost($2.00 / 1K turns)", - "Estimated spend at highest-tier model", + "Estimated baseline spend", ]); expect(values).toEqual(["$359.86", "$353.71", "$6.15", "$2,534.45"]); }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx index 063598bd46e..f532c2e4650 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx @@ -11,7 +11,7 @@ import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@ import { Separator } from "@/components/ui/separator"; import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table"; import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs"; -import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip"; +import { SimpleTooltip, Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip"; import { ApiError } from "@/lib/http/client"; import { formatNumberWithCommas } from "@/utils/dataUtils"; @@ -52,15 +52,17 @@ const Metric: React.FC<{ label: string; value: string; hint?: string }> = ({ lab ); -const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued?: boolean }> = ({ +const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued?: boolean; tooltip?: string }> = ({ label, value, hint, subdued, + tooltip, }) => (
- {completeCoverage ? "Total estimated savings" : "Estimated savings on covered turns"} + Total estimated savings
@@ -91,53 +96,57 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => { variant="secondary" className={`h-6 px-2.5 text-sm ${cheaper ? "bg-success/10 text-success" : "bg-destructive/10 text-destructive"}`} > - {stats.saved_spend !== 0 && (cheaper ? "-" : "+")} + {stats.saved_pct !== 0 && (cheaper ? "-" : "+")} {Math.abs(stats.saved_pct).toFixed(0)}% )}
- {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()} turns estimated -
- {!completeCoverage && ( + {stats.baseline_spend != null && !completeCoverage && (- Turns without a current estimate are excluded, including older estimates. + Savings based on {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()}{" "} + requests +
+ )} + {stats.saved_spend != null && stats.baseline_spend == null && ( ++ Historical savings are included. Matching cost details are unavailable.
)}Breakdown unavailable because some usage predates classification-cost tracking.
)}- Compares covered turns with the estimated cost of using the router's highest-tier baseline model. Estimates - use registered requests since tracking began, matching cache prefixes and expiry, and the actual response - length. Total actual spend includes every turn; savings and baseline spend include only turns with a current - estimate, including turns with zero savings. Savings are net of recorded LLM classification cost. Classification - cost per 1K turns is averaged over all auto-router turns, including those that skip classification. The range - counts whole sessions that overlap it, so totals can differ from savings views that group usage by UTC day. + Savings, actual spend, and baseline compare the same historical and newer requests with recorded estimates, + including zero or negative savings. Requests without estimates are excluded. Savings are net of recorded LLM + classification cost. If historical cost details are unavailable, recorded savings remain visible without a + baseline or percentage. The range counts whole sessions that overlap it, so totals can differ from savings views + that group usage by UTC day.