diff --git a/litellm/proxy/db/autorouter_savings_comparison.py b/litellm/proxy/db/autorouter_savings_comparison.py deleted file mode 100644 index 041496d63f2..00000000000 --- a/litellm/proxy/db/autorouter_savings_comparison.py +++ /dev/null @@ -1,147 +0,0 @@ -from collections.abc import Mapping -from contextlib import AbstractAsyncContextManager -from datetime import timedelta -from math import isclose -from types import MappingProxyType -from typing import TYPE_CHECKING, Final, Protocol, cast - -from pydantic import BaseModel, ConfigDict, TypeAdapter - -from litellm._logging import verbose_proxy_logger -from litellm.constants import MAX_SPENDLOG_ROWS_TO_QUERY -from litellm.proxy.db.autorouter_session_rollup import AUTOROUTER_SESSION_WINDOW_SQL -from litellm.proxy.db.create_views import SupportsRawQueries - -if TYPE_CHECKING: - from litellm.proxy.utils import PrismaClient - - -class SessionSavingsComparison(BaseModel): - model_config = ConfigDict(frozen=True, allow_inf_nan=False) - - router_name: str - router_type: str - turns: int - estimated_turns: int - actual_spend: float - classifier_cost: float | None - saved_spend: float - complete: bool - - def coverage_fields(self, recorded_savings: float, recorded_turns: int) -> Mapping[str, float | int]: - if self.turns != recorded_turns or not self.complete: - return MappingProxyType({}) - if not isclose(self.saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9): - return MappingProxyType({}) - return MappingProxyType( - { - "savings_estimated_turns": self.estimated_turns, - "savings_estimated_actual_spend": self.actual_spend, - "savings_estimated_saved_spend": self.saved_spend, - } - ) - - -class _ReadTransactions(Protocol): - def tx(self, *, timeout: timedelta, max_wait: timedelta) -> AbstractAsyncContextManager[SupportsRawQueries]: ... - - -_COMPARISONS: Final = TypeAdapter(tuple[SessionSavingsComparison, ...]) - - -async def historical_session_comparisons( - prisma_client: "PrismaClient", - start_date: str, - end_date: str, - api_key: str | None, - user_id: str | None, - session_id: str | None = None, -) -> Mapping[tuple[str, str], SessionSavingsComparison]: - try: - reader: Final = cast(_ReadTransactions, prisma_client.read_db) # cast-ok: untyped Prisma transaction delegate - async with reader.tx(timeout=timedelta(seconds=3), max_wait=timedelta(seconds=1)) as transaction: - await transaction.execute_raw("SET TRANSACTION READ ONLY") - await transaction.execute_raw("SET LOCAL statement_timeout = 2000") - rows: Final = await transaction.query_raw( - HISTORICAL_SESSION_COMPARISONS_SQL, - start_date, - end_date, - api_key, - user_id, - session_id, - ) - comparisons: Final = _COMPARISONS.validate_python(rows or ()) - return MappingProxyType({(row.router_name, row.router_type): row for row in comparisons}) - except Exception: # noqa: BLE001 # missing retained logs must not discard recorded dollar savings - verbose_proxy_logger.warning("Historical auto-router cost comparison unavailable; preserving recorded savings") - return MappingProxyType({}) - - -HISTORICAL_SESSION_COMPARISONS_SQL: Final = f""" -WITH {AUTOROUTER_SESSION_WINDOW_SQL}, scoped AS MATERIALIZED ( - SELECT * FROM windowed WHERE $5::text IS NULL OR session_id = $5::text -), limited_logs AS MATERIALIZED ( - SELECT session.api_key, session.session_id, session.router_name, session.router_type, session.comparison_user_id, - session.classifier_cost_recorded_turns = session.turns AS classifier_cost_tracked, - logs.spend, logs.prompt_tokens + logs.completion_tokens AS tokens, - logs.metadata::jsonb -> 'routing_decision' AS decision, - logs.metadata::jsonb -> 'autorouter_savings' AS savings, - logs.metadata::jsonb -> 'autorouter_savings_estimate' AS estimate - FROM scoped AS session JOIN "LiteLLM_SpendLogs" AS logs - ON logs.api_key = session.api_key - AND CASE WHEN char_length(logs.session_id) > 256 - THEN 'sha256:' || encode(sha256(convert_to(logs.session_id, 'UTF8')), 'hex') - ELSE logs.session_id END = session.session_id - AND (session.comparison_user_id IS NULL OR logs."user" = session.comparison_user_id) - AND logs."startTime" BETWEEN session.first_turn_at AND session.last_turn_at - AND COALESCE(logs.metadata::jsonb #>> '{{routing_decision,router_model_name}}', logs.model_group) - = session.router_name - WHERE session.savings_estimated_turns < session.turns - AND logs.status = 'success' AND COALESCE(logs.metadata::jsonb ->> 'internal_call_origin', '') = '' - LIMIT {MAX_SPENDLOG_ROWS_TO_QUERY + 1} -), facts AS ( - SELECT *, - CASE WHEN jsonb_typeof(decision -> 'classifier_cost') = 'number' - THEN (decision ->> 'classifier_cost')::float8 - WHEN classifier_cost_tracked THEN 0 END AS classifier, - CASE WHEN jsonb_typeof(savings) = 'number' AND ( - estimate IS NULL OR estimate = 'null'::jsonb OR ( - jsonb_typeof(estimate -> 'version') = 'number' AND estimate ->> 'version' IN ('1', '2', '3') - AND estimate ->> 'status' = 'estimated' - ) - ) THEN savings::text::float8 END AS saved - FROM limited_logs -), compared AS ( - SELECT api_key, session_id, router_name, router_type, comparison_user_id, - COUNT(*) AS turns, SUM(spend + COALESCE(classifier, 0)) AS spend, SUM(tokens) AS total_tokens, - COUNT(saved) AS estimated_turns, - COALESCE(SUM(spend + COALESCE(classifier, 0)) FILTER (WHERE saved IS NOT NULL), 0)::float8 AS actual_spend, - CASE WHEN COUNT(saved) = COUNT(classifier) FILTER (WHERE saved IS NOT NULL) - THEN COALESCE(SUM(classifier) FILTER (WHERE saved IS NOT NULL), 0)::float8 - END AS estimated_classifier_cost, - COALESCE(SUM(saved), 0)::float8 AS saved_spend - FROM facts GROUP BY 1, 2, 3, 4, 5 -), reconciled AS ( - SELECT session.*, logs.estimated_turns, logs.actual_spend, logs.estimated_classifier_cost, - COALESCE((SELECT COUNT(*) FROM limited_logs) <= {MAX_SPENDLOG_ROWS_TO_QUERY} - AND logs.turns = session.turns AND logs.total_tokens = session.total_tokens - AND ABS(logs.spend - session.spend) <= GREATEST(1e-9, ABS(session.spend) * 1e-9) - AND ABS(logs.saved_spend - session.saved_spend) <= GREATEST(1e-9, ABS(session.saved_spend) * 1e-9), FALSE - ) AS recovered - FROM scoped AS session LEFT JOIN compared AS logs - ON logs.api_key = session.api_key AND logs.session_id = session.session_id - AND logs.router_name = session.router_name AND logs.router_type = session.router_type - AND logs.comparison_user_id IS NOT DISTINCT FROM session.comparison_user_id -) -SELECT router_name, router_type, - SUM(turns)::bigint AS turns, - SUM(CASE WHEN recovered THEN estimated_turns ELSE savings_estimated_turns END)::bigint AS estimated_turns, - SUM(CASE WHEN recovered THEN actual_spend ELSE savings_estimated_actual_spend END)::float8 AS actual_spend, - CASE WHEN BOOL_AND(CASE WHEN recovered THEN estimated_classifier_cost IS NOT NULL - ELSE savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns END) - THEN SUM(CASE WHEN recovered THEN estimated_classifier_cost ELSE classifier_cost END)::float8 - END AS classifier_cost, - SUM(saved_spend)::float8 AS saved_spend, - BOOL_AND(recovered OR savings_estimated_turns = turns) AS complete -FROM reconciled GROUP BY router_name, router_type -""" diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index 3bd02fe8738..8b73b8177f4 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -8,7 +8,6 @@ POST /auto_router/validate_complexity_router_config - Dry-run the complexity-rou from collections.abc import Mapping, Sequence from datetime import datetime, timedelta, timezone from itertools import chain, groupby -from math import isclose from types import MappingProxyType from typing import TYPE_CHECKING, Annotated, Final, Protocol from uuid import uuid4 @@ -32,7 +31,6 @@ from litellm.proxy.auth.auth_checks import ( can_key_call_resolved_model, ) from litellm.proxy.auth.user_api_key_auth import user_api_key_auth -from litellm.proxy.db.autorouter_savings_comparison import historical_session_comparisons from litellm.proxy.db.autorouter_session_rollup import ( AUTOROUTER_BENCHMARKS_SQL, bounded_session_id, @@ -643,7 +641,6 @@ class _SessionAggRow(BaseModel): savings_estimated_actual_spend: float = 0.0 savings_estimated_classifier_cost: float | None = None savings_estimated_saved_spend: float = 0.0 - savings_comparison_complete: bool = True classifier_cost: float classifier_cost_recorded_turns: int session_seconds: float @@ -671,25 +668,35 @@ def _cache_bucket(turns: int, hits: int) -> AutoRouterCacheBucket: def _savings_cohort( - turns: int, estimated_turns: int, actual_spend: float, saved_spend: float, recorded_savings: float + turns: int, estimated_turns: int, spend: float, saved_spend: float ) -> tuple[float | None, float | None]: - if turns > 0 and estimated_turns == 0 and recorded_savings == 0: + if turns > 0 and estimated_turns == 0 and saved_spend == 0: return None, None - if not isclose(saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9): - return recorded_savings, None - return recorded_savings, actual_spend + recorded_savings + return saved_spend, spend + saved_spend + + +def _compared_row(row: _SessionAggRow) -> _SessionAggRow: + _, baseline_spend = _savings_cohort(row.turns, row.savings_estimated_turns, row.spend, row.saved_spend) + compared: Final = row.router_type == "complexity" and baseline_spend is not None + return row.model_copy( + update={ + "savings_estimated_turns": row.turns if compared else 0, + "savings_estimated_actual_spend": row.spend if compared else 0.0, + "savings_estimated_classifier_cost": ( + row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None + ) + if compared + else 0.0, + "savings_estimated_saved_spend": row.saved_spend if compared else 0.0, + } + ) def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: return_misses: Final = row.return_turns - row.return_hits - saved_spend, compared_baseline = _savings_cohort( - row.turns, - row.savings_estimated_turns, - row.savings_estimated_actual_spend, - row.savings_estimated_saved_spend, - row.saved_spend, + saved_spend, baseline_spend = _savings_cohort( + row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend ) - baseline_spend: Final = compared_baseline if row.savings_comparison_complete else None sessions: Final = row.sessions return AutoRouterBenchmarkTotals( sessions=sessions, @@ -700,7 +707,7 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: spend=row.spend, savings_estimated_turns=row.savings_estimated_turns, savings_estimated_actual_spend=row.savings_estimated_actual_spend, - savings_estimated_classifier_cost=row.savings_estimated_classifier_cost if baseline_spend is not None else None, + savings_estimated_classifier_cost=row.savings_estimated_classifier_cost, saved_spend=saved_spend, classifier_cost=row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None, baseline_spend=baseline_spend, @@ -777,7 +784,6 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow: else None ), savings_estimated_saved_spend=sum(row.savings_estimated_saved_spend for row in rows), - savings_comparison_complete=all(row.savings_comparison_complete for row in rows), classifier_cost=sum(row.classifier_cost for row in rows), classifier_cost_recorded_turns=sum(row.classifier_cost_recorded_turns for row in rows), session_seconds=sum(row.session_seconds for row in rows), @@ -852,8 +858,8 @@ async def get_auto_router_benchmarks( Benchmarks for the auto-router dashboard: session shape, savings against the configured baseline, and prompt-caching behaviour bucketed by what the router did. - Reads session rollups folded once per request at spend-write time, with bounded - retained-log recovery for historical comparisons. A user filter selects only turns attributed to that + Reads session rollups folded once per request at spend-write time, so this endpoint + never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that internal user when written; older key-only history remains outside user views. A session is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is @@ -887,44 +893,7 @@ async def get_auto_router_benchmarks( api_key, user_id, ) - recorded_rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ()) - comparisons: Final = ( - await historical_session_comparisons( - prisma_client, - start_day.isoformat(), - (end_day + timedelta(days=1)).isoformat(), - api_key, - user_id, - ) - if any(row.savings_estimated_turns < row.turns for row in recorded_rows) - else MappingProxyType({}) - ) - covered_rows: Final = tuple( - row.model_copy( - update={ - **comparison.coverage_fields(row.saved_spend, row.turns), - "savings_estimated_classifier_cost": comparison.classifier_cost, - "savings_comparison_complete": comparison.complete and comparison.turns == row.turns, - } - ) - if (comparison := comparisons.get((row.router_name, row.router_type))) - else row.model_copy(update={"savings_comparison_complete": row.savings_estimated_turns == row.turns}) - for row in recorded_rows - ) - rows: Final = tuple( - row.model_copy( - update={ - "savings_comparison_complete": row.savings_comparison_complete - and isclose( - row.saved_spend, - row.savings_estimated_saved_spend, - rel_tol=1e-9, - abs_tol=1e-9, - ), - } - ) - for row in covered_rows - ) + rows: Final = tuple(_compared_row(row) for row in _SESSION_AGG_ROWS.validate_python(raw_rows or ())) groups: Final = ( *(_benchmark_group(row) for row in rows), *_idle_router_groups(llm_router, frozenset((row.router_name, row.router_type) for row in rows)), @@ -962,43 +931,16 @@ async def get_auto_router_session( if prisma_client is None: raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value) - recorded: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key( + row: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key( user_api_key_dict.api_key, bounded_session_id(session_id) ) - if recorded is None: + if row is None: raise HTTPException( status_code=404, detail=f"No auto-routed turns recorded for session {session_id!r} under this key" ) - comparisons: Final = ( - await historical_session_comparisons( - prisma_client, - recorded.first_turn_at.isoformat(), - (recorded.last_turn_at + timedelta(microseconds=1)).isoformat(), - user_api_key_dict.api_key, - None, - bounded_session_id(session_id), - ) - if recorded.savings_estimated_turns < recorded.turns - else MappingProxyType({}) - ) - comparison: Final = comparisons.get((recorded.router_name, recorded.router_type)) - row: Final = ( - recorded.model_copy(update=comparison.coverage_fields(recorded.saved_spend, recorded.turns)) - if comparison - else recorded - ) - saved_spend, compared_baseline = _savings_cohort( - row.turns, - row.savings_estimated_turns, - row.savings_estimated_actual_spend, - row.savings_estimated_saved_spend, - row.saved_spend, - ) - baseline_spend: Final = ( - compared_baseline - if row.savings_estimated_turns == row.turns - or (comparison and comparison.complete and comparison.turns == row.turns) - else None + saved_spend, baseline_spend = _savings_cohort(row.turns, row.savings_estimated_turns, row.spend, row.saved_spend) + _, estimated_baseline_spend = _savings_cohort( + row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend ) return AutoRouterSessionResponse( session_id=session_id, @@ -1010,8 +952,8 @@ async def get_auto_router_session( savings_estimated_turns=row.savings_estimated_turns, savings_estimated_actual_spend=row.savings_estimated_actual_spend, saved_spend=saved_spend, - baseline_spend=baseline_spend if row.savings_estimated_turns == row.turns else None, - savings_estimated_baseline_spend=baseline_spend, + baseline_spend=baseline_spend, + savings_estimated_baseline_spend=estimated_baseline_spend, baseline_model=row.baseline_model, baseline_models=row.baseline_models, ) diff --git a/litellm/types/management_endpoints/auto_router_endpoints.py b/litellm/types/management_endpoints/auto_router_endpoints.py index 67f7ed424e7..00083e01f54 100644 --- a/litellm/types/management_endpoints/auto_router_endpoints.py +++ b/litellm/types/management_endpoints/auto_router_endpoints.py @@ -218,23 +218,24 @@ class AutoRouterBenchmarkTotals(BaseModel): "subtotal recording, and zero for an empty window" ) savings_estimated_turns: int = Field( - description="Requests with a matching savings comparison, including historical recorded estimates" + description="Requests compared against the baseline: every request on complexity routers that recorded savings" ) savings_estimated_actual_spend: float = Field( - description="Actual spend, including classifier cost, for covered turns only" + description="Actual spend, including classifier cost, for the compared requests" ) savings_estimated_classifier_cost: float | None = Field( default=None, - description="Classifier cost included in the matching historical and newer savings comparison; " + description="Classifier cost included in the compared actual spend; " "null when classification costs for those requests are unavailable", ) saved_spend: float | None = Field( description="Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates" ) - baseline_spend: float | None = Field(description="Estimated single-model cost for covered turns only") - saved_pct: float | None = Field( - description="Total recorded savings over the matching historical and current baseline; null when costs are unavailable" + baseline_spend: float | None = Field( + description="Estimated single-model cost: compared actual spend plus recorded savings; " + "null when traffic has no recorded savings" ) + saved_pct: float | None = Field(description="Recorded savings over baseline_spend, as a percentage") saved_per_session: float | None = Field(description="Recorded savings per session, including historical estimates") cache: AutoRouterCacheStats @@ -266,20 +267,16 @@ class AutoRouterSessionResponse(BaseModel): turns: int = Field(description="Auto-routed turns the rollup has recorded for this session so far") last_model: str = Field(description="The deployment model the most recent turn was routed to") spend: float = Field(description="What the session's routed traffic actually cost, classifier calls included") - savings_estimated_turns: int = Field( - description="Requests with a matching savings comparison, including historical recorded estimates" - ) + savings_estimated_turns: int = Field(description="Requests whose savings estimate recorded its baseline cost") savings_estimated_actual_spend: float = Field( - description="Actual spend, including classifier cost, for covered turns only" + description="Actual spend, including classifier cost, for requests whose estimate recorded its baseline cost" ) saved_spend: float | None = Field( description="Recorded historical savings plus newer estimates, net of classifier cost" ) - baseline_spend: float | None = Field( - description="Estimated single-model cost; unavailable unless every turn is covered" - ) + baseline_spend: float | None = Field(description="Estimated single-model cost: spend plus recorded savings") savings_estimated_baseline_spend: float | None = Field( - description="Estimated single-model cost for covered turns only" + description="Estimated single-model cost for requests whose estimate recorded its baseline cost" ) baseline_model: str | None = Field( description="The savings baseline recorded by most session turns, including historical turns, recorded turn by " diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py index effb5ca83ed..2c648f309f6 100644 --- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py +++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py @@ -6,7 +6,6 @@ tests/unit/proxy/db/test_autorouter_session_rollup.py. """ import asyncio -import json import time import uuid from datetime import datetime, timedelta, timezone @@ -25,10 +24,6 @@ from litellm.proxy.db.autorouter_session_rollup import ( flush_autorouter_turn_transactions, ) from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import SpendLogCleanup -from litellm.proxy.db.autorouter_savings_comparison import ( - HISTORICAL_SESSION_COMPARISONS_SQL, - SessionSavingsComparison, -) pytestmark = pytest.mark.asyncio(loop_scope="session") @@ -96,66 +91,6 @@ async def _row(db, key: str, session_id: str = "s1", router: str = "auto-1") -> return rows[0] -@pytest.mark.parametrize("historical_saved, damaged, user_id, split_sessions, current_classifier", [ - (29.5, None, None, False, 0.2), (29.5, None, "owner", False, 0.2), (0.0, None, None, False, 0.2), - (-3.0, None, None, False, 0.2), (29.5, "missing", None, False, 0.2), (29.5, "cost", None, False, 0.2), - (0.0, "missing", None, False, 0.2), (29.5, None, None, True, 0.2), (29.5, None, None, False, 0.0), -]) -async def test_historical_and_new_savings_compare_matching_costs_and_exclude_unknown_requests( - db: Prisma, historical_saved: float, damaged: str | None, user_id: str | None, split_sessions: bool, - current_classifier: float, -) -> None: - async with db.tx() as tx: - for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession", "LiteLLM_SpendLogs"): - await tx.execute_raw(f'CREATE TEMP TABLE "{table}" (LIKE public."{table}" INCLUDING ALL) ON COMMIT DROP') - for name, spend, saved, classifier, estimated in ( - ("historical", 9.0, historical_saved, 0.1, False), - ("current", 1.0, 0.5, current_classifier, True), - ("unknown", 99.0, 0.0, 3.0, False), - ): - session_id: Final = "s2" if split_sessions and name == "current" else "s1" - await _turn(tx, "key", "model", T0, spend=spend, saved=saved, classifier_cost=classifier, - estimated=estimated, session_id=session_id) - metadata: Final = { - "routing_decision": {"router_model_name": "auto-1", **({"classifier_cost": classifier} if classifier else {})}, - "autorouter_savings": saved if name != "unknown" else None, - **({"autorouter_savings_estimate": { - "version": 3, "status": "estimated" if estimated else "unknown", - }} if name != "historical" else {}), - } - await tx.execute_raw('''INSERT INTO "LiteLLM_SpendLogs" - (request_id,api_key,session_id,model,"user","startTime","endTime",call_type, - spend,prompt_tokens,completion_tokens,status,metadata) - VALUES ($1,'key',$5,'model','owner',$2::timestamp,$2::timestamp,'acompletion', - $3::float8,100,0,'success',$4::jsonb) - ''', name, T0.isoformat(), spend - classifier, json.dumps(metadata), session_id) - await tx.execute_raw('''INSERT INTO "LiteLLM_AutoRouterUserSession" - (user_id,api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model, - turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend, - savings_estimated_saved_spend) - SELECT 'owner',api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model, - turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend, - savings_estimated_saved_spend FROM "LiteLLM_AutoRouterSession" - ''') - if damaged == "missing": - await tx.execute_raw('DELETE FROM "LiteLLM_SpendLogs" WHERE request_id = \'historical\'') - elif damaged == "cost": - await tx.execute_raw('UPDATE "LiteLLM_SpendLogs" SET spend = 1 WHERE request_id = \'historical\'') - rows: Final = await tx.query_raw( - HISTORICAL_SESSION_COMPARISONS_SQL, "2026-08-01", "2026-08-02", "key", user_id, None, - ) - comparison: Final = SessionSavingsComparison.model_validate(rows[0]) - assert comparison.saved_spend == historical_saved + 0.5 - assert comparison.complete is (damaged is None) - assert comparison.classifier_cost == (pytest.approx(0.1 + current_classifier) if damaged is None else None) - assert comparison.coverage_fields(historical_saved + 0.5, 4) == {} - assert comparison.coverage_fields(historical_saved + 0.5, 3) == ({ - "savings_estimated_turns": 2, - "savings_estimated_actual_spend": 10.0, - "savings_estimated_saved_spend": historical_saved + 0.5, - } if damaged is None else {}) - - async def test_every_turn_lands_in_exactly_one_bucket(db): key = f"k-{uuid.uuid4()}" await _turn(db, key, "A", T0, ttl=300) diff --git a/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py index 385b2b1cc5b..2cbba9da8b3 100644 --- a/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py +++ b/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py @@ -721,26 +721,60 @@ class TestAutoRouterBenchmarks: assert totals.saved_pct == -100.0 assert totals.classifier_cost == 0.4 + @pytest.mark.asyncio @pytest.mark.parametrize("estimated_turns", [0, 4]) - def test_recorded_savings_survive_when_historical_comparison_costs_are_missing(self, estimated_turns: int) -> None: - from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals - + async def test_historical_savings_without_recorded_baselines_compare_against_all_spend( + self, estimated_turns: int, monkeypatch: pytest.MonkeyPatch + ) -> None: row: Final = self.ROW.model_copy( update={ "savings_estimated_turns": estimated_turns, "savings_estimated_actual_spend": 2.0 if estimated_turns else 0.0, + "savings_estimated_classifier_cost": None, "savings_estimated_saved_spend": -0.5 if estimated_turns else 0.0, } ) - totals: Final = _benchmark_totals(row) - assert totals.spend == 10.0 - assert totals.savings_estimated_turns == estimated_turns - assert totals.saved_spend == 30.0 - assert totals.baseline_spend is None - assert totals.savings_estimated_classifier_cost is None - assert totals.saved_pct is None + response: Final = await self._benchmarks(monkeypatch, rows=[row.model_dump()], model_list=[]) + assert response.groups[0].model_dump(exclude={"router_name", "router_type", "tier_turns"}) == ( + response.totals.model_dump() + ) + totals: Final = response.totals + assert (totals.spend, totals.saved_spend, totals.baseline_spend, totals.saved_pct) == (10.0, 30.0, 40.0, 75.0) + assert (totals.savings_estimated_turns, totals.savings_estimated_actual_spend) == (40, 10.0) + assert totals.savings_estimated_classifier_cost == 0.4 assert totals.saved_per_session == 7.5 + @pytest.mark.asyncio + @pytest.mark.parametrize("router_type, saved", [("adaptive", 0.0), ("quality", 0.0), ("quality", 2.0)]) + async def test_only_complexity_routers_enter_the_compared_totals( + self, router_type: str, saved: float, monkeypatch: pytest.MonkeyPatch + ) -> None: + adaptive: Final = self.ROW.model_copy( + update={ + "router_name": f"{router_type}-auto", + "router_type": router_type, + "turns": 10, + "spend": 3.0, + "saved_spend": saved, + "savings_estimated_turns": 0, + "savings_estimated_actual_spend": 0.0, + "savings_estimated_saved_spend": 0.0, + "classifier_cost": 0.0, + "classifier_cost_recorded_turns": 10, + } + ) + response: Final = await self._benchmarks( + monkeypatch, rows=[self.ROW.model_dump(), adaptive.model_dump()], model_list=[] + ) + unbaselined: Final = response.groups[1] + assert (unbaselined.saved_spend, unbaselined.baseline_spend, unbaselined.saved_pct) == (None, None, None) + assert (unbaselined.savings_estimated_turns, unbaselined.savings_estimated_classifier_cost) == (0, 0.0) + totals: Final = response.totals + assert (totals.turns, totals.spend) == (50, 13.0) + assert (totals.savings_estimated_turns, totals.savings_estimated_actual_spend) == (40, 10.0) + assert (totals.saved_spend, totals.baseline_spend, totals.saved_pct) == (30.0, 40.0, 75.0) + assert totals.savings_estimated_classifier_cost == 0.4 + def test_an_empty_window_folds_to_zeros(self): from litellm.proxy.management_endpoints.auto_router_endpoints import ( _benchmark_totals, @@ -1138,8 +1172,10 @@ class TestAutoRouterSession: "saved_spend": 0.24, "savings_estimated_turns": 3 if estimated else 0, "savings_estimated_actual_spend": 0.14 if estimated else 0.0, - "baseline_spend": pytest.approx(0.38) if turns == 3 else None, - "savings_estimated_baseline_spend": pytest.approx(0.38) if turns == 3 else None, + "baseline_spend": pytest.approx(spend + 0.24), + "savings_estimated_baseline_spend": ( + pytest.approx(0.38 if turns == 3 else 0.10) if estimated else None + ), "baseline_model": "anthropic/claude-opus-5", "baseline_models": {"anthropic/claude-opus-5": 3}, } diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx index 5e8533c8b82..b4b34dbfaf3 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx @@ -70,6 +70,7 @@ const totals = (overrides: Partial = {}): Totals => ({ spend: 359.86, savings_estimated_turns: overrides.turns ?? 3073, savings_estimated_actual_spend: overrides.spend ?? 359.86, + savings_estimated_classifier_cost: overrides.classifier_cost === undefined ? 6.146 : overrides.classifier_cost, classifier_cost: 6.146, saved_spend: 2174.59, baseline_spend: 2534.45, @@ -104,6 +105,7 @@ const zeroTotals: Totals = { spend: 0, savings_estimated_turns: 0, savings_estimated_actual_spend: 0, + savings_estimated_classifier_cost: 0, classifier_cost: 0, saved_spend: 0, baseline_spend: 0, @@ -159,11 +161,10 @@ describe("AutoRouterBenchmarksTab", () => { it.each([ { estimatedTurns: 0, actual: 0, saved: null, pct: null }, - { estimatedTurns: 0, actual: 0, saved: 30, pct: null }, { estimatedTurns: 10, actual: 2, saved: -0.5, pct: -33.3 }, { estimatedTurns: 10, actual: 2, saved: 0, pct: 0 }, - { estimatedTurns: 40, actual: 10, saved: 30, pct: 75 }, - ])("compares matching old and new requests with savings $saved", ({ estimatedTurns, actual, saved, pct }) => { + { estimatedTurns: 3073, actual: 10, saved: 30, pct: 75 }, + ])("compares the requests on routers that recorded savings $saved", ({ estimatedTurns, actual, saved, pct }) => { const comparison = { spend: actual + 99, savings_estimated_turns: estimatedTurns, @@ -189,18 +190,17 @@ describe("AutoRouterBenchmarksTab", () => { ] : ["Unavailable", "Unavailable", "Unavailable", "Unavailable"], ); - expect(screen.queryByText("Actual spend on covered turns")).not.toBeInTheDocument(); + expect(screen.queryByText(/Matching cost details are unavailable/)).not.toBeInTheDocument(); expect(screen.getByLabelText("question-circle")).toBeInTheDocument(); - if (estimatedTurns) { - expect(screen.getByText(`Savings based on ${estimatedTurns} of 3,073 requests`)).toBeInTheDocument(); - const sign = pct && pct > 0 ? "-" : "+"; - const badge = pct === 0 ? "0%" : `${sign}${Math.abs(pct ?? 0).toFixed(0)}%`; + const partial = estimatedTurns > 0 && estimatedTurns < 3073; + expect(screen.queryByText(/adaptive and quality routers are excluded/) != null).toBe(partial); + if (partial) { + expect(screen.getByText(/Compared on 10 of 3,073 requests/)).toBeInTheDocument(); + } + if (pct != null) { + const sign = pct > 0 ? "-" : "+"; + const badge = pct === 0 ? "0%" : `${sign}${Math.abs(pct).toFixed(0)}%`; expect(screen.getByText(badge)).toBeInTheDocument(); - } else if (saved != null) { - expect(screen.getByText("$30.00")).toBeInTheDocument(); - expect( - screen.getByText("Historical savings are included. Matching cost details are unavailable."), - ).toBeInTheDocument(); } }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx index f532c2e4650..24a97587e32 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx @@ -76,10 +76,8 @@ const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued? const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => { const stats = view.stats; const cheaper = stats.saved_pct != null && stats.saved_pct >= 0; - const completeCoverage = stats.savings_estimated_turns === stats.turns; - const coveredClassifierCost = - stats.savings_estimated_classifier_cost ?? (completeCoverage ? stats.classifier_cost : null); - const classifierCost = stats.baseline_spend == null ? null : coveredClassifierCost; + const classifierCost = stats.baseline_spend == null ? null : stats.savings_estimated_classifier_cost ?? null; + const comparedAll = stats.savings_estimated_turns === stats.turns; return (
@@ -101,15 +99,10 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => { )}
- {stats.baseline_spend != null && !completeCoverage && ( + {stats.baseline_spend != null && !comparedAll && (

- Savings based on {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()}{" "} - requests -

- )} - {stats.saved_spend != null && stats.baseline_spend == null && ( -

- Historical savings are included. Matching cost details are unavailable. + Compared on {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()} requests; + adaptive and quality routers are excluded

)} @@ -118,7 +111,7 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
= ({ isPending, error, data,

- Savings, actual spend, and baseline compare the same historical and newer requests with recorded estimates, - including zero or negative savings. Requests without estimates are excluded. Savings are net of recorded LLM - classification cost. If historical cost details are unavailable, recorded savings remain visible without a - baseline or percentage. The range counts whole sessions that overlap it, so totals can differ from savings views - that group usage by UTC day. + Actual spend covers every request on complexity routers, including LLM classification cost. Baseline is actual + spend plus recorded savings, so savings can be zero or negative. The range counts whole sessions that overlap + it, so totals can differ from savings views that group usage by UTC day.

diff --git a/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx b/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx index 807bd2f4f15..5c7b9b28789 100644 --- a/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/templates/KeyAutoRouterUsageTab.integration.test.tsx @@ -34,6 +34,7 @@ const stats = { spend: 1.25, savings_estimated_turns: 4, savings_estimated_actual_spend: 1.25, + savings_estimated_classifier_cost: 0.25, classifier_cost: 0.25, saved_spend: 8.75, baseline_spend: 10, diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index ab164a61ca1..c6bb9be41df 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -1263,8 +1263,8 @@ export interface paths { * @description Benchmarks for the auto-router dashboard: session shape, savings against the configured * baseline, and prompt-caching behaviour bucketed by what the router did. * - * Reads session rollups folded once per request at spend-write time, with bounded - * retained-log recovery for historical comparisons. A user filter selects only turns attributed to that + * Reads session rollups folded once per request at spend-write time, so this endpoint + * never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that * internal user when written; older key-only history remains outside user views. A session * is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before * end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is @@ -25464,7 +25464,7 @@ export interface components { avg_turns_per_session: number; /** * Baseline Spend - * @description Estimated single-model cost for covered turns only + * @description Estimated single-model cost: compared actual spend plus recorded savings; null when traffic has no recorded savings */ baseline_spend: number | null; cache: components["schemas"]["AutoRouterCacheStats"]; @@ -25485,7 +25485,7 @@ export interface components { router_type: string; /** * Saved Pct - * @description Total recorded savings over the matching historical and current baseline; null when costs are unavailable + * @description Recorded savings over baseline_spend, as a percentage */ saved_pct: number | null; /** @@ -25500,17 +25500,17 @@ export interface components { saved_spend: number | null; /** * Savings Estimated Actual Spend - * @description Actual spend, including classifier cost, for covered turns only + * @description Actual spend, including classifier cost, for the compared requests */ savings_estimated_actual_spend: number; /** * Savings Estimated Classifier Cost - * @description Classifier cost included in the matching historical and newer savings comparison; null when classification costs for those requests are unavailable + * @description Classifier cost included in the compared actual spend; null when classification costs for those requests are unavailable */ savings_estimated_classifier_cost?: number | null; /** * Savings Estimated Turns - * @description Requests with a matching savings comparison, including historical recorded estimates + * @description Requests compared against the baseline: every request on complexity routers that recorded savings */ savings_estimated_turns: number; /** Sessions */ @@ -25543,7 +25543,7 @@ export interface components { avg_turns_per_session: number; /** * Baseline Spend - * @description Estimated single-model cost for covered turns only + * @description Estimated single-model cost: compared actual spend plus recorded savings; null when traffic has no recorded savings */ baseline_spend: number | null; cache: components["schemas"]["AutoRouterCacheStats"]; @@ -25554,7 +25554,7 @@ export interface components { classifier_cost: number | null; /** * Saved Pct - * @description Total recorded savings over the matching historical and current baseline; null when costs are unavailable + * @description Recorded savings over baseline_spend, as a percentage */ saved_pct: number | null; /** @@ -25569,17 +25569,17 @@ export interface components { saved_spend: number | null; /** * Savings Estimated Actual Spend - * @description Actual spend, including classifier cost, for covered turns only + * @description Actual spend, including classifier cost, for the compared requests */ savings_estimated_actual_spend: number; /** * Savings Estimated Classifier Cost - * @description Classifier cost included in the matching historical and newer savings comparison; null when classification costs for those requests are unavailable + * @description Classifier cost included in the compared actual spend; null when classification costs for those requests are unavailable */ savings_estimated_classifier_cost?: number | null; /** * Savings Estimated Turns - * @description Requests with a matching savings comparison, including historical recorded estimates + * @description Requests compared against the baseline: every request on complexity routers that recorded savings */ savings_estimated_turns: number; /** Sessions */ @@ -25868,7 +25868,7 @@ export interface components { }; /** * Baseline Spend - * @description Estimated single-model cost; unavailable unless every turn is covered + * @description Estimated single-model cost: spend plus recorded savings */ baseline_spend: number | null; /** @@ -25893,17 +25893,17 @@ export interface components { saved_spend: number | null; /** * Savings Estimated Actual Spend - * @description Actual spend, including classifier cost, for covered turns only + * @description Actual spend, including classifier cost, for requests whose estimate recorded its baseline cost */ savings_estimated_actual_spend: number; /** * Savings Estimated Baseline Spend - * @description Estimated single-model cost for covered turns only + * @description Estimated single-model cost for requests whose estimate recorded its baseline cost */ savings_estimated_baseline_spend: number | null; /** * Savings Estimated Turns - * @description Requests with a matching savings comparison, including historical recorded estimates + * @description Requests whose savings estimate recorded its baseline cost */ savings_estimated_turns: number; /** Session Id */