fix(auto-router): show actual and baseline spend for historical savings (#44057)

The usage card hid actual and baseline spend unless every older session
could be rebuilt from SpendLogs within two seconds, which on a real
gateway it never was. Each complexity router's actual spend is now its
rollup spend and its baseline is spend plus recorded savings, for old
and new requests alike, so the benchmarks and session endpoints never
scan SpendLogs. Adaptive and quality routers record no savings baseline
and stay out of the compared totals; savings_estimated_classifier_cost
is kept and covers the same compared requests
This commit is contained in:
tin-berri 2026-10-01 13:44:32 -07:00 • committed by GitHub
parent eb103334ee
commit 163ebccad5
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
9 changed files with 131 additions and 376 deletions

View file

@ -1,147 +0,0 @@
from collections.abc import Mapping
from contextlib import AbstractAsyncContextManager
from datetime import timedelta
from math import isclose
from types import MappingProxyType
from typing import TYPE_CHECKING, Final, Protocol, cast
from pydantic import BaseModel, ConfigDict, TypeAdapter
from litellm._logging import verbose_proxy_logger
from litellm.constants import MAX_SPENDLOG_ROWS_TO_QUERY
from litellm.proxy.db.autorouter_session_rollup import AUTOROUTER_SESSION_WINDOW_SQL
from litellm.proxy.db.create_views import SupportsRawQueries
if TYPE_CHECKING:
from litellm.proxy.utils import PrismaClient
class SessionSavingsComparison(BaseModel):
model_config = ConfigDict(frozen=True, allow_inf_nan=False)
router_name: str
router_type: str
turns: int
estimated_turns: int
actual_spend: float
classifier_cost: float | None
saved_spend: float
complete: bool
def coverage_fields(self, recorded_savings: float, recorded_turns: int) -> Mapping[str, float | int]:
if self.turns != recorded_turns or not self.complete:
return MappingProxyType({})
if not isclose(self.saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9):
return MappingProxyType({})
return MappingProxyType(
{
"savings_estimated_turns": self.estimated_turns,
"savings_estimated_actual_spend": self.actual_spend,
"savings_estimated_saved_spend": self.saved_spend,
}
)
class _ReadTransactions(Protocol):
def tx(self, *, timeout: timedelta, max_wait: timedelta) -> AbstractAsyncContextManager[SupportsRawQueries]: ...
_COMPARISONS: Final = TypeAdapter(tuple[SessionSavingsComparison, ...])
async def historical_session_comparisons(
prisma_client: "PrismaClient",
start_date: str,
end_date: str,
api_key: str | None,
user_id: str | None,
session_id: str | None = None,
) -> Mapping[tuple[str, str], SessionSavingsComparison]:
try:
reader: Final = cast(_ReadTransactions, prisma_client.read_db) # cast-ok: untyped Prisma transaction delegate
async with reader.tx(timeout=timedelta(seconds=3), max_wait=timedelta(seconds=1)) as transaction:
await transaction.execute_raw("SET TRANSACTION READ ONLY")
await transaction.execute_raw("SET LOCAL statement_timeout = 2000")
rows: Final = await transaction.query_raw(
HISTORICAL_SESSION_COMPARISONS_SQL,
start_date,
end_date,
api_key,
user_id,
session_id,
)
comparisons: Final = _COMPARISONS.validate_python(rows or ())
return MappingProxyType({(row.router_name, row.router_type): row for row in comparisons})
except Exception: # noqa: BLE001 # missing retained logs must not discard recorded dollar savings
verbose_proxy_logger.warning("Historical auto-router cost comparison unavailable; preserving recorded savings")
return MappingProxyType({})
HISTORICAL_SESSION_COMPARISONS_SQL: Final = f"""
WITH {AUTOROUTER_SESSION_WINDOW_SQL}, scoped AS MATERIALIZED (
SELECT * FROM windowed WHERE $5::text IS NULL OR session_id = $5::text
), limited_logs AS MATERIALIZED (
SELECT session.api_key, session.session_id, session.router_name, session.router_type, session.comparison_user_id,
session.classifier_cost_recorded_turns = session.turns AS classifier_cost_tracked,
logs.spend, logs.prompt_tokens + logs.completion_tokens AS tokens,
logs.metadata::jsonb -> 'routing_decision' AS decision,
logs.metadata::jsonb -> 'autorouter_savings' AS savings,
logs.metadata::jsonb -> 'autorouter_savings_estimate' AS estimate
FROM scoped AS session JOIN "LiteLLM_SpendLogs" AS logs
ON logs.api_key = session.api_key
AND CASE WHEN char_length(logs.session_id) > 256
THEN 'sha256:' || encode(sha256(convert_to(logs.session_id, 'UTF8')), 'hex')
ELSE logs.session_id END = session.session_id
AND (session.comparison_user_id IS NULL OR logs."user" = session.comparison_user_id)
AND logs."startTime" BETWEEN session.first_turn_at AND session.last_turn_at
AND COALESCE(logs.metadata::jsonb #>> '{{routing_decision,router_model_name}}', logs.model_group)
= session.router_name
WHERE session.savings_estimated_turns < session.turns
AND logs.status = 'success' AND COALESCE(logs.metadata::jsonb ->> 'internal_call_origin', '') = ''
LIMIT {MAX_SPENDLOG_ROWS_TO_QUERY + 1}
), facts AS (
SELECT *,
CASE WHEN jsonb_typeof(decision -> 'classifier_cost') = 'number'
THEN (decision ->> 'classifier_cost')::float8
WHEN classifier_cost_tracked THEN 0 END AS classifier,
CASE WHEN jsonb_typeof(savings) = 'number' AND (
estimate IS NULL OR estimate = 'null'::jsonb OR (
jsonb_typeof(estimate -> 'version') = 'number' AND estimate ->> 'version' IN ('1', '2', '3')
AND estimate ->> 'status' = 'estimated'
)
) THEN savings::text::float8 END AS saved
FROM limited_logs
), compared AS (
SELECT api_key, session_id, router_name, router_type, comparison_user_id,
COUNT(*) AS turns, SUM(spend + COALESCE(classifier, 0)) AS spend, SUM(tokens) AS total_tokens,
COUNT(saved) AS estimated_turns,
COALESCE(SUM(spend + COALESCE(classifier, 0)) FILTER (WHERE saved IS NOT NULL), 0)::float8 AS actual_spend,
CASE WHEN COUNT(saved) = COUNT(classifier) FILTER (WHERE saved IS NOT NULL)
THEN COALESCE(SUM(classifier) FILTER (WHERE saved IS NOT NULL), 0)::float8
END AS estimated_classifier_cost,
COALESCE(SUM(saved), 0)::float8 AS saved_spend
FROM facts GROUP BY 1, 2, 3, 4, 5
), reconciled AS (
SELECT session.*, logs.estimated_turns, logs.actual_spend, logs.estimated_classifier_cost,
COALESCE((SELECT COUNT(*) FROM limited_logs) <= {MAX_SPENDLOG_ROWS_TO_QUERY}
AND logs.turns = session.turns AND logs.total_tokens = session.total_tokens
AND ABS(logs.spend - session.spend) <= GREATEST(1e-9, ABS(session.spend) * 1e-9)
AND ABS(logs.saved_spend - session.saved_spend) <= GREATEST(1e-9, ABS(session.saved_spend) * 1e-9), FALSE
) AS recovered
FROM scoped AS session LEFT JOIN compared AS logs
ON logs.api_key = session.api_key AND logs.session_id = session.session_id
AND logs.router_name = session.router_name AND logs.router_type = session.router_type
AND logs.comparison_user_id IS NOT DISTINCT FROM session.comparison_user_id
)
SELECT router_name, router_type,
SUM(turns)::bigint AS turns,
SUM(CASE WHEN recovered THEN estimated_turns ELSE savings_estimated_turns END)::bigint AS estimated_turns,
SUM(CASE WHEN recovered THEN actual_spend ELSE savings_estimated_actual_spend END)::float8 AS actual_spend,
CASE WHEN BOOL_AND(CASE WHEN recovered THEN estimated_classifier_cost IS NOT NULL
ELSE savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns END)
THEN SUM(CASE WHEN recovered THEN estimated_classifier_cost ELSE classifier_cost END)::float8
END AS classifier_cost,
SUM(saved_spend)::float8 AS saved_spend,
BOOL_AND(recovered OR savings_estimated_turns = turns) AS complete
FROM reconciled GROUP BY router_name, router_type
"""

View file

@ -8,7 +8,6 @@ POST /auto_router/validate_complexity_router_config - Dry-run the complexity-rou
from collections.abc import Mapping, Sequence
from datetime import datetime, timedelta, timezone
from itertools import chain, groupby
from math import isclose
from types import MappingProxyType
from typing import TYPE_CHECKING, Annotated, Final, Protocol
from uuid import uuid4
@ -32,7 +31,6 @@ from litellm.proxy.auth.auth_checks import (
can_key_call_resolved_model,
)
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
from litellm.proxy.db.autorouter_savings_comparison import historical_session_comparisons
from litellm.proxy.db.autorouter_session_rollup import (
AUTOROUTER_BENCHMARKS_SQL,
bounded_session_id,
@ -643,7 +641,6 @@ class _SessionAggRow(BaseModel):
savings_estimated_actual_spend: float = 0.0
savings_estimated_classifier_cost: float | None = None
savings_estimated_saved_spend: float = 0.0
savings_comparison_complete: bool = True
classifier_cost: float
classifier_cost_recorded_turns: int
session_seconds: float
@ -671,25 +668,35 @@ def _cache_bucket(turns: int, hits: int) -> AutoRouterCacheBucket:
def _savings_cohort(
turns: int, estimated_turns: int, actual_spend: float, saved_spend: float, recorded_savings: float
turns: int, estimated_turns: int, spend: float, saved_spend: float
) -> tuple[float | None, float | None]:
if turns > 0 and estimated_turns == 0 and recorded_savings == 0:
if turns > 0 and estimated_turns == 0 and saved_spend == 0:
return None, None
if not isclose(saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9):
return recorded_savings, None
return recorded_savings, actual_spend + recorded_savings
return saved_spend, spend + saved_spend
def _compared_row(row: _SessionAggRow) -> _SessionAggRow:
_, baseline_spend = _savings_cohort(row.turns, row.savings_estimated_turns, row.spend, row.saved_spend)
compared: Final = row.router_type == "complexity" and baseline_spend is not None
return row.model_copy(
update={
"savings_estimated_turns": row.turns if compared else 0,
"savings_estimated_actual_spend": row.spend if compared else 0.0,
"savings_estimated_classifier_cost": (
row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None
)
if compared
else 0.0,
"savings_estimated_saved_spend": row.saved_spend if compared else 0.0,
}
)
def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals:
return_misses: Final = row.return_turns - row.return_hits
saved_spend, compared_baseline = _savings_cohort(
row.turns,
row.savings_estimated_turns,
row.savings_estimated_actual_spend,
row.savings_estimated_saved_spend,
row.saved_spend,
saved_spend, baseline_spend = _savings_cohort(
row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend
)
baseline_spend: Final = compared_baseline if row.savings_comparison_complete else None
sessions: Final = row.sessions
return AutoRouterBenchmarkTotals(
sessions=sessions,
@ -700,7 +707,7 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals:
spend=row.spend,
savings_estimated_turns=row.savings_estimated_turns,
savings_estimated_actual_spend=row.savings_estimated_actual_spend,
savings_estimated_classifier_cost=row.savings_estimated_classifier_cost if baseline_spend is not None else None,
savings_estimated_classifier_cost=row.savings_estimated_classifier_cost,
saved_spend=saved_spend,
classifier_cost=row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None,
baseline_spend=baseline_spend,
@ -777,7 +784,6 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow:
else None
),
savings_estimated_saved_spend=sum(row.savings_estimated_saved_spend for row in rows),
savings_comparison_complete=all(row.savings_comparison_complete for row in rows),
classifier_cost=sum(row.classifier_cost for row in rows),
classifier_cost_recorded_turns=sum(row.classifier_cost_recorded_turns for row in rows),
session_seconds=sum(row.session_seconds for row in rows),
@ -852,8 +858,8 @@ async def get_auto_router_benchmarks(
Benchmarks for the auto-router dashboard: session shape, savings against the configured
baseline, and prompt-caching behaviour bucketed by what the router did.
Reads session rollups folded once per request at spend-write time, with bounded
retained-log recovery for historical comparisons. A user filter selects only turns attributed to that
Reads session rollups folded once per request at spend-write time, so this endpoint
never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that
internal user when written; older key-only history remains outside user views. A session
is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before
end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is
@ -887,44 +893,7 @@ async def get_auto_router_benchmarks(
api_key,
user_id,
)
recorded_rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ())
comparisons: Final = (
await historical_session_comparisons(
prisma_client,
start_day.isoformat(),
(end_day + timedelta(days=1)).isoformat(),
api_key,
user_id,
)
if any(row.savings_estimated_turns < row.turns for row in recorded_rows)
else MappingProxyType({})
)
covered_rows: Final = tuple(
row.model_copy(
update={
**comparison.coverage_fields(row.saved_spend, row.turns),
"savings_estimated_classifier_cost": comparison.classifier_cost,
"savings_comparison_complete": comparison.complete and comparison.turns == row.turns,
}
)
if (comparison := comparisons.get((row.router_name, row.router_type)))
else row.model_copy(update={"savings_comparison_complete": row.savings_estimated_turns == row.turns})
for row in recorded_rows
)
rows: Final = tuple(
row.model_copy(
update={
"savings_comparison_complete": row.savings_comparison_complete
and isclose(
row.saved_spend,
row.savings_estimated_saved_spend,
rel_tol=1e-9,
abs_tol=1e-9,
),
}
)
for row in covered_rows
)
rows: Final = tuple(_compared_row(row) for row in _SESSION_AGG_ROWS.validate_python(raw_rows or ()))
groups: Final = (
*(_benchmark_group(row) for row in rows),
*_idle_router_groups(llm_router, frozenset((row.router_name, row.router_type) for row in rows)),
@ -962,43 +931,16 @@ async def get_auto_router_session(
if prisma_client is None:
raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value)
recorded: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key(
row: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key(
user_api_key_dict.api_key, bounded_session_id(session_id)
)
if recorded is None:
if row is None:
raise HTTPException(
status_code=404, detail=f"No auto-routed turns recorded for session {session_id!r} under this key"
)
comparisons: Final = (
await historical_session_comparisons(
prisma_client,
recorded.first_turn_at.isoformat(),
(recorded.last_turn_at + timedelta(microseconds=1)).isoformat(),
user_api_key_dict.api_key,
None,
bounded_session_id(session_id),
)
if recorded.savings_estimated_turns < recorded.turns
else MappingProxyType({})
)
comparison: Final = comparisons.get((recorded.router_name, recorded.router_type))
row: Final = (
recorded.model_copy(update=comparison.coverage_fields(recorded.saved_spend, recorded.turns))
if comparison
else recorded
)
saved_spend, compared_baseline = _savings_cohort(
row.turns,
row.savings_estimated_turns,
row.savings_estimated_actual_spend,
row.savings_estimated_saved_spend,
row.saved_spend,
)
baseline_spend: Final = (
compared_baseline
if row.savings_estimated_turns == row.turns
or (comparison and comparison.complete and comparison.turns == row.turns)
else None
saved_spend, baseline_spend = _savings_cohort(row.turns, row.savings_estimated_turns, row.spend, row.saved_spend)
_, estimated_baseline_spend = _savings_cohort(
row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend
)
return AutoRouterSessionResponse(
session_id=session_id,
@ -1010,8 +952,8 @@ async def get_auto_router_session(
savings_estimated_turns=row.savings_estimated_turns,
savings_estimated_actual_spend=row.savings_estimated_actual_spend,
saved_spend=saved_spend,
baseline_spend=baseline_spend if row.savings_estimated_turns == row.turns else None,
savings_estimated_baseline_spend=baseline_spend,
baseline_spend=baseline_spend,
savings_estimated_baseline_spend=estimated_baseline_spend,
baseline_model=row.baseline_model,
baseline_models=row.baseline_models,
)

View file

@ -218,23 +218,24 @@ class AutoRouterBenchmarkTotals(BaseModel):
"subtotal recording, and zero for an empty window"
)
savings_estimated_turns: int = Field(
description="Requests with a matching savings comparison, including historical recorded estimates"
description="Requests compared against the baseline: every request on complexity routers that recorded savings"
)
savings_estimated_actual_spend: float = Field(
description="Actual spend, including classifier cost, for covered turns only"
description="Actual spend, including classifier cost, for the compared requests"
)
savings_estimated_classifier_cost: float | None = Field(
default=None,
description="Classifier cost included in the matching historical and newer savings comparison; "
description="Classifier cost included in the compared actual spend; "
"null when classification costs for those requests are unavailable",
)
saved_spend: float | None = Field(
description="Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates"
)
baseline_spend: float | None = Field(description="Estimated single-model cost for covered turns only")
saved_pct: float | None = Field(
description="Total recorded savings over the matching historical and current baseline; null when costs are unavailable"
baseline_spend: float | None = Field(
description="Estimated single-model cost: compared actual spend plus recorded savings; "
"null when traffic has no recorded savings"
)
saved_pct: float | None = Field(description="Recorded savings over baseline_spend, as a percentage")
saved_per_session: float | None = Field(description="Recorded savings per session, including historical estimates")
cache: AutoRouterCacheStats
@ -266,20 +267,16 @@ class AutoRouterSessionResponse(BaseModel):
turns: int = Field(description="Auto-routed turns the rollup has recorded for this session so far")
last_model: str = Field(description="The deployment model the most recent turn was routed to")
spend: float = Field(description="What the session's routed traffic actually cost, classifier calls included")
savings_estimated_turns: int = Field(
description="Requests with a matching savings comparison, including historical recorded estimates"
)
savings_estimated_turns: int = Field(description="Requests whose savings estimate recorded its baseline cost")
savings_estimated_actual_spend: float = Field(
description="Actual spend, including classifier cost, for covered turns only"
description="Actual spend, including classifier cost, for requests whose estimate recorded its baseline cost"
)
saved_spend: float | None = Field(
description="Recorded historical savings plus newer estimates, net of classifier cost"
)
baseline_spend: float | None = Field(
description="Estimated single-model cost; unavailable unless every turn is covered"
)
baseline_spend: float | None = Field(description="Estimated single-model cost: spend plus recorded savings")
savings_estimated_baseline_spend: float | None = Field(
description="Estimated single-model cost for covered turns only"
description="Estimated single-model cost for requests whose estimate recorded its baseline cost"
)
baseline_model: str | None = Field(
description="The savings baseline recorded by most session turns, including historical turns, recorded turn by "

View file

@ -6,7 +6,6 @@ tests/unit/proxy/db/test_autorouter_session_rollup.py.
"""
import asyncio
import json
import time
import uuid
from datetime import datetime, timedelta, timezone
@ -25,10 +24,6 @@ from litellm.proxy.db.autorouter_session_rollup import (
flush_autorouter_turn_transactions,
)
from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import SpendLogCleanup
from litellm.proxy.db.autorouter_savings_comparison import (
HISTORICAL_SESSION_COMPARISONS_SQL,
SessionSavingsComparison,
)
pytestmark = pytest.mark.asyncio(loop_scope="session")
@ -96,66 +91,6 @@ async def _row(db, key: str, session_id: str = "s1", router: str = "auto-1") ->
return rows[0]
@pytest.mark.parametrize("historical_saved, damaged, user_id, split_sessions, current_classifier", [
(29.5, None, None, False, 0.2), (29.5, None, "owner", False, 0.2), (0.0, None, None, False, 0.2),
(-3.0, None, None, False, 0.2), (29.5, "missing", None, False, 0.2), (29.5, "cost", None, False, 0.2),
(0.0, "missing", None, False, 0.2), (29.5, None, None, True, 0.2), (29.5, None, None, False, 0.0),
])
async def test_historical_and_new_savings_compare_matching_costs_and_exclude_unknown_requests(
db: Prisma, historical_saved: float, damaged: str | None, user_id: str | None, split_sessions: bool,
current_classifier: float,
) -> None:
async with db.tx() as tx:
for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession", "LiteLLM_SpendLogs"):
await tx.execute_raw(f'CREATE TEMP TABLE "{table}" (LIKE public."{table}" INCLUDING ALL) ON COMMIT DROP')
for name, spend, saved, classifier, estimated in (
("historical", 9.0, historical_saved, 0.1, False),
("current", 1.0, 0.5, current_classifier, True),
("unknown", 99.0, 0.0, 3.0, False),
):
session_id: Final = "s2" if split_sessions and name == "current" else "s1"
await _turn(tx, "key", "model", T0, spend=spend, saved=saved, classifier_cost=classifier,
estimated=estimated, session_id=session_id)
metadata: Final = {
"routing_decision": {"router_model_name": "auto-1", **({"classifier_cost": classifier} if classifier else {})},
"autorouter_savings": saved if name != "unknown" else None,
**({"autorouter_savings_estimate": {
"version": 3, "status": "estimated" if estimated else "unknown",
}} if name != "historical" else {}),
}
await tx.execute_raw('''INSERT INTO "LiteLLM_SpendLogs"
(request_id,api_key,session_id,model,"user","startTime","endTime",call_type,
spend,prompt_tokens,completion_tokens,status,metadata)
VALUES ($1,'key',$5,'model','owner',$2::timestamp,$2::timestamp,'acompletion',
$3::float8,100,0,'success',$4::jsonb)
''', name, T0.isoformat(), spend - classifier, json.dumps(metadata), session_id)
await tx.execute_raw('''INSERT INTO "LiteLLM_AutoRouterUserSession"
(user_id,api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model,
turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend,
savings_estimated_saved_spend)
SELECT 'owner',api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model,
turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend,
savings_estimated_saved_spend FROM "LiteLLM_AutoRouterSession"
''')
if damaged == "missing":
await tx.execute_raw('DELETE FROM "LiteLLM_SpendLogs" WHERE request_id = \'historical\'')
elif damaged == "cost":
await tx.execute_raw('UPDATE "LiteLLM_SpendLogs" SET spend = 1 WHERE request_id = \'historical\'')
rows: Final = await tx.query_raw(
HISTORICAL_SESSION_COMPARISONS_SQL, "2026-08-01", "2026-08-02", "key", user_id, None,
)
comparison: Final = SessionSavingsComparison.model_validate(rows[0])
assert comparison.saved_spend == historical_saved + 0.5
assert comparison.complete is (damaged is None)
assert comparison.classifier_cost == (pytest.approx(0.1 + current_classifier) if damaged is None else None)
assert comparison.coverage_fields(historical_saved + 0.5, 4) == {}
assert comparison.coverage_fields(historical_saved + 0.5, 3) == ({
"savings_estimated_turns": 2,
"savings_estimated_actual_spend": 10.0,
"savings_estimated_saved_spend": historical_saved + 0.5,
} if damaged is None else {})
async def test_every_turn_lands_in_exactly_one_bucket(db):
key = f"k-{uuid.uuid4()}"
await _turn(db, key, "A", T0, ttl=300)

View file

@ -721,26 +721,60 @@ class TestAutoRouterBenchmarks:
assert totals.saved_pct == -100.0
assert totals.classifier_cost == 0.4
@pytest.mark.asyncio
@pytest.mark.parametrize("estimated_turns", [0, 4])
def test_recorded_savings_survive_when_historical_comparison_costs_are_missing(self, estimated_turns: int) -> None:
from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals
async def test_historical_savings_without_recorded_baselines_compare_against_all_spend(
self, estimated_turns: int, monkeypatch: pytest.MonkeyPatch
) -> None:
row: Final = self.ROW.model_copy(
update={
"savings_estimated_turns": estimated_turns,
"savings_estimated_actual_spend": 2.0 if estimated_turns else 0.0,
"savings_estimated_classifier_cost": None,
"savings_estimated_saved_spend": -0.5 if estimated_turns else 0.0,
}
)
totals: Final = _benchmark_totals(row)
assert totals.spend == 10.0
assert totals.savings_estimated_turns == estimated_turns
assert totals.saved_spend == 30.0
assert totals.baseline_spend is None
assert totals.savings_estimated_classifier_cost is None
assert totals.saved_pct is None
response: Final = await self._benchmarks(monkeypatch, rows=[row.model_dump()], model_list=[])
assert response.groups[0].model_dump(exclude={"router_name", "router_type", "tier_turns"}) == (
response.totals.model_dump()
)
totals: Final = response.totals
assert (totals.spend, totals.saved_spend, totals.baseline_spend, totals.saved_pct) == (10.0, 30.0, 40.0, 75.0)
assert (totals.savings_estimated_turns, totals.savings_estimated_actual_spend) == (40, 10.0)
assert totals.savings_estimated_classifier_cost == 0.4
assert totals.saved_per_session == 7.5
@pytest.mark.asyncio
@pytest.mark.parametrize("router_type, saved", [("adaptive", 0.0), ("quality", 0.0), ("quality", 2.0)])
async def test_only_complexity_routers_enter_the_compared_totals(
self, router_type: str, saved: float, monkeypatch: pytest.MonkeyPatch
) -> None:
adaptive: Final = self.ROW.model_copy(
update={
"router_name": f"{router_type}-auto",
"router_type": router_type,
"turns": 10,
"spend": 3.0,
"saved_spend": saved,
"savings_estimated_turns": 0,
"savings_estimated_actual_spend": 0.0,
"savings_estimated_saved_spend": 0.0,
"classifier_cost": 0.0,
"classifier_cost_recorded_turns": 10,
}
)
response: Final = await self._benchmarks(
monkeypatch, rows=[self.ROW.model_dump(), adaptive.model_dump()], model_list=[]
)
unbaselined: Final = response.groups[1]
assert (unbaselined.saved_spend, unbaselined.baseline_spend, unbaselined.saved_pct) == (None, None, None)
assert (unbaselined.savings_estimated_turns, unbaselined.savings_estimated_classifier_cost) == (0, 0.0)
totals: Final = response.totals
assert (totals.turns, totals.spend) == (50, 13.0)
assert (totals.savings_estimated_turns, totals.savings_estimated_actual_spend) == (40, 10.0)
assert (totals.saved_spend, totals.baseline_spend, totals.saved_pct) == (30.0, 40.0, 75.0)
assert totals.savings_estimated_classifier_cost == 0.4
def test_an_empty_window_folds_to_zeros(self):
from litellm.proxy.management_endpoints.auto_router_endpoints import (
_benchmark_totals,
@ -1138,8 +1172,10 @@ class TestAutoRouterSession:
"saved_spend": 0.24,
"savings_estimated_turns": 3 if estimated else 0,
"savings_estimated_actual_spend": 0.14 if estimated else 0.0,
"baseline_spend": pytest.approx(0.38) if turns == 3 else None,
"savings_estimated_baseline_spend": pytest.approx(0.38) if turns == 3 else None,
"baseline_spend": pytest.approx(spend + 0.24),
"savings_estimated_baseline_spend": (
pytest.approx(0.38 if turns == 3 else 0.10) if estimated else None
),
"baseline_model": "anthropic/claude-opus-5",
"baseline_models": {"anthropic/claude-opus-5": 3},
}

View file

@ -70,6 +70,7 @@ const totals = (overrides: Partial<Totals> = {}): Totals => ({
spend: 359.86,
savings_estimated_turns: overrides.turns ?? 3073,
savings_estimated_actual_spend: overrides.spend ?? 359.86,
savings_estimated_classifier_cost: overrides.classifier_cost === undefined ? 6.146 : overrides.classifier_cost,
classifier_cost: 6.146,
saved_spend: 2174.59,
baseline_spend: 2534.45,
@ -104,6 +105,7 @@ const zeroTotals: Totals = {
spend: 0,
savings_estimated_turns: 0,
savings_estimated_actual_spend: 0,
savings_estimated_classifier_cost: 0,
classifier_cost: 0,
saved_spend: 0,
baseline_spend: 0,
@ -159,11 +161,10 @@ describe("AutoRouterBenchmarksTab", () => {
it.each([
{ estimatedTurns: 0, actual: 0, saved: null, pct: null },
{ estimatedTurns: 0, actual: 0, saved: 30, pct: null },
{ estimatedTurns: 10, actual: 2, saved: -0.5, pct: -33.3 },
{ estimatedTurns: 10, actual: 2, saved: 0, pct: 0 },
{ estimatedTurns: 40, actual: 10, saved: 30, pct: 75 },
])("compares matching old and new requests with savings $saved", ({ estimatedTurns, actual, saved, pct }) => {
{ estimatedTurns: 3073, actual: 10, saved: 30, pct: 75 },
])("compares the requests on routers that recorded savings $saved", ({ estimatedTurns, actual, saved, pct }) => {
const comparison = {
spend: actual + 99,
savings_estimated_turns: estimatedTurns,
@ -189,18 +190,17 @@ describe("AutoRouterBenchmarksTab", () => {
]
: ["Unavailable", "Unavailable", "Unavailable", "Unavailable"],
);
expect(screen.queryByText("Actual spend on covered turns")).not.toBeInTheDocument();
expect(screen.queryByText(/Matching cost details are unavailable/)).not.toBeInTheDocument();
expect(screen.getByLabelText("question-circle")).toBeInTheDocument();
if (estimatedTurns) {
expect(screen.getByText(`Savings based on ${estimatedTurns} of 3,073 requests`)).toBeInTheDocument();
const sign = pct && pct > 0 ? "-" : "+";
const badge = pct === 0 ? "0%" : `${sign}${Math.abs(pct ?? 0).toFixed(0)}%`;
const partial = estimatedTurns > 0 && estimatedTurns < 3073;
expect(screen.queryByText(/adaptive and quality routers are excluded/) != null).toBe(partial);
if (partial) {
expect(screen.getByText(/Compared on 10 of 3,073 requests/)).toBeInTheDocument();
}
if (pct != null) {
const sign = pct > 0 ? "-" : "+";
const badge = pct === 0 ? "0%" : `${sign}${Math.abs(pct).toFixed(0)}%`;
expect(screen.getByText(badge)).toBeInTheDocument();
} else if (saved != null) {
expect(screen.getByText("$30.00")).toBeInTheDocument();
expect(
screen.getByText("Historical savings are included. Matching cost details are unavailable."),
).toBeInTheDocument();
}
});

View file

@ -76,10 +76,8 @@ const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued?
const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
const stats = view.stats;
const cheaper = stats.saved_pct != null && stats.saved_pct >= 0;
const completeCoverage = stats.savings_estimated_turns === stats.turns;
const coveredClassifierCost =
stats.savings_estimated_classifier_cost ?? (completeCoverage ? stats.classifier_cost : null);
const classifierCost = stats.baseline_spend == null ? null : coveredClassifierCost;
const classifierCost = stats.baseline_spend == null ? null : stats.savings_estimated_classifier_cost ?? null;
const comparedAll = stats.savings_estimated_turns === stats.turns;
return (
<Card className="overflow-hidden py-0">
<div className="grid md:grid-cols-[minmax(0,1fr)_minmax(0,1fr)]">
@ -101,15 +99,10 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
</Badge>
)}
</div>
{stats.baseline_spend != null && !completeCoverage && (
{stats.baseline_spend != null && !comparedAll && (
<p className="text-center text-xs text-muted-foreground">
Savings based on {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()}{" "}
requests
</p>
)}
{stats.saved_spend != null && stats.baseline_spend == null && (
<p className="text-center text-xs text-muted-foreground">
Historical savings are included. Matching cost details are unavailable.
Compared on {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()} requests;
adaptive and quality routers are excluded
</p>
)}
</div>
@ -118,7 +111,7 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
<SpendRow
label="Actual auto-router spend"
value={stats.baseline_spend == null ? "Unavailable" : usd(stats.savings_estimated_actual_spend)}
tooltip="Savings, actual spend, and baseline include historical and newer requests with recorded savings estimates. Requests without estimates are excluded. Actual spend includes classification costs."
tooltip="Auto-routed spend in the range, including classification costs, for complexity routers. Adaptive and quality routers are excluded. Baseline is this spend plus recorded savings."
/>
<div className="mb-3 border-l-2 pl-4">
<SpendRow
@ -316,11 +309,9 @@ const BenchmarksBody: React.FC<BenchmarksBodyProps> = ({ isPending, error, data,
</div>
<p className="text-xs text-muted-foreground">
Savings, actual spend, and baseline compare the same historical and newer requests with recorded estimates,
including zero or negative savings. Requests without estimates are excluded. Savings are net of recorded LLM
classification cost. If historical cost details are unavailable, recorded savings remain visible without a
baseline or percentage. The range counts whole sessions that overlap it, so totals can differ from savings views
that group usage by UTC day.
Actual spend covers every request on complexity routers, including LLM classification cost. Baseline is actual
spend plus recorded savings, so savings can be zero or negative. The range counts whole sessions that overlap
it, so totals can differ from savings views that group usage by UTC day.
</p>
<div className="space-y-4">

View file

@ -34,6 +34,7 @@ const stats = {
spend: 1.25,
savings_estimated_turns: 4,
savings_estimated_actual_spend: 1.25,
savings_estimated_classifier_cost: 0.25,
classifier_cost: 0.25,
saved_spend: 8.75,
baseline_spend: 10,

View file

@ -1263,8 +1263,8 @@ export interface paths {
* @description Benchmarks for the auto-router dashboard: session shape, savings against the configured
* baseline, and prompt-caching behaviour bucketed by what the router did.
*
* Reads session rollups folded once per request at spend-write time, with bounded
* retained-log recovery for historical comparisons. A user filter selects only turns attributed to that
* Reads session rollups folded once per request at spend-write time, so this endpoint
* never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that
* internal user when written; older key-only history remains outside user views. A session
* is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before
* end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is
@ -25464,7 +25464,7 @@ export interface components {
avg_turns_per_session: number;
/**
* Baseline Spend
* @description Estimated single-model cost for covered turns only
* @description Estimated single-model cost: compared actual spend plus recorded savings; null when traffic has no recorded savings
*/
baseline_spend: number | null;
cache: components["schemas"]["AutoRouterCacheStats"];
@ -25485,7 +25485,7 @@ export interface components {
router_type: string;
/**
* Saved Pct
* @description Total recorded savings over the matching historical and current baseline; null when costs are unavailable
* @description Recorded savings over baseline_spend, as a percentage
*/
saved_pct: number | null;
/**
@ -25500,17 +25500,17 @@ export interface components {
saved_spend: number | null;
/**
* Savings Estimated Actual Spend
* @description Actual spend, including classifier cost, for covered turns only
* @description Actual spend, including classifier cost, for the compared requests
*/
savings_estimated_actual_spend: number;
/**
* Savings Estimated Classifier Cost
* @description Classifier cost included in the matching historical and newer savings comparison; null when classification costs for those requests are unavailable
* @description Classifier cost included in the compared actual spend; null when classification costs for those requests are unavailable
*/
savings_estimated_classifier_cost?: number | null;
/**
* Savings Estimated Turns
* @description Requests with a matching savings comparison, including historical recorded estimates
* @description Requests compared against the baseline: every request on complexity routers that recorded savings
*/
savings_estimated_turns: number;
/** Sessions */
@ -25543,7 +25543,7 @@ export interface components {
avg_turns_per_session: number;
/**
* Baseline Spend
* @description Estimated single-model cost for covered turns only
* @description Estimated single-model cost: compared actual spend plus recorded savings; null when traffic has no recorded savings
*/
baseline_spend: number | null;
cache: components["schemas"]["AutoRouterCacheStats"];
@ -25554,7 +25554,7 @@ export interface components {
classifier_cost: number | null;
/**
* Saved Pct
* @description Total recorded savings over the matching historical and current baseline; null when costs are unavailable
* @description Recorded savings over baseline_spend, as a percentage
*/
saved_pct: number | null;
/**
@ -25569,17 +25569,17 @@ export interface components {
saved_spend: number | null;
/**
* Savings Estimated Actual Spend
* @description Actual spend, including classifier cost, for covered turns only
* @description Actual spend, including classifier cost, for the compared requests
*/
savings_estimated_actual_spend: number;
/**
* Savings Estimated Classifier Cost
* @description Classifier cost included in the matching historical and newer savings comparison; null when classification costs for those requests are unavailable
* @description Classifier cost included in the compared actual spend; null when classification costs for those requests are unavailable
*/
savings_estimated_classifier_cost?: number | null;
/**
* Savings Estimated Turns
* @description Requests with a matching savings comparison, including historical recorded estimates
* @description Requests compared against the baseline: every request on complexity routers that recorded savings
*/
savings_estimated_turns: number;
/** Sessions */
@ -25868,7 +25868,7 @@ export interface components {
};
/**
* Baseline Spend
* @description Estimated single-model cost; unavailable unless every turn is covered
* @description Estimated single-model cost: spend plus recorded savings
*/
baseline_spend: number | null;
/**
@ -25893,17 +25893,17 @@ export interface components {
saved_spend: number | null;
/**
* Savings Estimated Actual Spend
* @description Actual spend, including classifier cost, for covered turns only
* @description Actual spend, including classifier cost, for requests whose estimate recorded its baseline cost
*/
savings_estimated_actual_spend: number;
/**
* Savings Estimated Baseline Spend
* @description Estimated single-model cost for covered turns only
* @description Estimated single-model cost for requests whose estimate recorded its baseline cost
*/
savings_estimated_baseline_spend: number | null;
/**
* Savings Estimated Turns
* @description Requests with a matching savings comparison, including historical recorded estimates
* @description Requests whose savings estimate recorded its baseline cost
*/
savings_estimated_turns: number;
/** Session Id */