mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix(autorouter): compare historical and new savings consistently (#43348)
* fix(autorouter): compare historical and new savings consistently * fix(autorouter): reject comparisons if request counts changed * fix(router): restore eligible LLM and classification breakdown * fix(router): avoid ambiguous baseline labels for partial comparisons
This commit is contained in:
parent
f4a7c04d99
commit
3b2a447fae
11 changed files with 482 additions and 136 deletions
|
|
@ -34,15 +34,12 @@ class LiteLLM_AutoRouterSession(LiteLLMPydanticObjectBase):
|
|||
|
||||
@property
|
||||
def baseline_model(self) -> str | None:
|
||||
"""The baseline most covered turns were priced against, or None when none were estimated.
|
||||
|
||||
A router reconfigured mid-session leaves turns priced against two baselines; the row keeps both
|
||||
counts, and the label is the one that priced the most money-carrying turns rather than whatever the
|
||||
router is configured with now.
|
||||
"""
|
||||
if not self.savings_estimated_baseline_models:
|
||||
"""A recorded baseline label when excluded turns cannot change the selected model."""
|
||||
if not self.baseline_models:
|
||||
return None
|
||||
if self.savings_estimated_turns < self.turns and len(self.baseline_models) > 1:
|
||||
return None
|
||||
return max(
|
||||
self.savings_estimated_baseline_models,
|
||||
key=lambda model: (self.savings_estimated_baseline_models[model], model),
|
||||
self.baseline_models,
|
||||
key=lambda model: (self.baseline_models[model], model),
|
||||
)
|
||||
|
|
|
|||
147
litellm/proxy/db/autorouter_savings_comparison.py
Normal file
147
litellm/proxy/db/autorouter_savings_comparison.py
Normal file
|
|
@ -0,0 +1,147 @@
|
|||
from collections.abc import Mapping
|
||||
from contextlib import AbstractAsyncContextManager
|
||||
from datetime import timedelta
|
||||
from math import isclose
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Final, Protocol, cast
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, TypeAdapter
|
||||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.constants import MAX_SPENDLOG_ROWS_TO_QUERY
|
||||
from litellm.proxy.db.autorouter_session_rollup import AUTOROUTER_SESSION_WINDOW_SQL
|
||||
from litellm.proxy.db.create_views import SupportsRawQueries
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from litellm.proxy.utils import PrismaClient
|
||||
|
||||
|
||||
class SessionSavingsComparison(BaseModel):
|
||||
model_config = ConfigDict(frozen=True, allow_inf_nan=False)
|
||||
|
||||
router_name: str
|
||||
router_type: str
|
||||
turns: int
|
||||
estimated_turns: int
|
||||
actual_spend: float
|
||||
classifier_cost: float | None
|
||||
saved_spend: float
|
||||
complete: bool
|
||||
|
||||
def coverage_fields(self, recorded_savings: float, recorded_turns: int) -> Mapping[str, float | int]:
|
||||
if self.turns != recorded_turns or not self.complete:
|
||||
return MappingProxyType({})
|
||||
if not isclose(self.saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9):
|
||||
return MappingProxyType({})
|
||||
return MappingProxyType(
|
||||
{
|
||||
"savings_estimated_turns": self.estimated_turns,
|
||||
"savings_estimated_actual_spend": self.actual_spend,
|
||||
"savings_estimated_saved_spend": self.saved_spend,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
class _ReadTransactions(Protocol):
|
||||
def tx(self, *, timeout: timedelta, max_wait: timedelta) -> AbstractAsyncContextManager[SupportsRawQueries]: ...
|
||||
|
||||
|
||||
_COMPARISONS: Final = TypeAdapter(tuple[SessionSavingsComparison, ...])
|
||||
|
||||
|
||||
async def historical_session_comparisons(
|
||||
prisma_client: "PrismaClient",
|
||||
start_date: str,
|
||||
end_date: str,
|
||||
api_key: str | None,
|
||||
user_id: str | None,
|
||||
session_id: str | None = None,
|
||||
) -> Mapping[tuple[str, str], SessionSavingsComparison]:
|
||||
try:
|
||||
reader: Final = cast(_ReadTransactions, prisma_client.read_db) # cast-ok: untyped Prisma transaction delegate
|
||||
async with reader.tx(timeout=timedelta(seconds=3), max_wait=timedelta(seconds=1)) as transaction:
|
||||
await transaction.execute_raw("SET TRANSACTION READ ONLY")
|
||||
await transaction.execute_raw("SET LOCAL statement_timeout = 2000")
|
||||
rows: Final = await transaction.query_raw(
|
||||
HISTORICAL_SESSION_COMPARISONS_SQL,
|
||||
start_date,
|
||||
end_date,
|
||||
api_key,
|
||||
user_id,
|
||||
session_id,
|
||||
)
|
||||
comparisons: Final = _COMPARISONS.validate_python(rows or ())
|
||||
return MappingProxyType({(row.router_name, row.router_type): row for row in comparisons})
|
||||
except Exception: # noqa: BLE001 # missing retained logs must not discard recorded dollar savings
|
||||
verbose_proxy_logger.warning("Historical auto-router cost comparison unavailable; preserving recorded savings")
|
||||
return MappingProxyType({})
|
||||
|
||||
|
||||
HISTORICAL_SESSION_COMPARISONS_SQL: Final = f"""
|
||||
WITH {AUTOROUTER_SESSION_WINDOW_SQL}, scoped AS MATERIALIZED (
|
||||
SELECT * FROM windowed WHERE $5::text IS NULL OR session_id = $5::text
|
||||
), limited_logs AS MATERIALIZED (
|
||||
SELECT session.api_key, session.session_id, session.router_name, session.router_type, session.comparison_user_id,
|
||||
session.classifier_cost_recorded_turns = session.turns AS classifier_cost_tracked,
|
||||
logs.spend, logs.prompt_tokens + logs.completion_tokens AS tokens,
|
||||
logs.metadata::jsonb -> 'routing_decision' AS decision,
|
||||
logs.metadata::jsonb -> 'autorouter_savings' AS savings,
|
||||
logs.metadata::jsonb -> 'autorouter_savings_estimate' AS estimate
|
||||
FROM scoped AS session JOIN "LiteLLM_SpendLogs" AS logs
|
||||
ON logs.api_key = session.api_key
|
||||
AND CASE WHEN char_length(logs.session_id) > 256
|
||||
THEN 'sha256:' || encode(sha256(convert_to(logs.session_id, 'UTF8')), 'hex')
|
||||
ELSE logs.session_id END = session.session_id
|
||||
AND (session.comparison_user_id IS NULL OR logs."user" = session.comparison_user_id)
|
||||
AND logs."startTime" BETWEEN session.first_turn_at AND session.last_turn_at
|
||||
AND COALESCE(logs.metadata::jsonb #>> '{{routing_decision,router_model_name}}', logs.model_group)
|
||||
= session.router_name
|
||||
WHERE session.savings_estimated_turns < session.turns
|
||||
AND logs.status = 'success' AND COALESCE(logs.metadata::jsonb ->> 'internal_call_origin', '') = ''
|
||||
LIMIT {MAX_SPENDLOG_ROWS_TO_QUERY + 1}
|
||||
), facts AS (
|
||||
SELECT *,
|
||||
CASE WHEN jsonb_typeof(decision -> 'classifier_cost') = 'number'
|
||||
THEN (decision ->> 'classifier_cost')::float8
|
||||
WHEN classifier_cost_tracked THEN 0 END AS classifier,
|
||||
CASE WHEN jsonb_typeof(savings) = 'number' AND (
|
||||
estimate IS NULL OR estimate = 'null'::jsonb OR (
|
||||
jsonb_typeof(estimate -> 'version') = 'number' AND estimate ->> 'version' IN ('1', '2', '3')
|
||||
AND estimate ->> 'status' = 'estimated'
|
||||
)
|
||||
) THEN savings::text::float8 END AS saved
|
||||
FROM limited_logs
|
||||
), compared AS (
|
||||
SELECT api_key, session_id, router_name, router_type, comparison_user_id,
|
||||
COUNT(*) AS turns, SUM(spend + COALESCE(classifier, 0)) AS spend, SUM(tokens) AS total_tokens,
|
||||
COUNT(saved) AS estimated_turns,
|
||||
COALESCE(SUM(spend + COALESCE(classifier, 0)) FILTER (WHERE saved IS NOT NULL), 0)::float8 AS actual_spend,
|
||||
CASE WHEN COUNT(saved) = COUNT(classifier) FILTER (WHERE saved IS NOT NULL)
|
||||
THEN COALESCE(SUM(classifier) FILTER (WHERE saved IS NOT NULL), 0)::float8
|
||||
END AS estimated_classifier_cost,
|
||||
COALESCE(SUM(saved), 0)::float8 AS saved_spend
|
||||
FROM facts GROUP BY 1, 2, 3, 4, 5
|
||||
), reconciled AS (
|
||||
SELECT session.*, logs.estimated_turns, logs.actual_spend, logs.estimated_classifier_cost,
|
||||
COALESCE((SELECT COUNT(*) FROM limited_logs) <= {MAX_SPENDLOG_ROWS_TO_QUERY}
|
||||
AND logs.turns = session.turns AND logs.total_tokens = session.total_tokens
|
||||
AND ABS(logs.spend - session.spend) <= GREATEST(1e-9, ABS(session.spend) * 1e-9)
|
||||
AND ABS(logs.saved_spend - session.saved_spend) <= GREATEST(1e-9, ABS(session.saved_spend) * 1e-9), FALSE
|
||||
) AS recovered
|
||||
FROM scoped AS session LEFT JOIN compared AS logs
|
||||
ON logs.api_key = session.api_key AND logs.session_id = session.session_id
|
||||
AND logs.router_name = session.router_name AND logs.router_type = session.router_type
|
||||
AND logs.comparison_user_id IS NOT DISTINCT FROM session.comparison_user_id
|
||||
)
|
||||
SELECT router_name, router_type,
|
||||
SUM(turns)::bigint AS turns,
|
||||
SUM(CASE WHEN recovered THEN estimated_turns ELSE savings_estimated_turns END)::bigint AS estimated_turns,
|
||||
SUM(CASE WHEN recovered THEN actual_spend ELSE savings_estimated_actual_spend END)::float8 AS actual_spend,
|
||||
CASE WHEN BOOL_AND(CASE WHEN recovered THEN estimated_classifier_cost IS NOT NULL
|
||||
ELSE savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns END)
|
||||
THEN SUM(CASE WHEN recovered THEN estimated_classifier_cost ELSE classifier_cost END)::float8
|
||||
END AS classifier_cost,
|
||||
SUM(saved_spend)::float8 AS saved_spend,
|
||||
BOOL_AND(recovered OR savings_estimated_turns = turns) AS complete
|
||||
FROM reconciled GROUP BY router_name, router_type
|
||||
"""
|
||||
|
|
@ -7,8 +7,8 @@ on the prisma client. The spend-log flush job drains the queue into
|
|||
key and user session rollups with one atomic statement per turn: each upsert classifies
|
||||
the turn (same model, first visit, return to a model the session already used, out of
|
||||
order) against the row's own columns, so nothing is read before the write and concurrent
|
||||
pods compose. The benchmarks endpoint aggregates these rows and never touches
|
||||
LiteLLM_SpendLogs.
|
||||
pods compose. The benchmarks endpoint aggregates these rows and can recover matching historical
|
||||
costs from retained spend logs when estimate coverage predates these columns.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -45,20 +45,24 @@ _SESSION_COLUMNS: Final = """
|
|||
savings_estimated_baseline_models
|
||||
"""
|
||||
|
||||
AUTOROUTER_BENCHMARKS_SQL: Final = f"""
|
||||
WITH windowed AS (
|
||||
SELECT {_SESSION_COLUMNS} FROM "LiteLLM_AutoRouterSession"
|
||||
AUTOROUTER_SESSION_WINDOW_SQL: Final = f"""
|
||||
windowed AS (
|
||||
SELECT {_SESSION_COLUMNS}, NULL::text AS comparison_user_id FROM "LiteLLM_AutoRouterSession"
|
||||
WHERE $4::text IS NULL
|
||||
AND last_turn_at >= $1::timestamp
|
||||
AND first_turn_at < $2::timestamp
|
||||
AND ($3::text IS NULL OR api_key = $3::text)
|
||||
UNION ALL
|
||||
SELECT {_SESSION_COLUMNS} FROM "LiteLLM_AutoRouterUserSession"
|
||||
SELECT {_SESSION_COLUMNS}, user_id AS comparison_user_id FROM "LiteLLM_AutoRouterUserSession"
|
||||
WHERE (($4::text IS NOT NULL AND user_id = $4::text) OR ($4::text IS NULL AND api_key = ''))
|
||||
AND last_turn_at >= $1::timestamp
|
||||
AND first_turn_at < $2::timestamp
|
||||
AND ($3::text IS NULL OR api_key = $3::text)
|
||||
),
|
||||
)
|
||||
"""
|
||||
|
||||
AUTOROUTER_BENCHMARKS_SQL: Final = f"""
|
||||
WITH {AUTOROUTER_SESSION_WINDOW_SQL},
|
||||
tier_maps AS (
|
||||
SELECT router_name, router_type, jsonb_object_agg(tier, tier_turns) AS tier_turns
|
||||
FROM (
|
||||
|
|
@ -95,6 +99,8 @@ SELECT
|
|||
COALESCE(SUM(saved_spend), 0)::float8 AS saved_spend,
|
||||
COALESCE(SUM(savings_estimated_turns), 0)::int AS savings_estimated_turns,
|
||||
COALESCE(SUM(savings_estimated_actual_spend), 0)::float8 AS savings_estimated_actual_spend,
|
||||
CASE WHEN BOOL_AND(savings_estimated_turns = turns AND classifier_cost_recorded_turns = turns)
|
||||
THEN SUM(classifier_cost)::float8 END AS savings_estimated_classifier_cost,
|
||||
COALESCE(SUM(savings_estimated_saved_spend), 0)::float8 AS savings_estimated_saved_spend,
|
||||
COALESCE(SUM(classifier_cost), 0)::float8 AS classifier_cost,
|
||||
COALESCE(SUM(classifier_cost_recorded_turns), 0)::int AS classifier_cost_recorded_turns,
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ POST /auto_router/validate_complexity_router_config - Dry-run the complexity-rou
|
|||
from collections.abc import Mapping, Sequence
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from itertools import chain, groupby
|
||||
from math import isclose
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Annotated, Final, Protocol
|
||||
from uuid import uuid4
|
||||
|
|
@ -31,6 +32,7 @@ from litellm.proxy.auth.auth_checks import (
|
|||
can_key_call_resolved_model,
|
||||
)
|
||||
from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
|
||||
from litellm.proxy.db.autorouter_savings_comparison import historical_session_comparisons
|
||||
from litellm.proxy.db.autorouter_session_rollup import (
|
||||
AUTOROUTER_BENCHMARKS_SQL,
|
||||
bounded_session_id,
|
||||
|
|
@ -651,7 +653,9 @@ class _SessionAggRow(BaseModel):
|
|||
saved_spend: float
|
||||
savings_estimated_turns: int = 0
|
||||
savings_estimated_actual_spend: float = 0.0
|
||||
savings_estimated_classifier_cost: float | None = None
|
||||
savings_estimated_saved_spend: float = 0.0
|
||||
savings_comparison_complete: bool = True
|
||||
classifier_cost: float
|
||||
classifier_cost_recorded_turns: int
|
||||
session_seconds: float
|
||||
|
|
@ -679,18 +683,25 @@ def _cache_bucket(turns: int, hits: int) -> AutoRouterCacheBucket:
|
|||
|
||||
|
||||
def _savings_cohort(
|
||||
turns: int, estimated_turns: int, actual_spend: float, saved_spend: float
|
||||
turns: int, estimated_turns: int, actual_spend: float, saved_spend: float, recorded_savings: float
|
||||
) -> tuple[float | None, float | None]:
|
||||
if turns > 0 and estimated_turns == 0:
|
||||
if turns > 0 and estimated_turns == 0 and recorded_savings == 0:
|
||||
return None, None
|
||||
return saved_spend, actual_spend + saved_spend
|
||||
if not isclose(saved_spend, recorded_savings, rel_tol=1e-9, abs_tol=1e-9):
|
||||
return recorded_savings, None
|
||||
return recorded_savings, actual_spend + recorded_savings
|
||||
|
||||
|
||||
def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals:
|
||||
return_misses: Final = row.return_turns - row.return_hits
|
||||
saved_spend, baseline_spend = _savings_cohort(
|
||||
row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend
|
||||
saved_spend, compared_baseline = _savings_cohort(
|
||||
row.turns,
|
||||
row.savings_estimated_turns,
|
||||
row.savings_estimated_actual_spend,
|
||||
row.savings_estimated_saved_spend,
|
||||
row.saved_spend,
|
||||
)
|
||||
baseline_spend: Final = compared_baseline if row.savings_comparison_complete else None
|
||||
sessions: Final = row.sessions
|
||||
return AutoRouterBenchmarkTotals(
|
||||
sessions=sessions,
|
||||
|
|
@ -701,13 +712,12 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals:
|
|||
spend=row.spend,
|
||||
savings_estimated_turns=row.savings_estimated_turns,
|
||||
savings_estimated_actual_spend=row.savings_estimated_actual_spend,
|
||||
savings_estimated_classifier_cost=row.savings_estimated_classifier_cost if baseline_spend is not None else None,
|
||||
saved_spend=saved_spend,
|
||||
classifier_cost=row.classifier_cost if row.classifier_cost_recorded_turns == row.turns else None,
|
||||
baseline_spend=baseline_spend,
|
||||
saved_pct=_pct(saved_spend, baseline_spend) if saved_spend is not None and baseline_spend is not None else None,
|
||||
saved_per_session=(row.savings_estimated_saved_spend / sessions if sessions else 0.0)
|
||||
if row.savings_estimated_turns == row.turns
|
||||
else None,
|
||||
saved_per_session=(saved_spend / sessions if sessions else 0.0) if saved_spend is not None else None,
|
||||
cache=AutoRouterCacheStats(
|
||||
coverage_pct=_pct(row.covered_turns, row.turns),
|
||||
hit_rate_pct=_pct(row.cache_hits, row.covered_turns),
|
||||
|
|
@ -739,6 +749,7 @@ def _benchmark_group(row: _SessionAggRow) -> AutoRouterBenchmarkGroup:
|
|||
saved_spend=totals.saved_spend,
|
||||
savings_estimated_turns=totals.savings_estimated_turns,
|
||||
savings_estimated_actual_spend=totals.savings_estimated_actual_spend,
|
||||
savings_estimated_classifier_cost=totals.savings_estimated_classifier_cost,
|
||||
classifier_cost=totals.classifier_cost,
|
||||
baseline_spend=totals.baseline_spend,
|
||||
saved_pct=totals.saved_pct,
|
||||
|
|
@ -772,7 +783,13 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow:
|
|||
saved_spend=sum(row.saved_spend for row in rows),
|
||||
savings_estimated_turns=sum(row.savings_estimated_turns for row in rows),
|
||||
savings_estimated_actual_spend=sum(row.savings_estimated_actual_spend for row in rows),
|
||||
savings_estimated_classifier_cost=(
|
||||
sum(row.savings_estimated_classifier_cost or 0.0 for row in rows)
|
||||
if all(row.savings_estimated_classifier_cost is not None for row in rows)
|
||||
else None
|
||||
),
|
||||
savings_estimated_saved_spend=sum(row.savings_estimated_saved_spend for row in rows),
|
||||
savings_comparison_complete=all(row.savings_comparison_complete for row in rows),
|
||||
classifier_cost=sum(row.classifier_cost for row in rows),
|
||||
classifier_cost_recorded_turns=sum(row.classifier_cost_recorded_turns for row in rows),
|
||||
session_seconds=sum(row.session_seconds for row in rows),
|
||||
|
|
@ -847,8 +864,8 @@ async def get_auto_router_benchmarks(
|
|||
Benchmarks for the auto-router dashboard: session shape, savings against the configured
|
||||
baseline, and prompt-caching behaviour bucketed by what the router did.
|
||||
|
||||
Reads session rollups folded once per request at spend-write time, so this endpoint
|
||||
never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that
|
||||
Reads session rollups folded once per request at spend-write time, with bounded
|
||||
retained-log recovery for historical comparisons. A user filter selects only turns attributed to that
|
||||
internal user when written; older key-only history remains outside user views. A session
|
||||
is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before
|
||||
end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is
|
||||
|
|
@ -882,7 +899,44 @@ async def get_auto_router_benchmarks(
|
|||
api_key,
|
||||
user_id,
|
||||
)
|
||||
rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ())
|
||||
recorded_rows: Final = _SESSION_AGG_ROWS.validate_python(raw_rows or ())
|
||||
comparisons: Final = (
|
||||
await historical_session_comparisons(
|
||||
prisma_client,
|
||||
start_day.isoformat(),
|
||||
(end_day + timedelta(days=1)).isoformat(),
|
||||
api_key,
|
||||
user_id,
|
||||
)
|
||||
if any(row.savings_estimated_turns < row.turns for row in recorded_rows)
|
||||
else MappingProxyType({})
|
||||
)
|
||||
covered_rows: Final = tuple(
|
||||
row.model_copy(
|
||||
update={
|
||||
**comparison.coverage_fields(row.saved_spend, row.turns),
|
||||
"savings_estimated_classifier_cost": comparison.classifier_cost,
|
||||
"savings_comparison_complete": comparison.complete and comparison.turns == row.turns,
|
||||
}
|
||||
)
|
||||
if (comparison := comparisons.get((row.router_name, row.router_type)))
|
||||
else row.model_copy(update={"savings_comparison_complete": row.savings_estimated_turns == row.turns})
|
||||
for row in recorded_rows
|
||||
)
|
||||
rows: Final = tuple(
|
||||
row.model_copy(
|
||||
update={
|
||||
"savings_comparison_complete": row.savings_comparison_complete
|
||||
and isclose(
|
||||
row.saved_spend,
|
||||
row.savings_estimated_saved_spend,
|
||||
rel_tol=1e-9,
|
||||
abs_tol=1e-9,
|
||||
),
|
||||
}
|
||||
)
|
||||
for row in covered_rows
|
||||
)
|
||||
groups: Final = (
|
||||
*(_benchmark_group(row) for row in rows),
|
||||
*_idle_router_groups(llm_router, frozenset((row.router_name, row.router_type) for row in rows)),
|
||||
|
|
@ -920,15 +974,43 @@ async def get_auto_router_session(
|
|||
|
||||
if prisma_client is None:
|
||||
raise HTTPException(status_code=500, detail=CommonProxyErrors.db_not_connected_error.value)
|
||||
row: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key(
|
||||
recorded: Final = await AutoRouterSessionRepository(prisma_client).find_latest_for_key(
|
||||
user_api_key_dict.api_key, bounded_session_id(session_id)
|
||||
)
|
||||
if row is None:
|
||||
if recorded is None:
|
||||
raise HTTPException(
|
||||
status_code=404, detail=f"No auto-routed turns recorded for session {session_id!r} under this key"
|
||||
)
|
||||
saved_spend, baseline_spend = _savings_cohort(
|
||||
row.turns, row.savings_estimated_turns, row.savings_estimated_actual_spend, row.savings_estimated_saved_spend
|
||||
comparisons: Final = (
|
||||
await historical_session_comparisons(
|
||||
prisma_client,
|
||||
recorded.first_turn_at.isoformat(),
|
||||
(recorded.last_turn_at + timedelta(microseconds=1)).isoformat(),
|
||||
user_api_key_dict.api_key,
|
||||
None,
|
||||
bounded_session_id(session_id),
|
||||
)
|
||||
if recorded.savings_estimated_turns < recorded.turns
|
||||
else MappingProxyType({})
|
||||
)
|
||||
comparison: Final = comparisons.get((recorded.router_name, recorded.router_type))
|
||||
row: Final = (
|
||||
recorded.model_copy(update=comparison.coverage_fields(recorded.saved_spend, recorded.turns))
|
||||
if comparison
|
||||
else recorded
|
||||
)
|
||||
saved_spend, compared_baseline = _savings_cohort(
|
||||
row.turns,
|
||||
row.savings_estimated_turns,
|
||||
row.savings_estimated_actual_spend,
|
||||
row.savings_estimated_saved_spend,
|
||||
row.saved_spend,
|
||||
)
|
||||
baseline_spend: Final = (
|
||||
compared_baseline
|
||||
if row.savings_estimated_turns == row.turns
|
||||
or (comparison and comparison.complete and comparison.turns == row.turns)
|
||||
else None
|
||||
)
|
||||
return AutoRouterSessionResponse(
|
||||
session_id=session_id,
|
||||
|
|
@ -943,7 +1025,7 @@ async def get_auto_router_session(
|
|||
baseline_spend=baseline_spend if row.savings_estimated_turns == row.turns else None,
|
||||
savings_estimated_baseline_spend=baseline_spend,
|
||||
baseline_model=row.baseline_model,
|
||||
baseline_models=row.savings_estimated_baseline_models,
|
||||
baseline_models=row.baseline_models,
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -224,19 +224,24 @@ class AutoRouterBenchmarkTotals(BaseModel):
|
|||
"subtotal recording, and zero for an empty window"
|
||||
)
|
||||
savings_estimated_turns: int = Field(
|
||||
description="Turns covered by the current savings estimator; legacy estimates are excluded"
|
||||
description="Requests with a matching savings comparison, including historical recorded estimates"
|
||||
)
|
||||
savings_estimated_actual_spend: float = Field(
|
||||
description="Actual spend, including classifier cost, for covered turns only"
|
||||
)
|
||||
savings_estimated_classifier_cost: float | None = Field(
|
||||
default=None,
|
||||
description="Classifier cost included in the matching historical and newer savings comparison; "
|
||||
"null when classification costs for those requests are unavailable",
|
||||
)
|
||||
saved_spend: float | None = Field(
|
||||
description="Signed savings for covered turns only; null when traffic has no current estimates"
|
||||
description="Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates"
|
||||
)
|
||||
baseline_spend: float | None = Field(description="Estimated single-model cost for covered turns only")
|
||||
saved_pct: float | None = Field(description="Covered savings over covered baseline spend, as a percentage")
|
||||
saved_per_session: float | None = Field(
|
||||
description="Average session savings; unavailable unless every turn is covered"
|
||||
saved_pct: float | None = Field(
|
||||
description="Total recorded savings over the matching historical and current baseline; null when costs are unavailable"
|
||||
)
|
||||
saved_per_session: float | None = Field(description="Recorded savings per session, including historical estimates")
|
||||
cache: AutoRouterCacheStats
|
||||
|
||||
|
||||
|
|
@ -268,12 +273,14 @@ class AutoRouterSessionResponse(BaseModel):
|
|||
last_model: str = Field(description="The deployment model the most recent turn was routed to")
|
||||
spend: float = Field(description="What the session's routed traffic actually cost, classifier calls included")
|
||||
savings_estimated_turns: int = Field(
|
||||
description="Turns covered by the current savings estimator; legacy estimates are excluded"
|
||||
description="Requests with a matching savings comparison, including historical recorded estimates"
|
||||
)
|
||||
savings_estimated_actual_spend: float = Field(
|
||||
description="Actual spend, including classifier cost, for covered turns only"
|
||||
)
|
||||
saved_spend: float | None = Field(description="Estimated savings for covered turns only, net of classifier cost")
|
||||
saved_spend: float | None = Field(
|
||||
description="Recorded historical savings plus newer estimates, net of classifier cost"
|
||||
)
|
||||
baseline_spend: float | None = Field(
|
||||
description="Estimated single-model cost; unavailable unless every turn is covered"
|
||||
)
|
||||
|
|
@ -281,14 +288,14 @@ class AutoRouterSessionResponse(BaseModel):
|
|||
description="Estimated single-model cost for covered turns only"
|
||||
)
|
||||
baseline_model: str | None = Field(
|
||||
description="The savings baseline most covered turns were priced against, recorded turn by "
|
||||
description="The savings baseline recorded by most session turns, including historical turns, recorded turn by "
|
||||
"turn, so it still names the counterfactual after the router is reconfigured or removed. None when no "
|
||||
"turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, "
|
||||
"which derive no baseline and so report no savings"
|
||||
)
|
||||
baseline_models: Mapping[str, int] = Field(
|
||||
description="Covered turns priced against each baseline model; more than one entry means the router's "
|
||||
"baseline changed mid-session and baseline_spend mixes both"
|
||||
description="Session turns recording each baseline model; more than one entry means the router's "
|
||||
"baseline changed mid-session; these counts do not imply savings coverage"
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ tests/test_litellm/proxy/db/test_autorouter_session_rollup.py.
|
|||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import time
|
||||
import uuid
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
|
@ -24,6 +25,10 @@ from litellm.proxy.db.autorouter_session_rollup import (
|
|||
flush_autorouter_turn_transactions,
|
||||
)
|
||||
from litellm.proxy.db.db_transaction_queue.spend_log_cleanup import SpendLogCleanup
|
||||
from litellm.proxy.db.autorouter_savings_comparison import (
|
||||
HISTORICAL_SESSION_COMPARISONS_SQL,
|
||||
SessionSavingsComparison,
|
||||
)
|
||||
|
||||
pytestmark = pytest.mark.asyncio(loop_scope="session")
|
||||
|
||||
|
|
@ -91,6 +96,66 @@ async def _row(db, key: str, session_id: str = "s1", router: str = "auto-1") ->
|
|||
return rows[0]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("historical_saved, damaged, user_id, split_sessions, current_classifier", [
|
||||
(29.5, None, None, False, 0.2), (29.5, None, "owner", False, 0.2), (0.0, None, None, False, 0.2),
|
||||
(-3.0, None, None, False, 0.2), (29.5, "missing", None, False, 0.2), (29.5, "cost", None, False, 0.2),
|
||||
(0.0, "missing", None, False, 0.2), (29.5, None, None, True, 0.2), (29.5, None, None, False, 0.0),
|
||||
])
|
||||
async def test_historical_and_new_savings_compare_matching_costs_and_exclude_unknown_requests(
|
||||
db: Prisma, historical_saved: float, damaged: str | None, user_id: str | None, split_sessions: bool,
|
||||
current_classifier: float,
|
||||
) -> None:
|
||||
async with db.tx() as tx:
|
||||
for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession", "LiteLLM_SpendLogs"):
|
||||
await tx.execute_raw(f'CREATE TEMP TABLE "{table}" (LIKE public."{table}" INCLUDING ALL) ON COMMIT DROP')
|
||||
for name, spend, saved, classifier, estimated in (
|
||||
("historical", 9.0, historical_saved, 0.1, False),
|
||||
("current", 1.0, 0.5, current_classifier, True),
|
||||
("unknown", 99.0, 0.0, 3.0, False),
|
||||
):
|
||||
session_id: Final = "s2" if split_sessions and name == "current" else "s1"
|
||||
await _turn(tx, "key", "model", T0, spend=spend, saved=saved, classifier_cost=classifier,
|
||||
estimated=estimated, session_id=session_id)
|
||||
metadata: Final = {
|
||||
"routing_decision": {"router_model_name": "auto-1", **({"classifier_cost": classifier} if classifier else {})},
|
||||
"autorouter_savings": saved if name != "unknown" else None,
|
||||
**({"autorouter_savings_estimate": {
|
||||
"version": 3, "status": "estimated" if estimated else "unknown",
|
||||
}} if name != "historical" else {}),
|
||||
}
|
||||
await tx.execute_raw('''INSERT INTO "LiteLLM_SpendLogs"
|
||||
(request_id,api_key,session_id,model,"user","startTime","endTime",call_type,
|
||||
spend,prompt_tokens,completion_tokens,status,metadata)
|
||||
VALUES ($1,'key',$5,'model','owner',$2::timestamp,$2::timestamp,'acompletion',
|
||||
$3::float8,100,0,'success',$4::jsonb)
|
||||
''', name, T0.isoformat(), spend - classifier, json.dumps(metadata), session_id)
|
||||
await tx.execute_raw('''INSERT INTO "LiteLLM_AutoRouterUserSession"
|
||||
(user_id,api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model,
|
||||
turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend,
|
||||
savings_estimated_saved_spend)
|
||||
SELECT 'owner',api_key,session_id,router_name,router_type,first_turn_at,last_turn_at,last_model,
|
||||
turns,total_tokens,spend,saved_spend,savings_estimated_turns,savings_estimated_actual_spend,
|
||||
savings_estimated_saved_spend FROM "LiteLLM_AutoRouterSession"
|
||||
''')
|
||||
if damaged == "missing":
|
||||
await tx.execute_raw('DELETE FROM "LiteLLM_SpendLogs" WHERE request_id = \'historical\'')
|
||||
elif damaged == "cost":
|
||||
await tx.execute_raw('UPDATE "LiteLLM_SpendLogs" SET spend = 1 WHERE request_id = \'historical\'')
|
||||
rows: Final = await tx.query_raw(
|
||||
HISTORICAL_SESSION_COMPARISONS_SQL, "2026-08-01", "2026-08-02", "key", user_id, None,
|
||||
)
|
||||
comparison: Final = SessionSavingsComparison.model_validate(rows[0])
|
||||
assert comparison.saved_spend == historical_saved + 0.5
|
||||
assert comparison.complete is (damaged is None)
|
||||
assert comparison.classifier_cost == (pytest.approx(0.1 + current_classifier) if damaged is None else None)
|
||||
assert comparison.coverage_fields(historical_saved + 0.5, 4) == {}
|
||||
assert comparison.coverage_fields(historical_saved + 0.5, 3) == ({
|
||||
"savings_estimated_turns": 2,
|
||||
"savings_estimated_actual_spend": 10.0,
|
||||
"savings_estimated_saved_spend": historical_saved + 0.5,
|
||||
} if damaged is None else {})
|
||||
|
||||
|
||||
async def test_every_turn_lands_in_exactly_one_bucket(db):
|
||||
key = f"k-{uuid.uuid4()}"
|
||||
await _turn(db, key, "A", T0, ttl=300)
|
||||
|
|
|
|||
|
|
@ -676,6 +676,7 @@ class TestAutoRouterBenchmarks:
|
|||
saved_spend=30.0,
|
||||
savings_estimated_turns=40,
|
||||
savings_estimated_actual_spend=10.0,
|
||||
savings_estimated_classifier_cost=0.4,
|
||||
savings_estimated_saved_spend=30.0,
|
||||
classifier_cost=0.4,
|
||||
classifier_cost_recorded_turns=40,
|
||||
|
|
@ -701,6 +702,7 @@ class TestAutoRouterBenchmarks:
|
|||
assert totals.avg_tokens_per_session == 1000.0
|
||||
assert totals.baseline_spend == 40.0
|
||||
assert totals.saved_pct == 75.0
|
||||
assert totals.savings_estimated_classifier_cost == 0.4
|
||||
assert totals.saved_per_session == 7.5
|
||||
assert totals.cache.coverage_pct == 95.0
|
||||
assert totals.cache.hit_rate_pct == pytest.approx(73.7)
|
||||
|
|
@ -720,7 +722,7 @@ class TestAutoRouterBenchmarks:
|
|||
assert totals.classifier_cost == 0.4
|
||||
|
||||
@pytest.mark.parametrize("estimated_turns", [0, 4])
|
||||
def test_savings_compare_only_the_current_estimated_cohort(self, estimated_turns: int) -> None:
|
||||
def test_recorded_savings_survive_when_historical_comparison_costs_are_missing(self, estimated_turns: int) -> None:
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals
|
||||
|
||||
row: Final = self.ROW.model_copy(
|
||||
|
|
@ -733,10 +735,11 @@ class TestAutoRouterBenchmarks:
|
|||
totals: Final = _benchmark_totals(row)
|
||||
assert totals.spend == 10.0
|
||||
assert totals.savings_estimated_turns == estimated_turns
|
||||
assert totals.saved_spend == (-0.5 if estimated_turns else None)
|
||||
assert totals.baseline_spend == (1.5 if estimated_turns else None)
|
||||
assert totals.saved_pct == (pytest.approx(-33.3) if estimated_turns else None)
|
||||
assert totals.saved_per_session is None
|
||||
assert totals.saved_spend == 30.0
|
||||
assert totals.baseline_spend is None
|
||||
assert totals.savings_estimated_classifier_cost is None
|
||||
assert totals.saved_pct is None
|
||||
assert totals.saved_per_session == 7.5
|
||||
|
||||
def test_an_empty_window_folds_to_zeros(self):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import (
|
||||
|
|
@ -765,6 +768,7 @@ class TestAutoRouterBenchmarks:
|
|||
"spend": 0.0,
|
||||
"savings_estimated_turns": 10,
|
||||
"savings_estimated_actual_spend": 0.0,
|
||||
"savings_estimated_classifier_cost": 0.0,
|
||||
}
|
||||
)
|
||||
summed = _summed_agg_row([self.ROW, other])
|
||||
|
|
@ -773,6 +777,9 @@ class TestAutoRouterBenchmarks:
|
|||
assert summed.turns == 50
|
||||
assert totals.avg_turns_per_session == 10.0
|
||||
assert totals.spend == 10.0
|
||||
assert totals.savings_estimated_classifier_cost == 0.4
|
||||
unknown_cost = other.model_copy(update={"savings_estimated_classifier_cost": None})
|
||||
assert _benchmark_totals(_summed_agg_row([self.ROW, unknown_cost])).savings_estimated_classifier_cost is None
|
||||
|
||||
def test_tier_names_stay_scoped_to_the_router_type_that_recorded_them(self):
|
||||
quality = self.ROW.model_copy(
|
||||
|
|
@ -1128,13 +1135,13 @@ class TestAutoRouterSession:
|
|||
"turns": turns,
|
||||
"last_model": "anthropic/claude-sonnet-5",
|
||||
"spend": spend,
|
||||
"saved_spend": (0.24 if turns == 3 else -0.04) if estimated else None,
|
||||
"saved_spend": 0.24,
|
||||
"savings_estimated_turns": 3 if estimated else 0,
|
||||
"savings_estimated_actual_spend": 0.14 if estimated else 0.0,
|
||||
"baseline_spend": pytest.approx(0.38) if turns == 3 else None,
|
||||
"savings_estimated_baseline_spend": pytest.approx(0.38 if turns == 3 else 0.1) if estimated else None,
|
||||
"baseline_model": "anthropic/claude-opus-5" if estimated else None,
|
||||
"baseline_models": {"anthropic/claude-opus-5": 3} if estimated else {},
|
||||
"savings_estimated_baseline_spend": pytest.approx(0.38) if turns == 3 else None,
|
||||
"baseline_model": "anthropic/claude-opus-5",
|
||||
"baseline_models": {"anthropic/claude-opus-5": 3},
|
||||
}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -1168,11 +1175,10 @@ class TestAutoRouterSession:
|
|||
assert response.router_name == "new-auto"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_reconfigured_router_keeps_the_label_the_money_was_priced_against(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
@pytest.mark.parametrize("mixed", [False, True])
|
||||
async def test_session_preserves_historical_baseline_labels(
|
||||
self, monkeypatch: pytest.MonkeyPatch, mixed: bool
|
||||
):
|
||||
# The proxy's router now prices against a different baseline, but the row's money was priced
|
||||
# against opus for two of three turns, and the label says so; the full split is on the response.
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import get_auto_router_session
|
||||
|
||||
priced = {"anthropic/claude-opus-5": 2, "anthropic/claude-sonnet-5": 1}
|
||||
|
|
@ -1183,14 +1189,15 @@ class TestAutoRouterSession:
|
|||
**self.ROW,
|
||||
"api_key": ADMIN.api_key,
|
||||
"session_id": "s",
|
||||
"baseline_models": {"old-baseline": 100},
|
||||
"baseline_models": {"old-baseline": 100, **({"unknown-baseline": 200} if mixed else {})},
|
||||
"savings_estimated_turns": 1,
|
||||
"savings_estimated_baseline_models": priced,
|
||||
}
|
||||
],
|
||||
)
|
||||
response = await get_auto_router_session(user_api_key_dict=ADMIN, session_id="s")
|
||||
assert response.baseline_model == "anthropic/claude-opus-5"
|
||||
assert response.baseline_models == priced
|
||||
assert response.baseline_model == (None if mixed else "old-baseline")
|
||||
assert response.baseline_models == {"old-baseline": 100, **({"unknown-baseline": 200} if mixed else {})}
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_an_oversized_client_session_id_is_bounded_like_the_writer_bounded_it(
|
||||
|
|
|
|||
|
|
@ -158,35 +158,49 @@ describe("AutoRouterBenchmarksTab", () => {
|
|||
});
|
||||
|
||||
it.each([
|
||||
{ estimatedTurns: 0, saved: null, pct: null },
|
||||
{ estimatedTurns: 10, saved: -0.5, pct: -33.3 },
|
||||
{ estimatedTurns: 10, saved: 0, pct: 0 },
|
||||
])("preserves costs for $estimatedTurns estimated turns with savings $saved", ({ estimatedTurns, saved, pct }) => {
|
||||
const cohort = {
|
||||
{ estimatedTurns: 0, actual: 0, saved: null, pct: null },
|
||||
{ estimatedTurns: 0, actual: 0, saved: 30, pct: null },
|
||||
{ estimatedTurns: 10, actual: 2, saved: -0.5, pct: -33.3 },
|
||||
{ estimatedTurns: 10, actual: 2, saved: 0, pct: 0 },
|
||||
{ estimatedTurns: 40, actual: 10, saved: 30, pct: 75 },
|
||||
])("compares matching old and new requests with savings $saved", ({ estimatedTurns, actual, saved, pct }) => {
|
||||
const comparison = {
|
||||
spend: actual + 99,
|
||||
savings_estimated_turns: estimatedTurns,
|
||||
savings_estimated_actual_spend: estimatedTurns ? 2 : 0,
|
||||
savings_estimated_actual_spend: actual,
|
||||
savings_estimated_classifier_cost: 0.1,
|
||||
saved_spend: saved,
|
||||
baseline_spend: estimatedTurns ? 2 + (saved ?? 0) : null,
|
||||
baseline_spend: estimatedTurns ? actual + (saved ?? 0) : null,
|
||||
saved_pct: pct,
|
||||
saved_per_session: null,
|
||||
};
|
||||
const partial = totals(cohort);
|
||||
mockHook({ data: response([], partial) });
|
||||
mockHook({
|
||||
data: response([], totals(comparison)),
|
||||
});
|
||||
renderTab();
|
||||
expect(screen.getByText("Estimated savings on covered turns")).toBeInTheDocument();
|
||||
expect(screen.getByText(`${estimatedTurns} of 3,073 turns estimated`)).toBeInTheDocument();
|
||||
expect(screen.getByText("$359.86")).toBeInTheDocument();
|
||||
expect(screen.getByText("Actual spend on covered turns")).toBeInTheDocument();
|
||||
expect(screen.getByText("Estimated baseline spend on covered turns")).toBeInTheDocument();
|
||||
expect(screen.getAllByText("Unavailable")).toHaveLength(estimatedTurns ? 1 : 3);
|
||||
if (saved === 0) {
|
||||
expect(screen.getByText("0%")).toBeInTheDocument();
|
||||
expect(screen.getAllByText("$2.00")).toHaveLength(2);
|
||||
} else if (estimatedTurns) {
|
||||
expect(screen.getByText("-$0.5000")).toBeInTheDocument();
|
||||
expect(screen.getByText("+33%")).toBeInTheDocument();
|
||||
} else {
|
||||
expect(screen.queryByText("+0%")).not.toBeInTheDocument();
|
||||
expect(screen.getByText("Total estimated savings")).toBeInTheDocument();
|
||||
expect(screen.getAllByRole("definition").map((row) => row.textContent)).toEqual(
|
||||
estimatedTurns
|
||||
? [
|
||||
`$${actual.toFixed(2)}`,
|
||||
`$${(actual - 0.1).toFixed(2)}`,
|
||||
"$0.1000",
|
||||
`$${(actual + (saved ?? 0)).toFixed(2)}`,
|
||||
]
|
||||
: ["Unavailable", "Unavailable", "Unavailable", "Unavailable"],
|
||||
);
|
||||
expect(screen.queryByText("Actual spend on covered turns")).not.toBeInTheDocument();
|
||||
expect(screen.getByLabelText("question-circle")).toBeInTheDocument();
|
||||
if (estimatedTurns) {
|
||||
expect(screen.getByText(`Savings based on ${estimatedTurns} of 3,073 requests`)).toBeInTheDocument();
|
||||
const sign = pct && pct > 0 ? "-" : "+";
|
||||
const badge = pct === 0 ? "0%" : `${sign}${Math.abs(pct ?? 0).toFixed(0)}%`;
|
||||
expect(screen.getByText(badge)).toBeInTheDocument();
|
||||
} else if (saved != null) {
|
||||
expect(screen.getByText("$30.00")).toBeInTheDocument();
|
||||
expect(
|
||||
screen.getByText("Historical savings are included. Matching cost details are unavailable."),
|
||||
).toBeInTheDocument();
|
||||
}
|
||||
});
|
||||
|
||||
|
|
@ -216,7 +230,7 @@ describe("AutoRouterBenchmarksTab", () => {
|
|||
expect(screen.getByText("-86%")).toBeInTheDocument();
|
||||
expect(screen.getByText("Actual auto-router spend")).toBeInTheDocument();
|
||||
expect(screen.getByText("$359.86")).toBeInTheDocument();
|
||||
expect(screen.getByText("Estimated spend at highest-tier model")).toBeInTheDocument();
|
||||
expect(screen.getByText("Estimated baseline spend")).toBeInTheDocument();
|
||||
expect(screen.getByText("$2,534.45")).toBeInTheDocument();
|
||||
expect(screen.getByText("32.7")).toBeInTheDocument();
|
||||
expect(screen.getByText("2.1h")).toBeInTheDocument();
|
||||
|
|
@ -242,17 +256,20 @@ describe("AutoRouterBenchmarksTab", () => {
|
|||
expect(screen.getAllByText("$10,126.28").length).toBeGreaterThan(0);
|
||||
});
|
||||
|
||||
it.each([null, undefined])("keeps totals when the classification breakdown is %s", (classifier_cost) => {
|
||||
const stats = totals({ classifier_cost });
|
||||
mockHook({ data: response([group(stats)], stats) });
|
||||
renderTab();
|
||||
it.each([null, undefined])(
|
||||
"keeps eligible totals when the classification breakdown is %s",
|
||||
(savings_estimated_classifier_cost) => {
|
||||
const stats = totals({ savings_estimated_turns: 30, savings_estimated_classifier_cost });
|
||||
mockHook({ data: response([group(stats)], stats) });
|
||||
renderTab();
|
||||
|
||||
expect(screen.getAllByText("Unavailable")).toHaveLength(2);
|
||||
expect(screen.queryByText(/\/ 1K turns/)).not.toBeInTheDocument();
|
||||
expect(screen.getByText("$359.86")).toBeInTheDocument();
|
||||
expect(screen.getByText("$2,174.59")).toBeInTheDocument();
|
||||
expect(screen.getByText(/some usage predates classification-cost tracking/)).toBeInTheDocument();
|
||||
});
|
||||
expect(screen.getAllByText("Unavailable")).toHaveLength(2);
|
||||
expect(screen.queryByText(/\/ 1K turns/)).not.toBeInTheDocument();
|
||||
expect(screen.getByText("$359.86")).toBeInTheDocument();
|
||||
expect(screen.getByText("$2,174.59")).toBeInTheDocument();
|
||||
expect(screen.getByText(/some usage predates classification-cost tracking/)).toBeInTheDocument();
|
||||
},
|
||||
);
|
||||
|
||||
it("pairs the savings with the session count it was earned over, in its own tile", () => {
|
||||
mockHook({ data: response([group(), group({ router_name: "gpt-auto" })]) });
|
||||
|
|
@ -275,7 +292,7 @@ describe("AutoRouterBenchmarksTab", () => {
|
|||
"Actual auto-router spend",
|
||||
"LLM spend",
|
||||
"Classification cost($2.00 / 1K turns)",
|
||||
"Estimated spend at highest-tier model",
|
||||
"Estimated baseline spend",
|
||||
]);
|
||||
expect(values).toEqual(["$359.86", "$353.71", "$6.15", "$2,534.45"]);
|
||||
});
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@
|
|||
import { Separator } from "@/components/ui/separator";
|
||||
import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table";
|
||||
import { Tabs, TabsContent, TabsList, TabsTrigger } from "@/components/ui/tabs";
|
||||
import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip";
|
||||
import { SimpleTooltip, Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from "@/components/ui/tooltip";
|
||||
import { ApiError } from "@/lib/http/client";
|
||||
import { formatNumberWithCommas } from "@/utils/dataUtils";
|
||||
|
||||
|
|
@ -52,15 +52,17 @@ const Metric: React.FC<{ label: string; value: string; hint?: string }> = ({ lab
|
|||
</Card>
|
||||
);
|
||||
|
||||
const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued?: boolean }> = ({
|
||||
const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued?: boolean; tooltip?: string }> = ({
|
||||
label,
|
||||
value,
|
||||
hint,
|
||||
subdued,
|
||||
tooltip,
|
||||
}) => (
|
||||
<dl className="flex flex-wrap items-baseline justify-between gap-x-6 gap-y-1 py-2">
|
||||
<dt className="flex min-w-0 flex-wrap items-baseline gap-x-2 text-sm text-muted-foreground">
|
||||
{label}
|
||||
{tooltip && <SimpleTooltip content={tooltip} />}
|
||||
{hint && <span className="text-xs">{hint}</span>}
|
||||
</dt>
|
||||
<dd
|
||||
|
|
@ -73,14 +75,17 @@ const SpendRow: React.FC<{ label: string; value: string; hint?: string; subdued?
|
|||
|
||||
const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
|
||||
const stats = view.stats;
|
||||
const cheaper = stats.saved_spend != null && stats.saved_spend >= 0;
|
||||
const cheaper = stats.saved_pct != null && stats.saved_pct >= 0;
|
||||
const completeCoverage = stats.savings_estimated_turns === stats.turns;
|
||||
const coveredClassifierCost =
|
||||
stats.savings_estimated_classifier_cost ?? (completeCoverage ? stats.classifier_cost : null);
|
||||
const classifierCost = stats.baseline_spend == null ? null : coveredClassifierCost;
|
||||
return (
|
||||
<Card className="overflow-hidden py-0">
|
||||
<div className="grid md:grid-cols-[minmax(0,1fr)_minmax(0,1fr)]">
|
||||
<div className="flex flex-col items-center justify-center gap-2 p-6">
|
||||
<p className="text-xs font-semibold uppercase tracking-wider text-muted-foreground">
|
||||
{completeCoverage ? "Total estimated savings" : "Estimated savings on covered turns"}
|
||||
Total estimated savings
|
||||
</p>
|
||||
<div className="flex flex-wrap items-center justify-center gap-3">
|
||||
<p className="min-w-0 break-all text-center text-4xl font-semibold tracking-tight text-foreground xl:text-6xl">
|
||||
|
|
@ -91,53 +96,57 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
|
|||
variant="secondary"
|
||||
className={`h-6 px-2.5 text-sm ${cheaper ? "bg-success/10 text-success" : "bg-destructive/10 text-destructive"}`}
|
||||
>
|
||||
{stats.saved_spend !== 0 && (cheaper ? "-" : "+")}
|
||||
{stats.saved_pct !== 0 && (cheaper ? "-" : "+")}
|
||||
{Math.abs(stats.saved_pct).toFixed(0)}%
|
||||
</Badge>
|
||||
)}
|
||||
</div>
|
||||
<p className="text-center text-xs text-muted-foreground">
|
||||
{stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()} turns estimated
|
||||
</p>
|
||||
{!completeCoverage && (
|
||||
{stats.baseline_spend != null && !completeCoverage && (
|
||||
<p className="text-center text-xs text-muted-foreground">
|
||||
Turns without a current estimate are excluded, including older estimates.
|
||||
Savings based on {stats.savings_estimated_turns.toLocaleString()} of {stats.turns.toLocaleString()}{" "}
|
||||
requests
|
||||
</p>
|
||||
)}
|
||||
{stats.saved_spend != null && stats.baseline_spend == null && (
|
||||
<p className="text-center text-xs text-muted-foreground">
|
||||
Historical savings are included. Matching cost details are unavailable.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="flex flex-col justify-center border-t p-6 md:border-t-0 md:border-l">
|
||||
<SpendRow label="Actual auto-router spend" value={usd(stats.spend)} />
|
||||
<SpendRow
|
||||
label="Actual auto-router spend"
|
||||
value={stats.baseline_spend == null ? "Unavailable" : usd(stats.savings_estimated_actual_spend)}
|
||||
tooltip="Savings, actual spend, and baseline include historical and newer requests with recorded savings estimates. Requests without estimates are excluded. Actual spend includes classification costs."
|
||||
/>
|
||||
<div className="mb-3 border-l-2 pl-4">
|
||||
<SpendRow
|
||||
subdued
|
||||
label="LLM spend"
|
||||
value={stats.classifier_cost == null ? "Unavailable" : usd(stats.spend - stats.classifier_cost)}
|
||||
value={
|
||||
classifierCost == null ? "Unavailable" : usd(stats.savings_estimated_actual_spend - classifierCost)
|
||||
}
|
||||
/>
|
||||
<SpendRow
|
||||
subdued
|
||||
label="Classification cost"
|
||||
value={stats.classifier_cost == null ? "Unavailable" : usd(stats.classifier_cost)}
|
||||
value={classifierCost == null ? "Unavailable" : usd(classifierCost)}
|
||||
hint={
|
||||
stats.classifier_cost == null
|
||||
classifierCost == null
|
||||
? undefined
|
||||
: classificationRatePer1kTurns(stats.classifier_cost, stats.turns)
|
||||
: classificationRatePer1kTurns(classifierCost, stats.savings_estimated_turns)
|
||||
}
|
||||
/>
|
||||
</div>
|
||||
{stats.classifier_cost == null && (
|
||||
{stats.baseline_spend != null && classifierCost == null && (
|
||||
<p className="mb-3 text-xs text-muted-foreground">
|
||||
Breakdown unavailable because some usage predates classification-cost tracking.
|
||||
</p>
|
||||
)}
|
||||
<Separator />
|
||||
{!completeCoverage && (
|
||||
<SpendRow label="Actual spend on covered turns" value={usd(stats.savings_estimated_actual_spend)} />
|
||||
)}
|
||||
<SpendRow
|
||||
label={
|
||||
completeCoverage ? "Estimated spend at highest-tier model" : "Estimated baseline spend on covered turns"
|
||||
}
|
||||
label="Estimated baseline spend"
|
||||
value={stats.baseline_spend == null ? "Unavailable" : usd(stats.baseline_spend)}
|
||||
/>
|
||||
</div>
|
||||
|
|
@ -307,12 +316,11 @@ const BenchmarksBody: React.FC<BenchmarksBodyProps> = ({ isPending, error, data,
|
|||
</div>
|
||||
|
||||
<p className="text-xs text-muted-foreground">
|
||||
Compares covered turns with the estimated cost of using the router's highest-tier baseline model. Estimates
|
||||
use registered requests since tracking began, matching cache prefixes and expiry, and the actual response
|
||||
length. Total actual spend includes every turn; savings and baseline spend include only turns with a current
|
||||
estimate, including turns with zero savings. Savings are net of recorded LLM classification cost. Classification
|
||||
cost per 1K turns is averaged over all auto-router turns, including those that skip classification. The range
|
||||
counts whole sessions that overlap it, so totals can differ from savings views that group usage by UTC day.
|
||||
Savings, actual spend, and baseline compare the same historical and newer requests with recorded estimates,
|
||||
including zero or negative savings. Requests without estimates are excluded. Savings are net of recorded LLM
|
||||
classification cost. If historical cost details are unavailable, recorded savings remain visible without a
|
||||
baseline or percentage. The range counts whole sessions that overlap it, so totals can differ from savings views
|
||||
that group usage by UTC day.
|
||||
</p>
|
||||
|
||||
<div className="space-y-4">
|
||||
|
|
|
|||
|
|
@ -92,7 +92,7 @@ describe("KeyAutoRouterUsageTab", () => {
|
|||
expect(screen.getByText("Classification cost")).toBeInTheDocument();
|
||||
expect(screen.getByText("$0.2500")).toBeInTheDocument();
|
||||
expect(screen.getByText("($62.50 / 1K turns)")).toBeInTheDocument();
|
||||
expect(screen.getByText("Estimated spend at highest-tier model")).toBeInTheDocument();
|
||||
expect(screen.getByText("Estimated baseline spend")).toBeInTheDocument();
|
||||
expect(screen.getByText("$10.00")).toBeInTheDocument();
|
||||
expect(screen.getByText("Auto-router prompt caching")).toBeInTheDocument();
|
||||
expect(screen.getAllByText("50.0%").length).toBeGreaterThan(0);
|
||||
|
|
|
|||
38
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
38
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -1263,8 +1263,8 @@ export interface paths {
|
|||
* @description Benchmarks for the auto-router dashboard: session shape, savings against the configured
|
||||
* baseline, and prompt-caching behaviour bucketed by what the router did.
|
||||
*
|
||||
* Reads session rollups folded once per request at spend-write time, so this endpoint
|
||||
* never scans LiteLLM_SpendLogs. A user filter selects only turns attributed to that
|
||||
* Reads session rollups folded once per request at spend-write time, with bounded
|
||||
* retained-log recovery for historical comparisons. A user filter selects only turns attributed to that
|
||||
* internal user when written; older key-only history remains outside user views. A session
|
||||
* is in the window when it overlaps it: its last turn is on or after start_date and its first turn is on or before
|
||||
* end_date. Overall hit rate is over telemetry-bearing turns; each bucket's hit rate is
|
||||
|
|
@ -24899,17 +24899,17 @@ export interface components {
|
|||
router_type: string;
|
||||
/**
|
||||
* Saved Pct
|
||||
* @description Covered savings over covered baseline spend, as a percentage
|
||||
* @description Total recorded savings over the matching historical and current baseline; null when costs are unavailable
|
||||
*/
|
||||
saved_pct: number | null;
|
||||
/**
|
||||
* Saved Per Session
|
||||
* @description Average session savings; unavailable unless every turn is covered
|
||||
* @description Recorded savings per session, including historical estimates
|
||||
*/
|
||||
saved_per_session: number | null;
|
||||
/**
|
||||
* Saved Spend
|
||||
* @description Signed savings for covered turns only; null when traffic has no current estimates
|
||||
* @description Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates
|
||||
*/
|
||||
saved_spend: number | null;
|
||||
/**
|
||||
|
|
@ -24917,9 +24917,14 @@ export interface components {
|
|||
* @description Actual spend, including classifier cost, for covered turns only
|
||||
*/
|
||||
savings_estimated_actual_spend: number;
|
||||
/**
|
||||
* Savings Estimated Classifier Cost
|
||||
* @description Classifier cost included in the matching historical and newer savings comparison; null when classification costs for those requests are unavailable
|
||||
*/
|
||||
savings_estimated_classifier_cost?: number | null;
|
||||
/**
|
||||
* Savings Estimated Turns
|
||||
* @description Turns covered by the current savings estimator; legacy estimates are excluded
|
||||
* @description Requests with a matching savings comparison, including historical recorded estimates
|
||||
*/
|
||||
savings_estimated_turns: number;
|
||||
/** Sessions */
|
||||
|
|
@ -24963,17 +24968,17 @@ export interface components {
|
|||
classifier_cost: number | null;
|
||||
/**
|
||||
* Saved Pct
|
||||
* @description Covered savings over covered baseline spend, as a percentage
|
||||
* @description Total recorded savings over the matching historical and current baseline; null when costs are unavailable
|
||||
*/
|
||||
saved_pct: number | null;
|
||||
/**
|
||||
* Saved Per Session
|
||||
* @description Average session savings; unavailable unless every turn is covered
|
||||
* @description Recorded savings per session, including historical estimates
|
||||
*/
|
||||
saved_per_session: number | null;
|
||||
/**
|
||||
* Saved Spend
|
||||
* @description Signed savings for covered turns only; null when traffic has no current estimates
|
||||
* @description Recorded historical savings plus newer estimates; null when traffic has no recorded savings estimates
|
||||
*/
|
||||
saved_spend: number | null;
|
||||
/**
|
||||
|
|
@ -24981,9 +24986,14 @@ export interface components {
|
|||
* @description Actual spend, including classifier cost, for covered turns only
|
||||
*/
|
||||
savings_estimated_actual_spend: number;
|
||||
/**
|
||||
* Savings Estimated Classifier Cost
|
||||
* @description Classifier cost included in the matching historical and newer savings comparison; null when classification costs for those requests are unavailable
|
||||
*/
|
||||
savings_estimated_classifier_cost?: number | null;
|
||||
/**
|
||||
* Savings Estimated Turns
|
||||
* @description Turns covered by the current savings estimator; legacy estimates are excluded
|
||||
* @description Requests with a matching savings comparison, including historical recorded estimates
|
||||
*/
|
||||
savings_estimated_turns: number;
|
||||
/** Sessions */
|
||||
|
|
@ -25260,12 +25270,12 @@ export interface components {
|
|||
AutoRouterSessionResponse: {
|
||||
/**
|
||||
* Baseline Model
|
||||
* @description The savings baseline most covered turns were priced against, recorded turn by turn, so it still names the counterfactual after the router is reconfigured or removed. None when no turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, which derive no baseline and so report no savings
|
||||
* @description The savings baseline recorded by most session turns, including historical turns, recorded turn by turn, so it still names the counterfactual after the router is reconfigured or removed. None when no turn recorded one: rows from before the baseline was recorded, and adaptive and quality routers, which derive no baseline and so report no savings
|
||||
*/
|
||||
baseline_model: string | null;
|
||||
/**
|
||||
* Baseline Models
|
||||
* @description Covered turns priced against each baseline model; more than one entry means the router's baseline changed mid-session and baseline_spend mixes both
|
||||
* @description Session turns recording each baseline model; more than one entry means the router's baseline changed mid-session; these counts do not imply savings coverage
|
||||
*/
|
||||
baseline_models: {
|
||||
[key: string]: number;
|
||||
|
|
@ -25292,7 +25302,7 @@ export interface components {
|
|||
router_type: string;
|
||||
/**
|
||||
* Saved Spend
|
||||
* @description Estimated savings for covered turns only, net of classifier cost
|
||||
* @description Recorded historical savings plus newer estimates, net of classifier cost
|
||||
*/
|
||||
saved_spend: number | null;
|
||||
/**
|
||||
|
|
@ -25307,7 +25317,7 @@ export interface components {
|
|||
savings_estimated_baseline_spend: number | null;
|
||||
/**
|
||||
* Savings Estimated Turns
|
||||
* @description Turns covered by the current savings estimator; legacy estimates are excluded
|
||||
* @description Requests with a matching savings comparison, including historical recorded estimates
|
||||
*/
|
||||
savings_estimated_turns: number;
|
||||
/** Session Id */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue