feat(ui): add auto-router usage and savings table (#45152)

This commit is contained in:
tin-berri 2026-10-07 15:04:07 -07:00 • committed by GitHub
parent 46d440ae96
commit 2803a16b36
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
19 changed files with 479 additions and 120 deletions

View file

@ -0,0 +1,19 @@
DO $$
BEGIN
IF NOT EXISTS (
SELECT 1 FROM pg_attribute
WHERE attrelid = to_regclass('"LiteLLM_AutoRouterDailySpend"')
AND attname = 'total_tokens' AND NOT attisdropped
) THEN
ALTER TABLE "LiteLLM_AutoRouterDailySpend"
ADD COLUMN IF NOT EXISTS "total_tokens" BIGINT NOT NULL DEFAULT 0;
END IF;
IF NOT EXISTS (
SELECT 1 FROM pg_attribute
WHERE attrelid = to_regclass('"LiteLLM_AutoRouterDailySpend"')
AND attname = 'token_recorded_turns' AND NOT attisdropped
) THEN
ALTER TABLE "LiteLLM_AutoRouterDailySpend"
ADD COLUMN IF NOT EXISTS "token_recorded_turns" INTEGER NOT NULL DEFAULT 0;
END IF;
END $$;

View file

@ -1756,6 +1756,8 @@ model LiteLLM_AutoRouterDailySpend {
router_name String
router_type String
turns Int @default(0)
total_tokens BigInt @default(0)
token_recorded_turns Int @default(0)
spend Float @default(0)
saved_spend Float @default(0)
savings_estimated_turns Int @default(0)

View file

@ -291,7 +291,7 @@ else:
_GENERIC_API_LOGGER_CLS: Final = GenericAPILogger
_in_memory_loggers: Final[list[CustomLogger]] = []
_STANDARD_LOGGING_METADATA_RESOLVED_KEYS: Final[frozenset[str]] = frozenset(("used_client_oauth_token",))
_STANDARD_LOGGING_METADATA_RESOLVED_KEYS: Final[frozenset[str]] = frozenset(("used_client_oauth_token", "usage_object"))
_STANDARD_LOGGING_METADATA_KEYS: Final[frozenset[str]] = (
frozenset(StandardLoggingMetadata.__annotations__.keys()) - _STANDARD_LOGGING_METADATA_RESOLVED_KEYS
)
@ -5993,7 +5993,7 @@ class StandardLoggingPayloadSetup:
Like get_usage_from_response_obj but returns a plain dict, skipping
the Pydantic Usage construction on the hot path.
"""
_empty: Final[dict] = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
_empty: Final[dict[str, object]] = {}
if combined_usage_object is not None:
return combined_usage_object.model_dump()
if not response_obj:

View file

@ -104,6 +104,8 @@ days AS (
router_name,
router_type,
SUM(turns)::int AS turns,
CASE WHEN SUM(token_recorded_turns) = SUM(turns)
THEN SUM(total_tokens)::bigint END AS day_total_tokens,
SUM(spend)::float8 AS spend,
SUM(saved_spend)::float8 AS saved_spend,
SUM(savings_estimated_turns)::int AS savings_estimated_turns,
@ -139,6 +141,7 @@ SELECT
COALESCE(sessions.total_tokens, 0) AS total_tokens,
COALESCE(sessions.session_seconds, 0) AS session_seconds,
COALESCE(days.turns, 0) AS turns,
CASE WHEN days.turns IS NULL THEN 0 ELSE days.day_total_tokens END AS day_total_tokens,
COALESCE(days.spend, 0) AS spend,
COALESCE(days.saved_spend, 0) AS saved_spend,
COALESCE(days.savings_estimated_turns, 0) AS savings_estimated_turns,
@ -175,6 +178,7 @@ class AutoRouterTurnTransaction:
savings_estimated_actual_spend: float = 0.0
savings_estimated_saved_spend: float = 0.0
user_id: str = ""
token_counts_recorded: bool = False
class TurnCacheFacts(NamedTuple):
@ -294,6 +298,11 @@ def build_autorouter_turn_transaction(
)
usage_object_raw: Final = metadata.get("usage_object")
token_counts: Final = (
(usage_object_raw.get("prompt_tokens"), usage_object_raw.get("completion_tokens"))
if isinstance(usage_object_raw, Mapping)
else ()
)
cache: Final = turn_cache_facts(usage_object_raw if isinstance(usage_object_raw, Mapping) else None)
tier_raw: Final = routing_decision.get("tier")
baseline_raw: Final = routing_decision.get("savings_baseline_model")
@ -311,6 +320,8 @@ def build_autorouter_turn_transaction(
model=model,
turn_at=turn_at,
total_tokens=int(payload.get("prompt_tokens") or 0) + int(payload.get("completion_tokens") or 0),
token_counts_recorded=len(token_counts) == 2
and all(isinstance(value, int) and not isinstance(value, bool) and value >= 0 for value in token_counts),
spend=actual_spend,
saved_spend=saved_spend,
classifier_cost=classifier_cost or 0.0,
@ -435,17 +446,21 @@ ON CONFLICT ({user_column}api_key, session_id, router_name) DO UPDATE SET
_DAY_UPSERT_SQL: Final = f"""
day_rollup AS (
INSERT INTO "LiteLLM_AutoRouterDailySpend" AS d (
date, api_key, user_id, router_name, router_type, turns, spend, saved_spend, savings_estimated_turns,
date, api_key, user_id, router_name, router_type, turns, total_tokens, token_recorded_turns,
spend, saved_spend, savings_estimated_turns,
savings_estimated_actual_spend, savings_estimated_saved_spend, classifier_cost, classifier_cost_recorded_turns
)
VALUES (
({_TURN_AT}::timestamp)::date::text, {_p("api_key")}::text, {_p("user_id")}::text, {_p("router_name")},
{_p("router_type")}, 1, {_p("spend")}::float8, {_p("saved_spend")}::float8, {_p("savings_estimated_turns")}::int,
{_p("router_type")}, 1, {_p("total_tokens")}::bigint, {_p("token_counts_recorded")}::int,
{_p("spend")}::float8, {_p("saved_spend")}::float8, {_p("savings_estimated_turns")}::int,
{_p("savings_estimated_actual_spend")}::float8, {_p("savings_estimated_saved_spend")}::float8,
{_p("classifier_cost")}::float8, 1
)
ON CONFLICT (date, api_key, user_id, router_name, router_type) DO UPDATE SET
turns = d.turns + 1,
total_tokens = d.total_tokens + EXCLUDED.total_tokens,
token_recorded_turns = d.token_recorded_turns + EXCLUDED.token_recorded_turns,
spend = d.spend + EXCLUDED.spend,
saved_spend = d.saved_spend + EXCLUDED.saved_spend,
savings_estimated_turns = d.savings_estimated_turns + EXCLUDED.savings_estimated_turns,

View file

@ -650,6 +650,7 @@ class _SessionAggRow(LiteLLMBaseModel):
ttl_5m_turns: int = 0
ttl_1h_turns: int = 0
total_tokens: int = 0
day_total_tokens: int | None = None
session_seconds: float = 0.0
turns: int = 0
spend: float = 0.0
@ -724,6 +725,7 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals:
return AutoRouterBenchmarkTotals(
sessions=sessions,
turns=row.turns,
total_tokens=row.day_total_tokens,
avg_turns_per_session=_per_session(row, row.session_turns),
avg_session_seconds=_per_session(row, row.session_seconds),
avg_tokens_per_session=_per_session(row, row.total_tokens),
@ -759,6 +761,7 @@ def _benchmark_group(row: _SessionAggRow) -> AutoRouterBenchmarkGroup:
tier_turns=row.tier_turns,
sessions=totals.sessions,
turns=totals.turns,
total_tokens=totals.total_tokens,
avg_turns_per_session=totals.avg_turns_per_session,
avg_session_seconds=totals.avg_session_seconds,
avg_tokens_per_session=totals.avg_tokens_per_session,
@ -796,6 +799,11 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow:
ttl_5m_turns=sum(row.ttl_5m_turns for row in rows),
ttl_1h_turns=sum(row.ttl_1h_turns for row in rows),
total_tokens=sum(row.total_tokens for row in rows),
day_total_tokens=(
sum(row.day_total_tokens or 0 for row in rows)
if all(row.day_total_tokens is not None for row in rows)
else None
),
spend=sum(row.spend for row in rows),
saved_spend=sum(row.saved_spend for row in rows),
savings_estimated_turns=sum(row.savings_estimated_turns for row in rows),

View file

@ -1756,6 +1756,8 @@ model LiteLLM_AutoRouterDailySpend {
router_name String
router_type String
turns Int @default(0)
total_tokens BigInt @default(0)
token_recorded_turns Int @default(0)
spend Float @default(0)
saved_spend Float @default(0)
savings_estimated_turns Int @default(0)

View file

@ -211,6 +211,11 @@ class AutoRouterBenchmarkTotals(LiteLLMBaseModel):
sessions: int = Field(description="Sessions overlapping the window, counted whole")
turns: int = Field(description="Auto-routed requests on the selected UTC days")
total_tokens: int | None = Field(
default=None,
description="Input and output tokens of routed generation requests on the selected UTC days, excluding "
"classifier tokens; null when any selected requests predate daily token recording",
)
avg_turns_per_session: float | None = Field(
description="Lifetime turns per overlapping session; null when the window has routed requests but no session "
"rows for this router type, such as an alias whose router type changed mid-session"

View file

@ -1756,6 +1756,8 @@ model LiteLLM_AutoRouterDailySpend {
router_name String
router_type String
turns Int @default(0)
total_tokens BigInt @default(0)
token_recorded_turns Int @default(0)
spend Float @default(0)
saved_spend Float @default(0)
savings_estimated_turns Int @default(0)

View file

@ -55,6 +55,7 @@ async def _turn(
baseline: "str | None" = None,
estimated: bool = True,
user_id: str = "",
token_counts_recorded: bool = True,
) -> None:
touched: Final = 1 if (hit or ttl is not None or not covered) else 0
await db.execute_raw(
@ -79,6 +80,7 @@ async def _turn(
spend if estimated else 0.0,
saved if estimated else 0.0,
user_id,
int(token_counts_recorded),
)
@ -646,9 +648,9 @@ async def test_a_cross_midnight_session_splits_its_money_by_request_day(db):
key = f"k-{uuid.uuid4()}"
router = f"auto-{uuid.uuid4()}"
midnight = datetime(2026, 9, 2)
await _turn(db, key, "A", midnight - timedelta(minutes=10), router=router, spend=1.0, saved=7.0, user_id="u1")
await _turn(db, key, "A", midnight + timedelta(minutes=10), router=router, spend=1.0, saved=3.0, user_id="u1")
await _turn(db, key, "B", midnight + timedelta(days=1), router=router, spend=1.0, saved=11.0, user_id="u1")
await _turn(db, key, "A", midnight - timedelta(minutes=10), router=router, tokens=100, spend=1.0, saved=7.0, user_id="u1")
await _turn(db, key, "A", midnight + timedelta(minutes=10), router=router, tokens=200, spend=1.0, saved=3.0, user_id="u1")
await _turn(db, key, "B", midnight + timedelta(days=1), router=router, tokens=300, spend=1.0, saved=11.0, user_id="u1")
assert (await _row(db, key, router=router))["saved_spend"] == 21.0
days = await db.query_raw(
@ -663,6 +665,7 @@ async def test_a_cross_midnight_session_splits_its_money_by_request_day(db):
(selected,) = await _benchmark_rows(db, midnight, midnight + timedelta(days=1), key, user_id)
assert (selected["sessions"], selected["session_turns"]) == (1, 3)
assert (selected["turns"], selected["spend"], selected["saved_spend"]) == (1, 1.0, 3.0)
assert (selected["day_total_tokens"], selected["total_tokens"]) == (200, 600)
async def test_a_router_type_change_within_a_day_keeps_each_types_money_apart(db):
@ -704,6 +707,7 @@ async def test_a_sessionless_turn_writes_its_router_day_row_and_no_session_row(d
model="A",
turn_at=T0 + timedelta(seconds=offset),
total_tokens=10,
token_counts_recorded=True,
spend=1.0,
saved_spend=2.0,
classifier_cost=0.1,
@ -720,11 +724,48 @@ async def test_a_sessionless_turn_writes_its_router_day_row_and_no_session_row(d
(day,) = await _days(db, key, router=router)
assert (day["turns"], day["spend"], day["saved_spend"], day["classifier_cost"]) == (2, 2.0, 4.0, 0.2)
assert day["day_total_tokens"] == 20
assert (day["sessions"], day["session_turns"]) == (0, 0)
for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession"):
assert await db.query_raw(f'SELECT 1 FROM "{table}" WHERE router_name = $1', router) == []
@pytest.mark.parametrize("historical", [True, False])
async def test_daily_token_coverage_stays_unknown_with_old_writers(db: Prisma, historical: bool) -> None:
key: Final = f"k-{uuid.uuid4()}"
router: Final = f"auto-{uuid.uuid4()}"
if historical:
await db.execute_raw(
'INSERT INTO "LiteLLM_AutoRouterDailySpend" '
'(date, api_key, user_id, router_name, router_type, turns, spend) '
"VALUES ($1, $2, 'u1', $3, 'complexity', 1, 1)",
T0.date().isoformat(), key, router,
)
await _turn(db, key, "A", T0, router=router, tokens=123, spend=1.0, user_id="u1")
if not historical:
await db.execute_raw(
'UPDATE "LiteLLM_AutoRouterDailySpend" SET turns = turns + 1, spend = spend + 1 '
'WHERE api_key = $1 AND router_name = $2', key, router,
)
for user_id in (None, "u1"):
(day,) = await _days(db, key, user_id, router)
assert (day["turns"], day["spend"], day["day_total_tokens"]) == (2, 2.0, None)
@pytest.mark.parametrize("missing_first", [True, False])
async def test_missing_usage_never_completes_daily_token_coverage(db: Prisma, missing_first: bool) -> None:
key: Final = f"k-{uuid.uuid4()}"
router: Final = f"auto-{uuid.uuid4()}"
for offset, recorded in enumerate((not missing_first, missing_first)):
await _turn(
db, key, "A", T0 + timedelta(seconds=offset), router=router,
tokens=100 if recorded else 0, spend=1.0, user_id="u1", token_counts_recorded=recorded,
)
for user_id in (None, "u1"):
(day,) = await _days(db, key, user_id, router)
assert (day["turns"], day["spend"], day["day_total_tokens"]) == (2, 2.0, None)
async def test_router_day_money_reconciles_with_the_overall_daily_total_including_sessionless_requests(db):
from litellm.proxy.db.daily_spend_bulk_upsert import DAILY_SPEND_TABLES, build_bulk_upsert, merge_by_conflict_key

View file

@ -212,8 +212,10 @@ async def test_logger_distinguishes_missing_usage_from_reported_zero(
now: Final = datetime.datetime.now()
monkeypatch.setattr(litellm, FLAG, True)
logger: Final = PrometheusLogger()
response_obj: Final = response.model_dump() if isinstance(response, litellm.ModelResponse) else response
usage: Final = StandardLoggingPayloadSetup.get_usage_as_dict(
response_obj=response if isinstance(response, dict) else None
response_obj=response_obj if isinstance(response_obj, dict) else None,
combined_usage_object=combined_usage if isinstance(combined_usage, litellm.Usage) else None,
)
payload: Final = _standard_logging_payload(now, usage.get("prompt_tokens", 0))

View file

@ -3714,11 +3714,11 @@ def test_get_usage_as_dict():
# Test case 1: None response_obj returns empty usage dict
result = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj=None)
assert result == {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
assert result == {}
# Test case 2: Empty response_obj returns empty usage dict
result = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj={})
assert result == {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
assert result == {}
# Test case 3: combined_usage_object takes priority
combined = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
@ -3738,7 +3738,35 @@ def test_get_usage_as_dict():
# Test case 5: response_obj with no usage key returns empty
result = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj={"id": "resp-1", "choices": []})
assert result == {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}
assert result == {}
@pytest.mark.parametrize(
"usage, include_usage",
[(None, False), (None, True), ({"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}, True)],
)
def test_logging_preserves_missing_usage_without_accepting_request_metadata(
logging_obj: Logging, usage: dict[str, int] | None, include_usage: bool
) -> None:
from litellm.litellm_core_utils.litellm_logging import get_standard_logging_object_payload
from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import convert_to_model_response_object
now: Final = datetime_unit_test(2026, 1, 1, 12, 0, 0)
response: Final[ModelResponse] = convert_to_model_response_object(
response_object={"id": "usage-coverage", "choices": [], **({"usage": usage} if include_usage else {})},
model_response_object=ModelResponse(),
)
payload: Final = get_standard_logging_object_payload(
kwargs={"litellm_params": {"metadata": {"usage_object": {"prompt_tokens": 99, "completion_tokens": 99}}}},
init_response_obj=response,
start_time=now,
end_time=now,
logging_obj=logging_obj,
status="success",
)
assert payload is not None
assert payload["metadata"]["usage_object"] == (response.usage.model_dump() if usage is not None else {})
assert (payload["prompt_tokens"], payload["completion_tokens"], payload["total_tokens"]) == (0, 0, 0)
def test_append_system_prompt_messages():

View file

@ -14,6 +14,7 @@ from typing import Final
import httpx
import pytest
from pydantic import TypeAdapter
from litellm.proxy.db.autorouter_session_rollup import (
UPSERT_AUTOROUTER_SESSION_SQL,
@ -45,7 +46,10 @@ def _payload(**overrides: object) -> dict:
def _metadata(**overrides: object) -> dict:
base: dict = {"routing_decision": dict(ROUTING_DECISION), "usage_object": {"prompt_tokens": 90}}
base: dict = {
"routing_decision": dict(ROUTING_DECISION),
"usage_object": {"prompt_tokens": 90, "completion_tokens": 10},
}
base.update(overrides)
return base
@ -88,7 +92,10 @@ class TestBuildTransaction:
transaction = _build(
metadata=_metadata(
routing_decision={**ROUTING_DECISION, "savings_baseline_model": "anthropic/claude-opus-5"},
usage_object={"prompt_tokens": 90, "cache_read_input_tokens": 5, "cache_creation_input_tokens": 7},
usage_object={
"prompt_tokens": 90, "completion_tokens": 10,
"cache_read_input_tokens": 5, "cache_creation_input_tokens": 7,
},
)
)
assert transaction == AutoRouterTurnTransaction(
@ -99,6 +106,7 @@ class TestBuildTransaction:
model="bedrock/haiku",
turn_at=datetime(2026, 8, 1, 12, 0, 0),
total_tokens=100,
token_counts_recorded=True,
spend=0.01,
saved_spend=0.02,
classifier_cost=0.0,
@ -213,6 +221,30 @@ class TestBuildTransaction:
assert transaction.cache_ttl_seconds is None
assert transaction.cache_touched is True
@pytest.mark.parametrize(
"usage, recorded",
[
(None, False), ({}, False), ({"prompt_tokens": 90}, False),
({"prompt_tokens": 90, "completion_tokens": 10}, True),
({"prompt_tokens": 0, "completion_tokens": 0}, True),
({"prompt_tokens": -1, "completion_tokens": 10}, False),
({"prompt_tokens": True, "completion_tokens": 10}, False),
({"prompt_tokens": "90", "completion_tokens": 10}, False),
],
)
def test_token_coverage_requires_complete_reported_counts(self, usage: object, recorded: bool) -> None:
transaction: Final = _build(metadata=_metadata(usage_object=usage))
assert transaction is not None
assert transaction.token_counts_recorded is recorded
def test_persisted_turns_preserve_coverage_and_default_old_records_to_unknown(self) -> None:
transaction: Final = _build()
adapter: Final = TypeAdapter(AutoRouterTurnTransaction)
assert transaction is not None
assert adapter.validate_json(adapter.dump_json(transaction)).token_counts_recorded is True
legacy: Final = adapter.dump_json(transaction, exclude={"token_counts_recorded"})
assert adapter.validate_json(legacy).token_counts_recorded is False
def test_a_covered_turn_that_neither_read_nor_wrote_did_not_touch_the_cache(self):
transaction = _build()
assert transaction is not None
@ -341,6 +373,7 @@ class TestFlush:
0.0,
0.0,
"canonical-user",
0,
)
def test_a_keys_turns_stay_chronological_when_its_canonical_user_changes(self) -> None:

View file

@ -770,6 +770,7 @@ class TestAutoRouterBenchmarks:
ttl_5m_turns=30,
ttl_1h_turns=5,
total_tokens=4000,
day_total_tokens=2500,
spend=10.0,
saved_spend=30.0,
savings_estimated_turns=40,
@ -798,6 +799,7 @@ class TestAutoRouterBenchmarks:
assert totals.avg_turns_per_session == 10.0
assert totals.avg_session_seconds == 100.0
assert totals.avg_tokens_per_session == 1000.0
assert totals.total_tokens == 2500
assert totals.baseline_spend == 40.0
assert totals.saved_pct == 75.0
assert totals.savings_estimated_classifier_cost == 0.4
@ -818,6 +820,20 @@ class TestAutoRouterBenchmarks:
assert totals.saved_pct == -100.0
assert totals.classifier_cost == 0.4
@pytest.mark.asyncio
@pytest.mark.parametrize("historical_tokens, expected_total", [(None, None), (0, 2500), (750, 3250)])
async def test_daily_token_totals_preserve_missing_coverage_in_any_router(
self, historical_tokens: int | None, expected_total: int | None, monkeypatch: pytest.MonkeyPatch
) -> None:
historical: Final = self.ROW.model_copy(
update={"router_name": "historical-auto", "day_total_tokens": historical_tokens}
)
response: Final = await self._benchmarks(
monkeypatch, rows=[self.ROW.model_dump(), historical.model_dump()], model_list=[]
)
assert [group.total_tokens for group in response.groups] == [2500, historical_tokens]
assert response.totals.total_tokens == expected_total
@pytest.mark.asyncio
@pytest.mark.parametrize("estimated_turns", [0, 4])
async def test_historical_savings_without_recorded_baselines_compare_against_all_spend(
@ -921,6 +937,7 @@ class TestAutoRouterBenchmarks:
totals = _benchmark_totals(_summed_agg_row([]))
assert totals.sessions == 0
assert totals.turns == 0
assert totals.total_tokens == 0
assert totals.saved_pct == 0.0
assert totals.cache.hit_rate_pct == 0.0
assert totals.classifier_cost == 0.0
@ -1164,6 +1181,7 @@ class TestAutoRouterBenchmarks:
assert idle.cache.same_model.turns == idle.cache.return_to_tier.hits == 0
assert idle.tier_turns == {}
assert idle.classifier_cost == 0.0
assert idle.total_tokens == 0
@pytest.mark.asyncio
@pytest.mark.parametrize(

View file

@ -8,6 +8,7 @@ import { ApiError } from "@/lib/http/client";
vi.mock("./useAutoRouterBenchmarks", () => ({ useAutoRouterBenchmarks: vi.fn() }));
vi.mock("@/app/(dashboard)/hooks/models/useModels", () => ({ useAutoRouters: vi.fn() }));
vi.mock("./AutoRouterSummaryTable", () => ({ default: () => <div data-testid="router-summary" /> }));
vi.mock("./ShadowEvalSection", () => ({ default: () => <div data-testid="shadow-eval-section" /> }));
vi.mock("@/components/shared/advanced_date_picker", () => ({
__esModule: true,

View file

@ -25,12 +25,14 @@ import {
groupLabel,
pctLabel,
viewFor,
viewGroup,
type AutoRouterBenchmarksResponse,
type AutoRouterCacheStats,
type BenchmarkView,
type BucketRow,
} from "./autoRouterBenchmarks";
import { classificationRatePer1kTurns, formatRangeLabel, usd } from "./costOptimizationUtils";
import AutoRouterSummaryTable from "./AutoRouterSummaryTable";
import ShadowEvalSection from "./ShadowEvalSection";
import TierTurnsChart from "./TierTurnsChart";
import { useAutoRouterBenchmarks } from "./useAutoRouterBenchmarks";
@ -79,7 +81,7 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => {
const classifierCost = stats.baseline_spend == null ? null : stats.savings_estimated_classifier_cost ?? null;
const comparedAll = stats.savings_estimated_turns === stats.turns;
return (
<Card className="overflow-hidden py-0">
<Card className="overflow-hidden py-0" role="region" aria-label="Auto-router savings">
<div className="grid md:grid-cols-[minmax(0,1fr)_minmax(0,1fr)]">
<div className="flex flex-col items-center justify-center gap-2 p-6">
<p className="text-xs font-semibold uppercase tracking-wider text-muted-foreground">
@ -332,6 +334,8 @@ const BenchmarksBody: React.FC<BenchmarksBodyProps> = ({ isPending, error, data,
/>
</div>
<AutoRouterSummaryTable groups={data.groups} selectedGroup={viewGroup(view)} />
<div className="space-y-4">
<div className="flex flex-wrap items-baseline gap-2">
<h3 className="text-lg font-semibold text-foreground">Auto-router prompt caching</h3>

View file

@ -0,0 +1,73 @@
import { Card, CardContent, CardDescription, CardHeader, CardTitle } from "@/components/ui/card";
import { Table, TableBody, TableCell, TableHead, TableHeader, TableRow } from "@/components/ui/table";
import { groupKey, groupLabel, pctLabel, type AutoRouterBenchmarkGroup } from "./autoRouterBenchmarks";
import { usd } from "./costOptimizationUtils";
interface AutoRouterSummaryTableProps {
groups: readonly AutoRouterBenchmarkGroup[];
selectedGroup: AutoRouterBenchmarkGroup | null;
}
export default function AutoRouterSummaryTable({ groups, selectedGroup }: AutoRouterSummaryTableProps) {
const visibleGroups = selectedGroup ? [selectedGroup] : groups;
return (
<Card>
<CardHeader>
<CardTitle>Router usage and savings</CardTitle>
<CardDescription>
Selected UTC days. Cost includes LLM calls and classification. Tokens count routed LLM input and output.
</CardDescription>
</CardHeader>
<CardContent>
<Table aria-label="Router usage and savings">
<TableHeader>
<TableRow>
<TableHead>Auto-router</TableHead>
<TableHead className="text-right whitespace-normal">Tokens through router</TableHead>
<TableHead className="text-right whitespace-normal">Cost via router</TableHead>
<TableHead className="text-right whitespace-normal">Cost per 1M tokens</TableHead>
<TableHead className="text-right whitespace-normal">Saved vs. premium model</TableHead>
<TableHead className="text-right">% saved</TableHead>
</TableRow>
</TableHeader>
<TableBody>
{visibleGroups.length === 0 ? (
<TableRow>
<TableCell colSpan={6} className="py-8 text-center text-muted-foreground">
No auto-routers in this range
</TableCell>
</TableRow>
) : (
visibleGroups.map((group) => (
<TableRow key={groupKey(group)}>
<TableCell className="font-medium">{groupLabel(group, groups)}</TableCell>
<TableCell className="text-right tabular-nums">
{group.total_tokens == null ? "Unavailable" : group.total_tokens.toLocaleString()}
</TableCell>
<TableCell className="text-right tabular-nums">{usd(group.spend)}</TableCell>
<TableCell className="text-right tabular-nums">
{group.total_tokens != null && group.total_tokens > 0
? usd((group.spend * 1_000_000) / group.total_tokens)
: "Unavailable"}
</TableCell>
<TableCell className="text-right tabular-nums">
{group.saved_spend == null ? "Unavailable" : usd(group.saved_spend)}
</TableCell>
<TableCell className="text-right tabular-nums">
{group.saved_pct == null ? "Unavailable" : pctLabel(group.saved_pct)}
</TableCell>
</TableRow>
))
)}
</TableBody>
</Table>
<p className="mt-3 text-xs text-muted-foreground">
Savings are estimated against each router&apos;s premium baseline. Token totals and unit costs are unavailable
for usage recorded before token tracking; unit cost also requires nonzero tokens.
</p>
</CardContent>
</Card>
);
}

View file

@ -0,0 +1,202 @@
import { screen, waitFor, within } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import { beforeEach, describe, expect, it, vi } from "vitest";
import AutoRouterBenchmarksTab from "@/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab";
import type {
AutoRouterBenchmarkGroup,
AutoRouterBenchmarksResponse,
} from "@/app/(dashboard)/cost-optimization/_components/autoRouterBenchmarks";
import { renderWithProviders, testQueryClient } from "../../../tests/test-utils";
import KeyAutoRouterUsageTab from "./KeyAutoRouterUsageTab";
vi.mock("@/app/(dashboard)/hooks/useAuthorized", () => ({
default: () => ({ accessToken: "test-token", userId: "admin-123", userRole: "Admin" }),
}));
const jsonResponse = (body: unknown) =>
new Response(JSON.stringify(body), { status: 200, headers: { "content-type": "application/json" } });
const cache = {
coverage_pct: 100,
hit_rate_pct: 50,
same_model: { turns: 2, hits: 1, hit_rate_pct: 50 },
first_visit: { turns: 1, hits: 0, hit_rate_pct: 0 },
return_to_tier: { turns: 1, hits: 1, hit_rate_pct: 100 },
unordered_turns: 0,
return_misses_expired: 0,
return_misses_within_ttl: 0,
return_misses_unknown: 0,
ttl_5m_turns: 0,
ttl_1h_turns: 0,
};
const stats = {
sessions: 2,
turns: 4,
avg_turns_per_session: 2,
avg_session_seconds: 30,
avg_tokens_per_session: 100,
spend: 1.25,
savings_estimated_turns: 4,
savings_estimated_actual_spend: 1.25,
savings_estimated_classifier_cost: 0.25,
classifier_cost: 0.25,
saved_spend: 8.75,
baseline_spend: 10,
saved_pct: 87.5,
cache,
};
const benchmarks: AutoRouterBenchmarksResponse = {
start_date: "2025-01-01",
end_date: "2025-01-31",
routers_in_scope: 2,
totals: stats,
groups: [
{ router_name: "router-one", router_type: "complexity", tier_turns: { SIMPLE: 4 }, ...stats },
{
router_name: "router-two",
router_type: "complexity",
tier_turns: { SIMPLE: 1 },
...stats,
spend: 0.25,
saved_spend: 0.75,
baseline_spend: 1,
},
],
};
const noDeployments = { data: [], total_count: 0, current_page: 1, total_pages: 1, size: 1000 };
const fetchMock = vi.fn<(request: Request | string) => Promise<Response>>();
const mockBenchmarks = (body: AutoRouterBenchmarksResponse) => {
fetchMock.mockImplementation(async (request) => {
const url = typeof request === "string" ? request : request.url;
return jsonResponse(url.includes("/auto_router/benchmarks") ? body : noDeployments);
});
};
const group = (overrides: Partial<AutoRouterBenchmarkGroup> = {}): AutoRouterBenchmarkGroup => ({
...stats,
router_name: "claude-auto",
router_type: "complexity",
...overrides,
});
const response = (groups: AutoRouterBenchmarkGroup[]): AutoRouterBenchmarksResponse => ({
...benchmarks,
routers_in_scope: groups.length,
groups,
});
const renderOverallTab = () => {
const activity = {
dateValue: { from: new Date(2025, 0, 1), to: new Date(2025, 0, 31) },
onDateChange: vi.fn(),
results: [],
loading: false,
isFetchingMore: false,
progress: { currentPage: 1, totalPages: 1 },
cancelled: false,
cancel: vi.fn(),
};
renderWithProviders(<AutoRouterBenchmarksTab accessToken="test-token" activity={activity} />);
};
const requestedUrls = () =>
fetchMock.mock.calls.map(([request]) => (typeof request === "string" ? request : request.url));
describe("Auto-router usage views", () => {
beforeEach(() => {
vi.clearAllMocks();
mockBenchmarks(benchmarks);
testQueryClient.clear();
vi.stubGlobal("fetch", fetchMock);
});
it("renders this key's spend, baseline, savings and per-router filter", async () => {
const activity = {
dateValue: { from: new Date(2025, 0, 1), to: new Date(2025, 0, 31) },
onDateChange: vi.fn(),
};
renderWithProviders(<KeyAutoRouterUsageTab accessToken="test-token" keyToken="key-hash-1" activity={activity} />);
const savings = within(await screen.findByRole("region", { name: "Auto-router savings" }));
expect(savings.getByText("$8.75")).toBeInTheDocument();
expect(savings.getByText("Actual auto-router spend")).toBeInTheDocument();
expect(savings.getByText("$1.25")).toBeInTheDocument();
expect(savings.getByText("LLM spend")).toBeInTheDocument();
expect(savings.getByText("$1.00")).toBeInTheDocument();
expect(savings.getByText("Classification cost")).toBeInTheDocument();
expect(savings.getByText("$0.2500")).toBeInTheDocument();
expect(savings.getByText("($62.50 / 1K turns)")).toBeInTheDocument();
expect(savings.getByText("Estimated baseline spend")).toBeInTheDocument();
expect(savings.getByText("$10.00")).toBeInTheDocument();
expect(screen.getByText("Auto-router prompt caching")).toBeInTheDocument();
expect(screen.getAllByText("50.0%").length).toBeGreaterThan(0);
expect(screen.getByText("All auto-routers")).toBeInTheDocument();
const summary = within(screen.getByRole("table", { name: "Router usage and savings" }));
expect(summary.getByRole("row", { name: /router-one.*\$1\.25.*\$8\.75/ })).toBeInTheDocument();
expect(summary.getByRole("row", { name: /router-two.*\$0\.25.*\$0\.75/ })).toBeInTheDocument();
const benchmarkUrl = new URL(requestedUrls().find((url) => url.includes("/auto_router/benchmarks")) ?? "");
expect(benchmarkUrl.searchParams.get("api_key")).toBe("key-hash-1");
expect(benchmarkUrl.searchParams.get("start_date")).toBe("2025-01-01");
expect(benchmarkUrl.searchParams.get("end_date")).toBe("2025-01-31");
});
it("shows per-router token costs above caching and follows the router picker", async () => {
const standard = { total_tokens: 100_000_000, spend: 20_000, saved_spend: 4_000, saved_pct: 16.7 };
const losing = { router_name: "gpt-auto", total_tokens: 2_000_000, spend: 15, saved_spend: -5, saved_pct: -50 };
mockBenchmarks(response([group(standard), group(losing)]));
renderOverallTab();
const table = await screen.findByRole("table", { name: "Router usage and savings" });
const rows = within(table).getAllByRole("row");
expect(
within(rows[1])
.getAllByRole("cell")
.map((cell) => cell.textContent),
).toEqual(["claude-auto", "100,000,000", "$20,000.00", "$200.00", "$4,000.00", "16.7%"]);
expect(
within(rows[2])
.getAllByRole("cell")
.map((cell) => cell.textContent),
).toEqual(["gpt-auto", "2,000,000", "$15.00", "$7.50", "-$5.00", "-50.0%"]);
expect(
table.compareDocumentPosition(screen.getByText("Auto-router prompt caching")) & Node.DOCUMENT_POSITION_FOLLOWING,
).toBeTruthy();
const user = userEvent.setup();
await user.click(screen.getByRole("combobox"));
await user.click(await screen.findByRole("option", { name: "gpt-auto" }));
await waitFor(() => expect(within(table).getAllByRole("row")).toHaveLength(2));
expect(within(table).queryByText("claude-auto")).not.toBeInTheDocument();
expect(within(table).getByText("-$5.00")).toBeInTheDocument();
});
it.each([null, undefined, 0])("keeps token coverage and unit costs honest for %s tokens", async (total_tokens) => {
const untracked = { total_tokens, saved_spend: null, saved_pct: null, baseline_spend: null };
mockBenchmarks(response([group(untracked)]));
renderOverallTab();
const row = within(await screen.findByRole("table", { name: "Router usage and savings" })).getAllByRole("row")[1];
expect(
within(row)
.getAllByRole("cell")
.map((cell) => cell.textContent),
).toEqual([
"claude-auto",
total_tokens === 0 ? "0" : "Unavailable",
"$1.25",
"Unavailable",
"Unavailable",
"Unavailable",
]);
});
it("shows a clear empty summary when no routers are present", async () => {
mockBenchmarks(response([]));
renderOverallTab();
expect(
within(await screen.findByRole("table", { name: "Router usage and savings" })).getByText(
"No auto-routers in this range",
),
).toBeInTheDocument();
});
});

View file

@ -1,106 +0,0 @@
import { screen } from "@testing-library/react";
import { beforeEach, describe, expect, it, vi } from "vitest";
import { renderWithProviders, testQueryClient } from "../../../tests/test-utils";
import KeyAutoRouterUsageTab from "./KeyAutoRouterUsageTab";
vi.mock("@/app/(dashboard)/hooks/useAuthorized", () => ({
default: () => ({ accessToken: "test-token", userId: "admin-123", userRole: "Admin" }),
}));
const jsonResponse = (body: unknown) =>
new Response(JSON.stringify(body), { status: 200, headers: { "content-type": "application/json" } });
const cache = {
coverage_pct: 100,
hit_rate_pct: 50,
same_model: { turns: 2, hits: 1, hit_rate_pct: 50 },
first_visit: { turns: 1, hits: 0, hit_rate_pct: 0 },
return_to_tier: { turns: 1, hits: 1, hit_rate_pct: 100 },
unordered_turns: 0,
return_misses_expired: 0,
return_misses_within_ttl: 0,
return_misses_unknown: 0,
ttl_5m_turns: 0,
ttl_1h_turns: 0,
};
const stats = {
sessions: 2,
turns: 4,
avg_turns_per_session: 2,
avg_session_seconds: 30,
avg_tokens_per_session: 100,
spend: 1.25,
savings_estimated_turns: 4,
savings_estimated_actual_spend: 1.25,
savings_estimated_classifier_cost: 0.25,
classifier_cost: 0.25,
saved_spend: 8.75,
baseline_spend: 10,
saved_pct: 87.5,
cache,
};
const benchmarks = {
start_date: "2025-01-01",
end_date: "2025-01-31",
routers_in_scope: 2,
totals: stats,
groups: [
{ router_name: "router-one", router_type: "complexity", tier_turns: { SIMPLE: 4 }, ...stats },
{
router_name: "router-two",
router_type: "complexity",
tier_turns: { SIMPLE: 1 },
...stats,
spend: 0.25,
saved_spend: 0.75,
baseline_spend: 1,
},
],
};
const noDeployments = { data: [], total_count: 0, current_page: 1, total_pages: 1, size: 1000 };
const fetchMock = vi.fn(async (request: Request | string) => {
const url = typeof request === "string" ? request : request.url;
if (url.includes("/auto_router/benchmarks")) return jsonResponse(benchmarks);
return jsonResponse(noDeployments);
});
const requestedUrls = () =>
fetchMock.mock.calls.map(([request]) => (typeof request === "string" ? request : request.url));
describe("KeyAutoRouterUsageTab", () => {
beforeEach(() => {
vi.clearAllMocks();
testQueryClient.clear();
vi.stubGlobal("fetch", fetchMock);
});
it("renders this key's spend, baseline, savings and per-router filter", async () => {
const activity = {
dateValue: { from: new Date(2025, 0, 1), to: new Date(2025, 0, 31) },
onDateChange: vi.fn(),
};
renderWithProviders(<KeyAutoRouterUsageTab accessToken="test-token" keyToken="key-hash-1" activity={activity} />);
expect(await screen.findByText("$8.75")).toBeInTheDocument();
expect(screen.getByText("Actual auto-router spend")).toBeInTheDocument();
expect(screen.getByText("$1.25")).toBeInTheDocument();
expect(screen.getByText("LLM spend")).toBeInTheDocument();
expect(screen.getByText("$1.00")).toBeInTheDocument();
expect(screen.getByText("Classification cost")).toBeInTheDocument();
expect(screen.getByText("$0.2500")).toBeInTheDocument();
expect(screen.getByText("($62.50 / 1K turns)")).toBeInTheDocument();
expect(screen.getByText("Estimated baseline spend")).toBeInTheDocument();
expect(screen.getByText("$10.00")).toBeInTheDocument();
expect(screen.getByText("Auto-router prompt caching")).toBeInTheDocument();
expect(screen.getAllByText("50.0%").length).toBeGreaterThan(0);
expect(screen.getByText("All auto-routers")).toBeInTheDocument();
const benchmarkUrl = new URL(requestedUrls().find((url) => url.includes("/auto_router/benchmarks")) ?? "");
expect(benchmarkUrl.searchParams.get("api_key")).toBe("key-hash-1");
expect(benchmarkUrl.searchParams.get("start_date")).toBe("2025-01-01");
expect(benchmarkUrl.searchParams.get("end_date")).toBe("2025-01-31");
});
});

View file

@ -26544,6 +26544,11 @@ export interface components {
tier_turns?: {
[key: string]: number;
};
/**
* Total Tokens
* @description Input and output tokens of routed generation requests on the selected UTC days, excluding classifier tokens; null when any selected requests predate daily token recording
*/
total_tokens?: number | null;
/**
* Turns
* @description Auto-routed requests on the selected UTC days
@ -26622,6 +26627,11 @@ export interface components {
* @description What the selected days' routed traffic actually cost
*/
spend: number;
/**
* Total Tokens
* @description Input and output tokens of routed generation requests on the selected UTC days, excluding classifier tokens; null when any selected requests predate daily token recording
*/
total_tokens?: number | null;
/**
* Turns
* @description Auto-routed requests on the selected UTC days