diff --git a/litellm-proxy-extras/litellm_proxy_extras/migrations/20261007120000_add_autorouter_daily_tokens/migration.sql b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261007120000_add_autorouter_daily_tokens/migration.sql new file mode 100644 index 00000000000..9259fa24fa6 --- /dev/null +++ b/litellm-proxy-extras/litellm_proxy_extras/migrations/20261007120000_add_autorouter_daily_tokens/migration.sql @@ -0,0 +1,19 @@ +DO $$ +BEGIN + IF NOT EXISTS ( + SELECT 1 FROM pg_attribute + WHERE attrelid = to_regclass('"LiteLLM_AutoRouterDailySpend"') + AND attname = 'total_tokens' AND NOT attisdropped + ) THEN + ALTER TABLE "LiteLLM_AutoRouterDailySpend" + ADD COLUMN IF NOT EXISTS "total_tokens" BIGINT NOT NULL DEFAULT 0; + END IF; + IF NOT EXISTS ( + SELECT 1 FROM pg_attribute + WHERE attrelid = to_regclass('"LiteLLM_AutoRouterDailySpend"') + AND attname = 'token_recorded_turns' AND NOT attisdropped + ) THEN + ALTER TABLE "LiteLLM_AutoRouterDailySpend" + ADD COLUMN IF NOT EXISTS "token_recorded_turns" INTEGER NOT NULL DEFAULT 0; + END IF; +END $$; diff --git a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma index 3b83c5b09cc..dbd44934ca8 100644 --- a/litellm-proxy-extras/litellm_proxy_extras/schema.prisma +++ b/litellm-proxy-extras/litellm_proxy_extras/schema.prisma @@ -1756,6 +1756,8 @@ model LiteLLM_AutoRouterDailySpend { router_name String router_type String turns Int @default(0) + total_tokens BigInt @default(0) + token_recorded_turns Int @default(0) spend Float @default(0) saved_spend Float @default(0) savings_estimated_turns Int @default(0) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 2ce05b9aa39..d0b8109441a 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -291,7 +291,7 @@ else: _GENERIC_API_LOGGER_CLS: Final = GenericAPILogger _in_memory_loggers: Final[list[CustomLogger]] = [] -_STANDARD_LOGGING_METADATA_RESOLVED_KEYS: Final[frozenset[str]] = frozenset(("used_client_oauth_token",)) +_STANDARD_LOGGING_METADATA_RESOLVED_KEYS: Final[frozenset[str]] = frozenset(("used_client_oauth_token", "usage_object")) _STANDARD_LOGGING_METADATA_KEYS: Final[frozenset[str]] = ( frozenset(StandardLoggingMetadata.__annotations__.keys()) - _STANDARD_LOGGING_METADATA_RESOLVED_KEYS ) @@ -5993,7 +5993,7 @@ class StandardLoggingPayloadSetup: Like get_usage_from_response_obj but returns a plain dict, skipping the Pydantic Usage construction on the hot path. """ - _empty: Final[dict] = {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0} + _empty: Final[dict[str, object]] = {} if combined_usage_object is not None: return combined_usage_object.model_dump() if not response_obj: diff --git a/litellm/proxy/db/autorouter_session_rollup.py b/litellm/proxy/db/autorouter_session_rollup.py index 21aa0ca4d29..75882bc3fb1 100644 --- a/litellm/proxy/db/autorouter_session_rollup.py +++ b/litellm/proxy/db/autorouter_session_rollup.py @@ -104,6 +104,8 @@ days AS ( router_name, router_type, SUM(turns)::int AS turns, + CASE WHEN SUM(token_recorded_turns) = SUM(turns) + THEN SUM(total_tokens)::bigint END AS day_total_tokens, SUM(spend)::float8 AS spend, SUM(saved_spend)::float8 AS saved_spend, SUM(savings_estimated_turns)::int AS savings_estimated_turns, @@ -139,6 +141,7 @@ SELECT COALESCE(sessions.total_tokens, 0) AS total_tokens, COALESCE(sessions.session_seconds, 0) AS session_seconds, COALESCE(days.turns, 0) AS turns, + CASE WHEN days.turns IS NULL THEN 0 ELSE days.day_total_tokens END AS day_total_tokens, COALESCE(days.spend, 0) AS spend, COALESCE(days.saved_spend, 0) AS saved_spend, COALESCE(days.savings_estimated_turns, 0) AS savings_estimated_turns, @@ -175,6 +178,7 @@ class AutoRouterTurnTransaction: savings_estimated_actual_spend: float = 0.0 savings_estimated_saved_spend: float = 0.0 user_id: str = "" + token_counts_recorded: bool = False class TurnCacheFacts(NamedTuple): @@ -294,6 +298,11 @@ def build_autorouter_turn_transaction( ) usage_object_raw: Final = metadata.get("usage_object") + token_counts: Final = ( + (usage_object_raw.get("prompt_tokens"), usage_object_raw.get("completion_tokens")) + if isinstance(usage_object_raw, Mapping) + else () + ) cache: Final = turn_cache_facts(usage_object_raw if isinstance(usage_object_raw, Mapping) else None) tier_raw: Final = routing_decision.get("tier") baseline_raw: Final = routing_decision.get("savings_baseline_model") @@ -311,6 +320,8 @@ def build_autorouter_turn_transaction( model=model, turn_at=turn_at, total_tokens=int(payload.get("prompt_tokens") or 0) + int(payload.get("completion_tokens") or 0), + token_counts_recorded=len(token_counts) == 2 + and all(isinstance(value, int) and not isinstance(value, bool) and value >= 0 for value in token_counts), spend=actual_spend, saved_spend=saved_spend, classifier_cost=classifier_cost or 0.0, @@ -435,17 +446,21 @@ ON CONFLICT ({user_column}api_key, session_id, router_name) DO UPDATE SET _DAY_UPSERT_SQL: Final = f""" day_rollup AS ( INSERT INTO "LiteLLM_AutoRouterDailySpend" AS d ( - date, api_key, user_id, router_name, router_type, turns, spend, saved_spend, savings_estimated_turns, + date, api_key, user_id, router_name, router_type, turns, total_tokens, token_recorded_turns, + spend, saved_spend, savings_estimated_turns, savings_estimated_actual_spend, savings_estimated_saved_spend, classifier_cost, classifier_cost_recorded_turns ) VALUES ( ({_TURN_AT}::timestamp)::date::text, {_p("api_key")}::text, {_p("user_id")}::text, {_p("router_name")}, - {_p("router_type")}, 1, {_p("spend")}::float8, {_p("saved_spend")}::float8, {_p("savings_estimated_turns")}::int, + {_p("router_type")}, 1, {_p("total_tokens")}::bigint, {_p("token_counts_recorded")}::int, + {_p("spend")}::float8, {_p("saved_spend")}::float8, {_p("savings_estimated_turns")}::int, {_p("savings_estimated_actual_spend")}::float8, {_p("savings_estimated_saved_spend")}::float8, {_p("classifier_cost")}::float8, 1 ) ON CONFLICT (date, api_key, user_id, router_name, router_type) DO UPDATE SET turns = d.turns + 1, + total_tokens = d.total_tokens + EXCLUDED.total_tokens, + token_recorded_turns = d.token_recorded_turns + EXCLUDED.token_recorded_turns, spend = d.spend + EXCLUDED.spend, saved_spend = d.saved_spend + EXCLUDED.saved_spend, savings_estimated_turns = d.savings_estimated_turns + EXCLUDED.savings_estimated_turns, diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index 617a76aaa04..de2a534a21d 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -650,6 +650,7 @@ class _SessionAggRow(LiteLLMBaseModel): ttl_5m_turns: int = 0 ttl_1h_turns: int = 0 total_tokens: int = 0 + day_total_tokens: int | None = None session_seconds: float = 0.0 turns: int = 0 spend: float = 0.0 @@ -724,6 +725,7 @@ def _benchmark_totals(row: _SessionAggRow) -> AutoRouterBenchmarkTotals: return AutoRouterBenchmarkTotals( sessions=sessions, turns=row.turns, + total_tokens=row.day_total_tokens, avg_turns_per_session=_per_session(row, row.session_turns), avg_session_seconds=_per_session(row, row.session_seconds), avg_tokens_per_session=_per_session(row, row.total_tokens), @@ -759,6 +761,7 @@ def _benchmark_group(row: _SessionAggRow) -> AutoRouterBenchmarkGroup: tier_turns=row.tier_turns, sessions=totals.sessions, turns=totals.turns, + total_tokens=totals.total_tokens, avg_turns_per_session=totals.avg_turns_per_session, avg_session_seconds=totals.avg_session_seconds, avg_tokens_per_session=totals.avg_tokens_per_session, @@ -796,6 +799,11 @@ def _summed_agg_row(rows: Sequence[_SessionAggRow]) -> _SessionAggRow: ttl_5m_turns=sum(row.ttl_5m_turns for row in rows), ttl_1h_turns=sum(row.ttl_1h_turns for row in rows), total_tokens=sum(row.total_tokens for row in rows), + day_total_tokens=( + sum(row.day_total_tokens or 0 for row in rows) + if all(row.day_total_tokens is not None for row in rows) + else None + ), spend=sum(row.spend for row in rows), saved_spend=sum(row.saved_spend for row in rows), savings_estimated_turns=sum(row.savings_estimated_turns for row in rows), diff --git a/litellm/proxy/schema.prisma b/litellm/proxy/schema.prisma index 3b83c5b09cc..dbd44934ca8 100644 --- a/litellm/proxy/schema.prisma +++ b/litellm/proxy/schema.prisma @@ -1756,6 +1756,8 @@ model LiteLLM_AutoRouterDailySpend { router_name String router_type String turns Int @default(0) + total_tokens BigInt @default(0) + token_recorded_turns Int @default(0) spend Float @default(0) saved_spend Float @default(0) savings_estimated_turns Int @default(0) diff --git a/litellm/types/management_endpoints/auto_router_endpoints.py b/litellm/types/management_endpoints/auto_router_endpoints.py index 0738f7bc737..e5e983ff48e 100644 --- a/litellm/types/management_endpoints/auto_router_endpoints.py +++ b/litellm/types/management_endpoints/auto_router_endpoints.py @@ -211,6 +211,11 @@ class AutoRouterBenchmarkTotals(LiteLLMBaseModel): sessions: int = Field(description="Sessions overlapping the window, counted whole") turns: int = Field(description="Auto-routed requests on the selected UTC days") + total_tokens: int | None = Field( + default=None, + description="Input and output tokens of routed generation requests on the selected UTC days, excluding " + "classifier tokens; null when any selected requests predate daily token recording", + ) avg_turns_per_session: float | None = Field( description="Lifetime turns per overlapping session; null when the window has routed requests but no session " "rows for this router type, such as an alias whose router type changed mid-session" diff --git a/schema.prisma b/schema.prisma index 3b83c5b09cc..dbd44934ca8 100644 --- a/schema.prisma +++ b/schema.prisma @@ -1756,6 +1756,8 @@ model LiteLLM_AutoRouterDailySpend { router_name String router_type String turns Int @default(0) + total_tokens BigInt @default(0) + token_recorded_turns Int @default(0) spend Float @default(0) saved_spend Float @default(0) savings_estimated_turns Int @default(0) diff --git a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py index d2511ba257b..3425bed22b2 100644 --- a/tests/proxy_behavior/spend/test_autorouter_session_rollup.py +++ b/tests/proxy_behavior/spend/test_autorouter_session_rollup.py @@ -55,6 +55,7 @@ async def _turn( baseline: "str | None" = None, estimated: bool = True, user_id: str = "", + token_counts_recorded: bool = True, ) -> None: touched: Final = 1 if (hit or ttl is not None or not covered) else 0 await db.execute_raw( @@ -79,6 +80,7 @@ async def _turn( spend if estimated else 0.0, saved if estimated else 0.0, user_id, + int(token_counts_recorded), ) @@ -646,9 +648,9 @@ async def test_a_cross_midnight_session_splits_its_money_by_request_day(db): key = f"k-{uuid.uuid4()}" router = f"auto-{uuid.uuid4()}" midnight = datetime(2026, 9, 2) - await _turn(db, key, "A", midnight - timedelta(minutes=10), router=router, spend=1.0, saved=7.0, user_id="u1") - await _turn(db, key, "A", midnight + timedelta(minutes=10), router=router, spend=1.0, saved=3.0, user_id="u1") - await _turn(db, key, "B", midnight + timedelta(days=1), router=router, spend=1.0, saved=11.0, user_id="u1") + await _turn(db, key, "A", midnight - timedelta(minutes=10), router=router, tokens=100, spend=1.0, saved=7.0, user_id="u1") + await _turn(db, key, "A", midnight + timedelta(minutes=10), router=router, tokens=200, spend=1.0, saved=3.0, user_id="u1") + await _turn(db, key, "B", midnight + timedelta(days=1), router=router, tokens=300, spend=1.0, saved=11.0, user_id="u1") assert (await _row(db, key, router=router))["saved_spend"] == 21.0 days = await db.query_raw( @@ -663,6 +665,7 @@ async def test_a_cross_midnight_session_splits_its_money_by_request_day(db): (selected,) = await _benchmark_rows(db, midnight, midnight + timedelta(days=1), key, user_id) assert (selected["sessions"], selected["session_turns"]) == (1, 3) assert (selected["turns"], selected["spend"], selected["saved_spend"]) == (1, 1.0, 3.0) + assert (selected["day_total_tokens"], selected["total_tokens"]) == (200, 600) async def test_a_router_type_change_within_a_day_keeps_each_types_money_apart(db): @@ -704,6 +707,7 @@ async def test_a_sessionless_turn_writes_its_router_day_row_and_no_session_row(d model="A", turn_at=T0 + timedelta(seconds=offset), total_tokens=10, + token_counts_recorded=True, spend=1.0, saved_spend=2.0, classifier_cost=0.1, @@ -720,11 +724,48 @@ async def test_a_sessionless_turn_writes_its_router_day_row_and_no_session_row(d (day,) = await _days(db, key, router=router) assert (day["turns"], day["spend"], day["saved_spend"], day["classifier_cost"]) == (2, 2.0, 4.0, 0.2) + assert day["day_total_tokens"] == 20 assert (day["sessions"], day["session_turns"]) == (0, 0) for table in ("LiteLLM_AutoRouterSession", "LiteLLM_AutoRouterUserSession"): assert await db.query_raw(f'SELECT 1 FROM "{table}" WHERE router_name = $1', router) == [] +@pytest.mark.parametrize("historical", [True, False]) +async def test_daily_token_coverage_stays_unknown_with_old_writers(db: Prisma, historical: bool) -> None: + key: Final = f"k-{uuid.uuid4()}" + router: Final = f"auto-{uuid.uuid4()}" + if historical: + await db.execute_raw( + 'INSERT INTO "LiteLLM_AutoRouterDailySpend" ' + '(date, api_key, user_id, router_name, router_type, turns, spend) ' + "VALUES ($1, $2, 'u1', $3, 'complexity', 1, 1)", + T0.date().isoformat(), key, router, + ) + await _turn(db, key, "A", T0, router=router, tokens=123, spend=1.0, user_id="u1") + if not historical: + await db.execute_raw( + 'UPDATE "LiteLLM_AutoRouterDailySpend" SET turns = turns + 1, spend = spend + 1 ' + 'WHERE api_key = $1 AND router_name = $2', key, router, + ) + for user_id in (None, "u1"): + (day,) = await _days(db, key, user_id, router) + assert (day["turns"], day["spend"], day["day_total_tokens"]) == (2, 2.0, None) + + +@pytest.mark.parametrize("missing_first", [True, False]) +async def test_missing_usage_never_completes_daily_token_coverage(db: Prisma, missing_first: bool) -> None: + key: Final = f"k-{uuid.uuid4()}" + router: Final = f"auto-{uuid.uuid4()}" + for offset, recorded in enumerate((not missing_first, missing_first)): + await _turn( + db, key, "A", T0 + timedelta(seconds=offset), router=router, + tokens=100 if recorded else 0, spend=1.0, user_id="u1", token_counts_recorded=recorded, + ) + for user_id in (None, "u1"): + (day,) = await _days(db, key, user_id, router) + assert (day["turns"], day["spend"], day["day_total_tokens"]) == (2, 2.0, None) + + async def test_router_day_money_reconciles_with_the_overall_daily_total_including_sessionless_requests(db): from litellm.proxy.db.daily_spend_bulk_upsert import DAILY_SPEND_TABLES, build_bulk_upsert, merge_by_conflict_key diff --git a/tests/unit/integrations/test_prometheus_input_sequence_length_label.py b/tests/unit/integrations/test_prometheus_input_sequence_length_label.py index bc922061544..9cfd4970580 100644 --- a/tests/unit/integrations/test_prometheus_input_sequence_length_label.py +++ b/tests/unit/integrations/test_prometheus_input_sequence_length_label.py @@ -212,8 +212,10 @@ async def test_logger_distinguishes_missing_usage_from_reported_zero( now: Final = datetime.datetime.now() monkeypatch.setattr(litellm, FLAG, True) logger: Final = PrometheusLogger() + response_obj: Final = response.model_dump() if isinstance(response, litellm.ModelResponse) else response usage: Final = StandardLoggingPayloadSetup.get_usage_as_dict( - response_obj=response if isinstance(response, dict) else None + response_obj=response_obj if isinstance(response_obj, dict) else None, + combined_usage_object=combined_usage if isinstance(combined_usage, litellm.Usage) else None, ) payload: Final = _standard_logging_payload(now, usage.get("prompt_tokens", 0)) diff --git a/tests/unit/litellm_core_utils/test_litellm_logging.py b/tests/unit/litellm_core_utils/test_litellm_logging.py index 66e54a9eba6..3fbc425da7f 100644 --- a/tests/unit/litellm_core_utils/test_litellm_logging.py +++ b/tests/unit/litellm_core_utils/test_litellm_logging.py @@ -3714,11 +3714,11 @@ def test_get_usage_as_dict(): # Test case 1: None response_obj returns empty usage dict result = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj=None) - assert result == {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0} + assert result == {} # Test case 2: Empty response_obj returns empty usage dict result = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj={}) - assert result == {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0} + assert result == {} # Test case 3: combined_usage_object takes priority combined = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15) @@ -3738,7 +3738,35 @@ def test_get_usage_as_dict(): # Test case 5: response_obj with no usage key returns empty result = StandardLoggingPayloadSetup.get_usage_as_dict(response_obj={"id": "resp-1", "choices": []}) - assert result == {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0} + assert result == {} + + +@pytest.mark.parametrize( + "usage, include_usage", + [(None, False), (None, True), ({"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}, True)], +) +def test_logging_preserves_missing_usage_without_accepting_request_metadata( + logging_obj: Logging, usage: dict[str, int] | None, include_usage: bool +) -> None: + from litellm.litellm_core_utils.litellm_logging import get_standard_logging_object_payload + from litellm.litellm_core_utils.llm_response_utils.convert_dict_to_response import convert_to_model_response_object + + now: Final = datetime_unit_test(2026, 1, 1, 12, 0, 0) + response: Final[ModelResponse] = convert_to_model_response_object( + response_object={"id": "usage-coverage", "choices": [], **({"usage": usage} if include_usage else {})}, + model_response_object=ModelResponse(), + ) + payload: Final = get_standard_logging_object_payload( + kwargs={"litellm_params": {"metadata": {"usage_object": {"prompt_tokens": 99, "completion_tokens": 99}}}}, + init_response_obj=response, + start_time=now, + end_time=now, + logging_obj=logging_obj, + status="success", + ) + assert payload is not None + assert payload["metadata"]["usage_object"] == (response.usage.model_dump() if usage is not None else {}) + assert (payload["prompt_tokens"], payload["completion_tokens"], payload["total_tokens"]) == (0, 0, 0) def test_append_system_prompt_messages(): diff --git a/tests/unit/proxy/db/test_autorouter_session_rollup.py b/tests/unit/proxy/db/test_autorouter_session_rollup.py index 659d29cda16..686c5a3a970 100644 --- a/tests/unit/proxy/db/test_autorouter_session_rollup.py +++ b/tests/unit/proxy/db/test_autorouter_session_rollup.py @@ -14,6 +14,7 @@ from typing import Final import httpx import pytest +from pydantic import TypeAdapter from litellm.proxy.db.autorouter_session_rollup import ( UPSERT_AUTOROUTER_SESSION_SQL, @@ -45,7 +46,10 @@ def _payload(**overrides: object) -> dict: def _metadata(**overrides: object) -> dict: - base: dict = {"routing_decision": dict(ROUTING_DECISION), "usage_object": {"prompt_tokens": 90}} + base: dict = { + "routing_decision": dict(ROUTING_DECISION), + "usage_object": {"prompt_tokens": 90, "completion_tokens": 10}, + } base.update(overrides) return base @@ -88,7 +92,10 @@ class TestBuildTransaction: transaction = _build( metadata=_metadata( routing_decision={**ROUTING_DECISION, "savings_baseline_model": "anthropic/claude-opus-5"}, - usage_object={"prompt_tokens": 90, "cache_read_input_tokens": 5, "cache_creation_input_tokens": 7}, + usage_object={ + "prompt_tokens": 90, "completion_tokens": 10, + "cache_read_input_tokens": 5, "cache_creation_input_tokens": 7, + }, ) ) assert transaction == AutoRouterTurnTransaction( @@ -99,6 +106,7 @@ class TestBuildTransaction: model="bedrock/haiku", turn_at=datetime(2026, 8, 1, 12, 0, 0), total_tokens=100, + token_counts_recorded=True, spend=0.01, saved_spend=0.02, classifier_cost=0.0, @@ -213,6 +221,30 @@ class TestBuildTransaction: assert transaction.cache_ttl_seconds is None assert transaction.cache_touched is True + @pytest.mark.parametrize( + "usage, recorded", + [ + (None, False), ({}, False), ({"prompt_tokens": 90}, False), + ({"prompt_tokens": 90, "completion_tokens": 10}, True), + ({"prompt_tokens": 0, "completion_tokens": 0}, True), + ({"prompt_tokens": -1, "completion_tokens": 10}, False), + ({"prompt_tokens": True, "completion_tokens": 10}, False), + ({"prompt_tokens": "90", "completion_tokens": 10}, False), + ], + ) + def test_token_coverage_requires_complete_reported_counts(self, usage: object, recorded: bool) -> None: + transaction: Final = _build(metadata=_metadata(usage_object=usage)) + assert transaction is not None + assert transaction.token_counts_recorded is recorded + + def test_persisted_turns_preserve_coverage_and_default_old_records_to_unknown(self) -> None: + transaction: Final = _build() + adapter: Final = TypeAdapter(AutoRouterTurnTransaction) + assert transaction is not None + assert adapter.validate_json(adapter.dump_json(transaction)).token_counts_recorded is True + legacy: Final = adapter.dump_json(transaction, exclude={"token_counts_recorded"}) + assert adapter.validate_json(legacy).token_counts_recorded is False + def test_a_covered_turn_that_neither_read_nor_wrote_did_not_touch_the_cache(self): transaction = _build() assert transaction is not None @@ -341,6 +373,7 @@ class TestFlush: 0.0, 0.0, "canonical-user", + 0, ) def test_a_keys_turns_stay_chronological_when_its_canonical_user_changes(self) -> None: diff --git a/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py index c0af21671ba..360fc5cc292 100644 --- a/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py +++ b/tests/unit/proxy/management_endpoints/test_auto_router_endpoints.py @@ -770,6 +770,7 @@ class TestAutoRouterBenchmarks: ttl_5m_turns=30, ttl_1h_turns=5, total_tokens=4000, + day_total_tokens=2500, spend=10.0, saved_spend=30.0, savings_estimated_turns=40, @@ -798,6 +799,7 @@ class TestAutoRouterBenchmarks: assert totals.avg_turns_per_session == 10.0 assert totals.avg_session_seconds == 100.0 assert totals.avg_tokens_per_session == 1000.0 + assert totals.total_tokens == 2500 assert totals.baseline_spend == 40.0 assert totals.saved_pct == 75.0 assert totals.savings_estimated_classifier_cost == 0.4 @@ -818,6 +820,20 @@ class TestAutoRouterBenchmarks: assert totals.saved_pct == -100.0 assert totals.classifier_cost == 0.4 + @pytest.mark.asyncio + @pytest.mark.parametrize("historical_tokens, expected_total", [(None, None), (0, 2500), (750, 3250)]) + async def test_daily_token_totals_preserve_missing_coverage_in_any_router( + self, historical_tokens: int | None, expected_total: int | None, monkeypatch: pytest.MonkeyPatch + ) -> None: + historical: Final = self.ROW.model_copy( + update={"router_name": "historical-auto", "day_total_tokens": historical_tokens} + ) + response: Final = await self._benchmarks( + monkeypatch, rows=[self.ROW.model_dump(), historical.model_dump()], model_list=[] + ) + assert [group.total_tokens for group in response.groups] == [2500, historical_tokens] + assert response.totals.total_tokens == expected_total + @pytest.mark.asyncio @pytest.mark.parametrize("estimated_turns", [0, 4]) async def test_historical_savings_without_recorded_baselines_compare_against_all_spend( @@ -921,6 +937,7 @@ class TestAutoRouterBenchmarks: totals = _benchmark_totals(_summed_agg_row([])) assert totals.sessions == 0 assert totals.turns == 0 + assert totals.total_tokens == 0 assert totals.saved_pct == 0.0 assert totals.cache.hit_rate_pct == 0.0 assert totals.classifier_cost == 0.0 @@ -1164,6 +1181,7 @@ class TestAutoRouterBenchmarks: assert idle.cache.same_model.turns == idle.cache.return_to_tier.hits == 0 assert idle.tier_turns == {} assert idle.classifier_cost == 0.0 + assert idle.total_tokens == 0 @pytest.mark.asyncio @pytest.mark.parametrize( diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx index 081eb7f6e09..ef55e26efc8 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.test.tsx @@ -8,6 +8,7 @@ import { ApiError } from "@/lib/http/client"; vi.mock("./useAutoRouterBenchmarks", () => ({ useAutoRouterBenchmarks: vi.fn() })); vi.mock("@/app/(dashboard)/hooks/models/useModels", () => ({ useAutoRouters: vi.fn() })); +vi.mock("./AutoRouterSummaryTable", () => ({ default: () =>
})); vi.mock("./ShadowEvalSection", () => ({ default: () => })); vi.mock("@/components/shared/advanced_date_picker", () => ({ __esModule: true, diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx index 27e8df6db87..b3acb8f9168 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx @@ -25,12 +25,14 @@ import { groupLabel, pctLabel, viewFor, + viewGroup, type AutoRouterBenchmarksResponse, type AutoRouterCacheStats, type BenchmarkView, type BucketRow, } from "./autoRouterBenchmarks"; import { classificationRatePer1kTurns, formatRangeLabel, usd } from "./costOptimizationUtils"; +import AutoRouterSummaryTable from "./AutoRouterSummaryTable"; import ShadowEvalSection from "./ShadowEvalSection"; import TierTurnsChart from "./TierTurnsChart"; import { useAutoRouterBenchmarks } from "./useAutoRouterBenchmarks"; @@ -79,7 +81,7 @@ const HeroCard: React.FC<{ view: BenchmarkView }> = ({ view }) => { const classifierCost = stats.baseline_spend == null ? null : stats.savings_estimated_classifier_cost ?? null; const comparedAll = stats.savings_estimated_turns === stats.turns; return ( -
@@ -332,6 +334,8 @@ const BenchmarksBody: React.FC
+ Savings are estimated against each router's premium baseline. Token totals and unit costs are unavailable + for usage recorded before token tracking; unit cost also requires nonzero tokens. +
+