feat(ui): lead shadow eval results with the models compared against

The expanded results led with the router's internal difficulty tiers
(SIMPLE, REASONING, …) — jargon that answers a question nobody asked —
while the natural question, 'which of my models was the router compared
to?', wasn't answered anywhere. A key's traffic can mix models, so the
comparison isn't against one incumbent.

The verdict rollup now groups by (tier, real_model) in one query, and
ShadowEvalResult carries by_current_model alongside the tier groups.
The primary results table is 'Compared against': one row per model that
actually served the sampled requests, shown even for a single-model key
so the incumbent is named. The tier table moves behind a 'By prompt
difficulty' toggle with a plain-language description; the column is
renamed from 'Router tier' with a tooltip explaining what tiers are.
Results predating the by-model slice fall back to the tier table.

Tier judge-confidence is now turn-weighted across the merged rollup
rows rather than a per-row average.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Abhimanyu Kapur 2026-08-08 20:12:52 -07:00
parent f339938587
commit b61404ec53
8 changed files with 304 additions and 18 deletions

View file

@ -4,7 +4,8 @@ AUTO ROUTER MANAGEMENT ENDPOINTS
POST /auto_router/test_routing - Route one prompt through an unsaved complexity-router config
"""
from collections.abc import Mapping, Sequence
from collections.abc import Callable, Mapping, Sequence
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
from types import MappingProxyType
from typing import TYPE_CHECKING, Annotated, Final
@ -41,6 +42,7 @@ from litellm.types.management_endpoints.auto_router_endpoints import (
AutoRouterRoutingTestResponse,
GetShadowEvalJobResponse,
RequestComplexityRouterConfig,
ShadowEvalModelResult,
ShadowEvalResult,
ShadowEvalStatus,
ShadowEvalTierResult,
@ -540,6 +542,7 @@ def _estimate_judge_cost_per_call(judge_model: str) -> float:
class _VerdictAggRow(BaseModel):
tier_classification: str | None
real_model: str
turn_count: int
real_wins: int
shadow_wins: int
@ -552,6 +555,7 @@ _VERDICT_AGG_ROWS: Final = TypeAdapter(list[_VerdictAggRow])
_VERDICT_AGG_SQL: Final = """
SELECT
tier_classification,
real_model,
COUNT(*)::int AS turn_count,
COUNT(*) FILTER (WHERE judge_preference = 'real')::int AS real_wins,
COUNT(*) FILTER (WHERE judge_preference = 'shadow')::int AS shadow_wins,
@ -559,29 +563,76 @@ SELECT
AVG(judge_confidence)::float AS avg_confidence
FROM "LiteLLM_ShadowEvalVerdict"
WHERE job_id = $1
GROUP BY tier_classification
GROUP BY tier_classification, real_model
"""
@dataclass
class _VerdictTally:
turns: int = 0
real_wins: int = 0
shadow_wins: int = 0
ties: int = 0
confidence_weighted: float = 0.0
def absorb(self, row: _VerdictAggRow) -> None:
self.turns += row.turn_count
self.real_wins += row.real_wins
self.shadow_wins += row.shadow_wins
self.ties += row.ties
self.confidence_weighted += (row.avg_confidence or 0.0) * row.turn_count
@property
def avg_confidence(self) -> float:
return round(self.confidence_weighted / self.turns, 3) if self.turns else 0.0
def _tally_by(rows: Sequence[_VerdictAggRow], key: Callable[[_VerdictAggRow], str]) -> Mapping[str, _VerdictTally]:
tallies: Final[dict[str, _VerdictTally]] = {} # mutable-ok: aggregation accumulator
for row in rows:
tallies.setdefault(key(row), _VerdictTally()).absorb(row)
return tallies
def _shadow_eval_results(rows: Sequence[_VerdictAggRow]) -> ShadowEvalResult | None:
"""Both stratifications of one job's verdicts, from a single (tier, model) rollup.
Tier answers "where does the router do well"; current-model answers "which of the
models this key uses today would the router beat", which only says anything when
the key's traffic is a mix rather than a single incumbent.
"""
if not rows:
return None
total_turns: Final = sum(r.turn_count for r in rows)
total_shadow_wins: Final = sum(r.shadow_wins for r in rows)
total_ties: Final = sum(r.ties for r in rows)
by_tier: Final = _tally_by(rows, lambda r: r.tier_classification or "UNCLASSIFIED")
groups: Final = tuple(
ShadowEvalTierResult(
tier=row.tier_classification or "UNCLASSIFIED",
turn_count=row.turn_count,
real_win_rate_pct=_pct(row.real_wins, row.turn_count),
shadow_win_rate_pct=_pct(row.shadow_wins, row.turn_count),
tie_rate_pct=_pct(row.ties, row.turn_count),
avg_judge_confidence=round(row.avg_confidence or 0.0, 3),
tier=tier,
turn_count=tally.turns,
real_win_rate_pct=_pct(tally.real_wins, tally.turns),
shadow_win_rate_pct=_pct(tally.shadow_wins, tally.turns),
tie_rate_pct=_pct(tally.ties, tally.turns),
avg_judge_confidence=tally.avg_confidence,
)
for row in sorted(rows, key=lambda r: r.turn_count, reverse=True)
for tier, tally in sorted(by_tier.items(), key=lambda kv: kv[1].turns, reverse=True)
)
by_model: Final = _tally_by(rows, lambda r: r.real_model)
model_groups: Final = tuple(
ShadowEvalModelResult(
current_model=model,
turn_count=tally.turns,
real_win_rate_pct=_pct(tally.real_wins, tally.turns),
shadow_win_rate_pct=_pct(tally.shadow_wins, tally.turns),
tie_rate_pct=_pct(tally.ties, tally.turns),
avg_judge_confidence=tally.avg_confidence,
)
for model, tally in sorted(by_model.items(), key=lambda kv: kv[1].turns, reverse=True)
)
return ShadowEvalResult(
groups=groups,
by_current_model=model_groups,
overall_shadow_win_rate_pct=_pct(total_shadow_wins, total_turns),
overall_tie_rate_pct=_pct(total_ties, total_turns),
)

View file

@ -214,10 +214,26 @@ class ShadowEvalTierResult(BaseModel):
avg_judge_confidence: float
class ShadowEvalModelResult(BaseModel):
"""Judge outcomes against one of the models the shadowed key currently uses.
A key's real traffic can be a mix of models; per-tier win rates blend those
incumbents together, so this slice answers which of them the router actually beat.
"""
current_model: str
turn_count: int
real_win_rate_pct: float
shadow_win_rate_pct: float
tie_rate_pct: float
avg_judge_confidence: float
class ShadowEvalResult(BaseModel):
"""Stratified results of a completed (or in-progress) shadow-eval job."""
groups: tuple[ShadowEvalTierResult, ...]
by_current_model: tuple[ShadowEvalModelResult, ...] = ()
overall_shadow_win_rate_pct: float
overall_tie_rate_pct: float

View file

@ -674,6 +674,81 @@ class TestShadowEvalJobsAreTimeBound:
assert response.ends_at == "2026-08-08T00:00:00+00:00"
class TestShadowEvalResultsStratifyByTierAndByCurrentModel:
"""A key's real traffic can mix models. Per-tier rates blend the incumbents, so
the results also slice by real_model — which of today's models did the router beat."""
@staticmethod
def _row(tier: str | None, real_model: str, real_wins: int, shadow_wins: int, ties: int, conf: float):
from litellm.proxy.management_endpoints.auto_router_endpoints import _VerdictAggRow
return _VerdictAggRow(
tier_classification=tier,
real_model=real_model,
turn_count=real_wins + shadow_wins + ties,
real_wins=real_wins,
shadow_wins=shadow_wins,
ties=ties,
avg_confidence=conf,
)
def test_same_rollup_produces_both_stratifications(self):
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
rows = [
self._row("SIMPLE", "gpt-4o", 1, 8, 1, 0.9),
self._row("SIMPLE", "my-finetune", 6, 2, 2, 0.7),
self._row("REASONING", "gpt-4o", 3, 5, 2, 0.8),
]
result = _shadow_eval_results(rows)
assert result is not None
simple = next(g for g in result.groups if g.tier == "SIMPLE")
assert simple.turn_count == 20
assert simple.shadow_win_rate_pct == 50.0
gpt4o = next(m for m in result.by_current_model if m.current_model == "gpt-4o")
finetune = next(m for m in result.by_current_model if m.current_model == "my-finetune")
assert gpt4o.turn_count == 20
assert gpt4o.shadow_win_rate_pct == 65.0
assert finetune.turn_count == 10
assert finetune.shadow_win_rate_pct == 20.0
def test_tier_confidence_is_turn_weighted_not_a_plain_average(self):
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
rows = [
self._row("SIMPLE", "gpt-4o", 0, 9, 0, 1.0),
self._row("SIMPLE", "my-finetune", 1, 0, 0, 0.0),
]
result = _shadow_eval_results(rows)
assert result is not None
assert result.groups[0].avg_judge_confidence == 0.9
def test_single_incumbent_still_reports_one_model_slice(self):
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
result = _shadow_eval_results([self._row("SIMPLE", "gpt-4o", 2, 6, 2, 0.8)])
assert result is not None
assert len(result.by_current_model) == 1
assert result.by_current_model[0].current_model == "gpt-4o"
def test_overall_rates_are_unchanged_by_the_extra_grouping_column(self):
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
rows = [
self._row("SIMPLE", "gpt-4o", 1, 8, 1, 0.9),
self._row("SIMPLE", "my-finetune", 6, 2, 2, 0.7),
]
result = _shadow_eval_results(rows)
assert result is not None
assert result.overall_shadow_win_rate_pct == 50.0
assert result.overall_tie_rate_pct == 15.0
class TestShadowEvalJobLifecycleEndpoints:
"""List shows every job newest-first, get returns one job with its results, and
stop flips only active jobs while keeping the verdicts already collected."""

View file

@ -82,6 +82,24 @@ const job = (overrides: Partial<ShadowEvalJob> = {}): ShadowEvalJob => ({
avg_judge_confidence: 0.74,
},
],
by_current_model: [
{
current_model: "gpt-4o",
turn_count: 30,
real_win_rate_pct: 30.0,
shadow_win_rate_pct: 45.0,
tie_rate_pct: 25.0,
avg_judge_confidence: 0.8,
},
{
current_model: "my-finetune",
turn_count: 12,
real_win_rate_pct: 25.0,
shadow_win_rate_pct: 58.3,
tie_rate_pct: 16.7,
avg_judge_confidence: 0.75,
},
],
overall_shadow_win_rate_pct: 48.0,
overall_tie_rate_pct: 22.0,
},
@ -132,15 +150,31 @@ describe("ShadowEvalSection", () => {
expect(screen.getByText("Start shadow eval")).toBeInTheDocument();
});
it("renders per-tier win rates for an active job", () => {
it("leads with the models the router was compared against", () => {
const j = job();
mockHooks({ jobs: [j], detail: j });
render(<ShadowEvalSection accessToken="token" />);
expect(screen.getByText("Compared against")).toBeInTheDocument();
expect(screen.getByText("gpt-4o")).toBeInTheDocument();
expect(screen.getByText("my-finetune")).toBeInTheDocument();
expect(screen.getByText("58.3%")).toBeInTheDocument();
// Overall matched-or-beat = shadow wins + ties = 70%
expect(screen.getByText("70.0%")).toBeInTheDocument();
// Tier rows are behind the difficulty toggle, not in the primary view.
expect(screen.queryByText("SIMPLE")).not.toBeInTheDocument();
});
it("reveals per-tier win rates behind the prompt-difficulty toggle", async () => {
const user = userEvent.setup();
const j = job();
mockHooks({ jobs: [j], detail: j });
render(<ShadowEvalSection accessToken="token" />);
await user.click(screen.getByText(/By prompt difficulty/));
expect(screen.getByText("SIMPLE")).toBeInTheDocument();
expect(screen.getByText("REASONING")).toBeInTheDocument();
expect(screen.getByText("55.0%")).toBeInTheDocument();
// Overall matched-or-beat = shadow wins + ties = 70%
expect(screen.getByText("70.0%")).toBeInTheDocument();
});
it("states the metric's denominator next to the headline number", () => {
@ -172,12 +206,23 @@ describe("ShadowEvalSection", () => {
expect(screen.getByText(/LLM judge compares the two answers blind/)).toBeInTheDocument();
});
it("flags low-sample tiers", () => {
it("flags low-sample rows", () => {
const j = job();
mockHooks({ jobs: [j], detail: j });
render(<ShadowEvalSection accessToken="token" />);
// REASONING tier has 12 turns < 30
expect(screen.getByText("(low sample)")).toBeInTheDocument();
// my-finetune has 12 turns < 30
expect(screen.getAllByText("(low sample)")).toHaveLength(1);
});
it("falls back to the tier table when results predate the by-model slice", () => {
const j = job();
const results = j.results!;
const legacy = job({ results: { ...results, by_current_model: [] } });
mockHooks({ jobs: [legacy], detail: legacy });
render(<ShadowEvalSection accessToken="token" />);
expect(screen.getByText("SIMPLE")).toBeInTheDocument();
expect(screen.queryByText("Compared against")).not.toBeInTheDocument();
expect(screen.queryByText(/By prompt difficulty/)).not.toBeInTheDocument();
});
it("shows a stop button for running jobs but not completed ones", () => {

View file

@ -25,6 +25,7 @@ import {
useStartShadowEval,
useStopShadowEval,
type ShadowEvalJob,
type ShadowEvalModelResult,
type ShadowEvalTierResult,
} from "./useShadowEval";
@ -116,7 +117,10 @@ const TierResultsTable: React.FC<{ groups: readonly ShadowEvalTierResult[] }> =
<Table>
<TableHeader>
<TableRow>
<TableHead>Router tier</TableHead>
<HeadTooltip
label="Prompt difficulty"
tooltip="The router sorts every prompt into a difficulty tier and picks a model sized to it. Rows where the router holds up on hard tiers matter most."
/>
<TableHead className="text-right">Judged turns</TableHead>
<TableHead className="text-right">Router wins</TableHead>
<TableHead className="text-right">Current model wins</TableHead>
@ -152,6 +156,69 @@ const TierResultsTable: React.FC<{ groups: readonly ShadowEvalTierResult[] }> =
</Table>
);
const ModelResultsTable: React.FC<{ models: readonly ShadowEvalModelResult[] }> = ({ models }) => (
<Table>
<TableHeader>
<TableRow>
<HeadTooltip
label="Compared against"
tooltip="The model that actually served each sampled request. Your traffic can mix models; each row is the router's head-to-head record against one of them."
/>
<TableHead className="text-right">Judged turns</TableHead>
<TableHead className="text-right">Router wins</TableHead>
<TableHead className="text-right">Current model wins</TableHead>
<HeadTooltip
className="text-right"
label="Ties"
tooltip="The judge saw the same quality from both. Ties count in the router's favor — same answer quality, usually at lower cost."
/>
<HeadTooltip
className="text-right"
label="Judge confidence"
tooltip="The judge's self-reported certainty in its verdicts (0 to 1), averaged over these turns."
/>
</TableRow>
</TableHeader>
<TableBody>
{models.map((m) => (
<TableRow key={m.current_model}>
<TableCell className="font-mono text-xs text-foreground">
{m.current_model}
{m.turn_count < MIN_TURNS_FOR_CONFIDENCE ? <LowSampleFlag /> : null}
</TableCell>
<TableCell className="text-right tabular-nums">{m.turn_count.toLocaleString()}</TableCell>
<TableCell className="text-right font-medium tabular-nums text-foreground">
{pct(m.shadow_win_rate_pct)}
</TableCell>
<TableCell className="text-right tabular-nums">{pct(m.real_win_rate_pct)}</TableCell>
<TableCell className="text-right tabular-nums">{pct(m.tie_rate_pct)}</TableCell>
<TableCell className="text-right tabular-nums">{m.avg_judge_confidence.toFixed(2)}</TableCell>
</TableRow>
))}
</TableBody>
</Table>
);
const TierBreakdown: React.FC<{ groups: readonly ShadowEvalTierResult[] }> = ({ groups }) => {
const [open, setOpen] = useState(false);
return (
<div className="border-t">
<button
type="button"
aria-expanded={open}
onClick={() => setOpen((prev) => !prev)}
className="flex w-full items-center justify-between gap-3 px-6 py-3 text-left hover:bg-muted/50"
>
<span className="text-xs text-muted-foreground">
By prompt difficulty — how the router scored across the difficulty tiers it sorted prompts into
</span>
<span className="text-xs text-muted-foreground">{open ? "Hide" : "Show"}</span>
</button>
{open ? <TierResultsTable groups={groups} /> : null}
</div>
);
};
const MetricLabel: React.FC<{ label: string; tooltip: string }> = ({ label, tooltip }) => (
<TooltipProvider delay={200}>
<Tooltip>
@ -242,7 +309,12 @@ const JobResults: React.FC<{
<p className="text-xs text-muted-foreground">of {job.completed_count.toLocaleString()} judged responses</p>
</div>
<VerdictBar results={results} />
<TierResultsTable groups={results.groups} />
{(results.by_current_model?.length ?? 0) > 0 ? (
<ModelResultsTable models={results.by_current_model ?? []} />
) : (
<TierResultsTable groups={results.groups} />
)}
{(results.by_current_model?.length ?? 0) > 0 ? <TierBreakdown groups={results.groups} /> : null}
</>
) : (
<p className="px-6 py-8 text-center text-sm text-muted-foreground">

View file

@ -6,6 +6,7 @@ import type { components } from "@/lib/http/schema";
export type ShadowEvalJob = components["schemas"]["GetShadowEvalJobResponse"];
export type ShadowEvalTierResult = components["schemas"]["ShadowEvalTierResult"];
export type ShadowEvalModelResult = components["schemas"]["ShadowEvalModelResult"];
export type StartShadowEvalRequest = components["schemas"]["StartShadowEvalRequest"];
const JOBS_PATH = "/auto_router/shadow_eval" as const;

View file

@ -32579,11 +32579,37 @@ export interface components {
/** Timeout */
timeout?: number | null;
};
/**
* ShadowEvalModelResult
* @description Judge outcomes against one of the models the shadowed key currently uses.
*
* A key's real traffic can be a mix of models; per-tier win rates blend those
* incumbents together, so this slice answers which of them the router actually beat.
*/
ShadowEvalModelResult: {
/** Avg Judge Confidence */
avg_judge_confidence: number;
/** Current Model */
current_model: string;
/** Real Win Rate Pct */
real_win_rate_pct: number;
/** Shadow Win Rate Pct */
shadow_win_rate_pct: number;
/** Tie Rate Pct */
tie_rate_pct: number;
/** Turn Count */
turn_count: number;
};
/**
* ShadowEvalResult
* @description Stratified results of a completed (or in-progress) shadow-eval job.
*/
ShadowEvalResult: {
/**
* By Current Model
* @default []
*/
by_current_model: components["schemas"]["ShadowEvalModelResult"][];
/** Groups */
groups: components["schemas"]["ShadowEvalTierResult"][];
/** Overall Shadow Win Rate Pct */

File diff suppressed because one or more lines are too long