mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
feat(ui): lead shadow eval results with the models compared against
The expanded results led with the router's internal difficulty tiers (SIMPLE, REASONING, …) — jargon that answers a question nobody asked — while the natural question, 'which of my models was the router compared to?', wasn't answered anywhere. A key's traffic can mix models, so the comparison isn't against one incumbent. The verdict rollup now groups by (tier, real_model) in one query, and ShadowEvalResult carries by_current_model alongside the tier groups. The primary results table is 'Compared against': one row per model that actually served the sampled requests, shown even for a single-model key so the incumbent is named. The tier table moves behind a 'By prompt difficulty' toggle with a plain-language description; the column is renamed from 'Router tier' with a tooltip explaining what tiers are. Results predating the by-model slice fall back to the tier table. Tier judge-confidence is now turn-weighted across the merged rollup rows rather than a per-row average. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
f339938587
commit
b61404ec53
8 changed files with 304 additions and 18 deletions
|
|
@ -4,7 +4,8 @@ AUTO ROUTER MANAGEMENT ENDPOINTS
|
|||
POST /auto_router/test_routing - Route one prompt through an unsaved complexity-router config
|
||||
"""
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from types import MappingProxyType
|
||||
from typing import TYPE_CHECKING, Annotated, Final
|
||||
|
|
@ -41,6 +42,7 @@ from litellm.types.management_endpoints.auto_router_endpoints import (
|
|||
AutoRouterRoutingTestResponse,
|
||||
GetShadowEvalJobResponse,
|
||||
RequestComplexityRouterConfig,
|
||||
ShadowEvalModelResult,
|
||||
ShadowEvalResult,
|
||||
ShadowEvalStatus,
|
||||
ShadowEvalTierResult,
|
||||
|
|
@ -540,6 +542,7 @@ def _estimate_judge_cost_per_call(judge_model: str) -> float:
|
|||
|
||||
class _VerdictAggRow(BaseModel):
|
||||
tier_classification: str | None
|
||||
real_model: str
|
||||
turn_count: int
|
||||
real_wins: int
|
||||
shadow_wins: int
|
||||
|
|
@ -552,6 +555,7 @@ _VERDICT_AGG_ROWS: Final = TypeAdapter(list[_VerdictAggRow])
|
|||
_VERDICT_AGG_SQL: Final = """
|
||||
SELECT
|
||||
tier_classification,
|
||||
real_model,
|
||||
COUNT(*)::int AS turn_count,
|
||||
COUNT(*) FILTER (WHERE judge_preference = 'real')::int AS real_wins,
|
||||
COUNT(*) FILTER (WHERE judge_preference = 'shadow')::int AS shadow_wins,
|
||||
|
|
@ -559,29 +563,76 @@ SELECT
|
|||
AVG(judge_confidence)::float AS avg_confidence
|
||||
FROM "LiteLLM_ShadowEvalVerdict"
|
||||
WHERE job_id = $1
|
||||
GROUP BY tier_classification
|
||||
GROUP BY tier_classification, real_model
|
||||
"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class _VerdictTally:
|
||||
turns: int = 0
|
||||
real_wins: int = 0
|
||||
shadow_wins: int = 0
|
||||
ties: int = 0
|
||||
confidence_weighted: float = 0.0
|
||||
|
||||
def absorb(self, row: _VerdictAggRow) -> None:
|
||||
self.turns += row.turn_count
|
||||
self.real_wins += row.real_wins
|
||||
self.shadow_wins += row.shadow_wins
|
||||
self.ties += row.ties
|
||||
self.confidence_weighted += (row.avg_confidence or 0.0) * row.turn_count
|
||||
|
||||
@property
|
||||
def avg_confidence(self) -> float:
|
||||
return round(self.confidence_weighted / self.turns, 3) if self.turns else 0.0
|
||||
|
||||
|
||||
def _tally_by(rows: Sequence[_VerdictAggRow], key: Callable[[_VerdictAggRow], str]) -> Mapping[str, _VerdictTally]:
|
||||
tallies: Final[dict[str, _VerdictTally]] = {} # mutable-ok: aggregation accumulator
|
||||
for row in rows:
|
||||
tallies.setdefault(key(row), _VerdictTally()).absorb(row)
|
||||
return tallies
|
||||
|
||||
|
||||
def _shadow_eval_results(rows: Sequence[_VerdictAggRow]) -> ShadowEvalResult | None:
|
||||
"""Both stratifications of one job's verdicts, from a single (tier, model) rollup.
|
||||
|
||||
Tier answers "where does the router do well"; current-model answers "which of the
|
||||
models this key uses today would the router beat", which only says anything when
|
||||
the key's traffic is a mix rather than a single incumbent.
|
||||
"""
|
||||
if not rows:
|
||||
return None
|
||||
total_turns: Final = sum(r.turn_count for r in rows)
|
||||
total_shadow_wins: Final = sum(r.shadow_wins for r in rows)
|
||||
total_ties: Final = sum(r.ties for r in rows)
|
||||
by_tier: Final = _tally_by(rows, lambda r: r.tier_classification or "UNCLASSIFIED")
|
||||
groups: Final = tuple(
|
||||
ShadowEvalTierResult(
|
||||
tier=row.tier_classification or "UNCLASSIFIED",
|
||||
turn_count=row.turn_count,
|
||||
real_win_rate_pct=_pct(row.real_wins, row.turn_count),
|
||||
shadow_win_rate_pct=_pct(row.shadow_wins, row.turn_count),
|
||||
tie_rate_pct=_pct(row.ties, row.turn_count),
|
||||
avg_judge_confidence=round(row.avg_confidence or 0.0, 3),
|
||||
tier=tier,
|
||||
turn_count=tally.turns,
|
||||
real_win_rate_pct=_pct(tally.real_wins, tally.turns),
|
||||
shadow_win_rate_pct=_pct(tally.shadow_wins, tally.turns),
|
||||
tie_rate_pct=_pct(tally.ties, tally.turns),
|
||||
avg_judge_confidence=tally.avg_confidence,
|
||||
)
|
||||
for row in sorted(rows, key=lambda r: r.turn_count, reverse=True)
|
||||
for tier, tally in sorted(by_tier.items(), key=lambda kv: kv[1].turns, reverse=True)
|
||||
)
|
||||
by_model: Final = _tally_by(rows, lambda r: r.real_model)
|
||||
model_groups: Final = tuple(
|
||||
ShadowEvalModelResult(
|
||||
current_model=model,
|
||||
turn_count=tally.turns,
|
||||
real_win_rate_pct=_pct(tally.real_wins, tally.turns),
|
||||
shadow_win_rate_pct=_pct(tally.shadow_wins, tally.turns),
|
||||
tie_rate_pct=_pct(tally.ties, tally.turns),
|
||||
avg_judge_confidence=tally.avg_confidence,
|
||||
)
|
||||
for model, tally in sorted(by_model.items(), key=lambda kv: kv[1].turns, reverse=True)
|
||||
)
|
||||
return ShadowEvalResult(
|
||||
groups=groups,
|
||||
by_current_model=model_groups,
|
||||
overall_shadow_win_rate_pct=_pct(total_shadow_wins, total_turns),
|
||||
overall_tie_rate_pct=_pct(total_ties, total_turns),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -214,10 +214,26 @@ class ShadowEvalTierResult(BaseModel):
|
|||
avg_judge_confidence: float
|
||||
|
||||
|
||||
class ShadowEvalModelResult(BaseModel):
|
||||
"""Judge outcomes against one of the models the shadowed key currently uses.
|
||||
|
||||
A key's real traffic can be a mix of models; per-tier win rates blend those
|
||||
incumbents together, so this slice answers which of them the router actually beat.
|
||||
"""
|
||||
|
||||
current_model: str
|
||||
turn_count: int
|
||||
real_win_rate_pct: float
|
||||
shadow_win_rate_pct: float
|
||||
tie_rate_pct: float
|
||||
avg_judge_confidence: float
|
||||
|
||||
|
||||
class ShadowEvalResult(BaseModel):
|
||||
"""Stratified results of a completed (or in-progress) shadow-eval job."""
|
||||
|
||||
groups: tuple[ShadowEvalTierResult, ...]
|
||||
by_current_model: tuple[ShadowEvalModelResult, ...] = ()
|
||||
overall_shadow_win_rate_pct: float
|
||||
overall_tie_rate_pct: float
|
||||
|
||||
|
|
|
|||
|
|
@ -674,6 +674,81 @@ class TestShadowEvalJobsAreTimeBound:
|
|||
assert response.ends_at == "2026-08-08T00:00:00+00:00"
|
||||
|
||||
|
||||
class TestShadowEvalResultsStratifyByTierAndByCurrentModel:
|
||||
"""A key's real traffic can mix models. Per-tier rates blend the incumbents, so
|
||||
the results also slice by real_model — which of today's models did the router beat."""
|
||||
|
||||
@staticmethod
|
||||
def _row(tier: str | None, real_model: str, real_wins: int, shadow_wins: int, ties: int, conf: float):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import _VerdictAggRow
|
||||
|
||||
return _VerdictAggRow(
|
||||
tier_classification=tier,
|
||||
real_model=real_model,
|
||||
turn_count=real_wins + shadow_wins + ties,
|
||||
real_wins=real_wins,
|
||||
shadow_wins=shadow_wins,
|
||||
ties=ties,
|
||||
avg_confidence=conf,
|
||||
)
|
||||
|
||||
def test_same_rollup_produces_both_stratifications(self):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
|
||||
|
||||
rows = [
|
||||
self._row("SIMPLE", "gpt-4o", 1, 8, 1, 0.9),
|
||||
self._row("SIMPLE", "my-finetune", 6, 2, 2, 0.7),
|
||||
self._row("REASONING", "gpt-4o", 3, 5, 2, 0.8),
|
||||
]
|
||||
result = _shadow_eval_results(rows)
|
||||
|
||||
assert result is not None
|
||||
simple = next(g for g in result.groups if g.tier == "SIMPLE")
|
||||
assert simple.turn_count == 20
|
||||
assert simple.shadow_win_rate_pct == 50.0
|
||||
|
||||
gpt4o = next(m for m in result.by_current_model if m.current_model == "gpt-4o")
|
||||
finetune = next(m for m in result.by_current_model if m.current_model == "my-finetune")
|
||||
assert gpt4o.turn_count == 20
|
||||
assert gpt4o.shadow_win_rate_pct == 65.0
|
||||
assert finetune.turn_count == 10
|
||||
assert finetune.shadow_win_rate_pct == 20.0
|
||||
|
||||
def test_tier_confidence_is_turn_weighted_not_a_plain_average(self):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
|
||||
|
||||
rows = [
|
||||
self._row("SIMPLE", "gpt-4o", 0, 9, 0, 1.0),
|
||||
self._row("SIMPLE", "my-finetune", 1, 0, 0, 0.0),
|
||||
]
|
||||
result = _shadow_eval_results(rows)
|
||||
|
||||
assert result is not None
|
||||
assert result.groups[0].avg_judge_confidence == 0.9
|
||||
|
||||
def test_single_incumbent_still_reports_one_model_slice(self):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
|
||||
|
||||
result = _shadow_eval_results([self._row("SIMPLE", "gpt-4o", 2, 6, 2, 0.8)])
|
||||
|
||||
assert result is not None
|
||||
assert len(result.by_current_model) == 1
|
||||
assert result.by_current_model[0].current_model == "gpt-4o"
|
||||
|
||||
def test_overall_rates_are_unchanged_by_the_extra_grouping_column(self):
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import _shadow_eval_results
|
||||
|
||||
rows = [
|
||||
self._row("SIMPLE", "gpt-4o", 1, 8, 1, 0.9),
|
||||
self._row("SIMPLE", "my-finetune", 6, 2, 2, 0.7),
|
||||
]
|
||||
result = _shadow_eval_results(rows)
|
||||
|
||||
assert result is not None
|
||||
assert result.overall_shadow_win_rate_pct == 50.0
|
||||
assert result.overall_tie_rate_pct == 15.0
|
||||
|
||||
|
||||
class TestShadowEvalJobLifecycleEndpoints:
|
||||
"""List shows every job newest-first, get returns one job with its results, and
|
||||
stop flips only active jobs while keeping the verdicts already collected."""
|
||||
|
|
|
|||
|
|
@ -82,6 +82,24 @@ const job = (overrides: Partial<ShadowEvalJob> = {}): ShadowEvalJob => ({
|
|||
avg_judge_confidence: 0.74,
|
||||
},
|
||||
],
|
||||
by_current_model: [
|
||||
{
|
||||
current_model: "gpt-4o",
|
||||
turn_count: 30,
|
||||
real_win_rate_pct: 30.0,
|
||||
shadow_win_rate_pct: 45.0,
|
||||
tie_rate_pct: 25.0,
|
||||
avg_judge_confidence: 0.8,
|
||||
},
|
||||
{
|
||||
current_model: "my-finetune",
|
||||
turn_count: 12,
|
||||
real_win_rate_pct: 25.0,
|
||||
shadow_win_rate_pct: 58.3,
|
||||
tie_rate_pct: 16.7,
|
||||
avg_judge_confidence: 0.75,
|
||||
},
|
||||
],
|
||||
overall_shadow_win_rate_pct: 48.0,
|
||||
overall_tie_rate_pct: 22.0,
|
||||
},
|
||||
|
|
@ -132,15 +150,31 @@ describe("ShadowEvalSection", () => {
|
|||
expect(screen.getByText("Start shadow eval")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders per-tier win rates for an active job", () => {
|
||||
it("leads with the models the router was compared against", () => {
|
||||
const j = job();
|
||||
mockHooks({ jobs: [j], detail: j });
|
||||
render(<ShadowEvalSection accessToken="token" />);
|
||||
expect(screen.getByText("Compared against")).toBeInTheDocument();
|
||||
expect(screen.getByText("gpt-4o")).toBeInTheDocument();
|
||||
expect(screen.getByText("my-finetune")).toBeInTheDocument();
|
||||
expect(screen.getByText("58.3%")).toBeInTheDocument();
|
||||
// Overall matched-or-beat = shadow wins + ties = 70%
|
||||
expect(screen.getByText("70.0%")).toBeInTheDocument();
|
||||
// Tier rows are behind the difficulty toggle, not in the primary view.
|
||||
expect(screen.queryByText("SIMPLE")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("reveals per-tier win rates behind the prompt-difficulty toggle", async () => {
|
||||
const user = userEvent.setup();
|
||||
const j = job();
|
||||
mockHooks({ jobs: [j], detail: j });
|
||||
render(<ShadowEvalSection accessToken="token" />);
|
||||
|
||||
await user.click(screen.getByText(/By prompt difficulty/));
|
||||
|
||||
expect(screen.getByText("SIMPLE")).toBeInTheDocument();
|
||||
expect(screen.getByText("REASONING")).toBeInTheDocument();
|
||||
expect(screen.getByText("55.0%")).toBeInTheDocument();
|
||||
// Overall matched-or-beat = shadow wins + ties = 70%
|
||||
expect(screen.getByText("70.0%")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("states the metric's denominator next to the headline number", () => {
|
||||
|
|
@ -172,12 +206,23 @@ describe("ShadowEvalSection", () => {
|
|||
expect(screen.getByText(/LLM judge compares the two answers blind/)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("flags low-sample tiers", () => {
|
||||
it("flags low-sample rows", () => {
|
||||
const j = job();
|
||||
mockHooks({ jobs: [j], detail: j });
|
||||
render(<ShadowEvalSection accessToken="token" />);
|
||||
// REASONING tier has 12 turns < 30
|
||||
expect(screen.getByText("(low sample)")).toBeInTheDocument();
|
||||
// my-finetune has 12 turns < 30
|
||||
expect(screen.getAllByText("(low sample)")).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("falls back to the tier table when results predate the by-model slice", () => {
|
||||
const j = job();
|
||||
const results = j.results!;
|
||||
const legacy = job({ results: { ...results, by_current_model: [] } });
|
||||
mockHooks({ jobs: [legacy], detail: legacy });
|
||||
render(<ShadowEvalSection accessToken="token" />);
|
||||
expect(screen.getByText("SIMPLE")).toBeInTheDocument();
|
||||
expect(screen.queryByText("Compared against")).not.toBeInTheDocument();
|
||||
expect(screen.queryByText(/By prompt difficulty/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows a stop button for running jobs but not completed ones", () => {
|
||||
|
|
|
|||
|
|
@ -25,6 +25,7 @@ import {
|
|||
useStartShadowEval,
|
||||
useStopShadowEval,
|
||||
type ShadowEvalJob,
|
||||
type ShadowEvalModelResult,
|
||||
type ShadowEvalTierResult,
|
||||
} from "./useShadowEval";
|
||||
|
||||
|
|
@ -116,7 +117,10 @@ const TierResultsTable: React.FC<{ groups: readonly ShadowEvalTierResult[] }> =
|
|||
<Table>
|
||||
<TableHeader>
|
||||
<TableRow>
|
||||
<TableHead>Router tier</TableHead>
|
||||
<HeadTooltip
|
||||
label="Prompt difficulty"
|
||||
tooltip="The router sorts every prompt into a difficulty tier and picks a model sized to it. Rows where the router holds up on hard tiers matter most."
|
||||
/>
|
||||
<TableHead className="text-right">Judged turns</TableHead>
|
||||
<TableHead className="text-right">Router wins</TableHead>
|
||||
<TableHead className="text-right">Current model wins</TableHead>
|
||||
|
|
@ -152,6 +156,69 @@ const TierResultsTable: React.FC<{ groups: readonly ShadowEvalTierResult[] }> =
|
|||
</Table>
|
||||
);
|
||||
|
||||
const ModelResultsTable: React.FC<{ models: readonly ShadowEvalModelResult[] }> = ({ models }) => (
|
||||
<Table>
|
||||
<TableHeader>
|
||||
<TableRow>
|
||||
<HeadTooltip
|
||||
label="Compared against"
|
||||
tooltip="The model that actually served each sampled request. Your traffic can mix models; each row is the router's head-to-head record against one of them."
|
||||
/>
|
||||
<TableHead className="text-right">Judged turns</TableHead>
|
||||
<TableHead className="text-right">Router wins</TableHead>
|
||||
<TableHead className="text-right">Current model wins</TableHead>
|
||||
<HeadTooltip
|
||||
className="text-right"
|
||||
label="Ties"
|
||||
tooltip="The judge saw the same quality from both. Ties count in the router's favor — same answer quality, usually at lower cost."
|
||||
/>
|
||||
<HeadTooltip
|
||||
className="text-right"
|
||||
label="Judge confidence"
|
||||
tooltip="The judge's self-reported certainty in its verdicts (0 to 1), averaged over these turns."
|
||||
/>
|
||||
</TableRow>
|
||||
</TableHeader>
|
||||
<TableBody>
|
||||
{models.map((m) => (
|
||||
<TableRow key={m.current_model}>
|
||||
<TableCell className="font-mono text-xs text-foreground">
|
||||
{m.current_model}
|
||||
{m.turn_count < MIN_TURNS_FOR_CONFIDENCE ? <LowSampleFlag /> : null}
|
||||
</TableCell>
|
||||
<TableCell className="text-right tabular-nums">{m.turn_count.toLocaleString()}</TableCell>
|
||||
<TableCell className="text-right font-medium tabular-nums text-foreground">
|
||||
{pct(m.shadow_win_rate_pct)}
|
||||
</TableCell>
|
||||
<TableCell className="text-right tabular-nums">{pct(m.real_win_rate_pct)}</TableCell>
|
||||
<TableCell className="text-right tabular-nums">{pct(m.tie_rate_pct)}</TableCell>
|
||||
<TableCell className="text-right tabular-nums">{m.avg_judge_confidence.toFixed(2)}</TableCell>
|
||||
</TableRow>
|
||||
))}
|
||||
</TableBody>
|
||||
</Table>
|
||||
);
|
||||
|
||||
const TierBreakdown: React.FC<{ groups: readonly ShadowEvalTierResult[] }> = ({ groups }) => {
|
||||
const [open, setOpen] = useState(false);
|
||||
return (
|
||||
<div className="border-t">
|
||||
<button
|
||||
type="button"
|
||||
aria-expanded={open}
|
||||
onClick={() => setOpen((prev) => !prev)}
|
||||
className="flex w-full items-center justify-between gap-3 px-6 py-3 text-left hover:bg-muted/50"
|
||||
>
|
||||
<span className="text-xs text-muted-foreground">
|
||||
By prompt difficulty — how the router scored across the difficulty tiers it sorted prompts into
|
||||
</span>
|
||||
<span className="text-xs text-muted-foreground">{open ? "Hide" : "Show"}</span>
|
||||
</button>
|
||||
{open ? <TierResultsTable groups={groups} /> : null}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
const MetricLabel: React.FC<{ label: string; tooltip: string }> = ({ label, tooltip }) => (
|
||||
<TooltipProvider delay={200}>
|
||||
<Tooltip>
|
||||
|
|
@ -242,7 +309,12 @@ const JobResults: React.FC<{
|
|||
<p className="text-xs text-muted-foreground">of {job.completed_count.toLocaleString()} judged responses</p>
|
||||
</div>
|
||||
<VerdictBar results={results} />
|
||||
<TierResultsTable groups={results.groups} />
|
||||
{(results.by_current_model?.length ?? 0) > 0 ? (
|
||||
<ModelResultsTable models={results.by_current_model ?? []} />
|
||||
) : (
|
||||
<TierResultsTable groups={results.groups} />
|
||||
)}
|
||||
{(results.by_current_model?.length ?? 0) > 0 ? <TierBreakdown groups={results.groups} /> : null}
|
||||
</>
|
||||
) : (
|
||||
<p className="px-6 py-8 text-center text-sm text-muted-foreground">
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ import type { components } from "@/lib/http/schema";
|
|||
|
||||
export type ShadowEvalJob = components["schemas"]["GetShadowEvalJobResponse"];
|
||||
export type ShadowEvalTierResult = components["schemas"]["ShadowEvalTierResult"];
|
||||
export type ShadowEvalModelResult = components["schemas"]["ShadowEvalModelResult"];
|
||||
export type StartShadowEvalRequest = components["schemas"]["StartShadowEvalRequest"];
|
||||
|
||||
const JOBS_PATH = "/auto_router/shadow_eval" as const;
|
||||
|
|
|
|||
26
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
26
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -32579,11 +32579,37 @@ export interface components {
|
|||
/** Timeout */
|
||||
timeout?: number | null;
|
||||
};
|
||||
/**
|
||||
* ShadowEvalModelResult
|
||||
* @description Judge outcomes against one of the models the shadowed key currently uses.
|
||||
*
|
||||
* A key's real traffic can be a mix of models; per-tier win rates blend those
|
||||
* incumbents together, so this slice answers which of them the router actually beat.
|
||||
*/
|
||||
ShadowEvalModelResult: {
|
||||
/** Avg Judge Confidence */
|
||||
avg_judge_confidence: number;
|
||||
/** Current Model */
|
||||
current_model: string;
|
||||
/** Real Win Rate Pct */
|
||||
real_win_rate_pct: number;
|
||||
/** Shadow Win Rate Pct */
|
||||
shadow_win_rate_pct: number;
|
||||
/** Tie Rate Pct */
|
||||
tie_rate_pct: number;
|
||||
/** Turn Count */
|
||||
turn_count: number;
|
||||
};
|
||||
/**
|
||||
* ShadowEvalResult
|
||||
* @description Stratified results of a completed (or in-progress) shadow-eval job.
|
||||
*/
|
||||
ShadowEvalResult: {
|
||||
/**
|
||||
* By Current Model
|
||||
* @default []
|
||||
*/
|
||||
by_current_model: components["schemas"]["ShadowEvalModelResult"][];
|
||||
/** Groups */
|
||||
groups: components["schemas"]["ShadowEvalTierResult"][];
|
||||
/** Overall Shadow Win Rate Pct */
|
||||
|
|
|
|||
File diff suppressed because one or more lines are too long
Loading…
Add table
Reference in a new issue