fix: widen judge output budget and surface prior shadow eval jobs

Three fixes from the live end-to-end run:

- The judge ran with max_tokens=200, which truncated roughly 12% of
  verdicts mid-JSON so they were lost to failed_count. Raise it to a
  named JUDGE_MAX_OUTPUT_TOKENS=500 and price the upfront cost estimate
  off the same constant, so the estimate can't silently drift from what
  the judge is actually allowed to emit.

- The UI only ever rendered the newest job, so starting a new eval hid
  the results of a populated older one. Prior jobs are now listed in a
  collapsible 'Previous evaluations' card, each expandable to its own
  per-tier results.

- Move the shadow eval section above the benchmarks body: pre-adoption
  keys have no router sessions, so it was buried under an empty state.
  It stays outside BenchmarksBody so it survives that early return.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Abhimanyu Kapur 2026-08-08 11:01:58 -07:00
parent ba4b52162b
commit 0b94603ca8
7 changed files with 208 additions and 13 deletions

View file

@ -57,6 +57,12 @@ _MAX_CONCURRENT_SHADOW_TASKS: Final = 16
# Truncation bound for text handed to the judge, to keep judge calls affordable.
_MAX_JUDGE_CHARS: Final = 16_000
# Output budget for a judge call. The judge answers with a small JSON object
# (preference + confidence + a one-sentence reasoning), which fits well inside
# this; a tighter budget truncates the JSON mid-object and the verdict is lost
# to failed_count, while a larger one buys nothing but cost exposure.
JUDGE_MAX_OUTPUT_TOKENS: Final = 500
_SEEN_FLUSH_INTERVAL_SECONDS: Final = 10.0
_EMPTY_METADATA: Final[Mapping[str, object]] = MappingProxyType({})
@ -436,7 +442,7 @@ class ShadowEvalLogger(CustomLogger):
{"role": "user", "content": user_prompt}, # mutable-ok: SDK message
],
temperature=0,
max_tokens=200,
max_tokens=JUDGE_MAX_OUTPUT_TOKENS,
metadata={SHADOW_EVAL_INTERNAL_MARKER: True}, # mutable-ok: SDK metadata
)
except Exception as e: # noqa: BLE001 # judge outages are a counted failure, not a crash

View file

@ -13,6 +13,7 @@ from pydantic import BaseModel, ConfigDict, TypeAdapter
from litellm._logging import verbose_proxy_logger
from litellm.exceptions import BudgetExceededError
from litellm.integrations.shadow_eval_logger import JUDGE_MAX_OUTPUT_TOKENS
from litellm.proxy._types import (
CommonProxyErrors,
LiteLLM_TeamTable,
@ -492,6 +493,9 @@ async def get_auto_router_benchmarks(
# entry in the price map: roughly one Sonnet-class call on a mid-sized prompt.
_FALLBACK_JUDGE_COST_PER_CALL: Final = 0.01
# Prompt side of the upfront estimate: the conversation plus both responses.
_JUDGE_PROMPT_TOKENS_ESTIMATE: Final = 4000
# The estimate projects from the key's request volume over this many trailing days.
_ESTIMATE_LOOKBACK_DAYS: Final = 7
@ -522,12 +526,12 @@ def _is_configured_pre_routing_strategy(llm_router: "Router", router_name: str)
def _estimate_judge_cost_per_call(judge_model: str) -> float:
"""Price one judge call: ~4k prompt tokens (two responses + conversation) + 200 output."""
"""Price one judge call: ~4k prompt tokens (two responses + conversation) + the judge's output budget."""
try:
import litellm as _litellm
prompt_cost, completion_cost = _litellm.cost_per_token(
model=judge_model, prompt_tokens=4000, completion_tokens=200
model=judge_model, prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, completion_tokens=JUDGE_MAX_OUTPUT_TOKENS
)
estimated: Final = prompt_cost + completion_cost
if estimated > 0:

View file

@ -7,6 +7,7 @@ from unittest.mock import AsyncMock, MagicMock
import pytest
from litellm.integrations.shadow_eval_logger import (
JUDGE_MAX_OUTPUT_TOKENS,
SHADOW_EVAL_INTERNAL_MARKER,
ActiveShadowEvalJob,
ShadowEvalLogger,
@ -81,6 +82,35 @@ class TestParsePairwiseVerdict:
_parse_pairwise_verdict("no json here at all")
@pytest.mark.asyncio
class TestJudgeOutputBudget:
async def test_judge_call_uses_the_named_output_budget(self, monkeypatch: pytest.MonkeyPatch):
"""Regression: a 200-token budget truncated ~12% of verdicts mid-JSON, losing them to failed_count."""
import litellm as litellm_module
acompletion = AsyncMock(
return_value={
"choices": [{"message": {"content": '{"preference": "A", "confidence": 0.8, "reasoning": "clearer"}'}}]
}
)
monkeypatch.setattr(litellm_module, "acompletion", acompletion)
logger = ShadowEvalLogger(router_provider=lambda: MagicMock(), prisma_provider=lambda: MagicMock())
verdict = await logger._call_judge(
"gpt-4o-mini", [{"role": "user", "content": "hi"}], "real text", "shadow text"
)
assert verdict is not None
_, kwargs = acompletion.call_args
assert kwargs["max_tokens"] == JUDGE_MAX_OUTPUT_TOKENS
class TestJudgeOutputBudgetStaticChecks:
def test_budget_leaves_room_for_the_reasoning_field(self):
"""A tight budget truncates the JSON object the judge is asked to emit."""
assert JUDGE_MAX_OUTPUT_TOKENS >= 500
def _logger_with_mocks(job=None):
prisma = MagicMock()
router = MagicMock()

View file

@ -318,9 +318,7 @@ class TestAutoRouterBenchmarks:
from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals
totals = _benchmark_totals(self.ROW)
bucket_hits = (
totals.cache.same_model.hits + totals.cache.first_visit.hits + totals.cache.return_to_tier.hits
)
bucket_hits = totals.cache.same_model.hits + totals.cache.first_visit.hits + totals.cache.return_to_tier.hits
assert bucket_hits == 27
assert totals.cache.hit_rate_pct == pytest.approx(100.0 * 28 / 38, abs=0.1)
@ -466,3 +464,39 @@ class TestAutoRouterBenchmarks:
end_date="2026-08-01",
)
assert response.groups[0].tier_turns == expected
class TestJudgeCostEstimate:
"""The upfront estimate must price the judge's real output budget, not a stale hardcoded one."""
JUDGE_MODEL = "gpt-4o-mini"
def test_estimate_prices_the_configured_judge_output_budget(self):
import litellm as litellm_module
from litellm.integrations.shadow_eval_logger import JUDGE_MAX_OUTPUT_TOKENS
from litellm.proxy.management_endpoints.auto_router_endpoints import (
_JUDGE_PROMPT_TOKENS_ESTIMATE,
_estimate_judge_cost_per_call,
)
prompt_cost, completion_cost = litellm_module.cost_per_token(
model=self.JUDGE_MODEL,
prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE,
completion_tokens=JUDGE_MAX_OUTPUT_TOKENS,
)
assert completion_cost > 0
assert _estimate_judge_cost_per_call(self.JUDGE_MODEL) == prompt_cost + completion_cost
def test_estimate_is_not_still_pinned_to_the_old_200_token_budget(self):
import litellm as litellm_module
from litellm.proxy.management_endpoints.auto_router_endpoints import (
_JUDGE_PROMPT_TOKENS_ESTIMATE,
_estimate_judge_cost_per_call,
)
stale = sum(
litellm_module.cost_per_token(
model=self.JUDGE_MODEL, prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, completion_tokens=200
)
)
assert _estimate_judge_cost_per_call(self.JUDGE_MODEL) > stale

View file

@ -320,12 +320,12 @@ const AutoRouterBenchmarksTab: React.FC<AutoRouterBenchmarksTabProps> = ({ acces
</div>
</div>
<BenchmarksBody isPending={isPending} error={error} data={data} selectedKey={selectedKey} />
{/* Outside BenchmarksBody on purpose: pre-adoption keys have no router
sessions yet, and the shadow-eval section must render even when the
benchmarks body early-returns its empty state. */}
{/* Sits above BenchmarksBody, and outside it on purpose: pre-adoption keys
have no router sessions yet, so the shadow-eval section must render even
when the benchmarks body early-returns its loading, error, or empty state. */}
<ShadowEvalSection accessToken={accessToken} />
<BenchmarksBody isPending={isPending} error={error} data={data} selectedKey={selectedKey} />
</div>
);
};

View file

@ -1,4 +1,5 @@
import { render, screen } from "@testing-library/react";
import userEvent from "@testing-library/user-event";
import React from "react";
import { describe, expect, it, vi } from "vitest";
@ -55,11 +56,24 @@ const job = (overrides: Partial<ShadowEvalJob> = {}): ShadowEvalJob => ({
...overrides,
});
const mockHooks = ({ jobs = [], detail = undefined }: { jobs?: ShadowEvalJob[]; detail?: ShadowEvalJob }) => {
const mockHooks = ({
jobs = [],
detail = undefined,
detailsById = undefined,
}: {
jobs?: ShadowEvalJob[];
detail?: ShadowEvalJob;
detailsById?: Record<string, ShadowEvalJob>;
}) => {
vi.mocked(useShadowEvalJobs).mockReturnValue({ data: jobs, error: null } as unknown as ReturnType<
typeof useShadowEvalJobs
>);
vi.mocked(useShadowEvalJob).mockReturnValue({ data: detail } as unknown as ReturnType<typeof useShadowEvalJob>);
vi.mocked(useShadowEvalJob).mockImplementation(
(_token, jobId) =>
({
data: detailsById ? (jobId ? detailsById[jobId] : undefined) : detail,
}) as unknown as ReturnType<typeof useShadowEvalJob>,
);
vi.mocked(useStartShadowEval).mockReturnValue({
mutate: vi.fn(),
isPending: false,
@ -116,4 +130,32 @@ describe("ShadowEvalSection", () => {
render(<ShadowEvalSection accessToken="token" />);
expect(screen.getByText("Start a shadow eval")).toBeInTheDocument();
});
it("keeps an older populated job reachable when the newest job has no verdicts", async () => {
const user = userEvent.setup();
const empty = job({
job_id: "job-new",
status: "pending",
completed_count: 0,
failed_count: 0,
results: null,
cost_actual: 0,
});
const older = job({ job_id: "job-old", status: "completed" });
mockHooks({
jobs: [empty, older],
detailsById: { "job-new": empty, "job-old": older },
});
render(<ShadowEvalSection accessToken="token" />);
// The newest job is empty, so its verdicts are absent from the primary card...
expect(screen.queryByText("SIMPLE")).not.toBeInTheDocument();
// ...but the older job with verdicts must still be reachable.
await user.click(screen.getByRole("button", { name: /Previous evaluations \(1\)/ }));
await user.click(screen.getByRole("button", { name: /job-old-row|10% via claude-auto/ }));
expect(await screen.findByText("SIMPLE")).toBeInTheDocument();
expect(screen.getByText("REASONING")).toBeInTheDocument();
});
});

View file

@ -210,6 +210,82 @@ const StartForm: React.FC<{ accessToken: string | null }> = ({ accessToken }) =>
);
};
const PreviousJob: React.FC<{
accessToken: string | null;
job: ShadowEvalJob;
}> = ({ accessToken, job }) => {
const [expanded, setExpanded] = useState(false);
const { data: detail } = useShadowEvalJob(accessToken, expanded ? job.job_id : null);
const shown = detail ?? job;
const results = shown.results;
const okOrBetter = results ? results.overall_shadow_win_rate_pct + results.overall_tie_rate_pct : null;
return (
<div className="border-b last:border-b-0">
<button
type="button"
aria-expanded={expanded}
onClick={() => setExpanded((open) => !open)}
className="flex w-full flex-wrap items-center justify-between gap-3 px-6 py-3 text-left hover:bg-muted/50"
>
<div className="flex items-center gap-3">
<StatusBadge status={shown.status} />
<div>
<p className="text-sm font-medium text-foreground">
{shown.shadow_percentage}% via <span className="font-mono text-xs">{shown.router_name}</span>
</p>
<p className="text-xs text-muted-foreground">
{shown.completed_count.toLocaleString()} judged · {shown.failed_count.toLocaleString()} failed ·{" "}
{shown.cost_actual != null ? `${usd(shown.cost_actual)} judge spend` : "no judge spend yet"}
{shown.created_at ? ` · ${new Date(shown.created_at).toLocaleDateString()}` : ""}
</p>
</div>
</div>
<span className="text-sm font-medium text-foreground">
{okOrBetter != null ? pct(okOrBetter) : "no verdicts"}
</span>
</button>
{expanded ? (
<div className="px-6 pb-4">
{results && results.groups.length > 0 ? (
<TierResultsTable groups={results.groups} />
) : (
<p className="text-xs text-muted-foreground">No verdicts were recorded for this evaluation.</p>
)}
</div>
) : null}
</div>
);
};
const PreviousJobs: React.FC<{
accessToken: string | null;
jobs: readonly ShadowEvalJob[];
}> = ({ accessToken, jobs }) => {
const [open, setOpen] = useState(false);
if (jobs.length === 0) return null;
return (
<Card className="overflow-hidden py-0">
<button
type="button"
aria-expanded={open}
onClick={() => setOpen((prev) => !prev)}
className="flex w-full items-center justify-between gap-3 px-6 py-3 text-left hover:bg-muted/50"
>
<span className="text-sm font-medium text-foreground">Previous evaluations ({jobs.length})</span>
<span className="text-xs text-muted-foreground">{open ? "Hide" : "Show"}</span>
</button>
{open ? (
<div className="border-t">
{jobs.map((job) => (
<PreviousJob key={job.job_id} accessToken={accessToken} job={job} />
))}
</div>
) : null}
</Card>
);
};
interface ShadowEvalSectionProps {
accessToken: string | null;
}
@ -220,6 +296,7 @@ const ShadowEvalSection: React.FC<ShadowEvalSectionProps> = ({ accessToken }) =>
// Most recent job carries the section; older jobs list below it.
const latest = useMemo(() => jobs?.[0] ?? null, [jobs]);
const previous = useMemo(() => jobs?.slice(1) ?? [], [jobs]);
const { data: latestDetail } = useShadowEvalJob(accessToken, latest?.job_id ?? null);
if (error instanceof ApiError && error.status === 403) return null; // admin-only section
@ -244,6 +321,8 @@ const ShadowEvalSection: React.FC<ShadowEvalSectionProps> = ({ accessToken }) =>
{!latest || (latestDetail && latestDetail.status !== "pending" && latestDetail.status !== "running") ? (
<StartForm accessToken={accessToken} />
) : null}
<PreviousJobs accessToken={accessToken} jobs={previous} />
</div>
);
};