mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-05 02:41:56 +00:00
fix: widen judge output budget and surface prior shadow eval jobs
Three fixes from the live end-to-end run: - The judge ran with max_tokens=200, which truncated roughly 12% of verdicts mid-JSON so they were lost to failed_count. Raise it to a named JUDGE_MAX_OUTPUT_TOKENS=500 and price the upfront cost estimate off the same constant, so the estimate can't silently drift from what the judge is actually allowed to emit. - The UI only ever rendered the newest job, so starting a new eval hid the results of a populated older one. Prior jobs are now listed in a collapsible 'Previous evaluations' card, each expandable to its own per-tier results. - Move the shadow eval section above the benchmarks body: pre-adoption keys have no router sessions, so it was buried under an empty state. It stays outside BenchmarksBody so it survives that early return. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
ba4b52162b
commit
0b94603ca8
7 changed files with 208 additions and 13 deletions
|
|
@ -57,6 +57,12 @@ _MAX_CONCURRENT_SHADOW_TASKS: Final = 16
|
|||
# Truncation bound for text handed to the judge, to keep judge calls affordable.
|
||||
_MAX_JUDGE_CHARS: Final = 16_000
|
||||
|
||||
# Output budget for a judge call. The judge answers with a small JSON object
|
||||
# (preference + confidence + a one-sentence reasoning), which fits well inside
|
||||
# this; a tighter budget truncates the JSON mid-object and the verdict is lost
|
||||
# to failed_count, while a larger one buys nothing but cost exposure.
|
||||
JUDGE_MAX_OUTPUT_TOKENS: Final = 500
|
||||
|
||||
_SEEN_FLUSH_INTERVAL_SECONDS: Final = 10.0
|
||||
|
||||
_EMPTY_METADATA: Final[Mapping[str, object]] = MappingProxyType({})
|
||||
|
|
@ -436,7 +442,7 @@ class ShadowEvalLogger(CustomLogger):
|
|||
{"role": "user", "content": user_prompt}, # mutable-ok: SDK message
|
||||
],
|
||||
temperature=0,
|
||||
max_tokens=200,
|
||||
max_tokens=JUDGE_MAX_OUTPUT_TOKENS,
|
||||
metadata={SHADOW_EVAL_INTERNAL_MARKER: True}, # mutable-ok: SDK metadata
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 # judge outages are a counted failure, not a crash
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ from pydantic import BaseModel, ConfigDict, TypeAdapter
|
|||
|
||||
from litellm._logging import verbose_proxy_logger
|
||||
from litellm.exceptions import BudgetExceededError
|
||||
from litellm.integrations.shadow_eval_logger import JUDGE_MAX_OUTPUT_TOKENS
|
||||
from litellm.proxy._types import (
|
||||
CommonProxyErrors,
|
||||
LiteLLM_TeamTable,
|
||||
|
|
@ -492,6 +493,9 @@ async def get_auto_router_benchmarks(
|
|||
# entry in the price map: roughly one Sonnet-class call on a mid-sized prompt.
|
||||
_FALLBACK_JUDGE_COST_PER_CALL: Final = 0.01
|
||||
|
||||
# Prompt side of the upfront estimate: the conversation plus both responses.
|
||||
_JUDGE_PROMPT_TOKENS_ESTIMATE: Final = 4000
|
||||
|
||||
# The estimate projects from the key's request volume over this many trailing days.
|
||||
_ESTIMATE_LOOKBACK_DAYS: Final = 7
|
||||
|
||||
|
|
@ -522,12 +526,12 @@ def _is_configured_pre_routing_strategy(llm_router: "Router", router_name: str)
|
|||
|
||||
|
||||
def _estimate_judge_cost_per_call(judge_model: str) -> float:
|
||||
"""Price one judge call: ~4k prompt tokens (two responses + conversation) + 200 output."""
|
||||
"""Price one judge call: ~4k prompt tokens (two responses + conversation) + the judge's output budget."""
|
||||
try:
|
||||
import litellm as _litellm
|
||||
|
||||
prompt_cost, completion_cost = _litellm.cost_per_token(
|
||||
model=judge_model, prompt_tokens=4000, completion_tokens=200
|
||||
model=judge_model, prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, completion_tokens=JUDGE_MAX_OUTPUT_TOKENS
|
||||
)
|
||||
estimated: Final = prompt_cost + completion_cost
|
||||
if estimated > 0:
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ from unittest.mock import AsyncMock, MagicMock
|
|||
import pytest
|
||||
|
||||
from litellm.integrations.shadow_eval_logger import (
|
||||
JUDGE_MAX_OUTPUT_TOKENS,
|
||||
SHADOW_EVAL_INTERNAL_MARKER,
|
||||
ActiveShadowEvalJob,
|
||||
ShadowEvalLogger,
|
||||
|
|
@ -81,6 +82,35 @@ class TestParsePairwiseVerdict:
|
|||
_parse_pairwise_verdict("no json here at all")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
class TestJudgeOutputBudget:
|
||||
async def test_judge_call_uses_the_named_output_budget(self, monkeypatch: pytest.MonkeyPatch):
|
||||
"""Regression: a 200-token budget truncated ~12% of verdicts mid-JSON, losing them to failed_count."""
|
||||
import litellm as litellm_module
|
||||
|
||||
acompletion = AsyncMock(
|
||||
return_value={
|
||||
"choices": [{"message": {"content": '{"preference": "A", "confidence": 0.8, "reasoning": "clearer"}'}}]
|
||||
}
|
||||
)
|
||||
monkeypatch.setattr(litellm_module, "acompletion", acompletion)
|
||||
|
||||
logger = ShadowEvalLogger(router_provider=lambda: MagicMock(), prisma_provider=lambda: MagicMock())
|
||||
verdict = await logger._call_judge(
|
||||
"gpt-4o-mini", [{"role": "user", "content": "hi"}], "real text", "shadow text"
|
||||
)
|
||||
|
||||
assert verdict is not None
|
||||
_, kwargs = acompletion.call_args
|
||||
assert kwargs["max_tokens"] == JUDGE_MAX_OUTPUT_TOKENS
|
||||
|
||||
|
||||
class TestJudgeOutputBudgetStaticChecks:
|
||||
def test_budget_leaves_room_for_the_reasoning_field(self):
|
||||
"""A tight budget truncates the JSON object the judge is asked to emit."""
|
||||
assert JUDGE_MAX_OUTPUT_TOKENS >= 500
|
||||
|
||||
|
||||
def _logger_with_mocks(job=None):
|
||||
prisma = MagicMock()
|
||||
router = MagicMock()
|
||||
|
|
|
|||
|
|
@ -318,9 +318,7 @@ class TestAutoRouterBenchmarks:
|
|||
from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals
|
||||
|
||||
totals = _benchmark_totals(self.ROW)
|
||||
bucket_hits = (
|
||||
totals.cache.same_model.hits + totals.cache.first_visit.hits + totals.cache.return_to_tier.hits
|
||||
)
|
||||
bucket_hits = totals.cache.same_model.hits + totals.cache.first_visit.hits + totals.cache.return_to_tier.hits
|
||||
assert bucket_hits == 27
|
||||
assert totals.cache.hit_rate_pct == pytest.approx(100.0 * 28 / 38, abs=0.1)
|
||||
|
||||
|
|
@ -466,3 +464,39 @@ class TestAutoRouterBenchmarks:
|
|||
end_date="2026-08-01",
|
||||
)
|
||||
assert response.groups[0].tier_turns == expected
|
||||
|
||||
|
||||
class TestJudgeCostEstimate:
|
||||
"""The upfront estimate must price the judge's real output budget, not a stale hardcoded one."""
|
||||
|
||||
JUDGE_MODEL = "gpt-4o-mini"
|
||||
|
||||
def test_estimate_prices_the_configured_judge_output_budget(self):
|
||||
import litellm as litellm_module
|
||||
from litellm.integrations.shadow_eval_logger import JUDGE_MAX_OUTPUT_TOKENS
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import (
|
||||
_JUDGE_PROMPT_TOKENS_ESTIMATE,
|
||||
_estimate_judge_cost_per_call,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = litellm_module.cost_per_token(
|
||||
model=self.JUDGE_MODEL,
|
||||
prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE,
|
||||
completion_tokens=JUDGE_MAX_OUTPUT_TOKENS,
|
||||
)
|
||||
assert completion_cost > 0
|
||||
assert _estimate_judge_cost_per_call(self.JUDGE_MODEL) == prompt_cost + completion_cost
|
||||
|
||||
def test_estimate_is_not_still_pinned_to_the_old_200_token_budget(self):
|
||||
import litellm as litellm_module
|
||||
from litellm.proxy.management_endpoints.auto_router_endpoints import (
|
||||
_JUDGE_PROMPT_TOKENS_ESTIMATE,
|
||||
_estimate_judge_cost_per_call,
|
||||
)
|
||||
|
||||
stale = sum(
|
||||
litellm_module.cost_per_token(
|
||||
model=self.JUDGE_MODEL, prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, completion_tokens=200
|
||||
)
|
||||
)
|
||||
assert _estimate_judge_cost_per_call(self.JUDGE_MODEL) > stale
|
||||
|
|
|
|||
|
|
@ -320,12 +320,12 @@ const AutoRouterBenchmarksTab: React.FC<AutoRouterBenchmarksTabProps> = ({ acces
|
|||
</div>
|
||||
</div>
|
||||
|
||||
<BenchmarksBody isPending={isPending} error={error} data={data} selectedKey={selectedKey} />
|
||||
|
||||
{/* Outside BenchmarksBody on purpose: pre-adoption keys have no router
|
||||
sessions yet, and the shadow-eval section must render even when the
|
||||
benchmarks body early-returns its empty state. */}
|
||||
{/* Sits above BenchmarksBody, and outside it on purpose: pre-adoption keys
|
||||
have no router sessions yet, so the shadow-eval section must render even
|
||||
when the benchmarks body early-returns its loading, error, or empty state. */}
|
||||
<ShadowEvalSection accessToken={accessToken} />
|
||||
|
||||
<BenchmarksBody isPending={isPending} error={error} data={data} selectedKey={selectedKey} />
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import { render, screen } from "@testing-library/react";
|
||||
import userEvent from "@testing-library/user-event";
|
||||
import React from "react";
|
||||
import { describe, expect, it, vi } from "vitest";
|
||||
|
||||
|
|
@ -55,11 +56,24 @@ const job = (overrides: Partial<ShadowEvalJob> = {}): ShadowEvalJob => ({
|
|||
...overrides,
|
||||
});
|
||||
|
||||
const mockHooks = ({ jobs = [], detail = undefined }: { jobs?: ShadowEvalJob[]; detail?: ShadowEvalJob }) => {
|
||||
const mockHooks = ({
|
||||
jobs = [],
|
||||
detail = undefined,
|
||||
detailsById = undefined,
|
||||
}: {
|
||||
jobs?: ShadowEvalJob[];
|
||||
detail?: ShadowEvalJob;
|
||||
detailsById?: Record<string, ShadowEvalJob>;
|
||||
}) => {
|
||||
vi.mocked(useShadowEvalJobs).mockReturnValue({ data: jobs, error: null } as unknown as ReturnType<
|
||||
typeof useShadowEvalJobs
|
||||
>);
|
||||
vi.mocked(useShadowEvalJob).mockReturnValue({ data: detail } as unknown as ReturnType<typeof useShadowEvalJob>);
|
||||
vi.mocked(useShadowEvalJob).mockImplementation(
|
||||
(_token, jobId) =>
|
||||
({
|
||||
data: detailsById ? (jobId ? detailsById[jobId] : undefined) : detail,
|
||||
}) as unknown as ReturnType<typeof useShadowEvalJob>,
|
||||
);
|
||||
vi.mocked(useStartShadowEval).mockReturnValue({
|
||||
mutate: vi.fn(),
|
||||
isPending: false,
|
||||
|
|
@ -116,4 +130,32 @@ describe("ShadowEvalSection", () => {
|
|||
render(<ShadowEvalSection accessToken="token" />);
|
||||
expect(screen.getByText("Start a shadow eval")).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("keeps an older populated job reachable when the newest job has no verdicts", async () => {
|
||||
const user = userEvent.setup();
|
||||
const empty = job({
|
||||
job_id: "job-new",
|
||||
status: "pending",
|
||||
completed_count: 0,
|
||||
failed_count: 0,
|
||||
results: null,
|
||||
cost_actual: 0,
|
||||
});
|
||||
const older = job({ job_id: "job-old", status: "completed" });
|
||||
mockHooks({
|
||||
jobs: [empty, older],
|
||||
detailsById: { "job-new": empty, "job-old": older },
|
||||
});
|
||||
render(<ShadowEvalSection accessToken="token" />);
|
||||
|
||||
// The newest job is empty, so its verdicts are absent from the primary card...
|
||||
expect(screen.queryByText("SIMPLE")).not.toBeInTheDocument();
|
||||
|
||||
// ...but the older job with verdicts must still be reachable.
|
||||
await user.click(screen.getByRole("button", { name: /Previous evaluations \(1\)/ }));
|
||||
await user.click(screen.getByRole("button", { name: /job-old-row|10% via claude-auto/ }));
|
||||
|
||||
expect(await screen.findByText("SIMPLE")).toBeInTheDocument();
|
||||
expect(screen.getByText("REASONING")).toBeInTheDocument();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -210,6 +210,82 @@ const StartForm: React.FC<{ accessToken: string | null }> = ({ accessToken }) =>
|
|||
);
|
||||
};
|
||||
|
||||
const PreviousJob: React.FC<{
|
||||
accessToken: string | null;
|
||||
job: ShadowEvalJob;
|
||||
}> = ({ accessToken, job }) => {
|
||||
const [expanded, setExpanded] = useState(false);
|
||||
const { data: detail } = useShadowEvalJob(accessToken, expanded ? job.job_id : null);
|
||||
const shown = detail ?? job;
|
||||
const results = shown.results;
|
||||
const okOrBetter = results ? results.overall_shadow_win_rate_pct + results.overall_tie_rate_pct : null;
|
||||
|
||||
return (
|
||||
<div className="border-b last:border-b-0">
|
||||
<button
|
||||
type="button"
|
||||
aria-expanded={expanded}
|
||||
onClick={() => setExpanded((open) => !open)}
|
||||
className="flex w-full flex-wrap items-center justify-between gap-3 px-6 py-3 text-left hover:bg-muted/50"
|
||||
>
|
||||
<div className="flex items-center gap-3">
|
||||
<StatusBadge status={shown.status} />
|
||||
<div>
|
||||
<p className="text-sm font-medium text-foreground">
|
||||
{shown.shadow_percentage}% via <span className="font-mono text-xs">{shown.router_name}</span>
|
||||
</p>
|
||||
<p className="text-xs text-muted-foreground">
|
||||
{shown.completed_count.toLocaleString()} judged · {shown.failed_count.toLocaleString()} failed ·{" "}
|
||||
{shown.cost_actual != null ? `${usd(shown.cost_actual)} judge spend` : "no judge spend yet"}
|
||||
{shown.created_at ? ` · ${new Date(shown.created_at).toLocaleDateString()}` : ""}
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
<span className="text-sm font-medium text-foreground">
|
||||
{okOrBetter != null ? pct(okOrBetter) : "no verdicts"}
|
||||
</span>
|
||||
</button>
|
||||
{expanded ? (
|
||||
<div className="px-6 pb-4">
|
||||
{results && results.groups.length > 0 ? (
|
||||
<TierResultsTable groups={results.groups} />
|
||||
) : (
|
||||
<p className="text-xs text-muted-foreground">No verdicts were recorded for this evaluation.</p>
|
||||
)}
|
||||
</div>
|
||||
) : null}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
const PreviousJobs: React.FC<{
|
||||
accessToken: string | null;
|
||||
jobs: readonly ShadowEvalJob[];
|
||||
}> = ({ accessToken, jobs }) => {
|
||||
const [open, setOpen] = useState(false);
|
||||
if (jobs.length === 0) return null;
|
||||
return (
|
||||
<Card className="overflow-hidden py-0">
|
||||
<button
|
||||
type="button"
|
||||
aria-expanded={open}
|
||||
onClick={() => setOpen((prev) => !prev)}
|
||||
className="flex w-full items-center justify-between gap-3 px-6 py-3 text-left hover:bg-muted/50"
|
||||
>
|
||||
<span className="text-sm font-medium text-foreground">Previous evaluations ({jobs.length})</span>
|
||||
<span className="text-xs text-muted-foreground">{open ? "Hide" : "Show"}</span>
|
||||
</button>
|
||||
{open ? (
|
||||
<div className="border-t">
|
||||
{jobs.map((job) => (
|
||||
<PreviousJob key={job.job_id} accessToken={accessToken} job={job} />
|
||||
))}
|
||||
</div>
|
||||
) : null}
|
||||
</Card>
|
||||
);
|
||||
};
|
||||
|
||||
interface ShadowEvalSectionProps {
|
||||
accessToken: string | null;
|
||||
}
|
||||
|
|
@ -220,6 +296,7 @@ const ShadowEvalSection: React.FC<ShadowEvalSectionProps> = ({ accessToken }) =>
|
|||
|
||||
// Most recent job carries the section; older jobs list below it.
|
||||
const latest = useMemo(() => jobs?.[0] ?? null, [jobs]);
|
||||
const previous = useMemo(() => jobs?.slice(1) ?? [], [jobs]);
|
||||
const { data: latestDetail } = useShadowEvalJob(accessToken, latest?.job_id ?? null);
|
||||
|
||||
if (error instanceof ApiError && error.status === 403) return null; // admin-only section
|
||||
|
|
@ -244,6 +321,8 @@ const ShadowEvalSection: React.FC<ShadowEvalSectionProps> = ({ accessToken }) =>
|
|||
{!latest || (latestDetail && latestDetail.status !== "pending" && latestDetail.status !== "running") ? (
|
||||
<StartForm accessToken={accessToken} />
|
||||
) : null}
|
||||
|
||||
<PreviousJobs accessToken={accessToken} jobs={previous} />
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue