diff --git a/litellm/integrations/shadow_eval_logger.py b/litellm/integrations/shadow_eval_logger.py index 2a4a2067882..66295ea2319 100644 --- a/litellm/integrations/shadow_eval_logger.py +++ b/litellm/integrations/shadow_eval_logger.py @@ -57,6 +57,12 @@ _MAX_CONCURRENT_SHADOW_TASKS: Final = 16 # Truncation bound for text handed to the judge, to keep judge calls affordable. _MAX_JUDGE_CHARS: Final = 16_000 +# Output budget for a judge call. The judge answers with a small JSON object +# (preference + confidence + a one-sentence reasoning), which fits well inside +# this; a tighter budget truncates the JSON mid-object and the verdict is lost +# to failed_count, while a larger one buys nothing but cost exposure. +JUDGE_MAX_OUTPUT_TOKENS: Final = 500 + _SEEN_FLUSH_INTERVAL_SECONDS: Final = 10.0 _EMPTY_METADATA: Final[Mapping[str, object]] = MappingProxyType({}) @@ -436,7 +442,7 @@ class ShadowEvalLogger(CustomLogger): {"role": "user", "content": user_prompt}, # mutable-ok: SDK message ], temperature=0, - max_tokens=200, + max_tokens=JUDGE_MAX_OUTPUT_TOKENS, metadata={SHADOW_EVAL_INTERNAL_MARKER: True}, # mutable-ok: SDK metadata ) except Exception as e: # noqa: BLE001 # judge outages are a counted failure, not a crash diff --git a/litellm/proxy/management_endpoints/auto_router_endpoints.py b/litellm/proxy/management_endpoints/auto_router_endpoints.py index 799b752c4a2..f20dc8dc76e 100644 --- a/litellm/proxy/management_endpoints/auto_router_endpoints.py +++ b/litellm/proxy/management_endpoints/auto_router_endpoints.py @@ -13,6 +13,7 @@ from pydantic import BaseModel, ConfigDict, TypeAdapter from litellm._logging import verbose_proxy_logger from litellm.exceptions import BudgetExceededError +from litellm.integrations.shadow_eval_logger import JUDGE_MAX_OUTPUT_TOKENS from litellm.proxy._types import ( CommonProxyErrors, LiteLLM_TeamTable, @@ -492,6 +493,9 @@ async def get_auto_router_benchmarks( # entry in the price map: roughly one Sonnet-class call on a mid-sized prompt. _FALLBACK_JUDGE_COST_PER_CALL: Final = 0.01 +# Prompt side of the upfront estimate: the conversation plus both responses. +_JUDGE_PROMPT_TOKENS_ESTIMATE: Final = 4000 + # The estimate projects from the key's request volume over this many trailing days. _ESTIMATE_LOOKBACK_DAYS: Final = 7 @@ -522,12 +526,12 @@ def _is_configured_pre_routing_strategy(llm_router: "Router", router_name: str) def _estimate_judge_cost_per_call(judge_model: str) -> float: - """Price one judge call: ~4k prompt tokens (two responses + conversation) + 200 output.""" + """Price one judge call: ~4k prompt tokens (two responses + conversation) + the judge's output budget.""" try: import litellm as _litellm prompt_cost, completion_cost = _litellm.cost_per_token( - model=judge_model, prompt_tokens=4000, completion_tokens=200 + model=judge_model, prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, completion_tokens=JUDGE_MAX_OUTPUT_TOKENS ) estimated: Final = prompt_cost + completion_cost if estimated > 0: diff --git a/tests/test_litellm/integrations/test_shadow_eval_logger.py b/tests/test_litellm/integrations/test_shadow_eval_logger.py index dd542d1b528..319834a3c5a 100644 --- a/tests/test_litellm/integrations/test_shadow_eval_logger.py +++ b/tests/test_litellm/integrations/test_shadow_eval_logger.py @@ -7,6 +7,7 @@ from unittest.mock import AsyncMock, MagicMock import pytest from litellm.integrations.shadow_eval_logger import ( + JUDGE_MAX_OUTPUT_TOKENS, SHADOW_EVAL_INTERNAL_MARKER, ActiveShadowEvalJob, ShadowEvalLogger, @@ -81,6 +82,35 @@ class TestParsePairwiseVerdict: _parse_pairwise_verdict("no json here at all") +@pytest.mark.asyncio +class TestJudgeOutputBudget: + async def test_judge_call_uses_the_named_output_budget(self, monkeypatch: pytest.MonkeyPatch): + """Regression: a 200-token budget truncated ~12% of verdicts mid-JSON, losing them to failed_count.""" + import litellm as litellm_module + + acompletion = AsyncMock( + return_value={ + "choices": [{"message": {"content": '{"preference": "A", "confidence": 0.8, "reasoning": "clearer"}'}}] + } + ) + monkeypatch.setattr(litellm_module, "acompletion", acompletion) + + logger = ShadowEvalLogger(router_provider=lambda: MagicMock(), prisma_provider=lambda: MagicMock()) + verdict = await logger._call_judge( + "gpt-4o-mini", [{"role": "user", "content": "hi"}], "real text", "shadow text" + ) + + assert verdict is not None + _, kwargs = acompletion.call_args + assert kwargs["max_tokens"] == JUDGE_MAX_OUTPUT_TOKENS + + +class TestJudgeOutputBudgetStaticChecks: + def test_budget_leaves_room_for_the_reasoning_field(self): + """A tight budget truncates the JSON object the judge is asked to emit.""" + assert JUDGE_MAX_OUTPUT_TOKENS >= 500 + + def _logger_with_mocks(job=None): prisma = MagicMock() router = MagicMock() diff --git a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py index 3a995e27697..357a2354cce 100644 --- a/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py +++ b/tests/test_litellm/proxy/management_endpoints/test_auto_router_endpoints.py @@ -318,9 +318,7 @@ class TestAutoRouterBenchmarks: from litellm.proxy.management_endpoints.auto_router_endpoints import _benchmark_totals totals = _benchmark_totals(self.ROW) - bucket_hits = ( - totals.cache.same_model.hits + totals.cache.first_visit.hits + totals.cache.return_to_tier.hits - ) + bucket_hits = totals.cache.same_model.hits + totals.cache.first_visit.hits + totals.cache.return_to_tier.hits assert bucket_hits == 27 assert totals.cache.hit_rate_pct == pytest.approx(100.0 * 28 / 38, abs=0.1) @@ -466,3 +464,39 @@ class TestAutoRouterBenchmarks: end_date="2026-08-01", ) assert response.groups[0].tier_turns == expected + + +class TestJudgeCostEstimate: + """The upfront estimate must price the judge's real output budget, not a stale hardcoded one.""" + + JUDGE_MODEL = "gpt-4o-mini" + + def test_estimate_prices_the_configured_judge_output_budget(self): + import litellm as litellm_module + from litellm.integrations.shadow_eval_logger import JUDGE_MAX_OUTPUT_TOKENS + from litellm.proxy.management_endpoints.auto_router_endpoints import ( + _JUDGE_PROMPT_TOKENS_ESTIMATE, + _estimate_judge_cost_per_call, + ) + + prompt_cost, completion_cost = litellm_module.cost_per_token( + model=self.JUDGE_MODEL, + prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, + completion_tokens=JUDGE_MAX_OUTPUT_TOKENS, + ) + assert completion_cost > 0 + assert _estimate_judge_cost_per_call(self.JUDGE_MODEL) == prompt_cost + completion_cost + + def test_estimate_is_not_still_pinned_to_the_old_200_token_budget(self): + import litellm as litellm_module + from litellm.proxy.management_endpoints.auto_router_endpoints import ( + _JUDGE_PROMPT_TOKENS_ESTIMATE, + _estimate_judge_cost_per_call, + ) + + stale = sum( + litellm_module.cost_per_token( + model=self.JUDGE_MODEL, prompt_tokens=_JUDGE_PROMPT_TOKENS_ESTIMATE, completion_tokens=200 + ) + ) + assert _estimate_judge_cost_per_call(self.JUDGE_MODEL) > stale diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx index f96d570bb43..96e34033d04 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/AutoRouterBenchmarksTab.tsx @@ -320,12 +320,12 @@ const AutoRouterBenchmarksTab: React.FC = ({ acces - - - {/* Outside BenchmarksBody on purpose: pre-adoption keys have no router - sessions yet, and the shadow-eval section must render even when the - benchmarks body early-returns its empty state. */} + {/* Sits above BenchmarksBody, and outside it on purpose: pre-adoption keys + have no router sessions yet, so the shadow-eval section must render even + when the benchmarks body early-returns its loading, error, or empty state. */} + + ); }; diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.test.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.test.tsx index 5efe23fa0cd..283dbfc7d0e 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.test.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.test.tsx @@ -1,4 +1,5 @@ import { render, screen } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; import React from "react"; import { describe, expect, it, vi } from "vitest"; @@ -55,11 +56,24 @@ const job = (overrides: Partial = {}): ShadowEvalJob => ({ ...overrides, }); -const mockHooks = ({ jobs = [], detail = undefined }: { jobs?: ShadowEvalJob[]; detail?: ShadowEvalJob }) => { +const mockHooks = ({ + jobs = [], + detail = undefined, + detailsById = undefined, +}: { + jobs?: ShadowEvalJob[]; + detail?: ShadowEvalJob; + detailsById?: Record; +}) => { vi.mocked(useShadowEvalJobs).mockReturnValue({ data: jobs, error: null } as unknown as ReturnType< typeof useShadowEvalJobs >); - vi.mocked(useShadowEvalJob).mockReturnValue({ data: detail } as unknown as ReturnType); + vi.mocked(useShadowEvalJob).mockImplementation( + (_token, jobId) => + ({ + data: detailsById ? (jobId ? detailsById[jobId] : undefined) : detail, + }) as unknown as ReturnType, + ); vi.mocked(useStartShadowEval).mockReturnValue({ mutate: vi.fn(), isPending: false, @@ -116,4 +130,32 @@ describe("ShadowEvalSection", () => { render(); expect(screen.getByText("Start a shadow eval")).toBeInTheDocument(); }); + + it("keeps an older populated job reachable when the newest job has no verdicts", async () => { + const user = userEvent.setup(); + const empty = job({ + job_id: "job-new", + status: "pending", + completed_count: 0, + failed_count: 0, + results: null, + cost_actual: 0, + }); + const older = job({ job_id: "job-old", status: "completed" }); + mockHooks({ + jobs: [empty, older], + detailsById: { "job-new": empty, "job-old": older }, + }); + render(); + + // The newest job is empty, so its verdicts are absent from the primary card... + expect(screen.queryByText("SIMPLE")).not.toBeInTheDocument(); + + // ...but the older job with verdicts must still be reachable. + await user.click(screen.getByRole("button", { name: /Previous evaluations \(1\)/ })); + await user.click(screen.getByRole("button", { name: /job-old-row|10% via claude-auto/ })); + + expect(await screen.findByText("SIMPLE")).toBeInTheDocument(); + expect(screen.getByText("REASONING")).toBeInTheDocument(); + }); }); diff --git a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.tsx b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.tsx index d9ddc519ebd..cdddfdd6bf9 100644 --- a/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.tsx +++ b/ui/litellm-dashboard/src/app/(dashboard)/cost-optimization/_components/ShadowEvalSection.tsx @@ -210,6 +210,82 @@ const StartForm: React.FC<{ accessToken: string | null }> = ({ accessToken }) => ); }; +const PreviousJob: React.FC<{ + accessToken: string | null; + job: ShadowEvalJob; +}> = ({ accessToken, job }) => { + const [expanded, setExpanded] = useState(false); + const { data: detail } = useShadowEvalJob(accessToken, expanded ? job.job_id : null); + const shown = detail ?? job; + const results = shown.results; + const okOrBetter = results ? results.overall_shadow_win_rate_pct + results.overall_tie_rate_pct : null; + + return ( +
+ + {expanded ? ( +
+ {results && results.groups.length > 0 ? ( + + ) : ( +

No verdicts were recorded for this evaluation.

+ )} +
+ ) : null} +
+ ); +}; + +const PreviousJobs: React.FC<{ + accessToken: string | null; + jobs: readonly ShadowEvalJob[]; +}> = ({ accessToken, jobs }) => { + const [open, setOpen] = useState(false); + if (jobs.length === 0) return null; + return ( + + + {open ? ( +
+ {jobs.map((job) => ( + + ))} +
+ ) : null} +
+ ); +}; + interface ShadowEvalSectionProps { accessToken: string | null; } @@ -220,6 +296,7 @@ const ShadowEvalSection: React.FC = ({ accessToken }) => // Most recent job carries the section; older jobs list below it. const latest = useMemo(() => jobs?.[0] ?? null, [jobs]); + const previous = useMemo(() => jobs?.slice(1) ?? [], [jobs]); const { data: latestDetail } = useShadowEvalJob(accessToken, latest?.job_id ?? null); if (error instanceof ApiError && error.status === 403) return null; // admin-only section @@ -244,6 +321,8 @@ const ShadowEvalSection: React.FC = ({ accessToken }) => {!latest || (latestDetail && latestDetail.status !== "pending" && latestDetail.status !== "running") ? ( ) : null} + + ); };