feat(shadow_eval): say which shape produced an unparseable judge verdict

The parser message alone cannot separate a judge that answered with nothing
from one truncated mid-object, and the two want opposite fixes. Records the
reply's shape, never its text, since no attempt row carries sampled content.
This commit is contained in:
moe-berri 2026-09-04 16:50:35 -07:00
parent da5e38ce9c
commit 5980055d7e
2 changed files with 113 additions and 1 deletions

View file

@ -345,6 +345,22 @@ def _failure_detail(e: BaseException) -> str:
return f"{type(e).__name__}{location}: {e}"
def _judge_reply_shape(response: object) -> str:
"""How an unparseable judge reply was shaped. The parser's own message cannot separate a
judge that answered with nothing from one truncated mid-object, and those want opposite
fixes. Shape only, never the reply text: the judge quotes the sampled turns it compares,
and no attempt row carries sampled content today."""
try:
choice: Final = response["choices"][0] # pyright: ignore[reportIndexIssue] # judge replies are subscriptable payloads
content: Final = choice["message"]["content"]
finish: Final = choice.get("finish_reason") or "unknown"
except (AttributeError, KeyError, IndexError, TypeError):
return "unreadable judge reply"
served: Final = str(getattr(response, "model", None) or "unknown")
body: Final = f"{len(str(content))} chars" if content else "no content"
return f"finish_reason={finish}, content={body}, model={served}"
def _call_cost(response: object) -> float:
"""Price one eval-arm call with the figure the spend pipeline bills: the router client
stamps _hidden_params.response_cost from the deployment's own pricing, which the public
@ -1139,7 +1155,9 @@ class ShadowEvalLogger(CustomLogger):
verdict: Final = PairwiseVerdict.model_validate(parse_json_verdict(raw))
except Exception as e: # noqa: BLE001 # malformed verdicts become error rows
verbose_logger.debug("shadow_eval: unparseable judge verdict: %s", e)
return _CallFailure(f"unparseable judge verdict: {e}", cost=_call_cost(response))
return _CallFailure(
f"unparseable judge verdict: {e}; {_judge_reply_shape(response)}", cost=_call_cost(response)
)
return _JudgeVerdict(
preference=_unmask_preference(verdict.preference, real_is_a),
confidence=max(0.0, min(1.0, verdict.confidence)),

View file

@ -141,6 +141,27 @@ def _reasoning_judge_router(
return router
def _judge_reply_router(content: str | None, finish_reason: str = "stop", served_model: str = "judge-pick") -> MagicMock:
"""A router whose judge arm returns a caller-shaped reply, so the shapes that all land
on the same parser error can be posed apart: no content at all, versus JSON cut off
mid-object."""
router = MagicMock()
router.model_group_alias = {}
router.get_model_list = MagicMock(return_value=[{"litellm_params": {"model": "openai/gpt-4o-mini"}}])
async def acompletion(**kwargs):
if kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) == SHADOW_EVAL_ROUTER_CALL_ORIGIN:
kwargs["metadata"]["routing_decision"] = {"tier_label": "SIMPLE", "routed_model": "cheap-model"}
return {"choices": [{"message": {"content": "shadow answer"}}]}
return ModelResponse(
model=served_model,
choices=[{"index": 0, "finish_reason": finish_reason, "message": {"role": "assistant", "content": content}}],
)
router.acompletion = MagicMock(side_effect=acompletion)
return router
def _spend_counter(store=None):
"""In-memory stand-in for the proxy's cross-pod spend counter: reads take the max of
the counter and the caller's fallback, exactly like get_current_spend does for a key
@ -1126,6 +1147,79 @@ class TestShadowPipeline:
assert row["judge_cost"] == expected_cost
assert row["shadow_cost"] == expected_shadow_cost
async def _judge_error(self, router: MagicMock, monkeypatch: pytest.MonkeyPatch) -> str:
import litellm as litellm_module
monkeypatch.setattr(litellm_module, "completion_cost", lambda completion_response: 0.007)
prisma = _prisma()
await _logger(router=router, prisma=prisma)._run_shadow_eval(
job=_job(),
request_id="req-1",
messages=({"role": "user", "content": "hi"},),
real_text="real answer",
real_model="claude-opus",
real_cost=0.0,
real_classifier_cost=0.0,
real_cache_hit=False,
control_tier=None,
shadow_params={},
parent_metadata={},
)
return prisma.db.litellm_shadowevalattempt.create.call_args.kwargs["data"]["error"]
async def test_a_judge_that_answered_nothing_is_told_apart_from_one_cut_off(
self, monkeypatch: pytest.MonkeyPatch
):
"""Both land on the same parser message, and they want opposite fixes: a judge
returning no content points at the reply never being text, while one cut off
mid-object points at the output cap. The row has to say which."""
truncated = '{"preference": "A", "confidence": 0.9, "reasoning": "'
answered_nothing = await self._judge_error(_judge_reply_router(None), monkeypatch)
cut_off = await self._judge_error(
_judge_reply_router(truncated, finish_reason="length"), monkeypatch
)
assert "content=no content" in answered_nothing
assert "finish_reason=stop" in answered_nothing
assert f"content={len(truncated)} chars" in cut_off
assert "finish_reason=length" in cut_off
async def test_an_unparseable_verdict_names_the_model_that_served_it(self, monkeypatch: pytest.MonkeyPatch):
"""A judge_model that fans out over deployments hides which one truncates: without
the served model the operator cannot tell a bad deployment from a bad cap."""
error = await self._judge_error(_judge_reply_router(None, served_model="claude-sonnet-5"), monkeypatch)
assert "model=claude-sonnet-5" in error
async def test_a_diagnosed_verdict_error_stays_groupable(self, monkeypatch: pytest.MonkeyPatch):
"""The customer groups attempt rows by error text. Every varying part has to sit
after the first semicolon or each row becomes its own group."""
first = await self._judge_error(_judge_reply_router(None, served_model="model-a"), monkeypatch)
second = await self._judge_error(_judge_reply_router(None, served_model="model-b"), monkeypatch)
assert first != second
assert first.split(";")[0] == second.split(";")[0]
async def test_a_judge_reply_that_cannot_be_read_still_records_an_error(self, monkeypatch: pytest.MonkeyPatch):
"""The shape reader runs inside the failure path: it must never raise a second time
and cost the row entirely."""
router = MagicMock()
router.model_group_alias = {}
router.get_model_list = MagicMock(return_value=[{"litellm_params": {"model": "openai/gpt-4o-mini"}}])
async def acompletion(**kwargs):
if kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) == SHADOW_EVAL_ROUTER_CALL_ORIGIN:
kwargs["metadata"]["routing_decision"] = {"tier_label": "SIMPLE", "routed_model": "cheap-model"}
return {"choices": [{"message": {"content": "shadow answer"}}]}
return {"choices": []}
router.acompletion = MagicMock(side_effect=acompletion)
error = await self._judge_error(router, monkeypatch)
assert "unparseable judge verdict" in error
assert "unreadable judge reply" in error
async def test_an_empty_shadow_reply_still_bills_its_cost(self, monkeypatch: pytest.MonkeyPatch):
"""A shadow call that returns no extractable text has still billed; pricing it at
zero would keep the dollar gate open while shadow calls keep charging the key."""