diff --git a/litellm/integrations/shadow_eval_logger.py b/litellm/integrations/shadow_eval_logger.py index 27da785331a..b28d21ba1ca 100644 --- a/litellm/integrations/shadow_eval_logger.py +++ b/litellm/integrations/shadow_eval_logger.py @@ -165,28 +165,91 @@ def _chat_request_from_responses( ) -def _chat_final_text(response_obj: object) -> str: - """The assistant's text, or empty when the turn carries tool calls: only text-final - turns produce a judgeable A/B comparison.""" +def _chat_choice(response_obj: object) -> object | None: + """The response's first choice, from a payload mapping or a duck-typed ModelResponse.""" try: - message: Final = ( - response_obj["choices"][0]["message"] - if isinstance(response_obj, Mapping) - else response_obj.choices[0].message # pyright: ignore[reportAttributeAccessIssue] # duck-typed ModelResponse - ) + if isinstance(response_obj, Mapping): + return response_obj["choices"][0] + return response_obj.choices[0] # pyright: ignore[reportAttributeAccessIssue] # duck-typed ModelResponse except (AttributeError, KeyError, IndexError, TypeError): + return None + + +def _field_reader(obj: object) -> Callable[[str], object]: + return obj.get if isinstance(obj, Mapping) else lambda key: getattr(obj, key, None) + + +def _chat_message_reader(response_obj: object) -> Callable[[str], object] | None: + """Field access over the assistant message of a chat response, or None for a payload + with no readable message.""" + choice: Final = _chat_choice(response_obj) + if choice is None: + return None + message: Final = _field_reader(choice)("message") + return _field_reader(message) if message is not None else None + + +def _chat_final_text(response_obj: object) -> str: + """The turn's judgeable text: prose, or every tool call serialized alongside it as + `[tool call] name(arguments)` when the assistant chose to act instead of, or as well + as, answering directly. A tool call is a real turn, not a gap, so this is what both + the real arm's sampling decision and the shadow arm's reply compare against.""" + read: Final = _chat_message_reader(response_obj) + if read is None: return "" - read: Final = message.get if isinstance(message, Mapping) else lambda key: getattr(message, key, None) - if read("tool_calls") or read("function_call"): - return "" - return extract_text_from_content(read("content")) + prose: Final = extract_text_from_content(read("content")) + if not (read("tool_calls") or read("function_call")): + return prose + serialized: Final = _serialize_tool_calls(read) + return f"{prose} {serialized}".strip() if prose else serialized + + +def _chat_finish_reason(response_obj: object) -> str: + choice: Final = _chat_choice(response_obj) + raw: Final = _field_reader(choice)("finish_reason") if choice is not None else None + return str(raw) if raw else "unknown" + + +_RESPONSES_TOOL_CALL_TYPES: Final = frozenset(("function_call", "custom_tool_call")) + + +def _tool_calls_list(read: Callable[[str], object]) -> tuple[object, ...]: + calls: Final = read("tool_calls") + listed: Final = tuple(calls) if isinstance(calls, Sequence) and not isinstance(calls, str) else () + single: Final = read("function_call") + return listed if listed else ((single,) if single is not None else ()) + + +def _tool_call_invocation(call: object) -> str: + """One tool call as `name(arguments)`. Custom tool calls name themselves and carry their + arguments under `custom` rather than `function`.""" + read_call: Final = _field_reader(call) + payload: Final = read_call("function") or read_call("custom") or call + read_payload: Final = _field_reader(payload) + name: Final = read_payload("name") + arguments: Final = read_payload("arguments") or read_payload("input") or "" + return f"{name or 'unnamed'}({arguments})" + + +def _serialize_tool_calls(read: Callable[[str], object]) -> str: + """Every tool call in a reply as text a judge built for prose can still read.""" + return ", ".join(f"[tool call] {_tool_call_invocation(call)}" for call in _tool_calls_list(read)) + + +def _shadow_empty_reply_error(response_obj: object, routed_model: str) -> str: + """Why a shadow reply yielded no judgeable text at all: no prose, and no tool call to + serialize either. The stable sentence comes first and every varying part after the + semicolon, so grouping rows by error still yields one row per cause.""" + detail: Final = f"finish_reason={_chat_finish_reason(response_obj)}, model={routed_model or 'unknown'}" + return f"shadow router returned an empty response; {detail}" def _responses_final_text(response_obj: object) -> str: - """The turn's aggregated output text, or empty when the turn carries tool calls. A - dict-shaped payload is validated into the owner type first, because ``output_text`` - is a derived property rather than a serialized field, so it never exists on a dict; - a dict the owner type rejects is unjudgeable and skipped.""" + """The turn's judgeable text: the aggregated output plus any tool call serialized + alongside it, the same way the chat surface renders one. A dict-shaped payload is + validated into the owner type first, because ``output_text`` is a derived property + rather than a serialized field, so it never exists on a dict; a dict the owner type + rejects is unjudgeable and skipped.""" from litellm.types.llms.openai import ResponsesAPIResponse try: @@ -199,11 +262,16 @@ def _responses_final_text(response_obj: object) -> str: if not isinstance(output, Sequence): return "" items: Final = tuple(item.model_dump() if isinstance(item, BaseModel) else item for item in output) - if any( - not isinstance(item, Mapping) or item.get("type") in ("function_call", "custom_tool_call") for item in items - ): + if any(not isinstance(item, Mapping) for item in items): return "" - return str(getattr(response, "output_text", "") or "") + calls: Final = tuple( + item for item in items if isinstance(item, Mapping) and item.get("type") in _RESPONSES_TOOL_CALL_TYPES + ) + prose: Final = str(getattr(response, "output_text", "") or "") + if not calls: + return prose + serialized: Final = ", ".join(f"[tool call] {_tool_call_invocation(call)}" for call in calls) + return f"{prose} {serialized}".strip() if prose else serialized class _SurfaceOps: @@ -273,8 +341,8 @@ def _judgeable_sample( response_obj: object, ) -> tuple[tuple[Mapping[str, object], ...], Mapping[str, object], str] | None: """The normalized chat conversation, the forwardable generation params, and the - judgeable final text; None when this request's shapes cannot be sampled (tool-final - turn, empty text, or a shape the owner transformations reject).""" + judgeable final text; None when this request's shapes cannot be sampled (no text and no + tool call to serialize, or a shape the owner transformations reject).""" try: request: Final = ops.chat_request(kwargs, model_parameters) items: Final = _MESSAGE_ITEMS_ADAPTER.validate_python(request.get("messages")) @@ -307,6 +375,11 @@ PAIRWISE_JUDGE_SYSTEM_PROMPT: Final = """You are an impartial quality judge comp The responses are labeled A and B in random order. You do not know which system produced which. +A response may be prose, or a tool call shown as `[tool call] name(arguments)` if the +assistant chose to act instead of answering directly. A tool call is not a defect: judge +whether calling that tool was the right response to the conversation, the same as you +would judge prose. + Criteria: correctness, completeness, clarity, conciseness. Return ONLY valid JSON in this exact format, no other text: @@ -376,14 +449,37 @@ def _unmask_preference(raw_preference: str, real_is_a: bool) -> str: return "tie" -def _judge_user_prompt(conversation: str, response_a: str, response_b: str) -> str: +_MAX_JUDGE_TOOL_DEFS_CHARS: Final = 2_000 + + +def _tool_definitions_text(tools: object) -> str: + """The tools available to both arms, name and description only: enough for the judge + to tell whether the chosen tool, and not some other one, was the right call, without + forwarding parameter schemas it does not need to score that.""" + if not isinstance(tools, Sequence) or isinstance(tools, str): + return "" + entries: Final = tuple( + _field_reader(t)("function") or _field_reader(t)("custom") or t for t in tools if not isinstance(t, str) + ) + lines: Final = tuple( + f"- {_field_reader(e)('name') or 'unnamed'}: {_field_reader(e)('description') or 'no description'}" + for e in entries + ) + if not lines: + return "" + return ("Tools available to both responses:\n" + "\n".join(lines))[:_MAX_JUDGE_TOOL_DEFS_CHARS] + + +def _judge_user_prompt(conversation: str, response_a: str, response_b: str, tool_definitions: str = "") -> str: """The judge prompt under one total character budget: each response is capped, and - the conversation tail gets whatever budget the responses left over.""" + the conversation tail gets whatever budget the responses and tool definitions left + over.""" a: Final = response_a[:_MAX_JUDGE_RESPONSE_CHARS] b: Final = response_b[:_MAX_JUDGE_RESPONSE_CHARS] - conversation_budget: Final = _MAX_JUDGE_PROMPT_CHARS - len(a) - len(b) + prefix: Final = f"{tool_definitions}\n\n" if tool_definitions else "" + conversation_budget: Final = _MAX_JUDGE_PROMPT_CHARS - len(a) - len(b) - len(prefix) return ( - f"Conversation:\n{conversation[-conversation_budget:]}\n\n" + f"{prefix}Conversation:\n{conversation[-conversation_budget:]}\n\n" f"Response A:\n{a}\n\n" f"Response B:\n{b}\n\n" "Which response is better?" @@ -942,6 +1038,7 @@ class ShadowEvalLogger(CustomLogger): messages=messages, real_text=real_text, shadow_text=shadow.text, + tools=shadow_params.get("tools"), parent_metadata=parent_metadata, ) if isinstance(verdict, _CallFailure): @@ -1080,15 +1177,18 @@ class ShadowEvalLogger(CustomLogger): classifier_cost=_decision_classifier_cost(shadow_metadata), ) text: Final = _chat_final_text(response) + routed_model: Final = str( + getattr(response, "model", None) or _routing_decision(shadow_metadata).get("routed_model") or "" + ) if not text: return _CallFailure( - "shadow router returned an empty response", + _shadow_empty_reply_error(response, routed_model), cost=_call_cost(response), classifier_cost=_decision_classifier_cost(shadow_metadata), ) return _ShadowResponse( text=text, - model=str(getattr(response, "model", None) or _routing_decision(shadow_metadata).get("routed_model") or ""), + model=routed_model, tier=_routed_tier(shadow_metadata), cost=_call_cost(response), classifier_cost=_decision_classifier_cost(shadow_metadata), @@ -1100,9 +1200,12 @@ class ShadowEvalLogger(CustomLogger): messages: Sequence[Mapping[str, object]], real_text: str, shadow_text: str, + tools: object, parent_metadata: Mapping[str, object], ) -> "_JudgeVerdict | _CallFailure": - """Blind pairwise judge with A/B labels randomized to cancel position bias.""" + """Blind pairwise judge with A/B labels randomized to cancel position bias. Both + arms were offered the same tools, so the judge is shown their definitions too: a + tool call is only assessable against what else was available to call instead.""" real_is_a: Final = random.random() < 0.5 response_a: Final = real_text if real_is_a else shadow_text response_b: Final = shadow_text if real_is_a else real_text @@ -1117,7 +1220,7 @@ class ShadowEvalLogger(CustomLogger): {"role": "system", "content": PAIRWISE_JUDGE_SYSTEM_PROMPT}, # mutable-ok: SDK message { "role": "user", - "content": _judge_user_prompt(conversation, response_a, response_b), + "content": _judge_user_prompt(conversation, response_a, response_b, _tool_definitions_text(tools)), }, # mutable-ok: SDK message ] try: diff --git a/tests/test_litellm/integrations/test_shadow_eval_logger.py b/tests/test_litellm/integrations/test_shadow_eval_logger.py index 5628d69de26..f273a285d49 100644 --- a/tests/test_litellm/integrations/test_shadow_eval_logger.py +++ b/tests/test_litellm/integrations/test_shadow_eval_logger.py @@ -24,7 +24,13 @@ from litellm.integrations.shadow_eval_logger import ( _unmask_preference, ) from litellm.types.guardrails import GuardrailEventHooks -from litellm.types.utils import SHADOW_EVAL_JUDGE_CALL_ORIGIN, SHADOW_EVAL_ROUTER_CALL_ORIGIN, ModelResponse +from litellm.types.utils import ( + SHADOW_EVAL_JUDGE_CALL_ORIGIN, + SHADOW_EVAL_ROUTER_CALL_ORIGIN, + ChatCompletionCustomToolCallPayload, + ChatCompletionMessageCustomToolCall, + ModelResponse, +) def _job(**overrides) -> ActiveShadowEvalJob: @@ -120,6 +126,39 @@ def _router( return router +def _shadow_reply_router(message, finish_reason="stop", routed_model="cheap-model"): + """A router whose shadow arm answers with a caller-supplied message, so a reply that + yields no judgeable text can be posed as the two different things it can be: an arm + that chose a tool, or an arm that returned nothing.""" + router = MagicMock() + router.model_group_alias = {} + router.get_model_list = MagicMock(return_value=[{"litellm_params": {"model": "openai/gpt-4o-mini"}}]) + + async def acompletion(**kwargs): + if kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN: + return {"choices": [{"message": {"content": '{"preference": "A", "confidence": 0.9}'}}]} + kwargs["metadata"]["routing_decision"] = {"tier_label": "SIMPLE", "routed_model": routed_model} + return {"choices": [{"message": message, "finish_reason": finish_reason}]} + + router.acompletion = MagicMock(side_effect=acompletion) + return router + + +TOOL_CALL_MESSAGE = { + "content": None, + "tool_calls": [{"id": "c1", "type": "function", "function": {"name": "Read", "arguments": "{}"}}], +} + +CUSTOM_TOOL_CALL_MESSAGE = { + "content": None, + "tool_calls": [ + ChatCompletionMessageCustomToolCall( + id="c2", custom=ChatCompletionCustomToolCallPayload(name="exec_sql", input="select 1") + ) + ], +} + + def _spend_counter(store=None): """In-memory stand-in for the proxy's cross-pod spend counter: reads take the max of the counter and the caller's fallback, exactly like get_current_spend does for a key @@ -368,7 +407,13 @@ class TestSurfaceNormalization: ], ids=["tool-final-chat-turn", "tool-final-responses-turn"], ) - async def test_unjudgeable_turns_are_skipped_without_consuming_budget(self, response_mutation, kwargs_mutation): + async def test_a_tool_final_turn_is_sampled_and_serialized_for_the_judge( + self, response_mutation, kwargs_mutation + ): + """A turn where the real model called a tool used to be dropped before sampling, on + every surface. On agentic traffic that is most of the traffic, so a job set to + sample 10% was really sampling 10% of the prose-only slice and calling it 10% of + the key. The turn is sampled like any other and the call is serialized as text.""" from litellm.types.llms.openai import ResponsesAPIResponse hook_kwargs = _success_kwargs(**({"call_type": "acompletion"} | kwargs_mutation)) @@ -406,6 +451,38 @@ class TestSurfaceNormalization: prisma, router = await self._drive(hook_kwargs, response) + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + assert "[tool call] f({})" in judge_prompt + prisma.db.litellm_shadowevalattempt.create.assert_called_once() + + @pytest.mark.parametrize( + "response_mutation,kwargs_mutation", + [ + ("chat-no-content", {}), + ("responses-no-output", {"call_type": "aresponses"}), + ], + ids=["empty-chat-turn", "empty-responses-turn"], + ) + async def test_turns_with_nothing_to_compare_are_skipped_without_consuming_budget( + self, response_mutation, kwargs_mutation + ): + """No prose and no tool call leaves the judge nothing to score, so the turn is + still skipped rather than billed.""" + from litellm.types.llms.openai import ResponsesAPIResponse + + hook_kwargs = _success_kwargs(**({"call_type": "acompletion"} | kwargs_mutation)) + if response_mutation == "chat-no-content": + response = {"choices": [{"message": {"content": ""}}]} + else: + hook_kwargs["messages"] = "do the thing" + response = ResponsesAPIResponse.model_validate(RESPONSES_API_RESPONSE | {"output": []}) + + prisma, router = await self._drive(hook_kwargs, response) + router.acompletion.assert_not_called() prisma.db.litellm_shadowevalattempt.create.assert_not_called() @@ -1134,6 +1211,206 @@ class TestShadowPipeline: assert row["shadow_cost"] == 0.007 assert logger._test_counter["spend:shadow_eval:job-1"] == 0.007 + async def _no_text_error(self, router) -> str: + prisma = _prisma() + await _logger(router=router, prisma=prisma)._run_shadow_eval( + job=_job(), + request_id="req-1", + messages=({"role": "user", "content": "hi"},), + real_text="real answer", + real_model="claude-opus", + real_cost=0.0, + real_classifier_cost=0.0, + real_cache_hit=False, + control_tier=None, + shadow_params={}, + parent_metadata={}, + ) + row = prisma.db.litellm_shadowevalattempt.create.call_args.kwargs["data"] + assert row["outcome"] == "error" + return row["error"] + + async def _judged_shadow_row(self, router: MagicMock, shadow_params: dict | None = None) -> dict: + prisma = _prisma() + await _logger(router=router, prisma=prisma)._run_shadow_eval( + job=_job(), + request_id="req-1", + messages=({"role": "user", "content": "hi"},), + real_text="real answer", + real_model="claude-opus", + real_cost=0.0, + real_classifier_cost=0.0, + real_cache_hit=False, + control_tier=None, + shadow_params=shadow_params or {}, + parent_metadata={}, + ) + return prisma.db.litellm_shadowevalattempt.create.call_args.kwargs["data"] + + async def test_a_tool_call_shadow_reply_is_judged_rather_than_discarded(self): + """An arm that calls a tool where the real model wrote prose has answered, it just + answered by acting. Dropping that turn threw away the comparison the job exists to + make, and on agentic traffic it threw away most of them, so the tool call is + serialized into text and judged like any other response.""" + row = await self._judged_shadow_row(_shadow_reply_router(TOOL_CALL_MESSAGE, finish_reason="tool_calls")) + + assert row["outcome"] != "error" + assert row["error"] is None + assert row["confidence"] == 0.9 + + async def test_a_tool_call_reaches_the_judge_as_readable_text(self): + """The judge only ever sees strings, so a tool call has to arrive as its name and + arguments. A serialization that dropped either would ask the judge to score a + response it cannot tell apart from any other tool call.""" + router = _shadow_reply_router(TOOL_CALL_MESSAGE, finish_reason="tool_calls") + await self._judged_shadow_row(router) + + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "[tool call] Read({})" in judge_prompt + + async def test_the_judge_sees_what_tools_were_available(self): + """Scoring whether a tool call was the right response needs to know what else the + arm could have called instead. Without the tool list, the judge can score the + arguments but not whether Read, specifically, was the correct choice.""" + router = _shadow_reply_router(TOOL_CALL_MESSAGE, finish_reason="tool_calls") + tools = [ + {"type": "function", "function": {"name": "Read", "description": "read a file from disk"}}, + {"type": "function", "function": {"name": "Bash", "description": "run a shell command"}}, + ] + await self._judged_shadow_row(router, shadow_params={"tools": tools}) + + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "Read: read a file from disk" in judge_prompt + assert "Bash: run a shell command" in judge_prompt + + async def test_a_custom_tool_definition_is_named_for_the_judge(self): + """A custom tool definition nests name and description under `custom`, not + `function`, so reading only `function` renders every one of them as unnamed and + tells the judge nothing about what the arm could have called.""" + from openai.types.chat import ChatCompletionCustomToolParam + + router = _shadow_reply_router(TOOL_CALL_MESSAGE, finish_reason="tool_calls") + tools = [ + ChatCompletionCustomToolParam( + type="custom", + custom={"name": "exec_sql", "description": "run a read-only sql query"}, + ) + ] + await self._judged_shadow_row(router, shadow_params={"tools": tools}) + + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "exec_sql: run a read-only sql query" in judge_prompt + assert "unnamed" not in judge_prompt + + @pytest.mark.parametrize("shadow_params", [{}, {"tools": []}], ids=["omitted", "empty-list"]) + async def test_no_tool_definitions_section_when_the_turn_offered_no_tools(self, shadow_params): + """Padding every judge prompt with an empty tools section wastes budget on the + turns, still the majority, that never offered one, whether tools was left out of + the request entirely or sent as an empty list.""" + router = _shadow_reply_router({"content": "hello"}, finish_reason="stop") + await self._judged_shadow_row(router, shadow_params=shadow_params) + + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "Tools available" not in judge_prompt + + async def test_a_custom_tool_call_serializes_its_name_and_input(self): + """Custom tool calls carry no `function` key: name and arguments live under + `custom`, so reading only `function` serializes every one of them as unnamed.""" + router = _shadow_reply_router(CUSTOM_TOOL_CALL_MESSAGE, finish_reason="tool_calls") + await self._judged_shadow_row(router) + + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "[tool call] exec_sql(select 1)" in judge_prompt + + async def test_the_judge_is_told_a_tool_call_is_not_a_defect(self): + """The judge scores on completeness and clarity. Handed a tool call with no + instruction, it marks it down for not reading like an answer, which would bias + every verdict against a tool-calling arm on exactly the traffic that calls tools.""" + router = _shadow_reply_router(TOOL_CALL_MESSAGE, finish_reason="tool_calls") + await self._judged_shadow_row(router) + + system_prompt = next( + call.kwargs["messages"][0]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "tool call" in system_prompt + assert "not a defect" in system_prompt + + async def test_prose_written_alongside_a_tool_call_survives_into_the_verdict(self): + """Some providers write a sentence before acting. Serializing only the call would + hide half of what the arm actually said from the judge.""" + router = _shadow_reply_router( + {"content": "Let me look that up.", "tool_calls": TOOL_CALL_MESSAGE["tool_calls"]}, + finish_reason="tool_calls", + ) + await self._judged_shadow_row(router) + + judge_prompt = next( + call.kwargs["messages"][-1]["content"] + for call in router.acompletion.call_args_list + if call.kwargs["metadata"].get(INTERNAL_CALL_ORIGIN_METADATA_KEY) != SHADOW_EVAL_ROUTER_CALL_ORIGIN + ) + + assert "Let me look that up. [tool call] Read({})" in judge_prompt + + async def test_an_empty_shadow_reply_names_the_finish_reason_and_the_routed_model(self): + """A reply that really carried no text is diagnosable only if the row says what + the arm was doing when it produced none: a truncated turn and a model that answers + with nothing are different faults with different fixes.""" + error = await self._no_text_error( + _shadow_reply_router({"content": ""}, finish_reason="length", routed_model="some-model") + ) + + assert "empty response" in error + assert "finish_reason=length" in error + assert "model=some-model" in error + + async def test_no_text_errors_stay_groupable_across_models_and_finish_reasons(self): + """Operators read these rows by grouping on the error text, which is how a job's + failures collapse to a handful of causes. Every varying part therefore has to sit + behind the first semicolon, or each row becomes its own group and the count that + made the problem visible stops existing.""" + first = await self._no_text_error( + _shadow_reply_router({"content": None}, finish_reason="length", routed_model="model-a") + ) + second = await self._no_text_error( + _shadow_reply_router( + {"content": ""}, + finish_reason="stop", + routed_model="model-b", + ) + ) + + assert first != second + assert first.split(";")[0] == second.split(";")[0] + async def test_a_pipeline_error_after_the_shadow_call_keeps_its_billed_cost(self, monkeypatch: pytest.MonkeyPatch): """An unexpected error between the billed shadow call and the attempt write must still record the shadow cost, or the per-key dollar gate undercounts forever.""" @@ -1691,11 +1968,13 @@ class TestSamplingFunnel: prisma.db.litellm_shadowevalattempt.create.assert_not_awaited() async def test_an_unjudgeable_sampled_request_counts_unjudgeable(self): + """A tool call still serializes into judgeable text; a turn with neither prose nor + a tool call to serialize is the one case left with nothing to compare.""" prisma = _prisma() logger = _logger(router=_router(), prisma=prisma, jobs=(_job(),)) - tool_final = {"choices": [{"message": {"content": None, "tool_calls": [{"type": "function", "function": {}}]}}]} + empty = {"choices": [{"message": {"content": None}}]} - await logger.async_log_success_event(_success_kwargs(), tool_final, None, None) + await logger.async_log_success_event(_success_kwargs(), empty, None, None) await _drain(logger) assert logger._test_funnel == [("job-1", "unjudgeable")]