From 355175a7c6550408a8bcca9ffd782fccd5b34f1d Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 8 Sep 2026 18:04:42 +0000 Subject: [PATCH] fix(eval): the usage adapter read a shape the callback never receives Running the gateway against the scripted provider proved the accounting merged in #3220 does not work, and the same run showed why nothing had caught it. LiteLLM does not hand a logger the upstream body. It normalises usage into its own Chat-Completions-shaped object first, so an OpenAI Responses reply reaches the callback as prompt_tokens / prompt_tokens_details.cached_tokens - never the input_tokens / input_tokens_details the shipped adapter reads. Every field came back unknown. The observed call_type is "anthropic_messages" as well, because Claude Code calls the Anthropic-shaped endpoint, so canonical_provider returned None and normalize_usage would have refused outright. Both were assumptions about a boundary I had only read about. The unit tests agreed with them because their fixture was written in the same wrong shape, so producer and consumer were consistent and both wrong - the exact failure the producer/consumer round trip exists to catch, one layer further out. Adds a LITELLM_NORMALIZED adapter for the object that actually arrives. The arithmetic is still OpenAI's - prompt_tokens is the whole, the details are subsets - so ordinary input is recovered by subtraction. The Responses adapter stays for a raw upstream body, which the mock still serves and tests directly. An unrecognised provider is still refused rather than guessed. The fixtures now carry the measured shape, and the end-to-end test asserts it through a real proxy: 48k prompt tokens with 44k cached is read back as 3k ordinary rather than as silence. 676 eval tests pass, 16 skipped, none failing. --- eval/tests/fixtures/fake_claude.py | 8 +++- eval/tests/test_mock_provider.py | 27 ++++++++--- .../tests/test_offline_session_integration.py | 46 +++++++++++++++++- eval/tests/test_provider_usage_capture.py | 39 ++++++++++----- eval/workflow_bench/litellm_usage_callback.py | 4 +- eval/workflow_bench/provider_usage.py | 47 ++++++++++++++++++- 6 files changed, 147 insertions(+), 24 deletions(-) diff --git a/eval/tests/fixtures/fake_claude.py b/eval/tests/fixtures/fake_claude.py index d0143219a..ba5f9bc71 100644 --- a/eval/tests/fixtures/fake_claude.py +++ b/eval/tests/fixtures/fake_claude.py @@ -74,7 +74,13 @@ def main() -> int: tool_results = [] for block in blocks: if block.get("type") == "tool_use": - output = _run_tool(block["name"], block.get("input", {})) + # A refused write is a tool ERROR the session reports and carries + # on from, not a crash. Letting it kill the process would lose the + # result event and misreport a working boundary as a broken run. + try: + output = _run_tool(block["name"], block.get("input", {})) + except OSError as exc: + output = f"error: {type(exc).__name__}: {exc}" tool_results.append({"type": "tool_result", "tool_use_id": block["id"], "content": output}) if tool_results: emit({"type": "user", "message": {"role": "user", "content": tool_results}}) diff --git a/eval/tests/test_mock_provider.py b/eval/tests/test_mock_provider.py index f300649c8..2d2e2eccf 100644 --- a/eval/tests/test_mock_provider.py +++ b/eval/tests/test_mock_provider.py @@ -6,7 +6,12 @@ import json import urllib.request from workflow_bench.mock_provider import MockProvider, Reply -from workflow_bench.provider_usage import ANTHROPIC, OPENAI_RESPONSES, normalize_usage +from workflow_bench.provider_usage import ( + ANTHROPIC, + LITELLM_NORMALIZED, + OPENAI_RESPONSES, + normalize_usage, +) def _post(url: str, payload: dict) -> tuple[int, bytes]: @@ -169,10 +174,20 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke assert usage_log.exists(), "the callback never wrote - the env did not reach the proxy" events = [json.loads(line) for line in usage_log.read_text().splitlines()] assert events, "the proxy started but recorded nothing" - native = events[-1]["native_usage"] - # The provider's own arithmetic survived the Anthropic-shaped translation. - assert native["input_tokens_details"]["cached_tokens"] == 7_000 - assert native["input_tokens_details"]["cache_write_tokens"] == 1_000 - usage = normalize_usage(events[-1]["provider"], native) + event = events[-1] + native = event["native_usage"] + # LiteLLM hands a callback its OWN normalised object, not the upstream body: + # an OpenAI Responses reply arrives as prompt_tokens / prompt_tokens_details. + # Asserting the wire shape here is what proved the shipped adapter read keys + # that are never present. + assert native["prompt_tokens_details"]["cached_tokens"] == 7_000 + assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000 + assert event["provider"] == LITELLM_NORMALIZED + assert event["call_type"] == "anthropic_messages", "the observed call type, not a Responses one" + + usage = normalize_usage(event["provider"], native) assert usage.total_input_tokens == 10_000 + assert usage.cache_read_input_tokens == 7_000 + assert usage.cache_write_input_tokens == 1_000 assert usage.ordinary_input_tokens == 2_000 + assert usage.complete, "a run that cannot interpret its own usage measured nothing" diff --git a/eval/tests/test_offline_session_integration.py b/eval/tests/test_offline_session_integration.py index d1a1912fb..148c6027e 100644 --- a/eval/tests/test_offline_session_integration.py +++ b/eval/tests/test_offline_session_integration.py @@ -15,7 +15,11 @@ from pathlib import Path import pytest from workflow_bench.mock_provider import MockProvider, Reply -from workflow_bench.proposer_sandbox import prepare_sandbox, prepare_review_workspace +from workflow_bench.proposer_sandbox import ( + host_workspace_write_boundary, + prepare_review_workspace, + prepare_sandbox, +) from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output from workflow_bench.runner_sessions import run_claude @@ -105,3 +109,43 @@ def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone assert record["ok"] is False assert record["error_kind"] is not None + + +def test_the_write_boundary_refuses_the_workspace_and_permits_the_artifact(clone: Path, tmp_path: Path) -> None: + """The contract the empty-artifact run violated, on the backend available here. + + A review must not change the workspace, and must still be able to write its + artifact ATOMICALLY - temp file beside the target, then rename - which is + what needs a writable parent DIRECTORY rather than a writable file. Both + halves are asserted through the real session, with the real boundary + applied, and the model scripted to attempt each one. + + Scope: this is the host-unsafe boundary, which its own docstring calls + best-effort because a session that can chmod can undo it. The kernel-enforced + version is bubblewrap's --ro-bind, which needs namespaces this machine cannot + create; that half stays with the real-sandbox canary in CI. + """ + + artifacts = tmp_path / "artifacts" + artifacts.mkdir() + target = artifacts / REVIEW_OUTPUT + protected = clone / "source.ts" + before = protected.read_text() + + write_artifact = {"name": "Write", "input": {"file_path": str(target), "content": REVIEW_JSON}} + tamper = {"name": "Write", "input": {"file_path": str(protected), "content": "tampered"}} + + # No writable= entry: the boundary only governs paths INSIDE the workspace + # (it refuses one that escapes), and the artifact directory deliberately + # lives outside it - that relocation is the fix for the empty-artifact run. + with host_workspace_write_boundary(clone): + with MockProvider(default=Reply(text="writing", tools=[write_artifact, tamper])) as provider: + record = _session(clone, provider) + + assert record["ok"] is True, record.get("error_detail") + # The artifact landed, written the way the agent's Write tool does it. + verdict, _findings = parse_review_output(target) + assert verdict == "approve" + assert not list(artifacts.glob("*.tmp.*")), "the rename landed rather than a copy" + # The workspace did not move. + assert protected.read_text() == before, "the read-only workspace was modified" diff --git a/eval/tests/test_provider_usage_capture.py b/eval/tests/test_provider_usage_capture.py index 99e542e4d..5493ec553 100644 --- a/eval/tests/test_provider_usage_capture.py +++ b/eval/tests/test_provider_usage_capture.py @@ -24,6 +24,7 @@ from workflow_bench.model_gateway import ( ) from workflow_bench.provider_usage import ( ANTHROPIC, + LITELLM_NORMALIZED, OPENAI_RESPONSES, USAGE_ENV_VARS, normalize_usage, @@ -49,11 +50,16 @@ def _openai_response(usage: dict) -> SimpleNamespace: ) +# The shape a callback actually receives: LiteLLM normalises usage into its own +# Chat-Completions-style object before any logger sees it, so an OpenAI reply +# arrives as prompt_tokens / prompt_tokens_details. Confirmed against a real +# proxy in tests/test_mock_provider.py; a fixture in the wire shape would test +# an object this code path never gets. NATIVE = { - "input_tokens": 48_000, - "input_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000}, - "output_tokens": 900, - "output_tokens_details": {"reasoning_tokens": 640}, + "prompt_tokens": 48_000, + "prompt_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000}, + "completion_tokens": 900, + "completion_tokens_details": {"reasoning_tokens": 640}, } @@ -79,9 +85,9 @@ def test_native_openai_usage_survives_the_anthropic_translation(logged) -> None: event = logged(NATIVE) native = event["native_usage"] # Verbatim: the fields an Anthropic-shaped response cannot carry. - assert native["input_tokens_details"]["cached_tokens"] == 44_000 - assert native["input_tokens_details"]["cache_write_tokens"] == 1_000 - assert native["output_tokens_details"]["reasoning_tokens"] == 640 + assert native["prompt_tokens_details"]["cached_tokens"] == 44_000 + assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000 + assert native["completion_tokens_details"]["reasoning_tokens"] == 640 assert event["response_id"] == "resp_68f2c1" @@ -100,7 +106,10 @@ def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None: event = logged(NATIVE) # The provider the LOG recorded, not one the test supplies - passing # OPENAI_RESPONSES by hand here is what hid the adapter-key mismatch. - assert event["provider"] == OPENAI_RESPONSES + # LITELLM_NORMALIZED, not OPENAI_RESPONSES: a proxy callback never sees the + # upstream body. Measured against a real gateway - the Responses adapter + # found none of its keys there and reported every field unknown. + assert event["provider"] == LITELLM_NORMALIZED assert event["provider_label"] == "openai" usage = normalize_usage(event["provider"], event["native_usage"]) assert usage.total_input_tokens == 48_000 @@ -112,7 +121,7 @@ def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None: def test_usage_without_details_normalizes_to_unknown_rather_than_zero(logged) -> None: """The mutation the accounting must not survive: dropped details, silent zeros.""" - stripped = {k: v for k, v in NATIVE.items() if k != "input_tokens_details"} + stripped = {k: v for k, v in NATIVE.items() if k != "prompt_tokens_details"} event = logged(stripped) usage = normalize_usage(event["provider"], event["native_usage"]) assert usage.cache_read_input_tokens is None @@ -238,10 +247,16 @@ def test_an_unresolvable_provider_is_refused_rather_than_guessed() -> None: from workflow_bench.provider_usage import canonical_provider - assert canonical_provider("openai", "responses") == OPENAI_RESPONSES - assert canonical_provider("openai", "completion") is None - assert canonical_provider("openai", None) is None + # Every openai call reaching this callback has already been normalised by + # LiteLLM, whatever endpoint it used - the observed call_type for a Claude + # Code request through the gateway is "anthropic_messages". The adapter has + # to match the object in hand, not the protocol on the wire. + assert canonical_provider("openai", "responses") == LITELLM_NORMALIZED + assert canonical_provider("openai", "anthropic_messages") == LITELLM_NORMALIZED assert canonical_provider("anthropic", "completion") == ANTHROPIC + # An unrecognised provider is still refused rather than guessed. + assert canonical_provider("some-new-provider", "responses") is None + assert canonical_provider(None, None) is None def test_request_identity_cannot_come_from_the_proxy_environment() -> None: diff --git a/eval/workflow_bench/litellm_usage_callback.py b/eval/workflow_bench/litellm_usage_callback.py index 843413765..223b4ccfb 100644 --- a/eval/workflow_bench/litellm_usage_callback.py +++ b/eval/workflow_bench/litellm_usage_callback.py @@ -42,8 +42,8 @@ def canonical_provider(label, call_type): # noqa: ANN001, ANN201 above for why this is a copy rather than an import. """ - if label == "openai" and call_type and "responses" in call_type: - return "openai-responses" + if label == "openai": + return "litellm-normalized" if label == "anthropic": return "anthropic" return None diff --git a/eval/workflow_bench/provider_usage.py b/eval/workflow_bench/provider_usage.py index 782eb18fa..4ae11b424 100644 --- a/eval/workflow_bench/provider_usage.py +++ b/eval/workflow_bench/provider_usage.py @@ -50,6 +50,14 @@ USAGE_ENV_VARS = (USAGE_LOG_ENV_VAR, SWEEP_ID_ENV_VAR) ANTHROPIC = "anthropic" OPENAI_RESPONSES = "openai-responses" +# What the gateway's callback actually receives. LiteLLM does not hand a logger +# the upstream body: it normalises usage into its own Chat-Completions-shaped +# object first, so an OpenAI Responses reply arrives as prompt_tokens / +# prompt_tokens_details even though the wire carried input_tokens / +# input_tokens_details. Measured against a real proxy, not assumed - the +# Responses adapter below found none of its keys and reported every field +# unknown. The arithmetic is still OpenAI's (the whole, with subsets). +LITELLM_NORMALIZED = "litellm-normalized" class UsageSemanticsError(ValueError): @@ -131,6 +139,35 @@ def _normalize_openai_responses(usage: Mapping[str, Any]) -> NormalizedUsage: ) +def _normalize_litellm(usage: Mapping[str, Any]) -> NormalizedUsage: + """prompt_tokens is the WHOLE; the details are subsets of it.""" + + total = _int_or_none(usage, "prompt_tokens") + details = usage.get("prompt_tokens_details") + cache_read = _int_or_none(details, "cached_tokens") + cache_write = _int_or_none(details, "cache_write_tokens") + if cache_write is None: + cache_write = _int_or_none(details, "cache_creation_tokens") + output_details = usage.get("completion_tokens_details") + + ordinary: int | None = None + if total is not None and cache_read is not None and cache_write is not None: + ordinary = total - cache_read - cache_write + if ordinary < 0: + raise UsageSemanticsError( + f"LiteLLM cached ({cache_read}) + cache_write ({cache_write}) " + f"exceed prompt_tokens ({total})" + ) + return NormalizedUsage( + ordinary_input_tokens=ordinary, + cache_read_input_tokens=cache_read, + cache_write_input_tokens=cache_write, + total_input_tokens=total, + output_tokens=_int_or_none(usage, "completion_tokens"), + reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"), + ) + + def _normalize_anthropic(usage: Mapping[str, Any]) -> NormalizedUsage: """input_tokens is the uncached REMAINDER; the cache fields add to it.""" @@ -162,8 +199,13 @@ def canonical_provider(label: str | None, call_type: str | None) -> str | None: the whole point of keeping the native object authoritative. """ - if label == "openai" and call_type and "responses" in call_type: - return OPENAI_RESPONSES + if label == "openai": + # Anything reaching a proxy callback has already been normalised by + # LiteLLM, whichever endpoint the caller used - the observed call_type + # for a Claude Code request through this gateway is "anthropic_messages", + # not a Responses one. The adapter has to match the object in hand, not + # the protocol on the wire. + return LITELLM_NORMALIZED if label in _ADAPTERS: return label return None @@ -172,6 +214,7 @@ def canonical_provider(label: str | None, call_type: str | None) -> str | None: _ADAPTERS = { ANTHROPIC: _normalize_anthropic, OPENAI_RESPONSES: _normalize_openai_responses, + LITELLM_NORMALIZED: _normalize_litellm, }