fix(eval): the usage adapter read a shape the callback never receives

Running the gateway against the scripted provider proved the accounting merged
in #3220 does not work, and the same run showed why nothing had caught it.

LiteLLM does not hand a logger the upstream body. It normalises usage into its
own Chat-Completions-shaped object first, so an OpenAI Responses reply reaches
the callback as prompt_tokens / prompt_tokens_details.cached_tokens - never the
input_tokens / input_tokens_details the shipped adapter reads. Every field came
back unknown. The observed call_type is "anthropic_messages" as well, because
Claude Code calls the Anthropic-shaped endpoint, so canonical_provider returned
None and normalize_usage would have refused outright.

Both were assumptions about a boundary I had only read about. The unit tests
agreed with them because their fixture was written in the same wrong shape, so
producer and consumer were consistent and both wrong - the exact failure the
producer/consumer round trip exists to catch, one layer further out.

Adds a LITELLM_NORMALIZED adapter for the object that actually arrives. The
arithmetic is still OpenAI's - prompt_tokens is the whole, the details are
subsets - so ordinary input is recovered by subtraction. The Responses adapter
stays for a raw upstream body, which the mock still serves and tests directly.
An unrecognised provider is still refused rather than guessed.

The fixtures now carry the measured shape, and the end-to-end test asserts it
through a real proxy: 48k prompt tokens with 44k cached is read back as 3k
ordinary rather than as silence.

676 eval tests pass, 16 skipped, none failing.
This commit is contained in:
Gergo Magyar 2026-09-08 18:04:42 +00:00
parent 73ad083b69
commit 355175a7c6
6 changed files with 147 additions and 24 deletions

View file

@ -74,7 +74,13 @@ def main() -> int:
tool_results = []
for block in blocks:
if block.get("type") == "tool_use":
output = _run_tool(block["name"], block.get("input", {}))
# A refused write is a tool ERROR the session reports and carries
# on from, not a crash. Letting it kill the process would lose the
# result event and misreport a working boundary as a broken run.
try:
output = _run_tool(block["name"], block.get("input", {}))
except OSError as exc:
output = f"error: {type(exc).__name__}: {exc}"
tool_results.append({"type": "tool_result", "tool_use_id": block["id"], "content": output})
if tool_results:
emit({"type": "user", "message": {"role": "user", "content": tool_results}})

View file

@ -6,7 +6,12 @@ import json
import urllib.request
from workflow_bench.mock_provider import MockProvider, Reply
from workflow_bench.provider_usage import ANTHROPIC, OPENAI_RESPONSES, normalize_usage
from workflow_bench.provider_usage import (
ANTHROPIC,
LITELLM_NORMALIZED,
OPENAI_RESPONSES,
normalize_usage,
)
def _post(url: str, payload: dict) -> tuple[int, bytes]:
@ -169,10 +174,20 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke
assert usage_log.exists(), "the callback never wrote - the env did not reach the proxy"
events = [json.loads(line) for line in usage_log.read_text().splitlines()]
assert events, "the proxy started but recorded nothing"
native = events[-1]["native_usage"]
# The provider's own arithmetic survived the Anthropic-shaped translation.
assert native["input_tokens_details"]["cached_tokens"] == 7_000
assert native["input_tokens_details"]["cache_write_tokens"] == 1_000
usage = normalize_usage(events[-1]["provider"], native)
event = events[-1]
native = event["native_usage"]
# LiteLLM hands a callback its OWN normalised object, not the upstream body:
# an OpenAI Responses reply arrives as prompt_tokens / prompt_tokens_details.
# Asserting the wire shape here is what proved the shipped adapter read keys
# that are never present.
assert native["prompt_tokens_details"]["cached_tokens"] == 7_000
assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000
assert event["provider"] == LITELLM_NORMALIZED
assert event["call_type"] == "anthropic_messages", "the observed call type, not a Responses one"
usage = normalize_usage(event["provider"], native)
assert usage.total_input_tokens == 10_000
assert usage.cache_read_input_tokens == 7_000
assert usage.cache_write_input_tokens == 1_000
assert usage.ordinary_input_tokens == 2_000
assert usage.complete, "a run that cannot interpret its own usage measured nothing"

View file

@ -15,7 +15,11 @@ from pathlib import Path
import pytest
from workflow_bench.mock_provider import MockProvider, Reply
from workflow_bench.proposer_sandbox import prepare_sandbox, prepare_review_workspace
from workflow_bench.proposer_sandbox import (
host_workspace_write_boundary,
prepare_review_workspace,
prepare_sandbox,
)
from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output
from workflow_bench.runner_sessions import run_claude
@ -105,3 +109,43 @@ def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone
assert record["ok"] is False
assert record["error_kind"] is not None
def test_the_write_boundary_refuses_the_workspace_and_permits_the_artifact(clone: Path, tmp_path: Path) -> None:
"""The contract the empty-artifact run violated, on the backend available here.
A review must not change the workspace, and must still be able to write its
artifact ATOMICALLY - temp file beside the target, then rename - which is
what needs a writable parent DIRECTORY rather than a writable file. Both
halves are asserted through the real session, with the real boundary
applied, and the model scripted to attempt each one.
Scope: this is the host-unsafe boundary, which its own docstring calls
best-effort because a session that can chmod can undo it. The kernel-enforced
version is bubblewrap's --ro-bind, which needs namespaces this machine cannot
create; that half stays with the real-sandbox canary in CI.
"""
artifacts = tmp_path / "artifacts"
artifacts.mkdir()
target = artifacts / REVIEW_OUTPUT
protected = clone / "source.ts"
before = protected.read_text()
write_artifact = {"name": "Write", "input": {"file_path": str(target), "content": REVIEW_JSON}}
tamper = {"name": "Write", "input": {"file_path": str(protected), "content": "tampered"}}
# No writable= entry: the boundary only governs paths INSIDE the workspace
# (it refuses one that escapes), and the artifact directory deliberately
# lives outside it - that relocation is the fix for the empty-artifact run.
with host_workspace_write_boundary(clone):
with MockProvider(default=Reply(text="writing", tools=[write_artifact, tamper])) as provider:
record = _session(clone, provider)
assert record["ok"] is True, record.get("error_detail")
# The artifact landed, written the way the agent's Write tool does it.
verdict, _findings = parse_review_output(target)
assert verdict == "approve"
assert not list(artifacts.glob("*.tmp.*")), "the rename landed rather than a copy"
# The workspace did not move.
assert protected.read_text() == before, "the read-only workspace was modified"

View file

@ -24,6 +24,7 @@ from workflow_bench.model_gateway import (
)
from workflow_bench.provider_usage import (
ANTHROPIC,
LITELLM_NORMALIZED,
OPENAI_RESPONSES,
USAGE_ENV_VARS,
normalize_usage,
@ -49,11 +50,16 @@ def _openai_response(usage: dict) -> SimpleNamespace:
)
# The shape a callback actually receives: LiteLLM normalises usage into its own
# Chat-Completions-style object before any logger sees it, so an OpenAI reply
# arrives as prompt_tokens / prompt_tokens_details. Confirmed against a real
# proxy in tests/test_mock_provider.py; a fixture in the wire shape would test
# an object this code path never gets.
NATIVE = {
"input_tokens": 48_000,
"input_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000},
"output_tokens": 900,
"output_tokens_details": {"reasoning_tokens": 640},
"prompt_tokens": 48_000,
"prompt_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000},
"completion_tokens": 900,
"completion_tokens_details": {"reasoning_tokens": 640},
}
@ -79,9 +85,9 @@ def test_native_openai_usage_survives_the_anthropic_translation(logged) -> None:
event = logged(NATIVE)
native = event["native_usage"]
# Verbatim: the fields an Anthropic-shaped response cannot carry.
assert native["input_tokens_details"]["cached_tokens"] == 44_000
assert native["input_tokens_details"]["cache_write_tokens"] == 1_000
assert native["output_tokens_details"]["reasoning_tokens"] == 640
assert native["prompt_tokens_details"]["cached_tokens"] == 44_000
assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000
assert native["completion_tokens_details"]["reasoning_tokens"] == 640
assert event["response_id"] == "resp_68f2c1"
@ -100,7 +106,10 @@ def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None:
event = logged(NATIVE)
# The provider the LOG recorded, not one the test supplies - passing
# OPENAI_RESPONSES by hand here is what hid the adapter-key mismatch.
assert event["provider"] == OPENAI_RESPONSES
# LITELLM_NORMALIZED, not OPENAI_RESPONSES: a proxy callback never sees the
# upstream body. Measured against a real gateway - the Responses adapter
# found none of its keys there and reported every field unknown.
assert event["provider"] == LITELLM_NORMALIZED
assert event["provider_label"] == "openai"
usage = normalize_usage(event["provider"], event["native_usage"])
assert usage.total_input_tokens == 48_000
@ -112,7 +121,7 @@ def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None:
def test_usage_without_details_normalizes_to_unknown_rather_than_zero(logged) -> None:
"""The mutation the accounting must not survive: dropped details, silent zeros."""
stripped = {k: v for k, v in NATIVE.items() if k != "input_tokens_details"}
stripped = {k: v for k, v in NATIVE.items() if k != "prompt_tokens_details"}
event = logged(stripped)
usage = normalize_usage(event["provider"], event["native_usage"])
assert usage.cache_read_input_tokens is None
@ -238,10 +247,16 @@ def test_an_unresolvable_provider_is_refused_rather_than_guessed() -> None:
from workflow_bench.provider_usage import canonical_provider
assert canonical_provider("openai", "responses") == OPENAI_RESPONSES
assert canonical_provider("openai", "completion") is None
assert canonical_provider("openai", None) is None
# Every openai call reaching this callback has already been normalised by
# LiteLLM, whatever endpoint it used - the observed call_type for a Claude
# Code request through the gateway is "anthropic_messages". The adapter has
# to match the object in hand, not the protocol on the wire.
assert canonical_provider("openai", "responses") == LITELLM_NORMALIZED
assert canonical_provider("openai", "anthropic_messages") == LITELLM_NORMALIZED
assert canonical_provider("anthropic", "completion") == ANTHROPIC
# An unrecognised provider is still refused rather than guessed.
assert canonical_provider("some-new-provider", "responses") is None
assert canonical_provider(None, None) is None
def test_request_identity_cannot_come_from_the_proxy_environment() -> None:

View file

@ -42,8 +42,8 @@ def canonical_provider(label, call_type): # noqa: ANN001, ANN201
above for why this is a copy rather than an import.
"""
if label == "openai" and call_type and "responses" in call_type:
return "openai-responses"
if label == "openai":
return "litellm-normalized"
if label == "anthropic":
return "anthropic"
return None

View file

@ -50,6 +50,14 @@ USAGE_ENV_VARS = (USAGE_LOG_ENV_VAR, SWEEP_ID_ENV_VAR)
ANTHROPIC = "anthropic"
OPENAI_RESPONSES = "openai-responses"
# What the gateway's callback actually receives. LiteLLM does not hand a logger
# the upstream body: it normalises usage into its own Chat-Completions-shaped
# object first, so an OpenAI Responses reply arrives as prompt_tokens /
# prompt_tokens_details even though the wire carried input_tokens /
# input_tokens_details. Measured against a real proxy, not assumed - the
# Responses adapter below found none of its keys and reported every field
# unknown. The arithmetic is still OpenAI's (the whole, with subsets).
LITELLM_NORMALIZED = "litellm-normalized"
class UsageSemanticsError(ValueError):
@ -131,6 +139,35 @@ def _normalize_openai_responses(usage: Mapping[str, Any]) -> NormalizedUsage:
)
def _normalize_litellm(usage: Mapping[str, Any]) -> NormalizedUsage:
"""prompt_tokens is the WHOLE; the details are subsets of it."""
total = _int_or_none(usage, "prompt_tokens")
details = usage.get("prompt_tokens_details")
cache_read = _int_or_none(details, "cached_tokens")
cache_write = _int_or_none(details, "cache_write_tokens")
if cache_write is None:
cache_write = _int_or_none(details, "cache_creation_tokens")
output_details = usage.get("completion_tokens_details")
ordinary: int | None = None
if total is not None and cache_read is not None and cache_write is not None:
ordinary = total - cache_read - cache_write
if ordinary < 0:
raise UsageSemanticsError(
f"LiteLLM cached ({cache_read}) + cache_write ({cache_write}) "
f"exceed prompt_tokens ({total})"
)
return NormalizedUsage(
ordinary_input_tokens=ordinary,
cache_read_input_tokens=cache_read,
cache_write_input_tokens=cache_write,
total_input_tokens=total,
output_tokens=_int_or_none(usage, "completion_tokens"),
reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"),
)
def _normalize_anthropic(usage: Mapping[str, Any]) -> NormalizedUsage:
"""input_tokens is the uncached REMAINDER; the cache fields add to it."""
@ -162,8 +199,13 @@ def canonical_provider(label: str | None, call_type: str | None) -> str | None:
the whole point of keeping the native object authoritative.
"""
if label == "openai" and call_type and "responses" in call_type:
return OPENAI_RESPONSES
if label == "openai":
# Anything reaching a proxy callback has already been normalised by
# LiteLLM, whichever endpoint the caller used - the observed call_type
# for a Claude Code request through this gateway is "anthropic_messages",
# not a Responses one. The adapter has to match the object in hand, not
# the protocol on the wire.
return LITELLM_NORMALIZED
if label in _ADAPTERS:
return label
return None
@ -172,6 +214,7 @@ def canonical_provider(label: str | None, call_type: str | None) -> str | None:
_ADAPTERS = {
ANTHROPIC: _normalize_anthropic,
OPENAI_RESPONSES: _normalize_openai_responses,
LITELLM_NORMALIZED: _normalize_litellm,
}