mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-06 02:49:56 +00:00
fix(eval): the usage adapter read a shape the callback never receives
Running the gateway against the scripted provider proved the accounting merged in #3220 does not work, and the same run showed why nothing had caught it. LiteLLM does not hand a logger the upstream body. It normalises usage into its own Chat-Completions-shaped object first, so an OpenAI Responses reply reaches the callback as prompt_tokens / prompt_tokens_details.cached_tokens - never the input_tokens / input_tokens_details the shipped adapter reads. Every field came back unknown. The observed call_type is "anthropic_messages" as well, because Claude Code calls the Anthropic-shaped endpoint, so canonical_provider returned None and normalize_usage would have refused outright. Both were assumptions about a boundary I had only read about. The unit tests agreed with them because their fixture was written in the same wrong shape, so producer and consumer were consistent and both wrong - the exact failure the producer/consumer round trip exists to catch, one layer further out. Adds a LITELLM_NORMALIZED adapter for the object that actually arrives. The arithmetic is still OpenAI's - prompt_tokens is the whole, the details are subsets - so ordinary input is recovered by subtraction. The Responses adapter stays for a raw upstream body, which the mock still serves and tests directly. An unrecognised provider is still refused rather than guessed. The fixtures now carry the measured shape, and the end-to-end test asserts it through a real proxy: 48k prompt tokens with 44k cached is read back as 3k ordinary rather than as silence. 676 eval tests pass, 16 skipped, none failing.
This commit is contained in:
parent
73ad083b69
commit
355175a7c6
6 changed files with 147 additions and 24 deletions
8
eval/tests/fixtures/fake_claude.py
vendored
8
eval/tests/fixtures/fake_claude.py
vendored
|
|
@ -74,7 +74,13 @@ def main() -> int:
|
|||
tool_results = []
|
||||
for block in blocks:
|
||||
if block.get("type") == "tool_use":
|
||||
output = _run_tool(block["name"], block.get("input", {}))
|
||||
# A refused write is a tool ERROR the session reports and carries
|
||||
# on from, not a crash. Letting it kill the process would lose the
|
||||
# result event and misreport a working boundary as a broken run.
|
||||
try:
|
||||
output = _run_tool(block["name"], block.get("input", {}))
|
||||
except OSError as exc:
|
||||
output = f"error: {type(exc).__name__}: {exc}"
|
||||
tool_results.append({"type": "tool_result", "tool_use_id": block["id"], "content": output})
|
||||
if tool_results:
|
||||
emit({"type": "user", "message": {"role": "user", "content": tool_results}})
|
||||
|
|
|
|||
|
|
@ -6,7 +6,12 @@ import json
|
|||
import urllib.request
|
||||
|
||||
from workflow_bench.mock_provider import MockProvider, Reply
|
||||
from workflow_bench.provider_usage import ANTHROPIC, OPENAI_RESPONSES, normalize_usage
|
||||
from workflow_bench.provider_usage import (
|
||||
ANTHROPIC,
|
||||
LITELLM_NORMALIZED,
|
||||
OPENAI_RESPONSES,
|
||||
normalize_usage,
|
||||
)
|
||||
|
||||
|
||||
def _post(url: str, payload: dict) -> tuple[int, bytes]:
|
||||
|
|
@ -169,10 +174,20 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke
|
|||
assert usage_log.exists(), "the callback never wrote - the env did not reach the proxy"
|
||||
events = [json.loads(line) for line in usage_log.read_text().splitlines()]
|
||||
assert events, "the proxy started but recorded nothing"
|
||||
native = events[-1]["native_usage"]
|
||||
# The provider's own arithmetic survived the Anthropic-shaped translation.
|
||||
assert native["input_tokens_details"]["cached_tokens"] == 7_000
|
||||
assert native["input_tokens_details"]["cache_write_tokens"] == 1_000
|
||||
usage = normalize_usage(events[-1]["provider"], native)
|
||||
event = events[-1]
|
||||
native = event["native_usage"]
|
||||
# LiteLLM hands a callback its OWN normalised object, not the upstream body:
|
||||
# an OpenAI Responses reply arrives as prompt_tokens / prompt_tokens_details.
|
||||
# Asserting the wire shape here is what proved the shipped adapter read keys
|
||||
# that are never present.
|
||||
assert native["prompt_tokens_details"]["cached_tokens"] == 7_000
|
||||
assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000
|
||||
assert event["provider"] == LITELLM_NORMALIZED
|
||||
assert event["call_type"] == "anthropic_messages", "the observed call type, not a Responses one"
|
||||
|
||||
usage = normalize_usage(event["provider"], native)
|
||||
assert usage.total_input_tokens == 10_000
|
||||
assert usage.cache_read_input_tokens == 7_000
|
||||
assert usage.cache_write_input_tokens == 1_000
|
||||
assert usage.ordinary_input_tokens == 2_000
|
||||
assert usage.complete, "a run that cannot interpret its own usage measured nothing"
|
||||
|
|
|
|||
|
|
@ -15,7 +15,11 @@ from pathlib import Path
|
|||
import pytest
|
||||
|
||||
from workflow_bench.mock_provider import MockProvider, Reply
|
||||
from workflow_bench.proposer_sandbox import prepare_sandbox, prepare_review_workspace
|
||||
from workflow_bench.proposer_sandbox import (
|
||||
host_workspace_write_boundary,
|
||||
prepare_review_workspace,
|
||||
prepare_sandbox,
|
||||
)
|
||||
from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output
|
||||
from workflow_bench.runner_sessions import run_claude
|
||||
|
||||
|
|
@ -105,3 +109,43 @@ def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone
|
|||
|
||||
assert record["ok"] is False
|
||||
assert record["error_kind"] is not None
|
||||
|
||||
|
||||
def test_the_write_boundary_refuses_the_workspace_and_permits_the_artifact(clone: Path, tmp_path: Path) -> None:
|
||||
"""The contract the empty-artifact run violated, on the backend available here.
|
||||
|
||||
A review must not change the workspace, and must still be able to write its
|
||||
artifact ATOMICALLY - temp file beside the target, then rename - which is
|
||||
what needs a writable parent DIRECTORY rather than a writable file. Both
|
||||
halves are asserted through the real session, with the real boundary
|
||||
applied, and the model scripted to attempt each one.
|
||||
|
||||
Scope: this is the host-unsafe boundary, which its own docstring calls
|
||||
best-effort because a session that can chmod can undo it. The kernel-enforced
|
||||
version is bubblewrap's --ro-bind, which needs namespaces this machine cannot
|
||||
create; that half stays with the real-sandbox canary in CI.
|
||||
"""
|
||||
|
||||
artifacts = tmp_path / "artifacts"
|
||||
artifacts.mkdir()
|
||||
target = artifacts / REVIEW_OUTPUT
|
||||
protected = clone / "source.ts"
|
||||
before = protected.read_text()
|
||||
|
||||
write_artifact = {"name": "Write", "input": {"file_path": str(target), "content": REVIEW_JSON}}
|
||||
tamper = {"name": "Write", "input": {"file_path": str(protected), "content": "tampered"}}
|
||||
|
||||
# No writable= entry: the boundary only governs paths INSIDE the workspace
|
||||
# (it refuses one that escapes), and the artifact directory deliberately
|
||||
# lives outside it - that relocation is the fix for the empty-artifact run.
|
||||
with host_workspace_write_boundary(clone):
|
||||
with MockProvider(default=Reply(text="writing", tools=[write_artifact, tamper])) as provider:
|
||||
record = _session(clone, provider)
|
||||
|
||||
assert record["ok"] is True, record.get("error_detail")
|
||||
# The artifact landed, written the way the agent's Write tool does it.
|
||||
verdict, _findings = parse_review_output(target)
|
||||
assert verdict == "approve"
|
||||
assert not list(artifacts.glob("*.tmp.*")), "the rename landed rather than a copy"
|
||||
# The workspace did not move.
|
||||
assert protected.read_text() == before, "the read-only workspace was modified"
|
||||
|
|
|
|||
|
|
@ -24,6 +24,7 @@ from workflow_bench.model_gateway import (
|
|||
)
|
||||
from workflow_bench.provider_usage import (
|
||||
ANTHROPIC,
|
||||
LITELLM_NORMALIZED,
|
||||
OPENAI_RESPONSES,
|
||||
USAGE_ENV_VARS,
|
||||
normalize_usage,
|
||||
|
|
@ -49,11 +50,16 @@ def _openai_response(usage: dict) -> SimpleNamespace:
|
|||
)
|
||||
|
||||
|
||||
# The shape a callback actually receives: LiteLLM normalises usage into its own
|
||||
# Chat-Completions-style object before any logger sees it, so an OpenAI reply
|
||||
# arrives as prompt_tokens / prompt_tokens_details. Confirmed against a real
|
||||
# proxy in tests/test_mock_provider.py; a fixture in the wire shape would test
|
||||
# an object this code path never gets.
|
||||
NATIVE = {
|
||||
"input_tokens": 48_000,
|
||||
"input_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000},
|
||||
"output_tokens": 900,
|
||||
"output_tokens_details": {"reasoning_tokens": 640},
|
||||
"prompt_tokens": 48_000,
|
||||
"prompt_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000},
|
||||
"completion_tokens": 900,
|
||||
"completion_tokens_details": {"reasoning_tokens": 640},
|
||||
}
|
||||
|
||||
|
||||
|
|
@ -79,9 +85,9 @@ def test_native_openai_usage_survives_the_anthropic_translation(logged) -> None:
|
|||
event = logged(NATIVE)
|
||||
native = event["native_usage"]
|
||||
# Verbatim: the fields an Anthropic-shaped response cannot carry.
|
||||
assert native["input_tokens_details"]["cached_tokens"] == 44_000
|
||||
assert native["input_tokens_details"]["cache_write_tokens"] == 1_000
|
||||
assert native["output_tokens_details"]["reasoning_tokens"] == 640
|
||||
assert native["prompt_tokens_details"]["cached_tokens"] == 44_000
|
||||
assert native["prompt_tokens_details"]["cache_write_tokens"] == 1_000
|
||||
assert native["completion_tokens_details"]["reasoning_tokens"] == 640
|
||||
assert event["response_id"] == "resp_68f2c1"
|
||||
|
||||
|
||||
|
|
@ -100,7 +106,10 @@ def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None:
|
|||
event = logged(NATIVE)
|
||||
# The provider the LOG recorded, not one the test supplies - passing
|
||||
# OPENAI_RESPONSES by hand here is what hid the adapter-key mismatch.
|
||||
assert event["provider"] == OPENAI_RESPONSES
|
||||
# LITELLM_NORMALIZED, not OPENAI_RESPONSES: a proxy callback never sees the
|
||||
# upstream body. Measured against a real gateway - the Responses adapter
|
||||
# found none of its keys there and reported every field unknown.
|
||||
assert event["provider"] == LITELLM_NORMALIZED
|
||||
assert event["provider_label"] == "openai"
|
||||
usage = normalize_usage(event["provider"], event["native_usage"])
|
||||
assert usage.total_input_tokens == 48_000
|
||||
|
|
@ -112,7 +121,7 @@ def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None:
|
|||
def test_usage_without_details_normalizes_to_unknown_rather_than_zero(logged) -> None:
|
||||
"""The mutation the accounting must not survive: dropped details, silent zeros."""
|
||||
|
||||
stripped = {k: v for k, v in NATIVE.items() if k != "input_tokens_details"}
|
||||
stripped = {k: v for k, v in NATIVE.items() if k != "prompt_tokens_details"}
|
||||
event = logged(stripped)
|
||||
usage = normalize_usage(event["provider"], event["native_usage"])
|
||||
assert usage.cache_read_input_tokens is None
|
||||
|
|
@ -238,10 +247,16 @@ def test_an_unresolvable_provider_is_refused_rather_than_guessed() -> None:
|
|||
|
||||
from workflow_bench.provider_usage import canonical_provider
|
||||
|
||||
assert canonical_provider("openai", "responses") == OPENAI_RESPONSES
|
||||
assert canonical_provider("openai", "completion") is None
|
||||
assert canonical_provider("openai", None) is None
|
||||
# Every openai call reaching this callback has already been normalised by
|
||||
# LiteLLM, whatever endpoint it used - the observed call_type for a Claude
|
||||
# Code request through the gateway is "anthropic_messages". The adapter has
|
||||
# to match the object in hand, not the protocol on the wire.
|
||||
assert canonical_provider("openai", "responses") == LITELLM_NORMALIZED
|
||||
assert canonical_provider("openai", "anthropic_messages") == LITELLM_NORMALIZED
|
||||
assert canonical_provider("anthropic", "completion") == ANTHROPIC
|
||||
# An unrecognised provider is still refused rather than guessed.
|
||||
assert canonical_provider("some-new-provider", "responses") is None
|
||||
assert canonical_provider(None, None) is None
|
||||
|
||||
|
||||
def test_request_identity_cannot_come_from_the_proxy_environment() -> None:
|
||||
|
|
|
|||
|
|
@ -42,8 +42,8 @@ def canonical_provider(label, call_type): # noqa: ANN001, ANN201
|
|||
above for why this is a copy rather than an import.
|
||||
"""
|
||||
|
||||
if label == "openai" and call_type and "responses" in call_type:
|
||||
return "openai-responses"
|
||||
if label == "openai":
|
||||
return "litellm-normalized"
|
||||
if label == "anthropic":
|
||||
return "anthropic"
|
||||
return None
|
||||
|
|
|
|||
|
|
@ -50,6 +50,14 @@ USAGE_ENV_VARS = (USAGE_LOG_ENV_VAR, SWEEP_ID_ENV_VAR)
|
|||
|
||||
ANTHROPIC = "anthropic"
|
||||
OPENAI_RESPONSES = "openai-responses"
|
||||
# What the gateway's callback actually receives. LiteLLM does not hand a logger
|
||||
# the upstream body: it normalises usage into its own Chat-Completions-shaped
|
||||
# object first, so an OpenAI Responses reply arrives as prompt_tokens /
|
||||
# prompt_tokens_details even though the wire carried input_tokens /
|
||||
# input_tokens_details. Measured against a real proxy, not assumed - the
|
||||
# Responses adapter below found none of its keys and reported every field
|
||||
# unknown. The arithmetic is still OpenAI's (the whole, with subsets).
|
||||
LITELLM_NORMALIZED = "litellm-normalized"
|
||||
|
||||
|
||||
class UsageSemanticsError(ValueError):
|
||||
|
|
@ -131,6 +139,35 @@ def _normalize_openai_responses(usage: Mapping[str, Any]) -> NormalizedUsage:
|
|||
)
|
||||
|
||||
|
||||
def _normalize_litellm(usage: Mapping[str, Any]) -> NormalizedUsage:
|
||||
"""prompt_tokens is the WHOLE; the details are subsets of it."""
|
||||
|
||||
total = _int_or_none(usage, "prompt_tokens")
|
||||
details = usage.get("prompt_tokens_details")
|
||||
cache_read = _int_or_none(details, "cached_tokens")
|
||||
cache_write = _int_or_none(details, "cache_write_tokens")
|
||||
if cache_write is None:
|
||||
cache_write = _int_or_none(details, "cache_creation_tokens")
|
||||
output_details = usage.get("completion_tokens_details")
|
||||
|
||||
ordinary: int | None = None
|
||||
if total is not None and cache_read is not None and cache_write is not None:
|
||||
ordinary = total - cache_read - cache_write
|
||||
if ordinary < 0:
|
||||
raise UsageSemanticsError(
|
||||
f"LiteLLM cached ({cache_read}) + cache_write ({cache_write}) "
|
||||
f"exceed prompt_tokens ({total})"
|
||||
)
|
||||
return NormalizedUsage(
|
||||
ordinary_input_tokens=ordinary,
|
||||
cache_read_input_tokens=cache_read,
|
||||
cache_write_input_tokens=cache_write,
|
||||
total_input_tokens=total,
|
||||
output_tokens=_int_or_none(usage, "completion_tokens"),
|
||||
reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"),
|
||||
)
|
||||
|
||||
|
||||
def _normalize_anthropic(usage: Mapping[str, Any]) -> NormalizedUsage:
|
||||
"""input_tokens is the uncached REMAINDER; the cache fields add to it."""
|
||||
|
||||
|
|
@ -162,8 +199,13 @@ def canonical_provider(label: str | None, call_type: str | None) -> str | None:
|
|||
the whole point of keeping the native object authoritative.
|
||||
"""
|
||||
|
||||
if label == "openai" and call_type and "responses" in call_type:
|
||||
return OPENAI_RESPONSES
|
||||
if label == "openai":
|
||||
# Anything reaching a proxy callback has already been normalised by
|
||||
# LiteLLM, whichever endpoint the caller used - the observed call_type
|
||||
# for a Claude Code request through this gateway is "anthropic_messages",
|
||||
# not a Responses one. The adapter has to match the object in hand, not
|
||||
# the protocol on the wire.
|
||||
return LITELLM_NORMALIZED
|
||||
if label in _ADAPTERS:
|
||||
return label
|
||||
return None
|
||||
|
|
@ -172,6 +214,7 @@ def canonical_provider(label: str | None, call_type: str | None) -> str | None:
|
|||
_ADAPTERS = {
|
||||
ANTHROPIC: _normalize_anthropic,
|
||||
OPENAI_RESPONSES: _normalize_openai_responses,
|
||||
LITELLM_NORMALIZED: _normalize_litellm,
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue