test(eval): run a real session against the scripted provider

The mock only proves something once the harness runs against it. This adds the
stand-in CLI and the first integration tests that use it, so a session goes
through the real code with only the model faked.

tests/fixtures/fake_claude.py does what the CLI does at the two boundaries the
harness depends on: it calls ANTHROPIC_BASE_URL for a turn, EXECUTES the tool
blocks that come back, and prints the stream-json sequence the parent parses.
Everything between - the session runner, the event-stream parse, the usage
extraction, the artifact capture, the scorer - stays real.

Four tests, chosen for the layers that have actually broken here: the usage a
provider reported survives to the row, a scripted Write produces an artifact
parse_review_output accepts, the prompt the harness meant to send is what
arrived, and an upstream 529 lands as a failed session rather than a usable
measurement.

Writing the stand-in found two things worth keeping. The prompt arrives on
STDIN under "-p --input-format text"; scanning argv for a non-flag token picks
up a flag's value instead, and the prompt-fidelity test is what caught it. And
three of these tests had been holding a sandbox they never applied, since no
command_prefix is passed - that implied coverage which was not there, so the
sandbox is gone from them and stays only in the artifact test, which needs its
review directory.

What these do NOT cover, checked rather than assumed: making the stand-in write
in place instead of atomically still passes. On the host-unsafe backend there is
no read-only mount to refuse it, so the atomic-write requirement remains a
bubblewrap mount property that only the real-sandbox canary can prove. Dropping
cache_read from the recorded usage does fail, so that half is genuinely pinned.

672 eval tests pass, 17 skipped; the two test_model_gateway.py failures are the
environmental ones.
This commit is contained in:
Gergo Magyar 2026-09-08 17:49:03 +00:00
parent 6fd443a941
commit 73ad083b69
2 changed files with 211 additions and 0 deletions

104
eval/tests/fixtures/fake_claude.py vendored Normal file
View file

@ -0,0 +1,104 @@
#!/usr/bin/env python3
"""A stand-in for the Claude Code CLI: real HTTP, real tool execution, real stream-json.
Not a mock of the harness's own code. It does what the CLI does at the two
boundaries the harness depends on - it calls ANTHROPIC_BASE_URL for a turn, it
EXECUTES the tool blocks that come back, and it prints the stream-json event
sequence the parent parses. Everything between those boundaries (the sandbox,
the artifact capture, the scoring, the row) stays real, which is the whole
point: those are the layers that shipped bugs no unit test could see.
Reads the prompt from argv or stdin, like the real CLI under --print.
"""
from __future__ import annotations
import json
import os
import pathlib
import sys
import urllib.request
def _turn(base_url: str, prompt: str) -> dict:
request = urllib.request.Request(
base_url.rstrip("/") + "/v1/messages",
data=json.dumps({"model": os.environ.get("ANTHROPIC_MODEL", "mock"), "max_tokens": 1024,
"messages": [{"role": "user", "content": prompt}]}).encode(),
headers={"Content-Type": "application/json",
"x-api-key": os.environ.get("ANTHROPIC_API_KEY", ""),
"anthropic-version": "2023-06-01"},
)
with urllib.request.urlopen(request, timeout=30) as response:
return json.load(response)
def _run_tool(name: str, params: dict) -> str:
"""Execute for real. A Write here is what produces the review artifact."""
if name == "Write":
target = pathlib.Path(params["file_path"])
target.parent.mkdir(parents=True, exist_ok=True)
# Atomic, exactly as the real Write tool does it: temp file beside the
# target, then rename. This is the operation the read-only workspace
# boundary has to permit for the artifact directory and refuse for the
# workspace, so a stand-in that wrote in place would prove nothing.
staging = target.with_name(target.name + ".tmp.fake")
staging.write_text(params.get("content", ""))
os.replace(staging, target)
return f"wrote {target}"
if name == "Bash":
return "(bash suppressed in the stand-in)"
return f"(unhandled tool {name})"
def main() -> int:
# stdin, because that is where the real CLI takes it under
# "-p --input-format text": the parent pipes prompt bytes in. Scanning argv
# for a non-flag token picks up a flag's VALUE instead ("text"), which is
# exactly what the prompt-fidelity test caught.
prompt = sys.stdin.read()
base_url = os.environ.get("ANTHROPIC_BASE_URL")
if not base_url:
print(json.dumps({"type": "result", "subtype": "error", "is_error": True,
"session_id": "fake-session", "num_turns": 0}), flush=True)
return 1
emit = lambda event: print(json.dumps(event), flush=True) # noqa: E731
emit({"type": "system", "subtype": "init", "session_id": "fake-session"})
message = _turn(base_url, prompt)
blocks = message.get("content", [])
emit({"type": "assistant", "message": {"role": "assistant", "content": blocks}})
tool_results = []
for block in blocks:
if block.get("type") == "tool_use":
output = _run_tool(block["name"], block.get("input", {}))
tool_results.append({"type": "tool_result", "tool_use_id": block["id"], "content": output})
if tool_results:
emit({"type": "user", "message": {"role": "user", "content": tool_results}})
usage = message.get("usage", {})
emit({
"type": "result",
"subtype": "success",
"is_error": False,
"session_id": "fake-session",
"num_turns": 1,
"duration_ms": 1200,
# A measured zero is not the same as unmeasured; the parent rejects a
# collapsed cost, so report a real one.
"total_cost_usd": 0.42,
"usage": {
"input_tokens": usage.get("input_tokens", 0),
"output_tokens": usage.get("output_tokens", 0),
"cache_read_input_tokens": usage.get("cache_read_input_tokens", 0),
"cache_creation_input_tokens": usage.get("cache_creation_input_tokens", 0),
},
})
return 0
if __name__ == "__main__":
raise SystemExit(main())

View file

@ -0,0 +1,107 @@
"""A session end to end with only the model faked.
The layers between the CLI and the row are where this harness has actually
shipped bugs - the artifact that could not be written, the usage that was never
recorded, the evidence that was scored from the wrong directory. Every one of
them sat below the level its tests exercised. These run the real session path
against a scripted provider, so the only thing not real is what the model says.
"""
from __future__ import annotations
import sys
from pathlib import Path
import pytest
from workflow_bench.mock_provider import MockProvider, Reply
from workflow_bench.proposer_sandbox import prepare_sandbox, prepare_review_workspace
from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output
from workflow_bench.runner_sessions import run_claude
FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py"
REVIEW_JSON = '{"schema_version": 1, "verdict": "approve", "findings": []}'
def _session(clone: Path, provider: MockProvider, **overrides):
return run_claude(
"review the change",
clone,
claude_bin=str(FAKE_CLI),
timeout=60,
env={
"ANTHROPIC_BASE_URL": provider.base_url,
"ANTHROPIC_API_KEY": "offline",
"PATH": "/usr/bin:/bin",
},
**overrides,
)
@pytest.fixture
def clone(tmp_path: Path) -> Path:
workspace = tmp_path / "clone"
workspace.mkdir()
(workspace / "source.ts").write_text("export const answer = 42;\n")
return workspace
def test_a_session_records_the_usage_the_provider_reported(clone: Path) -> None:
"""Token counts must survive the CLI boundary, not be invented after it."""
reply = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000)
with MockProvider(default=reply) as provider:
record = _session(clone, provider)
assert record["ok"] is True, record.get("error_detail")
assert record["input_tokens"] == 2_000
assert record["cache_read_input_tokens"] == 7_000
assert record["cache_creation_input_tokens"] == 1_000
assert record["output_tokens"] == 300
# A measured zero would be indistinguishable from an unmeasured one.
assert record["cost_usd"] == 0.42
assert record["num_turns"] == 1
def test_a_scripted_write_produces_a_review_artifact_the_scorer_accepts(clone: Path, tmp_path: Path) -> None:
"""The full artifact path: model asks, CLI writes atomically, scorer reads.
This is the operation that shipped empty for a whole run. Nothing here
fakes the write, the directory, or the parse - only the decision to write.
"""
with prepare_sandbox(
clone=clone, claude_bin=Path(sys.executable), backend="host-unsafe", preflight=False
) as sandbox:
artifact = prepare_review_workspace(sandbox, REVIEW_OUTPUT)
write = {"name": "Write", "input": {"file_path": str(artifact), "content": REVIEW_JSON}}
with MockProvider(default=Reply(text="reviewing", tools=[write])) as provider:
record = _session(clone, provider)
assert record["ok"] is True, record.get("error_detail")
# Read inside the scope: prepare_sandbox removes the private root on exit.
verdict, findings = parse_review_output(artifact)
assert verdict == "approve"
assert findings == ()
def test_the_provider_saw_the_prompt_the_harness_meant_to_send(clone: Path) -> None:
"""A run that measures the wrong prompt measures nothing."""
with MockProvider() as provider:
_session(clone, provider)
assert provider.requests, "the session never reached the provider"
sent = provider.requests[0].body["messages"][0]["content"]
assert "review the change" in sent
def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone: Path) -> None:
"""An upstream 529 must not be recorded as a usable measurement."""
failing = Reply(status_code=529, error_body={"error": {"type": "overloaded_error"}})
with MockProvider(default=failing) as provider:
record = _session(clone, provider)
assert record["ok"] is False
assert record["error_kind"] is not None