diff --git a/eval/tests/fixtures/fake_claude.py b/eval/tests/fixtures/fake_claude.py new file mode 100644 index 000000000..d0143219a --- /dev/null +++ b/eval/tests/fixtures/fake_claude.py @@ -0,0 +1,104 @@ +#!/usr/bin/env python3 +"""A stand-in for the Claude Code CLI: real HTTP, real tool execution, real stream-json. + +Not a mock of the harness's own code. It does what the CLI does at the two +boundaries the harness depends on - it calls ANTHROPIC_BASE_URL for a turn, it +EXECUTES the tool blocks that come back, and it prints the stream-json event +sequence the parent parses. Everything between those boundaries (the sandbox, +the artifact capture, the scoring, the row) stays real, which is the whole +point: those are the layers that shipped bugs no unit test could see. + +Reads the prompt from argv or stdin, like the real CLI under --print. +""" + +from __future__ import annotations + +import json +import os +import pathlib +import sys +import urllib.request + + +def _turn(base_url: str, prompt: str) -> dict: + request = urllib.request.Request( + base_url.rstrip("/") + "/v1/messages", + data=json.dumps({"model": os.environ.get("ANTHROPIC_MODEL", "mock"), "max_tokens": 1024, + "messages": [{"role": "user", "content": prompt}]}).encode(), + headers={"Content-Type": "application/json", + "x-api-key": os.environ.get("ANTHROPIC_API_KEY", ""), + "anthropic-version": "2023-06-01"}, + ) + with urllib.request.urlopen(request, timeout=30) as response: + return json.load(response) + + +def _run_tool(name: str, params: dict) -> str: + """Execute for real. A Write here is what produces the review artifact.""" + + if name == "Write": + target = pathlib.Path(params["file_path"]) + target.parent.mkdir(parents=True, exist_ok=True) + # Atomic, exactly as the real Write tool does it: temp file beside the + # target, then rename. This is the operation the read-only workspace + # boundary has to permit for the artifact directory and refuse for the + # workspace, so a stand-in that wrote in place would prove nothing. + staging = target.with_name(target.name + ".tmp.fake") + staging.write_text(params.get("content", "")) + os.replace(staging, target) + return f"wrote {target}" + if name == "Bash": + return "(bash suppressed in the stand-in)" + return f"(unhandled tool {name})" + + +def main() -> int: + # stdin, because that is where the real CLI takes it under + # "-p --input-format text": the parent pipes prompt bytes in. Scanning argv + # for a non-flag token picks up a flag's VALUE instead ("text"), which is + # exactly what the prompt-fidelity test caught. + prompt = sys.stdin.read() + base_url = os.environ.get("ANTHROPIC_BASE_URL") + if not base_url: + print(json.dumps({"type": "result", "subtype": "error", "is_error": True, + "session_id": "fake-session", "num_turns": 0}), flush=True) + return 1 + + emit = lambda event: print(json.dumps(event), flush=True) # noqa: E731 + emit({"type": "system", "subtype": "init", "session_id": "fake-session"}) + + message = _turn(base_url, prompt) + blocks = message.get("content", []) + emit({"type": "assistant", "message": {"role": "assistant", "content": blocks}}) + + tool_results = [] + for block in blocks: + if block.get("type") == "tool_use": + output = _run_tool(block["name"], block.get("input", {})) + tool_results.append({"type": "tool_result", "tool_use_id": block["id"], "content": output}) + if tool_results: + emit({"type": "user", "message": {"role": "user", "content": tool_results}}) + + usage = message.get("usage", {}) + emit({ + "type": "result", + "subtype": "success", + "is_error": False, + "session_id": "fake-session", + "num_turns": 1, + "duration_ms": 1200, + # A measured zero is not the same as unmeasured; the parent rejects a + # collapsed cost, so report a real one. + "total_cost_usd": 0.42, + "usage": { + "input_tokens": usage.get("input_tokens", 0), + "output_tokens": usage.get("output_tokens", 0), + "cache_read_input_tokens": usage.get("cache_read_input_tokens", 0), + "cache_creation_input_tokens": usage.get("cache_creation_input_tokens", 0), + }, + }) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/eval/tests/test_offline_session_integration.py b/eval/tests/test_offline_session_integration.py new file mode 100644 index 000000000..d1a1912fb --- /dev/null +++ b/eval/tests/test_offline_session_integration.py @@ -0,0 +1,107 @@ +"""A session end to end with only the model faked. + +The layers between the CLI and the row are where this harness has actually +shipped bugs - the artifact that could not be written, the usage that was never +recorded, the evidence that was scored from the wrong directory. Every one of +them sat below the level its tests exercised. These run the real session path +against a scripted provider, so the only thing not real is what the model says. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +import pytest + +from workflow_bench.mock_provider import MockProvider, Reply +from workflow_bench.proposer_sandbox import prepare_sandbox, prepare_review_workspace +from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output +from workflow_bench.runner_sessions import run_claude + +FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py" +REVIEW_JSON = '{"schema_version": 1, "verdict": "approve", "findings": []}' + + +def _session(clone: Path, provider: MockProvider, **overrides): + return run_claude( + "review the change", + clone, + claude_bin=str(FAKE_CLI), + timeout=60, + env={ + "ANTHROPIC_BASE_URL": provider.base_url, + "ANTHROPIC_API_KEY": "offline", + "PATH": "/usr/bin:/bin", + }, + **overrides, + ) + + +@pytest.fixture +def clone(tmp_path: Path) -> Path: + workspace = tmp_path / "clone" + workspace.mkdir() + (workspace / "source.ts").write_text("export const answer = 42;\n") + return workspace + + +def test_a_session_records_the_usage_the_provider_reported(clone: Path) -> None: + """Token counts must survive the CLI boundary, not be invented after it.""" + + reply = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000) + with MockProvider(default=reply) as provider: + record = _session(clone, provider) + + assert record["ok"] is True, record.get("error_detail") + assert record["input_tokens"] == 2_000 + assert record["cache_read_input_tokens"] == 7_000 + assert record["cache_creation_input_tokens"] == 1_000 + assert record["output_tokens"] == 300 + # A measured zero would be indistinguishable from an unmeasured one. + assert record["cost_usd"] == 0.42 + assert record["num_turns"] == 1 + + +def test_a_scripted_write_produces_a_review_artifact_the_scorer_accepts(clone: Path, tmp_path: Path) -> None: + """The full artifact path: model asks, CLI writes atomically, scorer reads. + + This is the operation that shipped empty for a whole run. Nothing here + fakes the write, the directory, or the parse - only the decision to write. + """ + + with prepare_sandbox( + clone=clone, claude_bin=Path(sys.executable), backend="host-unsafe", preflight=False + ) as sandbox: + artifact = prepare_review_workspace(sandbox, REVIEW_OUTPUT) + write = {"name": "Write", "input": {"file_path": str(artifact), "content": REVIEW_JSON}} + with MockProvider(default=Reply(text="reviewing", tools=[write])) as provider: + record = _session(clone, provider) + assert record["ok"] is True, record.get("error_detail") + # Read inside the scope: prepare_sandbox removes the private root on exit. + verdict, findings = parse_review_output(artifact) + + assert verdict == "approve" + assert findings == () + + +def test_the_provider_saw_the_prompt_the_harness_meant_to_send(clone: Path) -> None: + """A run that measures the wrong prompt measures nothing.""" + + with MockProvider() as provider: + _session(clone, provider) + + assert provider.requests, "the session never reached the provider" + sent = provider.requests[0].body["messages"][0]["content"] + assert "review the change" in sent + + +def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone: Path) -> None: + """An upstream 529 must not be recorded as a usable measurement.""" + + failing = Reply(status_code=529, error_body={"error": {"type": "overloaded_error"}}) + with MockProvider(default=failing) as provider: + record = _session(clone, provider) + + assert record["ok"] is False + assert record["error_kind"] is not None