mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-05 02:43:32 +00:00
test(eval): run a real session against the scripted provider
The mock only proves something once the harness runs against it. This adds the stand-in CLI and the first integration tests that use it, so a session goes through the real code with only the model faked. tests/fixtures/fake_claude.py does what the CLI does at the two boundaries the harness depends on: it calls ANTHROPIC_BASE_URL for a turn, EXECUTES the tool blocks that come back, and prints the stream-json sequence the parent parses. Everything between - the session runner, the event-stream parse, the usage extraction, the artifact capture, the scorer - stays real. Four tests, chosen for the layers that have actually broken here: the usage a provider reported survives to the row, a scripted Write produces an artifact parse_review_output accepts, the prompt the harness meant to send is what arrived, and an upstream 529 lands as a failed session rather than a usable measurement. Writing the stand-in found two things worth keeping. The prompt arrives on STDIN under "-p --input-format text"; scanning argv for a non-flag token picks up a flag's value instead, and the prompt-fidelity test is what caught it. And three of these tests had been holding a sandbox they never applied, since no command_prefix is passed - that implied coverage which was not there, so the sandbox is gone from them and stays only in the artifact test, which needs its review directory. What these do NOT cover, checked rather than assumed: making the stand-in write in place instead of atomically still passes. On the host-unsafe backend there is no read-only mount to refuse it, so the atomic-write requirement remains a bubblewrap mount property that only the real-sandbox canary can prove. Dropping cache_read from the recorded usage does fail, so that half is genuinely pinned. 672 eval tests pass, 17 skipped; the two test_model_gateway.py failures are the environmental ones.
This commit is contained in:
parent
6fd443a941
commit
73ad083b69
2 changed files with 211 additions and 0 deletions
104
eval/tests/fixtures/fake_claude.py
vendored
Normal file
104
eval/tests/fixtures/fake_claude.py
vendored
Normal file
|
|
@ -0,0 +1,104 @@
|
|||
#!/usr/bin/env python3
|
||||
"""A stand-in for the Claude Code CLI: real HTTP, real tool execution, real stream-json.
|
||||
|
||||
Not a mock of the harness's own code. It does what the CLI does at the two
|
||||
boundaries the harness depends on - it calls ANTHROPIC_BASE_URL for a turn, it
|
||||
EXECUTES the tool blocks that come back, and it prints the stream-json event
|
||||
sequence the parent parses. Everything between those boundaries (the sandbox,
|
||||
the artifact capture, the scoring, the row) stays real, which is the whole
|
||||
point: those are the layers that shipped bugs no unit test could see.
|
||||
|
||||
Reads the prompt from argv or stdin, like the real CLI under --print.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import sys
|
||||
import urllib.request
|
||||
|
||||
|
||||
def _turn(base_url: str, prompt: str) -> dict:
|
||||
request = urllib.request.Request(
|
||||
base_url.rstrip("/") + "/v1/messages",
|
||||
data=json.dumps({"model": os.environ.get("ANTHROPIC_MODEL", "mock"), "max_tokens": 1024,
|
||||
"messages": [{"role": "user", "content": prompt}]}).encode(),
|
||||
headers={"Content-Type": "application/json",
|
||||
"x-api-key": os.environ.get("ANTHROPIC_API_KEY", ""),
|
||||
"anthropic-version": "2023-06-01"},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=30) as response:
|
||||
return json.load(response)
|
||||
|
||||
|
||||
def _run_tool(name: str, params: dict) -> str:
|
||||
"""Execute for real. A Write here is what produces the review artifact."""
|
||||
|
||||
if name == "Write":
|
||||
target = pathlib.Path(params["file_path"])
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
# Atomic, exactly as the real Write tool does it: temp file beside the
|
||||
# target, then rename. This is the operation the read-only workspace
|
||||
# boundary has to permit for the artifact directory and refuse for the
|
||||
# workspace, so a stand-in that wrote in place would prove nothing.
|
||||
staging = target.with_name(target.name + ".tmp.fake")
|
||||
staging.write_text(params.get("content", ""))
|
||||
os.replace(staging, target)
|
||||
return f"wrote {target}"
|
||||
if name == "Bash":
|
||||
return "(bash suppressed in the stand-in)"
|
||||
return f"(unhandled tool {name})"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
# stdin, because that is where the real CLI takes it under
|
||||
# "-p --input-format text": the parent pipes prompt bytes in. Scanning argv
|
||||
# for a non-flag token picks up a flag's VALUE instead ("text"), which is
|
||||
# exactly what the prompt-fidelity test caught.
|
||||
prompt = sys.stdin.read()
|
||||
base_url = os.environ.get("ANTHROPIC_BASE_URL")
|
||||
if not base_url:
|
||||
print(json.dumps({"type": "result", "subtype": "error", "is_error": True,
|
||||
"session_id": "fake-session", "num_turns": 0}), flush=True)
|
||||
return 1
|
||||
|
||||
emit = lambda event: print(json.dumps(event), flush=True) # noqa: E731
|
||||
emit({"type": "system", "subtype": "init", "session_id": "fake-session"})
|
||||
|
||||
message = _turn(base_url, prompt)
|
||||
blocks = message.get("content", [])
|
||||
emit({"type": "assistant", "message": {"role": "assistant", "content": blocks}})
|
||||
|
||||
tool_results = []
|
||||
for block in blocks:
|
||||
if block.get("type") == "tool_use":
|
||||
output = _run_tool(block["name"], block.get("input", {}))
|
||||
tool_results.append({"type": "tool_result", "tool_use_id": block["id"], "content": output})
|
||||
if tool_results:
|
||||
emit({"type": "user", "message": {"role": "user", "content": tool_results}})
|
||||
|
||||
usage = message.get("usage", {})
|
||||
emit({
|
||||
"type": "result",
|
||||
"subtype": "success",
|
||||
"is_error": False,
|
||||
"session_id": "fake-session",
|
||||
"num_turns": 1,
|
||||
"duration_ms": 1200,
|
||||
# A measured zero is not the same as unmeasured; the parent rejects a
|
||||
# collapsed cost, so report a real one.
|
||||
"total_cost_usd": 0.42,
|
||||
"usage": {
|
||||
"input_tokens": usage.get("input_tokens", 0),
|
||||
"output_tokens": usage.get("output_tokens", 0),
|
||||
"cache_read_input_tokens": usage.get("cache_read_input_tokens", 0),
|
||||
"cache_creation_input_tokens": usage.get("cache_creation_input_tokens", 0),
|
||||
},
|
||||
})
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
107
eval/tests/test_offline_session_integration.py
Normal file
107
eval/tests/test_offline_session_integration.py
Normal file
|
|
@ -0,0 +1,107 @@
|
|||
"""A session end to end with only the model faked.
|
||||
|
||||
The layers between the CLI and the row are where this harness has actually
|
||||
shipped bugs - the artifact that could not be written, the usage that was never
|
||||
recorded, the evidence that was scored from the wrong directory. Every one of
|
||||
them sat below the level its tests exercised. These run the real session path
|
||||
against a scripted provider, so the only thing not real is what the model says.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from workflow_bench.mock_provider import MockProvider, Reply
|
||||
from workflow_bench.proposer_sandbox import prepare_sandbox, prepare_review_workspace
|
||||
from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output
|
||||
from workflow_bench.runner_sessions import run_claude
|
||||
|
||||
FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py"
|
||||
REVIEW_JSON = '{"schema_version": 1, "verdict": "approve", "findings": []}'
|
||||
|
||||
|
||||
def _session(clone: Path, provider: MockProvider, **overrides):
|
||||
return run_claude(
|
||||
"review the change",
|
||||
clone,
|
||||
claude_bin=str(FAKE_CLI),
|
||||
timeout=60,
|
||||
env={
|
||||
"ANTHROPIC_BASE_URL": provider.base_url,
|
||||
"ANTHROPIC_API_KEY": "offline",
|
||||
"PATH": "/usr/bin:/bin",
|
||||
},
|
||||
**overrides,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def clone(tmp_path: Path) -> Path:
|
||||
workspace = tmp_path / "clone"
|
||||
workspace.mkdir()
|
||||
(workspace / "source.ts").write_text("export const answer = 42;\n")
|
||||
return workspace
|
||||
|
||||
|
||||
def test_a_session_records_the_usage_the_provider_reported(clone: Path) -> None:
|
||||
"""Token counts must survive the CLI boundary, not be invented after it."""
|
||||
|
||||
reply = Reply(input_tokens=2_000, output_tokens=300, cache_read_input_tokens=7_000, cache_creation_input_tokens=1_000)
|
||||
with MockProvider(default=reply) as provider:
|
||||
record = _session(clone, provider)
|
||||
|
||||
assert record["ok"] is True, record.get("error_detail")
|
||||
assert record["input_tokens"] == 2_000
|
||||
assert record["cache_read_input_tokens"] == 7_000
|
||||
assert record["cache_creation_input_tokens"] == 1_000
|
||||
assert record["output_tokens"] == 300
|
||||
# A measured zero would be indistinguishable from an unmeasured one.
|
||||
assert record["cost_usd"] == 0.42
|
||||
assert record["num_turns"] == 1
|
||||
|
||||
|
||||
def test_a_scripted_write_produces_a_review_artifact_the_scorer_accepts(clone: Path, tmp_path: Path) -> None:
|
||||
"""The full artifact path: model asks, CLI writes atomically, scorer reads.
|
||||
|
||||
This is the operation that shipped empty for a whole run. Nothing here
|
||||
fakes the write, the directory, or the parse - only the decision to write.
|
||||
"""
|
||||
|
||||
with prepare_sandbox(
|
||||
clone=clone, claude_bin=Path(sys.executable), backend="host-unsafe", preflight=False
|
||||
) as sandbox:
|
||||
artifact = prepare_review_workspace(sandbox, REVIEW_OUTPUT)
|
||||
write = {"name": "Write", "input": {"file_path": str(artifact), "content": REVIEW_JSON}}
|
||||
with MockProvider(default=Reply(text="reviewing", tools=[write])) as provider:
|
||||
record = _session(clone, provider)
|
||||
assert record["ok"] is True, record.get("error_detail")
|
||||
# Read inside the scope: prepare_sandbox removes the private root on exit.
|
||||
verdict, findings = parse_review_output(artifact)
|
||||
|
||||
assert verdict == "approve"
|
||||
assert findings == ()
|
||||
|
||||
|
||||
def test_the_provider_saw_the_prompt_the_harness_meant_to_send(clone: Path) -> None:
|
||||
"""A run that measures the wrong prompt measures nothing."""
|
||||
|
||||
with MockProvider() as provider:
|
||||
_session(clone, provider)
|
||||
|
||||
assert provider.requests, "the session never reached the provider"
|
||||
sent = provider.requests[0].body["messages"][0]["content"]
|
||||
assert "review the change" in sent
|
||||
|
||||
|
||||
def test_a_provider_failure_surfaces_as_a_failed_session_not_a_silent_pass(clone: Path) -> None:
|
||||
"""An upstream 529 must not be recorded as a usable measurement."""
|
||||
|
||||
failing = Reply(status_code=529, error_body={"error": {"type": "overloaded_error"}})
|
||||
with MockProvider(default=failing) as provider:
|
||||
record = _session(clone, provider)
|
||||
|
||||
assert record["ok"] is False
|
||||
assert record["error_kind"] is not None
|
||||
Loading…
Add table
Reference in a new issue