test(eval): run the offline sweep unstubbed in the job that can, and probe CLI identity

Items 5 and 6 turned out to be one change. The containment (ubuntu) job already
installs bubblewrap, the pinned Claude CLI, node_modules and a built GitNexus -
everything the sweep's two provisioning stubs stand in for. So the stubs are not
a property of the test, only of a machine that lacks those things.

GITNEXUS_REQUIRE_FULL_SWEEP=1 makes the sweep run with nothing stubbed: real
containment instead of --unsafe-no-bwrap, the real runtime mounts, the real
sanitized graph. Set in that job, following the GITNEXUS_REQUIRE_BWRAP_CANARY
pattern already there. The gate FAILS on a missing piece rather than degrading
to the stubbed path, which is the point - a green tick that silently tested less
is what the bubblewrap canary was written to prevent.

Verified both states here: default green, and gate-on fails on this machine
rather than skipping, since it cannot create user namespaces.

Item 7 is an experiment, not an answer. Per-cell attribution needs an
identifier that travels WITH the request, because one proxy serves the whole
sweep and anything read from its environment is identical for every call. What
the real CLI sends is not documented anywhere I can check, and guessing a wire
format is exactly how the last three accounting bugs happened. So the probe
drives the REAL pinned CLI against the mock and records the identity-bearing
headers and body keys that arrive. It asserts only that a request was made; the
recorded evidence is the deliverable, and the job log preserves it. Skips
without CLAUDE_CANARY_BIN.

Two guards caught this rather than review: the repo pins the containment job's
env and its exact test list, so both had to be updated deliberately - which is
the guard working, not friction.

682 eval tests pass, 17 skipped.
This commit is contained in:
Gergo Magyar 2026-09-09 06:16:17 +00:00
parent a98ec461af
commit dbb292d012
4 changed files with 95 additions and 4 deletions

View file

@ -846,6 +846,11 @@ jobs:
timeout-minutes: 20
env:
GITNEXUS_REQUIRE_BWRAP_CANARY: '1'
# This job installs bubblewrap, the pinned runtime and a built GitNexus,
# so the offline sweep runs here with nothing provisioning-stubbed: real
# containment, real mounts, real graph. A missing piece fails the job
# rather than silently falling back to the stubbed path.
GITNEXUS_REQUIRE_FULL_SWEEP: '1'
GITNEXUS_REQUIRE_CLAUDE_CANARY: '1'
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
@ -904,7 +909,8 @@ jobs:
tests/test_process_control.py
tests/test_proposer_sandbox.py
tests/test_workflow_bench_sessions.py
tests/test_ce_plugin_runtime.py -q
tests/test_ce_plugin_runtime.py
tests/test_offline_sweep_integration.py -q
working-directory: eval
# Native Windows Job Object canary. POSIX-only tests skip by platform, while

View file

@ -3,7 +3,13 @@
from __future__ import annotations
import json
import os
import shutil
import subprocess
import urllib.request
from pathlib import Path
import pytest
from workflow_bench.mock_provider import MockProvider, Reply
from workflow_bench.provider_usage import (
@ -131,7 +137,6 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke
exactly the span this exercises.
"""
import shutil
import yaml
@ -191,3 +196,50 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke
assert usage.cache_write_input_tokens == 1_000
assert usage.ordinary_input_tokens == 2_000
assert usage.complete, "a run that cannot interpret its own usage measured nothing"
def test_probe_what_identity_the_real_cli_actually_sends(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""An experiment, not an assertion: which fields could correlate a request to a cell?
Per-cell usage attribution is unbuilt because one proxy serves the whole
sweep, so anything read from the proxy environment is identical for every
request. Attribution needs something that travels WITH the request, and
what the Claude Code CLI actually sends is not documented anywhere I can
check - guessing it is how the last three accounting bugs happened.
So this drives the REAL pinned CLI against the mock and prints the
identity-bearing fields that arrive. It asserts only that a request was
made; the value is the recorded evidence, which the job log preserves.
"""
claude = os.environ.get("CLAUDE_CANARY_BIN")
if not claude or not Path(claude).exists():
pytest.skip("no pinned Claude CLI here; the containment job supplies CLAUDE_CANARY_BIN")
with MockProvider(default=Reply(text="ok")) as provider:
subprocess.run(
[claude, "-p", "--input-format", "text", "--output-format", "stream-json", "--verbose"],
input=b"say ok",
capture_output=True,
timeout=120,
env={
**os.environ,
"ANTHROPIC_BASE_URL": provider.base_url,
"ANTHROPIC_API_KEY": "offline-probe",
"HOME": str(tmp_path),
},
)
assert provider.requests, "the real CLI never reached the mock provider"
request = provider.requests[0]
interesting = {
"header:" + name: value
for name, value in request.headers.items()
if any(k in name.lower() for k in ("session", "user", "trace", "request-id", "conversation", "metadata"))
}
interesting.update(
{f"body:{key}": request.body[key] for key in ("metadata", "user", "session_id") if key in request.body}
)
print("\nIDENTITY FIELDS THE REAL CLI SENDS:")
print(" body keys:", sorted(request.body))
print(" candidate correlators:", interesting or "NONE — per-cell attribution needs another mechanism")

View file

@ -26,6 +26,9 @@ import sys
from pathlib import Path
from types import SimpleNamespace
import os
import shutil
import pytest
from workflow_bench import oracle_assets, runner
@ -34,6 +37,14 @@ from workflow_bench.mock_provider import MockProvider, Reply
FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py"
ARMS = ("ce_review", "review", "candidate_review")
# When set, the sweep runs with NOTHING provisioning-stubbed: real bubblewrap
# containment, the real pinned runtime mounts, and the real sanitized graph
# build. The named CI job installs all three, so a missing one there is a
# regression rather than an unsupported machine - it FAILS instead of quietly
# degrading to the stubbed path, which is the whole point of the gate.
FULL_SWEEP_ENV = "GITNEXUS_REQUIRE_FULL_SWEEP"
FULL_SWEEP = os.environ.get(FULL_SWEEP_ENV) == "1"
# The review output and the hidden labels are DELIBERATELY different shapes -
# the labels carry line_start/line_end and no recommendation. Only a real run
# surfaces that; it is why these are written out rather than shared.
@ -104,6 +115,17 @@ def bench(tmp_path: Path):
def _stub_provisioning(monkeypatch: pytest.MonkeyPatch) -> None:
"""Replace what this machine cannot supply - and nothing else.
Under FULL_SWEEP nothing is replaced: the runtime mounts and the graph are
built for real, so the sweep exercises containment and provisioning too.
"""
if FULL_SWEEP:
if shutil.which("bwrap") is None:
pytest.fail(f"{FULL_SWEEP_ENV}=1 but bubblewrap is absent")
return
monkeypatch.setattr(runner, "trusted_gitnexus_runtime_mounts", lambda: ())
def materialize(worktree, *, sanitized_head=None, **_kwargs):
@ -163,11 +185,14 @@ def _sweep(bench, monkeypatch: pytest.MonkeyPatch, findings: list[dict], verdict
"runner", "--tasks", str(bench.tasks), "--arms", *ARMS,
"--runs", "1", "--workers", "1", "--out", str(bench.out),
"--base-url", provider.base_url, "--anthropic-api-key", "offline",
"--claude-bin", str(FAKE_CLI), "--unsafe-no-bwrap", "--model", "mock-model",
"--claude-bin", str(FAKE_CLI),
*([] if FULL_SWEEP else ["--unsafe-no-bwrap"]),
"--model", "mock-model",
"--ce-plugin-dir", str(bench.plugin), "--ce-plugin-version", "0.0.0-fixture",
"--candidate-overlay", str(bench.overlay),
])
monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI
if not FULL_SWEEP:
monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI
try:
code = runner.main()
except SystemExit as exc:

View file

@ -195,6 +195,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
assert containment["env"] == {
"GITNEXUS_REQUIRE_BWRAP_CANARY": "1",
"GITNEXUS_REQUIRE_CLAUDE_CANARY": "1",
# This job is the only place with bubblewrap, the pinned runtime and a
# built GitNexus together, so it is where the offline sweep runs with
# nothing provisioning-stubbed. Pinned here so the gate cannot be
# dropped and leave the sweep silently running the stubbed path.
"GITNEXUS_REQUIRE_FULL_SWEEP": "1",
}
assert containment["timeout-minutes"] == 20
assert containment_node_setup["with"] == {
@ -239,6 +244,9 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
"tests/test_proposer_sandbox.py",
"tests/test_workflow_bench_sessions.py",
"tests/test_ce_plugin_runtime.py",
# The offline sweep, run here with nothing stubbed: this job is the only
# one carrying bubblewrap, the pinned runtime and a built GitNexus.
"tests/test_offline_sweep_integration.py",
"-q",
]
bwrap_canary_marker = re.compile(