From dbb292d012c610a71b390d51a5a8aceb16611d31 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Wed, 9 Sep 2026 06:16:17 +0000 Subject: [PATCH] test(eval): run the offline sweep unstubbed in the job that can, and probe CLI identity Items 5 and 6 turned out to be one change. The containment (ubuntu) job already installs bubblewrap, the pinned Claude CLI, node_modules and a built GitNexus - everything the sweep's two provisioning stubs stand in for. So the stubs are not a property of the test, only of a machine that lacks those things. GITNEXUS_REQUIRE_FULL_SWEEP=1 makes the sweep run with nothing stubbed: real containment instead of --unsafe-no-bwrap, the real runtime mounts, the real sanitized graph. Set in that job, following the GITNEXUS_REQUIRE_BWRAP_CANARY pattern already there. The gate FAILS on a missing piece rather than degrading to the stubbed path, which is the point - a green tick that silently tested less is what the bubblewrap canary was written to prevent. Verified both states here: default green, and gate-on fails on this machine rather than skipping, since it cannot create user namespaces. Item 7 is an experiment, not an answer. Per-cell attribution needs an identifier that travels WITH the request, because one proxy serves the whole sweep and anything read from its environment is identical for every call. What the real CLI sends is not documented anywhere I can check, and guessing a wire format is exactly how the last three accounting bugs happened. So the probe drives the REAL pinned CLI against the mock and records the identity-bearing headers and body keys that arrive. It asserts only that a request was made; the recorded evidence is the deliverable, and the job log preserves it. Skips without CLAUDE_CANARY_BIN. Two guards caught this rather than review: the repo pins the containment job's env and its exact test list, so both had to be updated deliberately - which is the guard working, not friction. 682 eval tests pass, 17 skipped. --- .github/workflows/ci-tests.yml | 8 ++- eval/tests/test_mock_provider.py | 54 +++++++++++++++++++- eval/tests/test_offline_sweep_integration.py | 29 ++++++++++- eval/tests/test_workflow_bench.py | 8 +++ 4 files changed, 95 insertions(+), 4 deletions(-) diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 7b469948c..a4ac9710d 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -846,6 +846,11 @@ jobs: timeout-minutes: 20 env: GITNEXUS_REQUIRE_BWRAP_CANARY: '1' + # This job installs bubblewrap, the pinned runtime and a built GitNexus, + # so the offline sweep runs here with nothing provisioning-stubbed: real + # containment, real mounts, real graph. A missing piece fails the job + # rather than silently falling back to the stubbed path. + GITNEXUS_REQUIRE_FULL_SWEEP: '1' GITNEXUS_REQUIRE_CLAUDE_CANARY: '1' steps: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 @@ -904,7 +909,8 @@ jobs: tests/test_process_control.py tests/test_proposer_sandbox.py tests/test_workflow_bench_sessions.py - tests/test_ce_plugin_runtime.py -q + tests/test_ce_plugin_runtime.py + tests/test_offline_sweep_integration.py -q working-directory: eval # Native Windows Job Object canary. POSIX-only tests skip by platform, while diff --git a/eval/tests/test_mock_provider.py b/eval/tests/test_mock_provider.py index 2d2e2eccf..62e098fa7 100644 --- a/eval/tests/test_mock_provider.py +++ b/eval/tests/test_mock_provider.py @@ -3,7 +3,13 @@ from __future__ import annotations import json +import os +import shutil +import subprocess import urllib.request +from pathlib import Path + +import pytest from workflow_bench.mock_provider import MockProvider, Reply from workflow_bench.provider_usage import ( @@ -131,7 +137,6 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke exactly the span this exercises. """ - import shutil import yaml @@ -191,3 +196,50 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke assert usage.cache_write_input_tokens == 1_000 assert usage.ordinary_input_tokens == 2_000 assert usage.complete, "a run that cannot interpret its own usage measured nothing" + + +def test_probe_what_identity_the_real_cli_actually_sends(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """An experiment, not an assertion: which fields could correlate a request to a cell? + + Per-cell usage attribution is unbuilt because one proxy serves the whole + sweep, so anything read from the proxy environment is identical for every + request. Attribution needs something that travels WITH the request, and + what the Claude Code CLI actually sends is not documented anywhere I can + check - guessing it is how the last three accounting bugs happened. + + So this drives the REAL pinned CLI against the mock and prints the + identity-bearing fields that arrive. It asserts only that a request was + made; the value is the recorded evidence, which the job log preserves. + """ + + claude = os.environ.get("CLAUDE_CANARY_BIN") + if not claude or not Path(claude).exists(): + pytest.skip("no pinned Claude CLI here; the containment job supplies CLAUDE_CANARY_BIN") + + with MockProvider(default=Reply(text="ok")) as provider: + subprocess.run( + [claude, "-p", "--input-format", "text", "--output-format", "stream-json", "--verbose"], + input=b"say ok", + capture_output=True, + timeout=120, + env={ + **os.environ, + "ANTHROPIC_BASE_URL": provider.base_url, + "ANTHROPIC_API_KEY": "offline-probe", + "HOME": str(tmp_path), + }, + ) + + assert provider.requests, "the real CLI never reached the mock provider" + request = provider.requests[0] + interesting = { + "header:" + name: value + for name, value in request.headers.items() + if any(k in name.lower() for k in ("session", "user", "trace", "request-id", "conversation", "metadata")) + } + interesting.update( + {f"body:{key}": request.body[key] for key in ("metadata", "user", "session_id") if key in request.body} + ) + print("\nIDENTITY FIELDS THE REAL CLI SENDS:") + print(" body keys:", sorted(request.body)) + print(" candidate correlators:", interesting or "NONE — per-cell attribution needs another mechanism") diff --git a/eval/tests/test_offline_sweep_integration.py b/eval/tests/test_offline_sweep_integration.py index dd3f258a1..1e6362272 100644 --- a/eval/tests/test_offline_sweep_integration.py +++ b/eval/tests/test_offline_sweep_integration.py @@ -26,6 +26,9 @@ import sys from pathlib import Path from types import SimpleNamespace +import os +import shutil + import pytest from workflow_bench import oracle_assets, runner @@ -34,6 +37,14 @@ from workflow_bench.mock_provider import MockProvider, Reply FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py" ARMS = ("ce_review", "review", "candidate_review") +# When set, the sweep runs with NOTHING provisioning-stubbed: real bubblewrap +# containment, the real pinned runtime mounts, and the real sanitized graph +# build. The named CI job installs all three, so a missing one there is a +# regression rather than an unsupported machine - it FAILS instead of quietly +# degrading to the stubbed path, which is the whole point of the gate. +FULL_SWEEP_ENV = "GITNEXUS_REQUIRE_FULL_SWEEP" +FULL_SWEEP = os.environ.get(FULL_SWEEP_ENV) == "1" + # The review output and the hidden labels are DELIBERATELY different shapes - # the labels carry line_start/line_end and no recommendation. Only a real run # surfaces that; it is why these are written out rather than shared. @@ -104,6 +115,17 @@ def bench(tmp_path: Path): def _stub_provisioning(monkeypatch: pytest.MonkeyPatch) -> None: + """Replace what this machine cannot supply - and nothing else. + + Under FULL_SWEEP nothing is replaced: the runtime mounts and the graph are + built for real, so the sweep exercises containment and provisioning too. + """ + + if FULL_SWEEP: + if shutil.which("bwrap") is None: + pytest.fail(f"{FULL_SWEEP_ENV}=1 but bubblewrap is absent") + return + monkeypatch.setattr(runner, "trusted_gitnexus_runtime_mounts", lambda: ()) def materialize(worktree, *, sanitized_head=None, **_kwargs): @@ -163,11 +185,14 @@ def _sweep(bench, monkeypatch: pytest.MonkeyPatch, findings: list[dict], verdict "runner", "--tasks", str(bench.tasks), "--arms", *ARMS, "--runs", "1", "--workers", "1", "--out", str(bench.out), "--base-url", provider.base_url, "--anthropic-api-key", "offline", - "--claude-bin", str(FAKE_CLI), "--unsafe-no-bwrap", "--model", "mock-model", + "--claude-bin", str(FAKE_CLI), + *([] if FULL_SWEEP else ["--unsafe-no-bwrap"]), + "--model", "mock-model", "--ce-plugin-dir", str(bench.plugin), "--ce-plugin-version", "0.0.0-fixture", "--candidate-overlay", str(bench.overlay), ]) - monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI + if not FULL_SWEEP: + monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI try: code = runner.main() except SystemExit as exc: diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index b8b70c0d6..23128c189 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -195,6 +195,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): assert containment["env"] == { "GITNEXUS_REQUIRE_BWRAP_CANARY": "1", "GITNEXUS_REQUIRE_CLAUDE_CANARY": "1", + # This job is the only place with bubblewrap, the pinned runtime and a + # built GitNexus together, so it is where the offline sweep runs with + # nothing provisioning-stubbed. Pinned here so the gate cannot be + # dropped and leave the sweep silently running the stubbed path. + "GITNEXUS_REQUIRE_FULL_SWEEP": "1", } assert containment["timeout-minutes"] == 20 assert containment_node_setup["with"] == { @@ -239,6 +244,9 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): "tests/test_proposer_sandbox.py", "tests/test_workflow_bench_sessions.py", "tests/test_ce_plugin_runtime.py", + # The offline sweep, run here with nothing stubbed: this job is the only + # one carrying bubblewrap, the pinned runtime and a built GitNexus. + "tests/test_offline_sweep_integration.py", "-q", ] bwrap_canary_marker = re.compile(