diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 7b469948c..a4ac9710d 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -846,6 +846,11 @@ jobs: timeout-minutes: 20 env: GITNEXUS_REQUIRE_BWRAP_CANARY: '1' + # This job installs bubblewrap, the pinned runtime and a built GitNexus, + # so the offline sweep runs here with nothing provisioning-stubbed: real + # containment, real mounts, real graph. A missing piece fails the job + # rather than silently falling back to the stubbed path. + GITNEXUS_REQUIRE_FULL_SWEEP: '1' GITNEXUS_REQUIRE_CLAUDE_CANARY: '1' steps: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 @@ -904,7 +909,8 @@ jobs: tests/test_process_control.py tests/test_proposer_sandbox.py tests/test_workflow_bench_sessions.py - tests/test_ce_plugin_runtime.py -q + tests/test_ce_plugin_runtime.py + tests/test_offline_sweep_integration.py -q working-directory: eval # Native Windows Job Object canary. POSIX-only tests skip by platform, while diff --git a/eval/tests/test_mock_provider.py b/eval/tests/test_mock_provider.py index 2d2e2eccf..62e098fa7 100644 --- a/eval/tests/test_mock_provider.py +++ b/eval/tests/test_mock_provider.py @@ -3,7 +3,13 @@ from __future__ import annotations import json +import os +import shutil +import subprocess import urllib.request +from pathlib import Path + +import pytest from workflow_bench.mock_provider import MockProvider, Reply from workflow_bench.provider_usage import ( @@ -131,7 +137,6 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke exactly the span this exercises. """ - import shutil import yaml @@ -191,3 +196,50 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke assert usage.cache_write_input_tokens == 1_000 assert usage.ordinary_input_tokens == 2_000 assert usage.complete, "a run that cannot interpret its own usage measured nothing" + + +def test_probe_what_identity_the_real_cli_actually_sends(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """An experiment, not an assertion: which fields could correlate a request to a cell? + + Per-cell usage attribution is unbuilt because one proxy serves the whole + sweep, so anything read from the proxy environment is identical for every + request. Attribution needs something that travels WITH the request, and + what the Claude Code CLI actually sends is not documented anywhere I can + check - guessing it is how the last three accounting bugs happened. + + So this drives the REAL pinned CLI against the mock and prints the + identity-bearing fields that arrive. It asserts only that a request was + made; the value is the recorded evidence, which the job log preserves. + """ + + claude = os.environ.get("CLAUDE_CANARY_BIN") + if not claude or not Path(claude).exists(): + pytest.skip("no pinned Claude CLI here; the containment job supplies CLAUDE_CANARY_BIN") + + with MockProvider(default=Reply(text="ok")) as provider: + subprocess.run( + [claude, "-p", "--input-format", "text", "--output-format", "stream-json", "--verbose"], + input=b"say ok", + capture_output=True, + timeout=120, + env={ + **os.environ, + "ANTHROPIC_BASE_URL": provider.base_url, + "ANTHROPIC_API_KEY": "offline-probe", + "HOME": str(tmp_path), + }, + ) + + assert provider.requests, "the real CLI never reached the mock provider" + request = provider.requests[0] + interesting = { + "header:" + name: value + for name, value in request.headers.items() + if any(k in name.lower() for k in ("session", "user", "trace", "request-id", "conversation", "metadata")) + } + interesting.update( + {f"body:{key}": request.body[key] for key in ("metadata", "user", "session_id") if key in request.body} + ) + print("\nIDENTITY FIELDS THE REAL CLI SENDS:") + print(" body keys:", sorted(request.body)) + print(" candidate correlators:", interesting or "NONE — per-cell attribution needs another mechanism") diff --git a/eval/tests/test_offline_sweep_integration.py b/eval/tests/test_offline_sweep_integration.py index dd3f258a1..1e6362272 100644 --- a/eval/tests/test_offline_sweep_integration.py +++ b/eval/tests/test_offline_sweep_integration.py @@ -26,6 +26,9 @@ import sys from pathlib import Path from types import SimpleNamespace +import os +import shutil + import pytest from workflow_bench import oracle_assets, runner @@ -34,6 +37,14 @@ from workflow_bench.mock_provider import MockProvider, Reply FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py" ARMS = ("ce_review", "review", "candidate_review") +# When set, the sweep runs with NOTHING provisioning-stubbed: real bubblewrap +# containment, the real pinned runtime mounts, and the real sanitized graph +# build. The named CI job installs all three, so a missing one there is a +# regression rather than an unsupported machine - it FAILS instead of quietly +# degrading to the stubbed path, which is the whole point of the gate. +FULL_SWEEP_ENV = "GITNEXUS_REQUIRE_FULL_SWEEP" +FULL_SWEEP = os.environ.get(FULL_SWEEP_ENV) == "1" + # The review output and the hidden labels are DELIBERATELY different shapes - # the labels carry line_start/line_end and no recommendation. Only a real run # surfaces that; it is why these are written out rather than shared. @@ -104,6 +115,17 @@ def bench(tmp_path: Path): def _stub_provisioning(monkeypatch: pytest.MonkeyPatch) -> None: + """Replace what this machine cannot supply - and nothing else. + + Under FULL_SWEEP nothing is replaced: the runtime mounts and the graph are + built for real, so the sweep exercises containment and provisioning too. + """ + + if FULL_SWEEP: + if shutil.which("bwrap") is None: + pytest.fail(f"{FULL_SWEEP_ENV}=1 but bubblewrap is absent") + return + monkeypatch.setattr(runner, "trusted_gitnexus_runtime_mounts", lambda: ()) def materialize(worktree, *, sanitized_head=None, **_kwargs): @@ -163,11 +185,14 @@ def _sweep(bench, monkeypatch: pytest.MonkeyPatch, findings: list[dict], verdict "runner", "--tasks", str(bench.tasks), "--arms", *ARMS, "--runs", "1", "--workers", "1", "--out", str(bench.out), "--base-url", provider.base_url, "--anthropic-api-key", "offline", - "--claude-bin", str(FAKE_CLI), "--unsafe-no-bwrap", "--model", "mock-model", + "--claude-bin", str(FAKE_CLI), + *([] if FULL_SWEEP else ["--unsafe-no-bwrap"]), + "--model", "mock-model", "--ce-plugin-dir", str(bench.plugin), "--ce-plugin-version", "0.0.0-fixture", "--candidate-overlay", str(bench.overlay), ]) - monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI + if not FULL_SWEEP: + monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI try: code = runner.main() except SystemExit as exc: diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index b8b70c0d6..23128c189 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -195,6 +195,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): assert containment["env"] == { "GITNEXUS_REQUIRE_BWRAP_CANARY": "1", "GITNEXUS_REQUIRE_CLAUDE_CANARY": "1", + # This job is the only place with bubblewrap, the pinned runtime and a + # built GitNexus together, so it is where the offline sweep runs with + # nothing provisioning-stubbed. Pinned here so the gate cannot be + # dropped and leave the sweep silently running the stubbed path. + "GITNEXUS_REQUIRE_FULL_SWEEP": "1", } assert containment["timeout-minutes"] == 20 assert containment_node_setup["with"] == { @@ -239,6 +244,9 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): "tests/test_proposer_sandbox.py", "tests/test_workflow_bench_sessions.py", "tests/test_ce_plugin_runtime.py", + # The offline sweep, run here with nothing stubbed: this job is the only + # one carrying bubblewrap, the pinned runtime and a built GitNexus. + "tests/test_offline_sweep_integration.py", "-q", ] bwrap_canary_marker = re.compile(