mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-05 02:43:32 +00:00
test(eval): run the offline sweep unstubbed in the job that can, and probe CLI identity
Items 5 and 6 turned out to be one change. The containment (ubuntu) job already installs bubblewrap, the pinned Claude CLI, node_modules and a built GitNexus - everything the sweep's two provisioning stubs stand in for. So the stubs are not a property of the test, only of a machine that lacks those things. GITNEXUS_REQUIRE_FULL_SWEEP=1 makes the sweep run with nothing stubbed: real containment instead of --unsafe-no-bwrap, the real runtime mounts, the real sanitized graph. Set in that job, following the GITNEXUS_REQUIRE_BWRAP_CANARY pattern already there. The gate FAILS on a missing piece rather than degrading to the stubbed path, which is the point - a green tick that silently tested less is what the bubblewrap canary was written to prevent. Verified both states here: default green, and gate-on fails on this machine rather than skipping, since it cannot create user namespaces. Item 7 is an experiment, not an answer. Per-cell attribution needs an identifier that travels WITH the request, because one proxy serves the whole sweep and anything read from its environment is identical for every call. What the real CLI sends is not documented anywhere I can check, and guessing a wire format is exactly how the last three accounting bugs happened. So the probe drives the REAL pinned CLI against the mock and records the identity-bearing headers and body keys that arrive. It asserts only that a request was made; the recorded evidence is the deliverable, and the job log preserves it. Skips without CLAUDE_CANARY_BIN. Two guards caught this rather than review: the repo pins the containment job's env and its exact test list, so both had to be updated deliberately - which is the guard working, not friction. 682 eval tests pass, 17 skipped.
This commit is contained in:
parent
a98ec461af
commit
dbb292d012
4 changed files with 95 additions and 4 deletions
8
.github/workflows/ci-tests.yml
vendored
8
.github/workflows/ci-tests.yml
vendored
|
|
@ -846,6 +846,11 @@ jobs:
|
|||
timeout-minutes: 20
|
||||
env:
|
||||
GITNEXUS_REQUIRE_BWRAP_CANARY: '1'
|
||||
# This job installs bubblewrap, the pinned runtime and a built GitNexus,
|
||||
# so the offline sweep runs here with nothing provisioning-stubbed: real
|
||||
# containment, real mounts, real graph. A missing piece fails the job
|
||||
# rather than silently falling back to the stubbed path.
|
||||
GITNEXUS_REQUIRE_FULL_SWEEP: '1'
|
||||
GITNEXUS_REQUIRE_CLAUDE_CANARY: '1'
|
||||
steps:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
|
|
@ -904,7 +909,8 @@ jobs:
|
|||
tests/test_process_control.py
|
||||
tests/test_proposer_sandbox.py
|
||||
tests/test_workflow_bench_sessions.py
|
||||
tests/test_ce_plugin_runtime.py -q
|
||||
tests/test_ce_plugin_runtime.py
|
||||
tests/test_offline_sweep_integration.py -q
|
||||
working-directory: eval
|
||||
|
||||
# Native Windows Job Object canary. POSIX-only tests skip by platform, while
|
||||
|
|
|
|||
|
|
@ -3,7 +3,13 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from workflow_bench.mock_provider import MockProvider, Reply
|
||||
from workflow_bench.provider_usage import (
|
||||
|
|
@ -131,7 +137,6 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke
|
|||
exactly the span this exercises.
|
||||
"""
|
||||
|
||||
import shutil
|
||||
|
||||
import yaml
|
||||
|
||||
|
|
@ -191,3 +196,50 @@ def test_a_request_through_the_real_gateway_records_native_usage(tmp_path, monke
|
|||
assert usage.cache_write_input_tokens == 1_000
|
||||
assert usage.ordinary_input_tokens == 2_000
|
||||
assert usage.complete, "a run that cannot interpret its own usage measured nothing"
|
||||
|
||||
|
||||
def test_probe_what_identity_the_real_cli_actually_sends(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""An experiment, not an assertion: which fields could correlate a request to a cell?
|
||||
|
||||
Per-cell usage attribution is unbuilt because one proxy serves the whole
|
||||
sweep, so anything read from the proxy environment is identical for every
|
||||
request. Attribution needs something that travels WITH the request, and
|
||||
what the Claude Code CLI actually sends is not documented anywhere I can
|
||||
check - guessing it is how the last three accounting bugs happened.
|
||||
|
||||
So this drives the REAL pinned CLI against the mock and prints the
|
||||
identity-bearing fields that arrive. It asserts only that a request was
|
||||
made; the value is the recorded evidence, which the job log preserves.
|
||||
"""
|
||||
|
||||
claude = os.environ.get("CLAUDE_CANARY_BIN")
|
||||
if not claude or not Path(claude).exists():
|
||||
pytest.skip("no pinned Claude CLI here; the containment job supplies CLAUDE_CANARY_BIN")
|
||||
|
||||
with MockProvider(default=Reply(text="ok")) as provider:
|
||||
subprocess.run(
|
||||
[claude, "-p", "--input-format", "text", "--output-format", "stream-json", "--verbose"],
|
||||
input=b"say ok",
|
||||
capture_output=True,
|
||||
timeout=120,
|
||||
env={
|
||||
**os.environ,
|
||||
"ANTHROPIC_BASE_URL": provider.base_url,
|
||||
"ANTHROPIC_API_KEY": "offline-probe",
|
||||
"HOME": str(tmp_path),
|
||||
},
|
||||
)
|
||||
|
||||
assert provider.requests, "the real CLI never reached the mock provider"
|
||||
request = provider.requests[0]
|
||||
interesting = {
|
||||
"header:" + name: value
|
||||
for name, value in request.headers.items()
|
||||
if any(k in name.lower() for k in ("session", "user", "trace", "request-id", "conversation", "metadata"))
|
||||
}
|
||||
interesting.update(
|
||||
{f"body:{key}": request.body[key] for key in ("metadata", "user", "session_id") if key in request.body}
|
||||
)
|
||||
print("\nIDENTITY FIELDS THE REAL CLI SENDS:")
|
||||
print(" body keys:", sorted(request.body))
|
||||
print(" candidate correlators:", interesting or "NONE — per-cell attribution needs another mechanism")
|
||||
|
|
|
|||
|
|
@ -26,6 +26,9 @@ import sys
|
|||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import os
|
||||
import shutil
|
||||
|
||||
import pytest
|
||||
|
||||
from workflow_bench import oracle_assets, runner
|
||||
|
|
@ -34,6 +37,14 @@ from workflow_bench.mock_provider import MockProvider, Reply
|
|||
FAKE_CLI = Path(__file__).parent / "fixtures" / "fake_claude.py"
|
||||
ARMS = ("ce_review", "review", "candidate_review")
|
||||
|
||||
# When set, the sweep runs with NOTHING provisioning-stubbed: real bubblewrap
|
||||
# containment, the real pinned runtime mounts, and the real sanitized graph
|
||||
# build. The named CI job installs all three, so a missing one there is a
|
||||
# regression rather than an unsupported machine - it FAILS instead of quietly
|
||||
# degrading to the stubbed path, which is the whole point of the gate.
|
||||
FULL_SWEEP_ENV = "GITNEXUS_REQUIRE_FULL_SWEEP"
|
||||
FULL_SWEEP = os.environ.get(FULL_SWEEP_ENV) == "1"
|
||||
|
||||
# The review output and the hidden labels are DELIBERATELY different shapes -
|
||||
# the labels carry line_start/line_end and no recommendation. Only a real run
|
||||
# surfaces that; it is why these are written out rather than shared.
|
||||
|
|
@ -104,6 +115,17 @@ def bench(tmp_path: Path):
|
|||
|
||||
|
||||
def _stub_provisioning(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""Replace what this machine cannot supply - and nothing else.
|
||||
|
||||
Under FULL_SWEEP nothing is replaced: the runtime mounts and the graph are
|
||||
built for real, so the sweep exercises containment and provisioning too.
|
||||
"""
|
||||
|
||||
if FULL_SWEEP:
|
||||
if shutil.which("bwrap") is None:
|
||||
pytest.fail(f"{FULL_SWEEP_ENV}=1 but bubblewrap is absent")
|
||||
return
|
||||
|
||||
monkeypatch.setattr(runner, "trusted_gitnexus_runtime_mounts", lambda: ())
|
||||
|
||||
def materialize(worktree, *, sanitized_head=None, **_kwargs):
|
||||
|
|
@ -163,11 +185,14 @@ def _sweep(bench, monkeypatch: pytest.MonkeyPatch, findings: list[dict], verdict
|
|||
"runner", "--tasks", str(bench.tasks), "--arms", *ARMS,
|
||||
"--runs", "1", "--workers", "1", "--out", str(bench.out),
|
||||
"--base-url", provider.base_url, "--anthropic-api-key", "offline",
|
||||
"--claude-bin", str(FAKE_CLI), "--unsafe-no-bwrap", "--model", "mock-model",
|
||||
"--claude-bin", str(FAKE_CLI),
|
||||
*([] if FULL_SWEEP else ["--unsafe-no-bwrap"]),
|
||||
"--model", "mock-model",
|
||||
"--ce-plugin-dir", str(bench.plugin), "--ce-plugin-version", "0.0.0-fixture",
|
||||
"--candidate-overlay", str(bench.overlay),
|
||||
])
|
||||
monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI
|
||||
if not FULL_SWEEP:
|
||||
monkeypatch.delenv("CI", raising=False) # --unsafe-no-bwrap is forbidden under CI
|
||||
try:
|
||||
code = runner.main()
|
||||
except SystemExit as exc:
|
||||
|
|
|
|||
|
|
@ -195,6 +195,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
|
|||
assert containment["env"] == {
|
||||
"GITNEXUS_REQUIRE_BWRAP_CANARY": "1",
|
||||
"GITNEXUS_REQUIRE_CLAUDE_CANARY": "1",
|
||||
# This job is the only place with bubblewrap, the pinned runtime and a
|
||||
# built GitNexus together, so it is where the offline sweep runs with
|
||||
# nothing provisioning-stubbed. Pinned here so the gate cannot be
|
||||
# dropped and leave the sweep silently running the stubbed path.
|
||||
"GITNEXUS_REQUIRE_FULL_SWEEP": "1",
|
||||
}
|
||||
assert containment["timeout-minutes"] == 20
|
||||
assert containment_node_setup["with"] == {
|
||||
|
|
@ -239,6 +244,9 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
|
|||
"tests/test_proposer_sandbox.py",
|
||||
"tests/test_workflow_bench_sessions.py",
|
||||
"tests/test_ce_plugin_runtime.py",
|
||||
# The offline sweep, run here with nothing stubbed: this job is the only
|
||||
# one carrying bubblewrap, the pinned runtime and a built GitNexus.
|
||||
"tests/test_offline_sweep_integration.py",
|
||||
"-q",
|
||||
]
|
||||
bwrap_canary_marker = re.compile(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue