mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-10 03:27:59 +00:00
fix(workflow-bench): keep proposer hooks and JSONL evidence intact
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
bc7f4a907e
commit
aea20ccf72
3 changed files with 204 additions and 19 deletions
|
|
@ -6,6 +6,7 @@ import os
|
|||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from contextlib import contextmanager
|
||||
from datetime import UTC, datetime, timedelta
|
||||
|
||||
import pytest
|
||||
|
|
@ -205,7 +206,7 @@ def test_proposer_reads_only_digest_bound_transcripts_below_results(tmp_path, mo
|
|||
gate_summary=[],
|
||||
)
|
||||
|
||||
assert entries["transcript-0-0.jsonl"] == payload.decode()
|
||||
assert [json.loads(line) for line in entries["transcript-0-0.jsonl"].splitlines()] == [json.loads(payload)]
|
||||
assert entries["patch-0.diff"] == patch.read_text()
|
||||
staged_rows = entries["selected-rows.json"]
|
||||
assert staged_rows[0]["patch_file"] == "patch-0.diff"
|
||||
|
|
@ -222,6 +223,61 @@ def test_proposer_reads_only_digest_bound_transcripts_below_results(tmp_path, mo
|
|||
)
|
||||
|
||||
|
||||
def test_proposer_compacts_transcripts_as_complete_json_events(tmp_path):
|
||||
results = tmp_path / "results"
|
||||
transcripts = results / "transcripts"
|
||||
transcripts.mkdir(parents=True, mode=0o700)
|
||||
events = [
|
||||
{
|
||||
"type": "assistant",
|
||||
"message": {
|
||||
"content": [
|
||||
{
|
||||
"type": "thinking",
|
||||
"thinking": "analysis-" + ("x" * 100_000),
|
||||
"signature": "opaque-base64-signature",
|
||||
}
|
||||
]
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "result",
|
||||
"session_id": "session-1",
|
||||
"usage": {"input_tokens": 1, "output_tokens": 2},
|
||||
},
|
||||
]
|
||||
payload = "".join(json.dumps(event) + "\n" for event in events).encode()
|
||||
artifact = transcripts / "session.jsonl"
|
||||
artifact.write_bytes(payload)
|
||||
artifact.chmod(0o600)
|
||||
|
||||
entries = proposer_evidence_entries(
|
||||
results_dir=results,
|
||||
evidence=[
|
||||
row(
|
||||
transcript_artifacts=[
|
||||
{
|
||||
"path": "transcripts/session.jsonl",
|
||||
"sha256": hashlib.sha256(payload).hexdigest(),
|
||||
"bytes": len(payload),
|
||||
"source": PARENT_EVENT_STREAM_SOURCE,
|
||||
}
|
||||
]
|
||||
)
|
||||
],
|
||||
learnings=[],
|
||||
gate_summary=[],
|
||||
artifact_limit=8192,
|
||||
)
|
||||
|
||||
staged = entries["transcript-0-0.jsonl"]
|
||||
parsed = [json.loads(line) for line in staged.splitlines()]
|
||||
assert len(staged.encode()) <= 8192
|
||||
assert parsed[-1]["type"] == "result"
|
||||
assert parsed[0]["message"]["content"][0]["signature"] == "[OMITTED]"
|
||||
assert "opaque-base64-signature" not in staged
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name == "nt", reason="transcript symlink containment is POSIX-only")
|
||||
def test_proposer_rejects_symlink_and_foreign_transcript_artifacts(tmp_path):
|
||||
results = tmp_path / "results"
|
||||
|
|
@ -412,14 +468,24 @@ def test_stage_proposer_evidence_bundle_drops_prior_proposal_to_fit_budget(tmp_p
|
|||
|
||||
|
||||
def test_stage_proposer_evidence_bundle_compacts_artifacts_before_dropping_rows(tmp_path, monkeypatch, capsys):
|
||||
monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 300_000)
|
||||
monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 300_000)
|
||||
monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 150_000)
|
||||
monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 150_000)
|
||||
results = tmp_path / "results"
|
||||
transcripts = results / "transcripts"
|
||||
transcripts.mkdir(parents=True, mode=0o700)
|
||||
rows = []
|
||||
for index in range(2):
|
||||
payload = f"session-{index}\n".encode() + b"x" * 100_000
|
||||
payload = (
|
||||
json.dumps(
|
||||
{
|
||||
"type": "assistant",
|
||||
"message": {"content": [{"type": "text", "text": "x" * 100_000}]},
|
||||
}
|
||||
)
|
||||
+ "\n"
|
||||
+ json.dumps({"type": "result", "session_id": f"session-{index}"})
|
||||
+ "\n"
|
||||
).encode()
|
||||
transcript = transcripts / f"session-{index}.jsonl"
|
||||
transcript.write_bytes(payload)
|
||||
transcript.chmod(0o600)
|
||||
|
|
@ -494,6 +560,56 @@ def test_proposer_refuses_a_prior_proposal_that_lost_its_trust_boundary(tmp_path
|
|||
)
|
||||
|
||||
|
||||
def test_run_proposer_keeps_trusted_input_hook_enabled(monkeypatch, tmp_path):
|
||||
evidence = tmp_path / "evidence"
|
||||
evidence.mkdir()
|
||||
transcript_projects = tmp_path / "transcript-projects"
|
||||
transcript_projects.mkdir()
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
def fake_make_worktree(_repo, _ref, destination):
|
||||
clone = destination / "clone"
|
||||
clone.mkdir()
|
||||
return clone
|
||||
|
||||
class FakeSandbox:
|
||||
claude_bin = "claude"
|
||||
command_prefix: list[str] = []
|
||||
settings_json = '{"hooks":{"PreToolUse":[]}}'
|
||||
|
||||
@property
|
||||
def transcript_projects(self):
|
||||
return transcript_projects
|
||||
|
||||
@contextmanager
|
||||
def fake_prepare_sandbox(**_kwargs):
|
||||
yield FakeSandbox()
|
||||
|
||||
def fake_run_claude(*_args, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return {"ok": False, "error_kind": "session-error"}
|
||||
|
||||
monkeypatch.setattr(evolve.runner, "make_worktree", fake_make_worktree)
|
||||
monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None)
|
||||
monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox)
|
||||
monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude)
|
||||
args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"])
|
||||
|
||||
record = evolve.run_proposer(
|
||||
"prompt",
|
||||
args,
|
||||
overlay_dir=tmp_path / "overlay",
|
||||
proposal_path=tmp_path / "proposal.md",
|
||||
evidence_bundle=evidence,
|
||||
bwrap_bin=tmp_path / "bwrap",
|
||||
)
|
||||
|
||||
assert record["ok"] is False
|
||||
assert captured.get("bare", False) is False
|
||||
assert captured["allowed_tools"] == evolve.PROPOSER_ALLOWED_TOOLS
|
||||
assert captured["settings_json"] == FakeSandbox.settings_json
|
||||
|
||||
|
||||
def test_parser_defaults_match_the_gate_minimums():
|
||||
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
|
||||
assert args.runs == 3
|
||||
|
|
|
|||
|
|
@ -883,13 +883,14 @@ def test_clone_controlled_mcp_replacement_is_never_executed_or_credentialed(tmp_
|
|||
os.environ.get("GITNEXUS_REQUIRE_CLAUDE_CANARY") != "1",
|
||||
reason="real Claude/Bash/MCP canary is mandatory in the named Ubuntu CI job",
|
||||
)
|
||||
def test_real_claude_bare_auth_inner_sandbox_and_mcp_permissions(tmp_path: Path) -> None:
|
||||
def test_real_claude_hooks_auth_inner_sandbox_and_mcp_permissions(tmp_path: Path) -> None:
|
||||
"""Exercise the exact CLI boundary without contacting a paid model."""
|
||||
|
||||
claude = Path(os.environ["CLAUDE_CANARY_BIN"]).resolve()
|
||||
assert claude.is_file()
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
(clone / "canary.txt").write_text("hook-readable")
|
||||
fake_mcp = clone / "fake_mcp.py"
|
||||
fake_mcp.write_text(
|
||||
"""import json
|
||||
|
|
@ -954,7 +955,20 @@ for line in sys.stdin:
|
|||
and isinstance(block.get("tool_use_id"), str)
|
||||
}
|
||||
)
|
||||
if "toolu_mcp_canary" not in tool_result_ids:
|
||||
if "toolu_read_canary" not in tool_result_ids:
|
||||
blocks = [
|
||||
{
|
||||
"type": "tool_use",
|
||||
"id": "toolu_read_canary",
|
||||
"name": "Read",
|
||||
"input": {
|
||||
"file_path": "/workspace/canary.txt",
|
||||
"pages": "",
|
||||
},
|
||||
}
|
||||
]
|
||||
stop_reason = "tool_use"
|
||||
elif "toolu_mcp_canary" not in tool_result_ids:
|
||||
blocks = [
|
||||
{
|
||||
"type": "tool_use",
|
||||
|
|
@ -1076,7 +1090,6 @@ for line in sys.stdin:
|
|||
"text",
|
||||
"--output-format",
|
||||
"json",
|
||||
"--bare",
|
||||
"--settings",
|
||||
sandbox.settings_json,
|
||||
"--strict-mcp-config",
|
||||
|
|
@ -1088,7 +1101,12 @@ for line in sys.stdin:
|
|||
# authoritative empirical gate for that behavior.
|
||||
"--model",
|
||||
"claude-canary-20260718",
|
||||
"--tools",
|
||||
"Read",
|
||||
"Bash",
|
||||
"mcp__gitnexus__list_repos",
|
||||
"--allowedTools",
|
||||
"Read",
|
||||
"Bash",
|
||||
"mcp__gitnexus__list_repos",
|
||||
],
|
||||
|
|
@ -1097,7 +1115,7 @@ for line in sys.stdin:
|
|||
auth_token="offline-canary-key",
|
||||
base_url=f"http://127.0.0.1:{server.server_port}",
|
||||
),
|
||||
stdin_data=b"Use both available tools, then finish.",
|
||||
stdin_data=b"Use all three available tools, then finish.",
|
||||
)
|
||||
finally:
|
||||
server.shutdown()
|
||||
|
|
@ -1107,6 +1125,9 @@ for line in sys.stdin:
|
|||
assert result.ok, result.stderr_tail + result.stdout_tail
|
||||
report = json.loads(result.stdout_tail)
|
||||
assert report["subtype"] == "success" and report["is_error"] is False, report
|
||||
read_result = observed_tool_results["toolu_read_canary"]
|
||||
assert read_result.get("is_error") is not True, read_result
|
||||
assert "hook-readable" in json.dumps(read_result)
|
||||
bash_result = observed_tool_results["toolu_bash_canary"]
|
||||
assert bash_result.get("is_error") is not True, bash_result
|
||||
assert (clone / "bash-called").read_text() == "canary"
|
||||
|
|
|
|||
|
|
@ -451,8 +451,6 @@ def _bound_transcript_artifact(
|
|||
while chunk := os.read(descriptor, 64 * 1024):
|
||||
digest.update(chunk)
|
||||
content.extend(chunk)
|
||||
if len(content) > limit:
|
||||
del content[: len(content) - limit]
|
||||
after = os.fstat(descriptor)
|
||||
if (opened.st_size, opened.st_mtime_ns) != (after.st_size, after.st_mtime_ns):
|
||||
raise SandboxError(f"transcript artifact changed while reading: {path}")
|
||||
|
|
@ -460,7 +458,57 @@ def _bound_transcript_artifact(
|
|||
os.close(descriptor)
|
||||
if digest.hexdigest() != expected_digest:
|
||||
raise SandboxError(f"transcript artifact digest does not match its results row: {path}")
|
||||
return bytes(content).decode(errors="replace")
|
||||
return _compact_transcript_jsonl(bytes(content), limit)
|
||||
|
||||
|
||||
def _compact_transcript_value(value: Any, *, key: str | None = None) -> Any:
|
||||
"""Bound large event fields while retaining valid, useful JSON."""
|
||||
|
||||
if key == "signature":
|
||||
return "[OMITTED]"
|
||||
if isinstance(value, str):
|
||||
field_limit = 4096
|
||||
if len(value) <= field_limit:
|
||||
return value
|
||||
half = field_limit // 2
|
||||
return f"{value[:half]}…[compacted {len(value) - field_limit} chars]…{value[-half:]}"
|
||||
if isinstance(value, list):
|
||||
return [_compact_transcript_value(item) for item in value]
|
||||
if isinstance(value, dict):
|
||||
return {str(item_key): _compact_transcript_value(item, key=str(item_key)) for item_key, item in value.items()}
|
||||
return value
|
||||
|
||||
|
||||
def _compact_transcript_jsonl(raw: bytes, limit: int) -> str:
|
||||
"""Select complete recent events; never cut through a JSON record."""
|
||||
|
||||
try:
|
||||
source_events = [json.loads(line) for line in raw.decode("utf-8", errors="strict").splitlines() if line.strip()]
|
||||
except (UnicodeError, json.JSONDecodeError) as exc:
|
||||
raise SandboxError(f"transcript artifact is not valid JSONL: {exc}") from exc
|
||||
|
||||
selected: list[bytes] = []
|
||||
total = 0
|
||||
for event in reversed(source_events):
|
||||
encoded = (
|
||||
json.dumps(
|
||||
_compact_transcript_value(event),
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
ensure_ascii=False,
|
||||
allow_nan=False,
|
||||
)
|
||||
+ "\n"
|
||||
).encode("utf-8")
|
||||
if len(encoded) > limit or total + len(encoded) > limit:
|
||||
continue
|
||||
selected.append(encoded)
|
||||
total += len(encoded)
|
||||
|
||||
if not selected:
|
||||
raise SandboxError("transcript artifact has no complete event within the evidence limit")
|
||||
selected.reverse()
|
||||
return b"".join(selected).decode("utf-8")
|
||||
|
||||
|
||||
def _prior_proposal_text(path: Path) -> str:
|
||||
|
|
@ -590,12 +638,11 @@ def stage_proposer_evidence_bundle(
|
|||
|
||||
# The proposer's exact tool surface. Read/Grep/Glob observe the read-only
|
||||
# evidence bundle and the incumbent skills; Bash writes the candidate overlay.
|
||||
# The proposer session runs --bare, which hard-disables the Write/Edit tools
|
||||
# ("Write exists but is not enabled in this context"), so Bash is the only
|
||||
# writable tool it enables — the settings pre-authorize it via
|
||||
# autoAllowBashIfSandboxed, and the sandbox filesystem policy confines writes to
|
||||
# the workspace/tmp/home. Exported so the containment canary tests the real
|
||||
# allowlist and cannot drift from production.
|
||||
# `--tools` restricts non-bare Claude to this list, so Write/Edit/Skill/Web are
|
||||
# unavailable while the trusted PreToolUse normalizer remains enabled. Settings
|
||||
# pre-authorize Bash via autoAllowBashIfSandboxed, and the sandbox filesystem
|
||||
# policy confines writes to workspace/tmp/home. Exported so containment tests
|
||||
# exercise the production allowlist without drift.
|
||||
PROPOSER_ALLOWED_TOOLS = ["Read", "Grep", "Glob", "Bash"]
|
||||
|
||||
|
||||
|
|
@ -646,10 +693,11 @@ def run_proposer(
|
|||
# No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB
|
||||
# forces "default", so requesting dontAsk only warns. Tools
|
||||
# are pre-approved via settings permissions.allow
|
||||
# (proposer_sandbox.build_claude_settings).
|
||||
# (proposer_sandbox.build_claude_settings). Do not use
|
||||
# Claude's --bare flag here: it disables that trusted
|
||||
# PreToolUse hook along with untrusted hooks/plugins.
|
||||
command_prefix=sandbox.command_prefix,
|
||||
require_pid_namespace=True,
|
||||
bare=True,
|
||||
settings_json=sandbox.settings_json,
|
||||
strict_mcp_config=True,
|
||||
mcp_config_json='{"mcpServers":{}}',
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue