From aea20ccf726f1eca91a1f20ee10a547b4c127cbc Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Thu, 3 Sep 2026 19:44:58 +0000 Subject: [PATCH] fix(workflow-bench): keep proposer hooks and JSONL evidence intact Co-authored-by: Cursor --- eval/tests/test_evolve.py | 124 +++++++++++++++++++++++++++- eval/tests/test_proposer_sandbox.py | 29 ++++++- eval/workflow_bench/evolve.py | 70 +++++++++++++--- 3 files changed, 204 insertions(+), 19 deletions(-) diff --git a/eval/tests/test_evolve.py b/eval/tests/test_evolve.py index 5cddbc0d2..5d1ee8e5d 100644 --- a/eval/tests/test_evolve.py +++ b/eval/tests/test_evolve.py @@ -6,6 +6,7 @@ import os import subprocess import sys import time +from contextlib import contextmanager from datetime import UTC, datetime, timedelta import pytest @@ -205,7 +206,7 @@ def test_proposer_reads_only_digest_bound_transcripts_below_results(tmp_path, mo gate_summary=[], ) - assert entries["transcript-0-0.jsonl"] == payload.decode() + assert [json.loads(line) for line in entries["transcript-0-0.jsonl"].splitlines()] == [json.loads(payload)] assert entries["patch-0.diff"] == patch.read_text() staged_rows = entries["selected-rows.json"] assert staged_rows[0]["patch_file"] == "patch-0.diff" @@ -222,6 +223,61 @@ def test_proposer_reads_only_digest_bound_transcripts_below_results(tmp_path, mo ) +def test_proposer_compacts_transcripts_as_complete_json_events(tmp_path): + results = tmp_path / "results" + transcripts = results / "transcripts" + transcripts.mkdir(parents=True, mode=0o700) + events = [ + { + "type": "assistant", + "message": { + "content": [ + { + "type": "thinking", + "thinking": "analysis-" + ("x" * 100_000), + "signature": "opaque-base64-signature", + } + ] + }, + }, + { + "type": "result", + "session_id": "session-1", + "usage": {"input_tokens": 1, "output_tokens": 2}, + }, + ] + payload = "".join(json.dumps(event) + "\n" for event in events).encode() + artifact = transcripts / "session.jsonl" + artifact.write_bytes(payload) + artifact.chmod(0o600) + + entries = proposer_evidence_entries( + results_dir=results, + evidence=[ + row( + transcript_artifacts=[ + { + "path": "transcripts/session.jsonl", + "sha256": hashlib.sha256(payload).hexdigest(), + "bytes": len(payload), + "source": PARENT_EVENT_STREAM_SOURCE, + } + ] + ) + ], + learnings=[], + gate_summary=[], + artifact_limit=8192, + ) + + staged = entries["transcript-0-0.jsonl"] + parsed = [json.loads(line) for line in staged.splitlines()] + assert len(staged.encode()) <= 8192 + assert parsed[-1]["type"] == "result" + assert parsed[0]["message"]["content"][0]["signature"] == "[OMITTED]" + assert "opaque-base64-signature" not in staged + + @pytest.mark.skipif(os.name == "nt", reason="transcript symlink containment is POSIX-only") def test_proposer_rejects_symlink_and_foreign_transcript_artifacts(tmp_path): results = tmp_path / "results" @@ -412,14 +468,24 @@ def test_stage_proposer_evidence_bundle_drops_prior_proposal_to_fit_budget(tmp_p def test_stage_proposer_evidence_bundle_compacts_artifacts_before_dropping_rows(tmp_path, monkeypatch, capsys): - monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 300_000) - monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 300_000) + monkeypatch.setattr(evolve, "MAX_BUNDLE_BYTES", 150_000) + monkeypatch.setattr("workflow_bench.proposer_sandbox.MAX_BUNDLE_BYTES", 150_000) results = tmp_path / "results" transcripts = results / "transcripts" transcripts.mkdir(parents=True, mode=0o700) rows = [] for index in range(2): - payload = f"session-{index}\n".encode() + b"x" * 100_000 + payload = ( + json.dumps( + { + "type": "assistant", + "message": {"content": [{"type": "text", "text": "x" * 100_000}]}, + } + ) + + "\n" + + json.dumps({"type": "result", "session_id": f"session-{index}"}) + + "\n" + ).encode() transcript = transcripts / f"session-{index}.jsonl" transcript.write_bytes(payload) transcript.chmod(0o600) @@ -494,6 +560,56 @@ def test_proposer_refuses_a_prior_proposal_that_lost_its_trust_boundary(tmp_path ) +def test_run_proposer_keeps_trusted_input_hook_enabled(monkeypatch, tmp_path): + evidence = tmp_path / "evidence" + evidence.mkdir() + transcript_projects = tmp_path / "transcript-projects" + transcript_projects.mkdir() + captured: dict[str, object] = {} + + def fake_make_worktree(_repo, _ref, destination): + clone = destination / "clone" + clone.mkdir() + return clone + + class FakeSandbox: + claude_bin = "claude" + command_prefix: list[str] = [] + settings_json = '{"hooks":{"PreToolUse":[]}}' + + @property + def transcript_projects(self): + return transcript_projects + + @contextmanager + def fake_prepare_sandbox(**_kwargs): + yield FakeSandbox() + + def fake_run_claude(*_args, **kwargs): + captured.update(kwargs) + return {"ok": False, "error_kind": "session-error"} + + monkeypatch.setattr(evolve.runner, "make_worktree", fake_make_worktree) + monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None) + monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox) + monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude) + args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"]) + + record = evolve.run_proposer( + "prompt", + args, + overlay_dir=tmp_path / "overlay", + proposal_path=tmp_path / "proposal.md", + evidence_bundle=evidence, + bwrap_bin=tmp_path / "bwrap", + ) + + assert record["ok"] is False + assert captured.get("bare", False) is False + assert captured["allowed_tools"] == evolve.PROPOSER_ALLOWED_TOOLS + assert captured["settings_json"] == FakeSandbox.settings_json + + def test_parser_defaults_match_the_gate_minimums(): args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"]) assert args.runs == 3 diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index 29767d6b7..6c69c53cb 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -883,13 +883,14 @@ def test_clone_controlled_mcp_replacement_is_never_executed_or_credentialed(tmp_ os.environ.get("GITNEXUS_REQUIRE_CLAUDE_CANARY") != "1", reason="real Claude/Bash/MCP canary is mandatory in the named Ubuntu CI job", ) -def test_real_claude_bare_auth_inner_sandbox_and_mcp_permissions(tmp_path: Path) -> None: +def test_real_claude_hooks_auth_inner_sandbox_and_mcp_permissions(tmp_path: Path) -> None: """Exercise the exact CLI boundary without contacting a paid model.""" claude = Path(os.environ["CLAUDE_CANARY_BIN"]).resolve() assert claude.is_file() clone = tmp_path / "clone" clone.mkdir() + (clone / "canary.txt").write_text("hook-readable") fake_mcp = clone / "fake_mcp.py" fake_mcp.write_text( """import json @@ -954,7 +955,20 @@ for line in sys.stdin: and isinstance(block.get("tool_use_id"), str) } ) - if "toolu_mcp_canary" not in tool_result_ids: + if "toolu_read_canary" not in tool_result_ids: + blocks = [ + { + "type": "tool_use", + "id": "toolu_read_canary", + "name": "Read", + "input": { + "file_path": "/workspace/canary.txt", + "pages": "", + }, + } + ] + stop_reason = "tool_use" + elif "toolu_mcp_canary" not in tool_result_ids: blocks = [ { "type": "tool_use", @@ -1076,7 +1090,6 @@ for line in sys.stdin: "text", "--output-format", "json", - "--bare", "--settings", sandbox.settings_json, "--strict-mcp-config", @@ -1088,7 +1101,12 @@ for line in sys.stdin: # authoritative empirical gate for that behavior. "--model", "claude-canary-20260718", + "--tools", + "Read", + "Bash", + "mcp__gitnexus__list_repos", "--allowedTools", + "Read", "Bash", "mcp__gitnexus__list_repos", ], @@ -1097,7 +1115,7 @@ for line in sys.stdin: auth_token="offline-canary-key", base_url=f"http://127.0.0.1:{server.server_port}", ), - stdin_data=b"Use both available tools, then finish.", + stdin_data=b"Use all three available tools, then finish.", ) finally: server.shutdown() @@ -1107,6 +1125,9 @@ for line in sys.stdin: assert result.ok, result.stderr_tail + result.stdout_tail report = json.loads(result.stdout_tail) assert report["subtype"] == "success" and report["is_error"] is False, report + read_result = observed_tool_results["toolu_read_canary"] + assert read_result.get("is_error") is not True, read_result + assert "hook-readable" in json.dumps(read_result) bash_result = observed_tool_results["toolu_bash_canary"] assert bash_result.get("is_error") is not True, bash_result assert (clone / "bash-called").read_text() == "canary" diff --git a/eval/workflow_bench/evolve.py b/eval/workflow_bench/evolve.py index 2dbd25d0a..a9c9c977a 100644 --- a/eval/workflow_bench/evolve.py +++ b/eval/workflow_bench/evolve.py @@ -451,8 +451,6 @@ def _bound_transcript_artifact( while chunk := os.read(descriptor, 64 * 1024): digest.update(chunk) content.extend(chunk) - if len(content) > limit: - del content[: len(content) - limit] after = os.fstat(descriptor) if (opened.st_size, opened.st_mtime_ns) != (after.st_size, after.st_mtime_ns): raise SandboxError(f"transcript artifact changed while reading: {path}") @@ -460,7 +458,57 @@ def _bound_transcript_artifact( os.close(descriptor) if digest.hexdigest() != expected_digest: raise SandboxError(f"transcript artifact digest does not match its results row: {path}") - return bytes(content).decode(errors="replace") + return _compact_transcript_jsonl(bytes(content), limit) + + +def _compact_transcript_value(value: Any, *, key: str | None = None) -> Any: + """Bound large event fields while retaining valid, useful JSON.""" + + if key == "signature": + return "[OMITTED]" + if isinstance(value, str): + field_limit = 4096 + if len(value) <= field_limit: + return value + half = field_limit // 2 + return f"{value[:half]}…[compacted {len(value) - field_limit} chars]…{value[-half:]}" + if isinstance(value, list): + return [_compact_transcript_value(item) for item in value] + if isinstance(value, dict): + return {str(item_key): _compact_transcript_value(item, key=str(item_key)) for item_key, item in value.items()} + return value + + +def _compact_transcript_jsonl(raw: bytes, limit: int) -> str: + """Select complete recent events; never cut through a JSON record.""" + + try: + source_events = [json.loads(line) for line in raw.decode("utf-8", errors="strict").splitlines() if line.strip()] + except (UnicodeError, json.JSONDecodeError) as exc: + raise SandboxError(f"transcript artifact is not valid JSONL: {exc}") from exc + + selected: list[bytes] = [] + total = 0 + for event in reversed(source_events): + encoded = ( + json.dumps( + _compact_transcript_value(event), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ) + + "\n" + ).encode("utf-8") + if len(encoded) > limit or total + len(encoded) > limit: + continue + selected.append(encoded) + total += len(encoded) + + if not selected: + raise SandboxError("transcript artifact has no complete event within the evidence limit") + selected.reverse() + return b"".join(selected).decode("utf-8") def _prior_proposal_text(path: Path) -> str: @@ -590,12 +638,11 @@ def stage_proposer_evidence_bundle( # The proposer's exact tool surface. Read/Grep/Glob observe the read-only # evidence bundle and the incumbent skills; Bash writes the candidate overlay. -# The proposer session runs --bare, which hard-disables the Write/Edit tools -# ("Write exists but is not enabled in this context"), so Bash is the only -# writable tool it enables — the settings pre-authorize it via -# autoAllowBashIfSandboxed, and the sandbox filesystem policy confines writes to -# the workspace/tmp/home. Exported so the containment canary tests the real -# allowlist and cannot drift from production. +# `--tools` restricts non-bare Claude to this list, so Write/Edit/Skill/Web are +# unavailable while the trusted PreToolUse normalizer remains enabled. Settings +# pre-authorize Bash via autoAllowBashIfSandboxed, and the sandbox filesystem +# policy confines writes to workspace/tmp/home. Exported so containment tests +# exercise the production allowlist without drift. PROPOSER_ALLOWED_TOOLS = ["Read", "Grep", "Glob", "Bash"] @@ -646,10 +693,11 @@ def run_proposer( # No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB # forces "default", so requesting dontAsk only warns. Tools # are pre-approved via settings permissions.allow - # (proposer_sandbox.build_claude_settings). + # (proposer_sandbox.build_claude_settings). Do not use + # Claude's --bare flag here: it disables that trusted + # PreToolUse hook along with untrusted hooks/plugins. command_prefix=sandbox.command_prefix, require_pid_namespace=True, - bare=True, settings_json=sandbox.settings_json, strict_mcp_config=True, mcp_config_json='{"mcpServers":{}}',