mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-10 03:27:59 +00:00
fix(eval): close CI and remaining review gaps
This commit is contained in:
parent
a957c5d757
commit
3598a69188
11 changed files with 135 additions and 53 deletions
|
|
@ -8,6 +8,7 @@ dependencies = [
|
|||
"mini-swe-agent>=2.0.0",
|
||||
"litellm[proxy]!=1.82.7,!=1.82.8,>=1.99.0",
|
||||
"cryptography>=50.0.0",
|
||||
"python-multipart>=0.0.30",
|
||||
"datasets>=3.0.0",
|
||||
"typer>=0.12.0",
|
||||
"rich>=13.0.0",
|
||||
|
|
|
|||
|
|
@ -32,6 +32,35 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed
|
|||
from workflow_bench.proposer_sandbox import pid_namespace_command, preflight_bubblewrap
|
||||
|
||||
|
||||
def test_runner_environment_does_not_forward_the_openai_key() -> None:
|
||||
from workflow_bench.model_gateway import credential_secrets
|
||||
|
||||
args = build_parser().parse_args(
|
||||
[
|
||||
"--tasks",
|
||||
"t.yaml",
|
||||
"--model",
|
||||
"gpt-4.1",
|
||||
"--anthropic-api-key",
|
||||
"loopback-master",
|
||||
"--openai-api-key",
|
||||
"sk-openai-secret",
|
||||
]
|
||||
)
|
||||
env = evolve.runner_environment(args)
|
||||
assert env["GITNEXUS_BENCH_ANTHROPIC_API_KEY"] == "loopback-master"
|
||||
assert "GITNEXUS_BENCH_AUTH_TOKEN" not in env
|
||||
assert "OPENAI_API_KEY" not in env
|
||||
assert "GITNEXUS_BENCH_OPENAI_API_KEY" not in env
|
||||
assert "sk-openai-secret" not in env.values()
|
||||
assert credential_secrets(args) == ["loopback-master", "sk-openai-secret"]
|
||||
|
||||
|
||||
def test_parser_keeps_the_legacy_auth_token_alias() -> None:
|
||||
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned", "--auth-token", "alias-secret"])
|
||||
assert args.auth_token == "alias-secret"
|
||||
|
||||
|
||||
def row(**overrides):
|
||||
base = {
|
||||
"task": "demo-task",
|
||||
|
|
|
|||
|
|
@ -27,7 +27,6 @@ from workflow_bench.model_gateway import (
|
|||
gateway_ready_timeout_s,
|
||||
anthropic_api_key_from_environ,
|
||||
claude_gateway_model_env,
|
||||
credential_secrets,
|
||||
is_openai_model,
|
||||
litellm_proxy_argv,
|
||||
openai_backend_model,
|
||||
|
|
@ -35,7 +34,6 @@ from workflow_bench.model_gateway import (
|
|||
resolve_model_access,
|
||||
write_openai_litellm_config,
|
||||
)
|
||||
from workflow_bench.evolve import build_parser, runner_environment
|
||||
|
||||
|
||||
def test_locked_litellm_translates_messages_to_offline_responses(monkeypatch, tmp_path):
|
||||
|
|
@ -196,6 +194,7 @@ time.sleep(60)
|
|||
else:
|
||||
os.killpg(int(proxy_pid.read_text()), signal.SIGKILL)
|
||||
except OSError:
|
||||
# Successful ownership cleanup has already removed this process.
|
||||
pass
|
||||
|
||||
|
||||
|
|
@ -358,28 +357,6 @@ def test_gateway_readiness_timeout_reports_the_proxy_log_and_the_override(tmp_pa
|
|||
assert GATEWAY_READY_TIMEOUT_ENV in message
|
||||
|
||||
|
||||
def test_runner_environment_does_not_forward_the_openai_key() -> None:
|
||||
args = build_parser().parse_args(
|
||||
[
|
||||
"--tasks",
|
||||
"t.yaml",
|
||||
"--model",
|
||||
"gpt-4.1",
|
||||
"--anthropic-api-key",
|
||||
"loopback-master",
|
||||
"--openai-api-key",
|
||||
"sk-openai-secret",
|
||||
]
|
||||
)
|
||||
env = runner_environment(args)
|
||||
assert env["GITNEXUS_BENCH_ANTHROPIC_API_KEY"] == "loopback-master"
|
||||
assert "GITNEXUS_BENCH_AUTH_TOKEN" not in env
|
||||
assert "OPENAI_API_KEY" not in env
|
||||
assert "GITNEXUS_BENCH_OPENAI_API_KEY" not in env
|
||||
assert "sk-openai-secret" not in env.values()
|
||||
assert credential_secrets(args) == ["loopback-master", "sk-openai-secret"]
|
||||
|
||||
|
||||
def test_openai_backend_model_preserves_openai_prefix() -> None:
|
||||
assert openai_backend_model("gpt-4.1") == "openai/gpt-4.1"
|
||||
assert openai_backend_model("openai/gpt-4.1") == "openai/gpt-4.1"
|
||||
|
|
@ -420,7 +397,3 @@ def test_anthropic_api_key_prefers_the_named_env_and_keeps_the_legacy_alias(monk
|
|||
assert anthropic_api_key_from_environ() == "legacy-secret"
|
||||
monkeypatch.setenv("GITNEXUS_BENCH_ANTHROPIC_API_KEY", "named-secret")
|
||||
assert anthropic_api_key_from_environ() == "named-secret"
|
||||
args = build_parser().parse_args(
|
||||
["--tasks", "t.yaml", "--model", "pinned", "--auth-token", "alias-secret"]
|
||||
)
|
||||
assert args.auth_token == "alias-secret"
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ import hashlib
|
|||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
|
|
@ -374,6 +375,26 @@ def test_vacuous_authored_test_cannot_self_certify_resolution(
|
|||
assert record["error_kind"] == "oracle-failed"
|
||||
|
||||
|
||||
def test_host_unsafe_oracle_executes_beside_candidate_and_is_removed(tmp_path: Path) -> None:
|
||||
from workflow_bench.proposer_sandbox import prepare_sandbox
|
||||
|
||||
source = tmp_path / "oracles"
|
||||
write_oracle(
|
||||
source,
|
||||
b"from pathlib import Path\nassert (Path(__file__).resolve().parents[2] / 'candidate.txt').read_text() == 'candidate'\n",
|
||||
)
|
||||
snapshot = capture_task_oracle(
|
||||
oracle_task(command='python3 "$GITNEXUS_BENCH_ORACLE_ROOT/nested/oracle.test.ts"'), root=source
|
||||
)
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
(clone / "candidate.txt").write_text("candidate")
|
||||
with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as session:
|
||||
passed, detail = runner._run_hidden_oracle(snapshot, clone, bench_args(), session)
|
||||
assert passed, detail
|
||||
assert not list(clone.glob(".wfbench-oracle-*"))
|
||||
|
||||
|
||||
def test_oracle_path_and_bytes_appear_only_after_the_model_session(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
tmp_path: Path,
|
||||
|
|
|
|||
|
|
@ -118,6 +118,7 @@ print(json.dumps({{'rows': rows, 'stopped': stopped}}))
|
|||
try:
|
||||
os.killpg(int(ready.read_text()), signal.SIGKILL)
|
||||
except ProcessLookupError:
|
||||
# Successful cancellation has already reaped this process group.
|
||||
pass
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1348,6 +1348,8 @@ PY"""
|
|||
}
|
||||
)
|
||||
with prepare_sandbox(clone=clone, claude_bin=claude, preflight=True) as sandbox:
|
||||
output = None
|
||||
before = {}
|
||||
if review_layout:
|
||||
output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json")
|
||||
before = runner_artifacts.workspace_snapshot(clone)
|
||||
|
|
@ -1393,8 +1395,17 @@ PY"""
|
|||
),
|
||||
stdin_data=b"Use all three available tools, then finish.",
|
||||
)
|
||||
assert result.ok, result.stderr_tail + result.stdout_tail
|
||||
report = json.loads(result.stdout_tail)
|
||||
assert report["subtype"] == "success" and report["is_error"] is False, report
|
||||
read_result = observed_tool_results["toolu_read_canary"]
|
||||
assert read_result.get("is_error") is not True, read_result
|
||||
assert "hook-readable" in json.dumps(read_result)
|
||||
bash_result = observed_tool_results["toolu_bash_canary"]
|
||||
assert bash_result.get("is_error") is not True, bash_result
|
||||
assert (sandbox.temp / "mcp-called").read_text() == "ok"
|
||||
if review_layout:
|
||||
assert output is not None
|
||||
runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=output)
|
||||
assert json.loads(output.read_text())["verdict"] == "approve"
|
||||
assert (clone / "canary.txt").read_text() == "hook-readable\nchanged for review\n"
|
||||
|
|
@ -1403,13 +1414,5 @@ PY"""
|
|||
server.server_close()
|
||||
thread.join(timeout=5)
|
||||
|
||||
assert result.ok, result.stderr_tail + result.stdout_tail
|
||||
report = json.loads(result.stdout_tail)
|
||||
assert report["subtype"] == "success" and report["is_error"] is False, report
|
||||
read_result = observed_tool_results["toolu_read_canary"]
|
||||
assert read_result.get("is_error") is not True, read_result
|
||||
assert "hook-readable" in json.dumps(read_result)
|
||||
bash_result = observed_tool_results["toolu_bash_canary"]
|
||||
assert bash_result.get("is_error") is not True, bash_result
|
||||
if not review_layout:
|
||||
assert (clone / "bash-called").read_text() == "canary"
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""Live progress reporting for long headless sessions.
|
||||
|
||||
The session event stream is evidence and is redacted before it is written
|
||||
anywhere, so progress may only report metadata derived from it. These tests pin
|
||||
Progress includes event metadata and bounded, redacted tool argument/result
|
||||
previews. Model prose and raw event streams are never echoed. These tests pin
|
||||
that boundary along with the signals that distinguish work from a wedged run.
|
||||
"""
|
||||
|
||||
|
|
@ -23,6 +23,46 @@ def _observe(progress: SessionProgress, chunk: bytes) -> None:
|
|||
progress._emit_pending()
|
||||
|
||||
|
||||
def test_progress_bounds_unanswered_tools_and_undrained_messages() -> None:
|
||||
progress = SessionProgress("bounded", stream=io.StringIO())
|
||||
for index in range(2000):
|
||||
progress.observe(
|
||||
(
|
||||
json.dumps(
|
||||
{
|
||||
"type": "assistant",
|
||||
"message": {
|
||||
"content": [
|
||||
{"type": "tool_use", "id": str(index), "name": "Bash", "input": {"command": "true"}}
|
||||
]
|
||||
},
|
||||
}
|
||||
)
|
||||
+ "\n"
|
||||
).encode()
|
||||
)
|
||||
assert len(progress._pending_tools) <= 256
|
||||
assert len(progress._pending_messages) <= 256
|
||||
assert "1999" in progress._pending_tools
|
||||
assert "0" not in progress._pending_tools
|
||||
progress.observe(
|
||||
(
|
||||
json.dumps(
|
||||
{
|
||||
"type": "user",
|
||||
"message": {
|
||||
"content": [{"type": "tool_result", "tool_use_id": "1999", "content": "recent result"}]
|
||||
},
|
||||
}
|
||||
)
|
||||
+ "\n"
|
||||
).encode()
|
||||
)
|
||||
progress._emit_pending()
|
||||
assert "recent result" in progress._stream.getvalue()
|
||||
assert "1999" not in progress._pending_tools
|
||||
|
||||
|
||||
def test_progress_reports_bounded_redacted_tool_io_but_never_model_prose() -> None:
|
||||
stream = io.StringIO()
|
||||
progress = SessionProgress(
|
||||
|
|
|
|||
8
eval/uv.lock
generated
8
eval/uv.lock
generated
|
|
@ -976,6 +976,7 @@ dependencies = [
|
|||
{ name = "mini-swe-agent" },
|
||||
{ name = "pandas" },
|
||||
{ name = "python-dotenv" },
|
||||
{ name = "python-multipart" },
|
||||
{ name = "pyyaml" },
|
||||
{ name = "rich" },
|
||||
{ name = "tabulate" },
|
||||
|
|
@ -1001,6 +1002,7 @@ requires-dist = [
|
|||
{ name = "pandas", specifier = ">=2.0.0" },
|
||||
{ name = "pytest", marker = "extra == 'dev'", specifier = ">=9.0.3" },
|
||||
{ name = "python-dotenv", specifier = ">=1.2.2" },
|
||||
{ name = "python-multipart", specifier = ">=0.0.30" },
|
||||
{ name = "pyyaml", specifier = ">=6.0" },
|
||||
{ name = "rich", specifier = ">=13.0.0" },
|
||||
{ name = "ruff", marker = "extra == 'dev'", specifier = ">=0.5.0" },
|
||||
|
|
@ -2618,11 +2620,11 @@ wheels = [
|
|||
|
||||
[[package]]
|
||||
name = "python-multipart"
|
||||
version = "0.0.27"
|
||||
version = "0.0.32"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/69/9b/f23807317a113dc36e74e75eb265a02dd1a4d9082abc3c1064acd22997c4/python_multipart-0.0.27.tar.gz", hash = "sha256:9870a6a8c5a20a5bf4f07c017bd1489006ff8836cff097b6933355ee2b49b602", size = 44043, upload-time = "2026-04-27T10:51:26.649Z" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/5b/42/55c32bb9b12693c092ad250a0e82edb5b31ddeda6eb772de5f308b3804ad/python_multipart-0.0.32.tar.gz", hash = "sha256:be54b7f3fa167bb83e4fcd936b887b708f4e57fe75911c02aebf53efaf8d938e", size = 46881, upload-time = "2026-06-04T16:18:58.647Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/99/78/4126abcbdbd3c559d43e0db7f7b9173fc6befe45d39a2856cc0b8ec2a5a6/python_multipart-0.0.27-py3-none-any.whl", hash = "sha256:6fccfad17a27334bd0193681b369f476eda3409f17381a2d65aa7df3f7275645", size = 29254, upload-time = "2026-04-27T10:51:24.997Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/e1/04/e8135ebd1ad02c56ec633277529b2602ff99ff634be76cdba5744cf554fd/python_multipart-0.0.32-py3-none-any.whl", hash = "sha256:ff6d3f776f16878c894e52e107296ffc890e913c611b1a4ec6c44e2821fe2e23", size = 30042, upload-time = "2026-06-04T16:18:57.319Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
|
|
|||
|
|
@ -315,14 +315,18 @@ def _run_hidden_oracle(
|
|||
mount_point.mkdir(mode=0o700)
|
||||
primary: BaseException | None = None
|
||||
try:
|
||||
with staged_task_oracle(sandbox.private_root, snapshot) as stage_root:
|
||||
host_unsafe = getattr(sandbox, "backend", "bwrap") == "host-unsafe"
|
||||
# Host mode has no bind mounts. Keep relative candidate imports valid
|
||||
# by staging beside the candidate, still only after the model exits.
|
||||
stage_parent = worktree if host_unsafe else sandbox.private_root
|
||||
with staged_task_oracle(stage_parent, snapshot) as stage_root:
|
||||
oracle_env = build_sandbox_environment()
|
||||
# A private RO bind at a random workspace sibling preserves each
|
||||
# oracle's ../gitnexus import as the candidate implementation. The
|
||||
# empty mountpoint exists only post-model and is removed before the
|
||||
# credited patch is captured.
|
||||
oracle_mount = f"{SANDBOX_WORKSPACE}/{mount_name}"
|
||||
oracle_env[ORACLE_ENV_VAR] = oracle_mount
|
||||
oracle_env[ORACLE_ENV_VAR] = str(stage_root) if host_unsafe else oracle_mount
|
||||
passed, _output = _verification_outcome(
|
||||
run_verify(
|
||||
snapshot.command,
|
||||
|
|
@ -331,7 +335,9 @@ def _run_hidden_oracle(
|
|||
command_prefix=sandbox.command_prefix_for(
|
||||
read_only_workspace=True,
|
||||
unshare_network=True,
|
||||
extra_read_only_mounts=(ReadOnlyMount(source=stage_root, target=oracle_mount),),
|
||||
extra_read_only_mounts=()
|
||||
if host_unsafe
|
||||
else (ReadOnlyMount(source=stage_root, target=oracle_mount),),
|
||||
),
|
||||
env=oracle_env,
|
||||
require_pid_namespace=getattr(sandbox, "require_pid_namespace", True),
|
||||
|
|
@ -754,7 +760,7 @@ def _run_wave(
|
|||
for run_idx, arm in wave:
|
||||
futures.append(pool.submit(copy_context().run, run, run_idx, arm))
|
||||
wait(futures)
|
||||
except BaseException as exc:
|
||||
except (Exception, KeyboardInterrupt, SystemExit) as exc:
|
||||
interruption = exc
|
||||
cancel_event.set()
|
||||
finally:
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ import stat
|
|||
import sys
|
||||
import threading
|
||||
import time
|
||||
from collections import deque
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path, PurePosixPath
|
||||
from typing import Any
|
||||
|
|
@ -52,6 +53,8 @@ PARENT_EVENT_STREAM_SOURCE = "parent-captured-stream-json"
|
|||
# reporter also speaks up on its own to distinguish "thinking" from "wedged".
|
||||
PROGRESS_HEARTBEAT_SECONDS = 60.0
|
||||
MAX_PROGRESS_LINE_BYTES = 1024 * 1024
|
||||
MAX_PROGRESS_PENDING = 256
|
||||
MAX_PROGRESS_TOOL_ID_CHARS = 256
|
||||
MAX_TOOL_PREVIEW_CHARS = 800
|
||||
_SAFE_TOOL_NAME = re.compile(r"[A-Za-z0-9._:-]{1,64}")
|
||||
|
||||
|
|
@ -116,13 +119,12 @@ def _tool_result_log_status(name: str, block: dict[str, Any]) -> str:
|
|||
|
||||
|
||||
class SessionProgress:
|
||||
"""Narrate a live Claude session without ever echoing its output.
|
||||
"""Report session metadata and bounded, redacted tool I/O previews.
|
||||
|
||||
A session's stdout is evidence: it is redacted before anything is written
|
||||
out, so it can never be streamed to the log. This reports only what the
|
||||
parent can derive safely — turn counts, tool names, API retries, and how
|
||||
long the session has been quiet — which is what tells a watcher whether a
|
||||
long run is working or stuck.
|
||||
out, so raw events and model prose are never streamed to the log. Turn
|
||||
counts, tool activity, API retries, and quiet time distinguish work from
|
||||
a wedged session. Tool arguments and results are content, not metadata.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
|
|
@ -150,7 +152,7 @@ class SessionProgress:
|
|||
self._tools = 0
|
||||
self._pending_tools: dict[str, str] = {}
|
||||
self._last_activity = "starting"
|
||||
self._pending_messages: list[str] = []
|
||||
self._pending_messages: deque[str] = deque(maxlen=MAX_PROGRESS_PENDING)
|
||||
self._timer: threading.Thread | None = None
|
||||
self._done = threading.Event()
|
||||
|
||||
|
|
@ -237,8 +239,11 @@ class SessionProgress:
|
|||
self._say(f"turn {self._turns} · {self._last_activity}")
|
||||
for block, name in zip(uses, names, strict=True):
|
||||
tool_id = block.get("id")
|
||||
if isinstance(tool_id, str):
|
||||
if isinstance(tool_id, str) and len(tool_id) <= MAX_PROGRESS_TOOL_ID_CHARS:
|
||||
self._pending_tools.pop(tool_id, None)
|
||||
self._pending_tools[tool_id] = name
|
||||
if len(self._pending_tools) > MAX_PROGRESS_PENDING:
|
||||
del self._pending_tools[next(iter(self._pending_tools))]
|
||||
if _debuggable_tool(name):
|
||||
self._say(f"tool {name} input={_tool_preview(block.get('input'), self._secrets)}")
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -261,7 +261,8 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
|
|||
label="primary GitNexus shared runtime",
|
||||
)
|
||||
except SandboxError:
|
||||
primary_shared = None
|
||||
# Keep the already validated local shared runtime when primary is unavailable.
|
||||
pass
|
||||
else:
|
||||
reuse_primary_shared = node_modules.source == (primary / "gitnexus" / "node_modules")
|
||||
if not reuse_primary_shared:
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue