Merge current main CI and indexing fixes into fallback split

This commit is contained in:
Abhinav Pandey 2026-09-09 06:23:46 +05:30
commit c9cb49038f
No known key found for this signature in database
91 changed files with 10852 additions and 617 deletions

View file

@ -587,6 +587,18 @@ jobs:
run: node --import tsx bench/finalize-reexport/measure.mjs --check
working-directory: gitnexus
- name: Parse dispatch-round cadence guards (#3194, #3196)
if: ${{ !cancelled() }}
# Build-free: asserts parse-cache pack membership is unchanged
# (fingerprint — every cache key derives from it), that a fixed corpus
# still batches into a fixed number of dispatch rounds, and that the
# round budget counts UTF-8 bytes rather than UTF-16 code units. Round
# boundaries are deliberately invisible to graph output, so no test can
# see these regress. Rationale and history: see the header of
# bench/parse-dispatch-rounds/measure.mjs.
run: node --import tsx bench/parse-dispatch-rounds/measure.mjs --check
working-directory: gitnexus
- name: C++ qualified-namespace resolution guards (#2788)
if: ${{ !cancelled() }}
# Build-free: asserts resolveCppQualifiedNamespaceMember resolves an
@ -745,6 +757,13 @@ jobs:
run: node --import tsx bench/scope-emission/measure.mjs --check
working-directory: gitnexus
- name: Zig cross-file static-gating guards (#3162)
if: ${{ !cancelled() }}
# Build-free: fingerprints cross-file dead-call classification and
# guards the workspace enrichment pass across file-count scaling.
run: node --import tsx bench/zig-cross-file-resolution/measure.mjs --check
working-directory: gitnexus
- name: CFG construction time / disk / memory guards (#2081 M1)
if: ${{ !cancelled() }}
# Build-free: asserts collectFunctionCfgs output is unchanged

View file

@ -63,13 +63,17 @@
# uploads, and a promotion (if any) opens a well-formed PR. Run
# 29907431284 (2026-07-22) went green end to end in 14h45m and reached a
# gate decision (`insufficient_evidence`, no promotion).
# [ ] After resizing the runner, prove a manual workers=3 run has zero excluded
# runs and does not stretch the 48-minute serial mean toward the session
# ceiling; then set GITNEXUS_EVOLUTION_WORKERS=3 and
# [ ] Confirm a workers=3 dispatch has zero excluded runs (review sessions in
# 33962002890 averaged ~19m serial, well under the 90m session ceiling).
# Then set GITNEXUS_EVOLUTION_WORKERS=3 and
# GITNEXUS_EVOLUTION_ENABLED=true for scheduled runs. Scheduled runs
# require both values, so leaving workers unset/1 is an immediate rollback;
# workflow_dispatch remains available for the proof and bills real API
# usage on GITNEXUS_BENCH_ANTHROPIC_API_KEY or GITNEXUS_BENCH_OPENAI_API_KEY.
# require both values, so leaving the var unset is an immediate rollback.
# Dispatch defaults to 3; pass workers=1 only to debug a contended host.
# Weekly generations reuse matching incumbent/CE cells from the previous
# artifact so the paid matrix is the new candidate, not a 54-cell replay.
# Wall clock is quantised by ceil(cells_per_task / workers), and a review
# task is 9 cells cold, so 4 costs host contention for exactly the wall
# clock of 3. The next step up that buys anything is 5 (3 waves -> 2).
name: GitNexus skill evolution
on:
@ -92,9 +96,9 @@ on:
default: '3'
type: string
workers:
description: 'Benchmark cells of one task to run at once — raise only to match the runner’s vCPUs'
description: 'Benchmark cells of one task to run at once — 3 fits the evolution box; drop to 1 only if siblings hit the session ceiling'
required: false
default: '1'
default: '3'
type: string
model:
description: 'Model for the benchmark arms (match the model your skill users run)'
@ -172,6 +176,13 @@ jobs:
# stops the runner just disappears mid-step. Scheduled runs can start well
# after the cron (the 2026-08-01 run was queued 65min late), so the job
# budget has to absorb that delay and still land inside the uptime window.
# A Friday workflow_dispatch on a box that already booted for Saturday's
# cron inherits leftover uptime, not a fresh 24h. Run 33962002890 started
# Friday 10:57 UTC and vanished at the Saturday 03:00 stop — 51 finished
# sessions never uploaded. run-evolution.sh therefore passes
# --max-runtime-from-instance-window, and the CLI derives its cap from
# /proc/uptime at startup, so the sweep fails in-process and this always()
# upload still runs.
timeout-minutes: 1260
permissions:
contents: read # The promotion PR uses a short-lived App token minted below.

View file

@ -26,7 +26,7 @@
</a>
</p>
<p><strong>The nervous system for agent context.</strong></p>
<p><strong>The context engine for Enterprise Codebases</strong></p>
<p>
Indexes any codebase into a knowledge graph — every dependency, call chain, cluster, and execution flow —
@ -102,6 +102,10 @@ The proxy strips `Origin` before forwarding, so the server's CSRF guard does not
Indexing is memory-bound. If `gitnexus-server` runs out of memory on a large repo, raise its `plan`, which sets available RAM: `standard` is 2 GB, `pro` is 4 GB. Raise `sizeGB` only if the disk fills with clones and indexes.
### Deploy to RepoCloud
[![Deploy on RepoCloud](https://d16t0pc4846x52.cloudfront.net/deploylobe.svg)](https://repocloud.io/details/gitnexus/)
## Two Ways to Use GitNexus
| | **CLI + MCP** (recommended) | **Web UI** |

1
eval/.gitignore vendored
View file

@ -14,3 +14,4 @@ build/
# Environment
.env
.venv/
.venv

View file

@ -0,0 +1,75 @@
"""Shared row shapes for the sweep tests.
Building the finalization tests turned up what a real scored review row must
carry: the report renders the whole review metric set, so an incomplete row
fails in string formatting rather than in the logic under test. That is a
property of the fixture, not of production - the shape lives here once so each
test does not rediscover it.
"""
from __future__ import annotations
from typing import Any
def scored_review_row(**overrides: Any) -> dict[str, Any]:
"""One admissible review cell, with zero-valued metrics written out."""
row: dict[str, Any] = {
"ok": True,
"error_kind": None,
"error_detail": None,
"resolved": True,
"review_evidence_valid": True,
"review_score": {"weighted_f1": 0.5},
"review_weighted_f1": 0.5,
"review_true_positives": 1,
"review_false_positives": 0,
"review_false_negatives": 0,
"review_precision": 0.5,
"review_recall": 0.5,
"review_f1": 0.5,
"review_weighted_precision": 0.5,
"review_weighted_recall": 0.5,
"review_blocker_recall": 1.0,
"review_severity_accuracy": 1.0,
"review_category_accuracy": 1.0,
"review_grounded_evidence": 1.0,
"review_verdict_correct": True,
"review_clean_control": True,
"review_clean_pass": True,
"transcript_missing": False,
"transcript_artifacts": [],
"num_turns": 3,
"duration_s": 1.0,
"cost_usd": 0.5,
"input_tokens": 1,
"output_tokens": 1,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"diff_files": 0,
"diff_insertions": 0,
"diff_deletions": 0,
}
row.update(overrides)
return row
def unusable_review_row(**overrides: Any) -> dict[str, Any]:
"""A cell that ran but produced evidence nothing can be scored from."""
# Merged into one mapping rather than passed as explicit keywords beside
# **overrides: Python rejects a duplicate keyword in the call expression
# itself, so unusable_review_row(error_kind=...) raised TypeError before
# scored_review_row could apply the override this helper advertises.
return scored_review_row(
**{
"ok": False,
"resolved": False,
"review_evidence_valid": False,
"error_kind": "review-evidence-invalid",
"review_score": None,
"review_weighted_f1": None,
**overrides,
}
)

View file

@ -0,0 +1,467 @@
"""Comparator-row reuse: skip unchanged incumbent/CE cells, never candidates."""
from __future__ import annotations
import hashlib
import os
from datetime import UTC, datetime, timedelta
from pathlib import Path
import pytest
from workflow_bench import comparator_reuse
from workflow_bench.comparator_reuse import (
ComparatorReuseExpectation,
TaskReuseBinding,
materialize_reused_row,
row_is_reusable_comparator,
select_reusable_comparator_rows,
)
from workflow_bench.proposer_sandbox import SandboxError
from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE
requires_openat = pytest.mark.skipif(
os.open not in os.supports_dir_fd,
reason="comparator reuse resolves every artifact against a pinned directory descriptor",
)
def _digest(text: str = "blob") -> str:
return hashlib.sha256(text.encode()).hexdigest()
def _artifact(name: str = "session-1.jsonl", payload: bytes = b'{"type":"ok"}\n') -> dict:
return {
"path": f"transcripts/{name}",
"sha256": hashlib.sha256(payload).hexdigest(),
"bytes": len(payload),
"source": PARENT_EVENT_STREAM_SOURCE,
}
def _row(**overrides) -> dict:
base = {
"task": "review-pr-2718-defect",
"arm": "review",
"run": 0,
"ok": True,
"error_kind": None,
"model": "gpt-5.6-sol",
"benchmark_model": "gpt-5.6-sol",
"effort": "xhigh",
"sandbox_backend": "bwrap",
"task_base_sha": "a" * 40,
"task_prompt_digest": _digest("prompt"),
"oracle_digest": _digest("oracle"),
"oracle_command_digest": _digest("oracle-cmd"),
"oracle_manifest_digest": _digest("oracle-man"),
"skill_digest": _digest("skill"),
"candidate_overlay_digest": None,
"review_evidence_valid": True,
# Production sets this whenever the review source exists, which is the
# normal path for a valid review; the fixture predated the requirement.
"review_artifact": "review-pr-2718-defect-review-run0.review.json",
"review_score": {"weighted_f1": 0.4},
"review_weighted_f1": 0.4,
"transcript_missing": False,
"transcript_artifacts": [_artifact()],
"recorded_at": datetime.now(UTC).isoformat(),
"runtime_digest": _digest("cli"),
"task_asset_manifest_digest": _digest("assets"),
"sandbox_dependency_manifest_digest": _digest("deps"),
}
base.update(overrides)
return base
def _expected(**overrides) -> ComparatorReuseExpectation:
now = datetime.now(UTC)
values = dict(
model="gpt-5.6-sol",
effort="xhigh",
sandbox_backend="bwrap",
runtime_digest=_digest("cli"),
now=now,
max_age=timedelta(days=90),
tasks={
"review-pr-2718-defect": TaskReuseBinding(
task_base_sha="a" * 40,
task_prompt_digest=_digest("prompt"),
oracle_digest=_digest("oracle"),
oracle_command_digest=_digest("oracle-cmd"),
oracle_manifest_digest=_digest("oracle-man"),
task_asset_manifest_digest=_digest("assets"),
sandbox_dependency_manifest_digest=_digest("deps"),
)
},
skill_digests={"review": _digest("skill"), "ce_review": None},
ce_plugin_version="3.24.0",
ce_plugin_manifest_digest=_digest("ce"),
)
values.update(overrides)
return ComparatorReuseExpectation(**values)
def test_matching_incumbent_review_row_is_reusable() -> None:
assert row_is_reusable_comparator(_row(), _expected()) is True
def test_candidate_rows_are_never_reusable() -> None:
assert row_is_reusable_comparator(_row(arm="candidate_review"), _expected()) is False
def test_skill_digest_drift_rejects_reuse() -> None:
assert row_is_reusable_comparator(_row(), _expected(skill_digests={"review": _digest("other")})) is False
def test_excluded_or_failed_rows_are_not_reusable() -> None:
expected = _expected()
assert row_is_reusable_comparator(_row(error_kind="session-error", ok=False), expected) is False
assert row_is_reusable_comparator(_row(ok=False), expected) is False
assert row_is_reusable_comparator(_row(review_evidence_valid=False), expected) is False
assert row_is_reusable_comparator(_row(recorded_at=(datetime.now(UTC) - timedelta(days=91)).isoformat()), expected) is False
def test_runtime_digest_mismatch_rejects_when_both_sides_are_bound() -> None:
row = _row(runtime_digest=_digest("old-cli"))
assert row_is_reusable_comparator(row, _expected(runtime_digest=_digest("new-cli"))) is False
assert row_is_reusable_comparator(row, _expected(runtime_digest=_digest("old-cli"))) is True
# A row with no runtime_digest was measured by a harness that recorded none,
# which is the drift this lock exists to catch - not evidence of agreement.
assert row_is_reusable_comparator(_row(runtime_digest=None), _expected()) is False
# And a sweep that cannot determine its own digest must not reuse either.
assert row_is_reusable_comparator(_row(), _expected(runtime_digest=None)) is False
def test_ce_review_matches_plugin_digest_not_repo_skill() -> None:
row = _row(
arm="ce_review",
skill_digest=None,
ce_plugin_version="3.24.0",
ce_plugin_manifest_digest=_digest("ce"),
)
assert row_is_reusable_comparator(row, _expected()) is True
assert (
row_is_reusable_comparator(row, _expected(ce_plugin_manifest_digest=_digest("other")))
is False
)
def test_select_drops_conflicting_duplicates() -> None:
first = _row(review_weighted_f1=0.4)
second = _row(review_weighted_f1=0.9, recorded_at=datetime.now(UTC).isoformat())
selected = select_reusable_comparator_rows([first, second], expected=_expected())
assert selected == {}
same = select_reusable_comparator_rows([first, dict(first)], expected=_expected())
assert ("review-pr-2718-defect", "review", 0) in same
@requires_openat
def test_materialize_copies_transcript_and_review_artifacts(tmp_path: Path) -> None:
payload = b'{"type":"result"}\n'
source = tmp_path / "prior"
dest = tmp_path / "fresh"
(source / "transcripts").mkdir(parents=True)
dest.mkdir()
transcript = source / "transcripts" / "session-1.jsonl"
transcript.write_bytes(payload)
transcript.chmod(0o600)
review = source / "review-pr-2718-defect-review-run0.review.json"
review.write_text('{"verdict":"comment"}\n')
patch = source / "review-pr-2718-defect-review-run0.patch"
patch.write_text("diff\n")
row = _row(
review_artifact=review.name,
transcript_artifacts=[_artifact(payload=payload)],
)
copied = materialize_reused_row(row, source_dir=source, dest_dir=dest)
assert copied["reused"] is True
assert copied["reused_from_recorded_at"] == row["recorded_at"]
assert (dest / "transcripts" / "session-1.jsonl").read_bytes() == payload
assert (dest / review.name).read_text() == review.read_text()
assert (dest / patch.name).read_text() == "diff\n"
assert copied["transcript_artifacts"][0]["sha256"] == hashlib.sha256(payload).hexdigest()
@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges")
@requires_openat
def test_a_reused_artifact_is_copied_from_the_inode_that_was_checked(tmp_path: Path) -> None:
"""The reuse source is a directory another sweep wrote and may still write.
Validating a path and then re-opening it hands a concurrent writer the gap:
replace the checked file with a symlink and the copy follows it out of the
results directory. Swapping the path while the descriptor is held is that
same substitution, made deterministic.
"""
(tmp_path / "transcript.jsonl").write_bytes(b"verified\n")
decoy = tmp_path / "decoy.jsonl"
decoy.write_bytes(b"substituted\n")
with comparator_reuse._open_real_directory(tmp_path, label="reuse source") as dir_fd:
with comparator_reuse._open_regular("transcript.jsonl", dir_fd=dir_fd, label="transcript") as descriptor:
(tmp_path / "transcript.jsonl").unlink()
(tmp_path / "transcript.jsonl").symlink_to(decoy)
comparator_reuse._copy_owner_only(descriptor, "copy.jsonl", dir_fd=dir_fd)
assert (tmp_path / "copy.jsonl").read_bytes() == b"verified\n"
with pytest.raises(SandboxError, match="regular non-symlink"):
with comparator_reuse._open_regular("transcript.jsonl", dir_fd=dir_fd, label="transcript"):
pass
@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges")
@requires_openat
def test_a_symlinked_transcripts_directory_is_refused_on_both_sides(tmp_path: Path) -> None:
"""`O_NOFOLLOW` refuses the leaf, not the directory above it.
A `transcripts` symlink on the source side makes reuse read a file outside
the results directory; one on the destination side writes the copy outside
this sweep's evidence. Neither is covered by the per-file guards that let
_resolved_directory tolerate a symlinked root.
"""
payload = b'{"type":"result"}\n'
outside = tmp_path / "outside"
(outside / "transcripts").mkdir(parents=True)
(outside / "transcripts" / "session-1.jsonl").write_bytes(payload)
row = _row(transcript_artifacts=[_artifact(payload=payload)])
linked_source = tmp_path / "linked-source"
linked_source.mkdir()
(linked_source / "transcripts").symlink_to(outside / "transcripts", target_is_directory=True)
dest = tmp_path / "fresh"
dest.mkdir()
with pytest.raises(SandboxError, match="transcript source must be a real directory"):
materialize_reused_row(row, source_dir=linked_source, dest_dir=dest)
source = tmp_path / "prior"
(source / "transcripts").mkdir(parents=True)
(source / "transcripts" / "session-1.jsonl").write_bytes(payload)
linked_dest = tmp_path / "linked-dest"
linked_dest.mkdir()
(linked_dest / "transcripts").symlink_to(outside / "transcripts", target_is_directory=True)
with pytest.raises(SandboxError, match="transcript destination must be a real directory"):
materialize_reused_row(row, source_dir=source, dest_dir=linked_dest)
@requires_openat
@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges")
def test_a_renamed_transcripts_directory_cannot_redirect_a_copy(tmp_path: Path) -> None:
"""The directory is pinned, not re-walked from its name.
An lstat that passed and a pathname used afterwards are two different
directories the moment a concurrent writer renames the first one away. This
performs exactly that substitution — rename, then leave a symlink in its
place — while the descriptor is held, which is what makes the race testable
without timing.
"""
payload = b'{"type":"result"}\n'
results = tmp_path / "results"
transcripts = results / "transcripts"
transcripts.mkdir(parents=True)
(transcripts / "session-1.jsonl").write_bytes(payload)
outside = tmp_path / "outside"
outside.mkdir()
with comparator_reuse._open_real_directory(results, label="reuse source") as root_fd:
with comparator_reuse._open_real_directory(
"transcripts", dir_fd=root_fd, label="transcript source"
) as dir_fd:
transcripts.rename(results / "moved")
(results / "transcripts").symlink_to(outside, target_is_directory=True)
with comparator_reuse._open_regular(
"session-1.jsonl", dir_fd=dir_fd, label="transcript"
) as artifact_fd:
comparator_reuse._copy_owner_only(artifact_fd, "copy.jsonl", dir_fd=dir_fd)
assert (results / "moved" / "copy.jsonl").read_bytes() == payload
assert not (outside / "copy.jsonl").exists()
@requires_openat
def test_a_transcript_rewritten_mid_copy_is_refused_not_recorded(tmp_path: Path, monkeypatch) -> None:
"""The digest has to describe the bytes that were written.
A held descriptor stops the pathname being substituted; it does not stop the
inode being rewritten, and the prior sweep's directory is one this sweep
treats as concurrently writable. Hashing the source and then reading it
again to copy let the row keep the expected digest while the destination
held different bytes.
"""
payload = b'{"type":"result"}\n'
source = tmp_path / "prior"
(source / "transcripts").mkdir(parents=True)
transcript = source / "transcripts" / "session-1.jsonl"
transcript.write_bytes(payload)
dest = tmp_path / "fresh"
dest.mkdir()
row = _row(transcript_artifacts=[_artifact(payload=payload)])
# Rewrite the inode in the window the copy reads through — same length, so
# only the digest can tell, which is the point.
real_read = comparator_reuse.os.read
rewritten = {"done": False}
def rewrite_then_read(fd: int, size: int) -> bytes:
if not rewritten["done"]:
rewritten["done"] = True
with open(transcript, "r+b") as handle:
handle.write(b'{"type":"TAMPER"}')
return real_read(fd, size)
monkeypatch.setattr(comparator_reuse.os, "read", rewrite_then_read)
with pytest.raises(SandboxError, match="drifted"):
materialize_reused_row(row, source_dir=source, dest_dir=dest)
monkeypatch.undo()
# And nothing unvouched-for is left behind for the proposer to read.
assert not (dest / "transcripts" / "session-1.jsonl").exists()
@requires_openat
def test_materialize_rejects_same_directory_and_missing_transcript(tmp_path: Path) -> None:
source = tmp_path / "prior"
source.mkdir()
row = _row()
with pytest.raises(SandboxError, match="same results directory"):
materialize_reused_row(row, source_dir=source, dest_dir=source)
dest = tmp_path / "fresh"
dest.mkdir()
with pytest.raises(SandboxError, match="missing"):
materialize_reused_row(row, source_dir=source, dest_dir=dest)
@requires_openat
def test_a_reused_row_ages_from_its_first_measurement_not_the_copy():
"""Reuse chains must not refresh the clock.
materialize_reused_row restamps recorded_at with the copy time, so aging
against that field let a row be copied forward every generation and outlive
max_age forever. The original measurement time is the one that counts.
"""
original = (datetime.now(UTC) - timedelta(days=91)).isoformat()
chained = _row(recorded_at=datetime.now(UTC).isoformat(), reused_from_recorded_at=original)
assert row_is_reusable_comparator(chained, _expected()) is False
# The same row inside the window is still reusable.
fresh = _row(
recorded_at=datetime.now(UTC).isoformat(),
reused_from_recorded_at=(datetime.now(UTC) - timedelta(days=1)).isoformat(),
)
assert row_is_reusable_comparator(fresh, _expected()) is True
def test_a_future_dated_row_is_corrupt_not_fresh():
ahead = (datetime.now(UTC) + timedelta(days=2)).isoformat()
assert row_is_reusable_comparator(_row(recorded_at=ahead), _expected()) is False
def test_a_changed_sandbox_dependency_is_not_the_same_baseline():
"""The environment is part of the measurement.
This branch itself changes `sandbox_dependencies` in the review corpus, so a
prior row measured against the old set is a measurement of a different
machine. Reusing it would compare a fresh candidate to a baseline built
somewhere else and hand the promotion gate a false comparison.
"""
assert row_is_reusable_comparator(
_row(sandbox_dependency_manifest_digest=_digest("other-deps")), _expected()
) is False
assert row_is_reusable_comparator(
_row(task_asset_manifest_digest=_digest("other-assets")), _expected()
) is False
# A row that predates the field is not evidence of agreement either.
assert row_is_reusable_comparator(_row(sandbox_dependency_manifest_digest=None), _expected()) is False
@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges")
def test_reuse_directories_allow_a_symlinked_parent_but_not_a_symlinked_leaf(tmp_path: Path):
"""Pins a deliberate difference from the sandbox's mount-root check.
proposer_sandbox refuses every symlink hop because a hop changes what an
untrusted session is handed. A reuse directory is data, and every file
inside it is validated on its own, so a symlinked parent is allowed -
rejecting it would break a symlinked artifacts directory or macOS's /var
for no gain. The leaf itself must still be a real directory.
"""
real = tmp_path / "real"
real.mkdir()
(real / "inner").mkdir()
linked_parent = tmp_path / "linked"
linked_parent.symlink_to(real, target_is_directory=True)
# Reached through a symlinked parent: allowed, and resolved to the real path.
# The identity returned alongside it is what pins the root against a swap
# between the check and the open; the symlink policy itself is unchanged.
resolved, identity = comparator_reuse._resolved_directory(linked_parent / "inner", label="probe")
assert resolved == (real / "inner").resolve()
inner_stat = (real / "inner").stat()
assert identity == (inner_stat.st_dev, inner_stat.st_ino)
# The leaf itself being a symlink is still refused.
with pytest.raises(SandboxError, match="must be a real directory"):
comparator_reuse._resolved_directory(linked_parent, label="probe")
def test_a_review_row_without_its_artifact_is_not_reusable() -> None:
"""A score is a claim about evidence, not the evidence itself.
materialize_reused_row copies the review artifact only when the row names
one, so accepting a row without it would carry a scored review forward with
nothing for a proposer to read.
"""
row = _row()
assert row_is_reusable_comparator(row, _expected()) is True
without = {**row, "review_artifact": ""}
assert row_is_reusable_comparator(without, _expected()) is False
missing = {k: v for k, v in row.items() if k != "review_artifact"}
assert row_is_reusable_comparator(missing, _expected()) is False
@requires_openat
def test_a_reuse_root_replaced_after_the_check_is_refused(tmp_path: Path, monkeypatch) -> None:
"""Check and use must name the same directory, not the same string.
_resolved_directory lstats a name and the open re-walks that same name, so
a prior sweep that swaps its results root in between is opened somewhere
else. The leaf-symlink rule does not cover it - a replacement that is
itself a real directory passes every check the policy makes - and the
failure is silent, folding another directory's rows into this sweep's
comparator baseline.
"""
original = tmp_path / "results"
original.mkdir()
resolved, stale_identity = comparator_reuse._resolved_directory(original, label="probe")
# Replaced by a different REAL directory: the name still resolves and still
# passes the symlink policy, but it is not the inode that was checked.
original.rename(tmp_path / "moved")
original.mkdir()
assert comparator_reuse._resolved_directory(original, label="probe")[1] != stale_identity
monkeypatch.setattr(
comparator_reuse, "_resolved_directory", lambda *_a, **_k: (resolved, stale_identity)
)
with pytest.raises(SandboxError, match="replaced between the check and the open"):
with comparator_reuse._open_pinned_root(original, label="probe"):
pass
@requires_openat
def test_a_stable_reuse_root_opens_normally(tmp_path: Path) -> None:
"""The guard rejects nothing that holds still - a directory matches itself."""
root = tmp_path / "results"
root.mkdir()
with comparator_reuse._open_pinned_root(root, label="probe") as fd:
assert os.fstat(fd).st_ino == root.stat().st_ino

View file

@ -9,19 +9,24 @@ import time
from contextlib import contextmanager
from datetime import UTC, datetime, timedelta
from pathlib import Path
from types import SimpleNamespace
import pytest
from workflow_bench import evolve, evolution
from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE
from workflow_bench.evolve import (
MIN_INSTANCE_SWEEP_SECONDS,
build_parser,
build_proposer_prompt,
capped_timeout_seconds,
executed_benchmark_arms,
generation_timeout_seconds,
instance_window_budget_seconds,
load_jsonl,
proposer_evidence_entries,
read_learnings,
remaining_runtime_seconds,
resolve_incumbent_arms,
runner_argv,
select_evidence,
@ -662,6 +667,127 @@ def test_run_proposer_hides_the_hidden_harness_and_keeps_the_full_tool_surface(m
assert captured["settings_json"] == FakeSandbox.settings_json
def test_proposer_session_cannot_outlive_the_remaining_instance_window(monkeypatch, tmp_path):
"""Clearing the sweep minimum is not a licence to run a full session.
--timeout is sized for a whole generation, so a proposer started with the
minimum left would run far past --max-runtime-seconds and the box would take
the evidence with it. The budget is sampled after the clone, the sanitize
pass and the sandbox setup, because a reading taken before them is already
stale by the time the session it bounds actually starts.
"""
captured: dict[str, object] = {}
# Pinned clock: real setup duration would make this assert on scheduling.
clock = {"now": 1000.0}
monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"])
setup_seconds = 100.0
@contextmanager
def fake_prepare_sandbox(**_kwargs):
yield SimpleNamespace(
claude_bin="claude",
command_prefix=[],
settings_json="{}",
transcript_projects=tmp_path / "transcript-projects",
)
def fake_run_claude(*_args, **kwargs):
captured.update(kwargs)
return {"ok": False, "error_kind": "session-error"}
def slow_sanitize(_clone):
# Stands in for the clone, the sanitize pass and the sandbox build —
# all of which run between the caller's decision and the session.
clock["now"] += setup_seconds
return "0" * 40
monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination)
monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None)
monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", slow_sanitize)
monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox)
monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude)
args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"])
assert args.timeout > evolve.MIN_INSTANCE_SWEEP_SECONDS, "otherwise this test proves nothing"
common = {
"overlay_dir": tmp_path / "overlay",
"proposal_path": tmp_path / "proposal.md",
"evidence_bundle": tmp_path / "evidence",
"bwrap_bin": tmp_path / "bwrap",
}
budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1
args.max_runtime_seconds = budget
started = clock["now"]
evolve.run_proposer("prompt", args, **common, started_monotonic=started)
# The setup time is charged, not handed back: a value sampled at `started`
# would have allowed the whole budget.
assert captured["timeout"] == budget - setup_seconds
# No cap configured means no budget to overrun: the session keeps its own.
args.max_runtime_seconds = None
evolve.run_proposer("prompt", args, **common, started_monotonic=started)
assert captured["timeout"] == args.timeout
evolve.run_proposer("prompt", args, **common)
assert captured["timeout"] == args.timeout
def test_a_budget_spent_during_setup_stops_the_proposer_rather_than_buying_a_second(
monkeypatch, tmp_path
):
"""An exhausted cap must end the generation, not start a one-second session.
remaining_runtime_seconds floors at 0, and the call site wrapped it in
max(1, ...) - so a cap fully consumed by the clone, the sanitize pass and
the sandbox build produced a paid session with a one-second allowance
instead of stopping before the upload reserve the cap exists to protect.
"""
captured: dict[str, object] = {}
clock = {"now": 1000.0}
monkeypatch.setattr(evolve.time, "monotonic", lambda: clock["now"])
@contextmanager
def fake_prepare_sandbox(**_kwargs):
yield SimpleNamespace(
claude_bin="claude",
command_prefix=[],
settings_json="{}",
transcript_projects=tmp_path / "transcript-projects",
)
def fake_run_claude(*_args, **kwargs):
captured.update(kwargs)
return {"ok": True}
def setup_that_spends_the_whole_budget(_clone):
clock["now"] += budget
return "0" * 40
monkeypatch.setattr(evolve.runner, "make_worktree", lambda _repo, _ref, destination: destination)
monkeypatch.setattr(evolve.runner, "remove_clone", lambda _clone: None)
monkeypatch.setattr(evolve, "sanitize_clone_for_hidden_oracles", setup_that_spends_the_whole_budget)
monkeypatch.setattr(evolve, "prepare_sandbox", fake_prepare_sandbox)
monkeypatch.setattr(evolve.runner, "run_claude", fake_run_claude)
args = build_parser().parse_args(["--tasks", "tasks.yaml", "--model", "model"])
budget = evolve.MIN_INSTANCE_SWEEP_SECONDS + 1
args.max_runtime_seconds = budget
record = evolve.run_proposer(
"prompt",
args,
overlay_dir=tmp_path / "overlay",
proposal_path=tmp_path / "proposal.md",
evidence_bundle=tmp_path / "evidence",
bwrap_bin=tmp_path / "bwrap",
started_monotonic=clock["now"],
)
assert not captured, "no session may start once the cap is exhausted"
assert record["ok"] is False
assert record["error_kind"] == "runtime-cap-exhausted"
def test_parser_defaults_match_the_gate_minimums():
args = build_parser().parse_args(["--tasks", "t.yaml", "--model", "pinned"])
assert args.runs == 3
@ -922,6 +1048,27 @@ def test_runner_argv_inserts_ce_review_for_review_overlay(tmp_path):
)
arms = argv[argv.index("--arms") + 1 : argv.index("--promotion-metric")]
assert arms == ["ce_review", "review", "candidate_review"]
assert "--reuse-results" not in argv
def test_runner_argv_forwards_prior_results_for_comparator_reuse(tmp_path):
args = build_parser().parse_args(
["--tasks", "t.yaml", "--model", "pinned", "--arms", "review"]
)
overlay = tmp_path / "overlay"
skill = overlay / ".claude" / "skills" / "gitnexus-review" / "SKILL.md"
skill.parent.mkdir(parents=True)
skill.write_text("candidate")
prior = tmp_path / "prior-bench"
argv = runner_argv(
args,
tmp_path / "bench",
overlay,
task_bindings=[{"id": "task"}],
target_base_digests={},
reuse_results=prior,
)
assert argv[argv.index("--reuse-results") + 1] == str(prior)
def test_runner_argv_omits_proposer_for_manual_overlay(tmp_path):
@ -1118,6 +1265,161 @@ def test_generation_timeout_rejects_unknown_arm() -> None:
)
def test_instance_window_budget_leaves_upload_reserve() -> None:
# Friday 10:57 on a box that booted 02:45 Saturday-window: ~8.2h uptime.
leftover = instance_window_budget_seconds(8.2 * 3600)
assert leftover == int(86_400 - 8.2 * 3600 - 5_400)
assert leftover >= MIN_INSTANCE_SWEEP_SECONDS
with pytest.raises(ValueError, match="only .*s left"):
instance_window_budget_seconds(23.5 * 3600)
with pytest.raises(ValueError, match="uptime must be"):
instance_window_budget_seconds(float("nan"))
def test_capped_timeout_clamps_to_leftover_window(monkeypatch) -> None:
assert capped_timeout_seconds(10_000, None) == 10_000
assert capped_timeout_seconds(10_000, 90) == 90
with pytest.raises(ValueError, match="no time remains"):
capped_timeout_seconds(10_000, 0)
# Pinned clock: the helper is pure arithmetic, so a real elapsed-time window
# would assert on scheduling rather than on the behaviour under test.
monkeypatch.setattr(evolve.time, "monotonic", lambda: 1040.0)
started = 1000.0
assert remaining_runtime_seconds(max_runtime_seconds=None, started_monotonic=started) is None
leftover = remaining_runtime_seconds(max_runtime_seconds=100, started_monotonic=started)
assert leftover is not None
assert leftover == 60
def test_parser_rejects_non_positive_max_runtime() -> None:
with pytest.raises(SystemExit):
build_parser().parse_args(
["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "0"]
)
args = build_parser().parse_args(
["--tasks", "t.yaml", "--model", "pinned", "--max-runtime-seconds", "7200"]
)
assert args.max_runtime_seconds == 7200
assert args.max_runtime_from_instance_window is False
def _task_file(tmp_path: Path) -> Path:
tasks = tmp_path / "tasks.yaml"
tasks.write_text(
"""tasks:
- id: demo
class: test
repo: .
prompt: implement
verify: "true"
oracle:
command: "true"
files:
- source: hidden.test.ts
target: hidden.test.ts
"""
)
return tasks
def _stub_main_preflight(monkeypatch, tmp_path) -> None:
"""Everything main() shells out to before it reaches _run_generations."""
monkeypatch.setattr(evolve.runner, "selected_task_bindings", lambda _tasks: [{"id": "demo"}])
monkeypatch.setattr(evolve, "preflight_bubblewrap", lambda: tmp_path / "bwrap")
monkeypatch.setattr(evolve, "require_claude_sandbox_helpers", lambda: None)
def test_the_runtime_cap_is_derived_where_its_clock_starts(monkeypatch, tmp_path, capsys) -> None:
"""The budget and the clock it is measured against must be one instant.
run-evolution.sh used to compute the budget in a separate `uv run python -c`
and pass a number, so the script's remaining provenance work and this
interpreter's startup were charged to the sweep — out of the upload reserve
the cap exists to protect. main() reads /proc/uptime itself now, next to its
own clock, so no interval exists to lose.
"""
monkeypatch.setenv("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", "20000")
monkeypatch.setenv("EVENTBRIDGE_STOP_RESERVE_SECONDS", "1000")
monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0)
captured: dict[str, object] = {}
def record(args, **kwargs):
captured["max_runtime_seconds"] = args.max_runtime_seconds
captured["started_monotonic"] = kwargs["started_monotonic"]
return 0
monkeypatch.setattr(evolve, "_run_generations", record)
_stub_main_preflight(monkeypatch, tmp_path)
monkeypatch.setattr(
sys,
"argv",
[
"evolve",
"--tasks",
str(_task_file(tmp_path)),
"--model",
"pinned",
"--out-root",
str(tmp_path / "out"),
"--max-runtime-from-instance-window",
],
)
assert evolve.main() == 0
assert captured["max_runtime_seconds"] == 20000 - 3600 - 1000
# Derived here, not passed in: the clock handed to the sweep is the one
# taken beside the uptime read.
assert isinstance(captured["started_monotonic"], float)
assert "capping the sweep to 15400s" in capsys.readouterr().out
def test_the_runtime_cap_refuses_two_sources_of_truth(monkeypatch, tmp_path) -> None:
monkeypatch.setattr(evolve, "read_instance_uptime_seconds", lambda: 3600.0)
monkeypatch.setattr(
sys,
"argv",
[
"evolve",
"--tasks",
str(_task_file(tmp_path)),
"--model",
"pinned",
"--max-runtime-from-instance-window",
"--max-runtime-seconds",
"7200",
],
)
with pytest.raises(SystemExit):
evolve.main()
def test_the_runtime_cap_fails_closed_without_a_readable_uptime(monkeypatch, tmp_path) -> None:
def unreadable():
raise ValueError("cannot read instance uptime from /proc/uptime")
monkeypatch.setattr(evolve, "read_instance_uptime_seconds", unreadable)
monkeypatch.setattr(
sys,
"argv",
[
"evolve",
"--tasks",
str(_task_file(tmp_path)),
"--model",
"pinned",
"--max-runtime-from-instance-window",
],
)
# Better to refuse than to run a box-stopped sweep believing it is uncapped.
with pytest.raises(SystemExit):
evolve.main()
@pytest.mark.skipif(sys.platform != "linux", reason="Bubblewrap PID namespaces require Linux")
def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path):
try:

View file

@ -0,0 +1,122 @@
"""Cost model for the evolution wall clock: measured cells, real schedules."""
from __future__ import annotations
import pytest
from workflow_bench.measure_evolution_cost import (
CANDIDATE_ARM,
SHA_OVERHEAD_SECONDS,
DURATIONS_BY_ARM,
PROPOSER_SECONDS,
REVIEW_ARMS,
expected_task_seconds,
fed_makespan,
fed_pool_enabled,
generation_seconds,
graph_pipeline_enabled,
paid_arms,
task_cells,
wave_makespan,
)
def test_every_arm_has_its_own_unsorted_sample():
assert set(DURATIONS_BY_ARM) == set(REVIEW_ARMS)
for arm, sample in DURATIONS_BY_ARM.items():
assert len(sample) >= 10, arm
# Sorting would hand each task a uniform block and hide the variance
# the whole model exists to price.
assert list(sample) != sorted(sample), arm
assert PROPOSER_SECONDS > 0
assert SHA_OVERHEAD_SECONDS > 0
def test_weekly_reuse_pays_the_candidate_arm_only():
assert paid_arms(weekly=True, reuse_enabled=True) == (CANDIDATE_ARM,)
assert paid_arms(weekly=False, reuse_enabled=True) == REVIEW_ARMS
assert paid_arms(weekly=True, reuse_enabled=False) == REVIEW_ARMS
def test_cells_are_submitted_run_major_arm_minor():
# runner.py: [(run_idx, arm) for run_idx in range(runs) for arm in arms].
# At workers=3 that puts one cell of each arm in every wave.
cells = task_cells(2, REVIEW_ARMS, 0)
assert len(cells) == 6
expected = [DURATIONS_BY_ARM[arm][run] for run in range(2) for arm in REVIEW_ARMS]
assert cells == expected
def test_overhead_is_charged_per_sha_and_outside_the_pool():
# Two properties at once: the residual sits outside the schedule, where more
# workers cannot dissolve it, and it scales with SHAs rather than cells.
assert task_cells(1, (CANDIDATE_ARM,), 0) == [DURATIONS_BY_ARM[CANDIDATE_ARM][0]]
wide = generation_seconds(
task_count=1, runs=3, arms=REVIEW_ARMS, workers=9, fed_pool=True, unique_shas=5
)
assert wide >= PROPOSER_SECONDS + 5 * SHA_OVERHEAD_SECONDS
def test_sweep_overhead_does_not_shrink_with_the_arm_count():
"""The bias that made weekly look cheaper than it is.
A seeded weekly generation pays one arm instead of three but builds exactly
the same graphs. Charging the residual per cell billed it a third of a cost
the real sweep still pays; per SHA, the two attribute the same setup.
"""
kwargs = dict(task_count=6, runs=3, workers=3, fed_pool=False, unique_shas=5)
weekly = generation_seconds(arms=(CANDIDATE_ARM,), **kwargs)
cold = generation_seconds(arms=REVIEW_ARMS, **kwargs)
weekly_sessions = 6 * expected_task_seconds(3, (CANDIDATE_ARM,), 3, fed_pool=False)
cold_sessions = 6 * expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False)
# Whatever each wall is, the non-session part is identical.
assert round(weekly - weekly_sessions) == round(cold - cold_sessions)
# Cycling wraps, so a task can ask for more runs than the sample holds.
long_sample = task_cells(len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2, (CANDIDATE_ARM,), 0)
assert len(long_sample) == len(DURATIONS_BY_ARM[CANDIDATE_ARM]) + 2
def test_a_wave_costs_its_slowest_cell_and_a_fed_pool_does_not():
slow = [10.0, 1.0, 1.0, 10.0, 1.0, 1.0]
assert wave_makespan(slow, 3) == 20.0
# Fed: one worker takes the first 10; the second 10 lands on a worker that
# has already cleared a 1, and the remaining 1s fill the third.
assert fed_makespan(slow, 3) == 11.0
assert fed_makespan(slow, 1) == wave_makespan(slow, 1) == 24.0
def test_expected_task_seconds_is_alignment_averaged_and_deterministic():
waved = expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False)
assert waved == expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=False)
assert expected_task_seconds(0, REVIEW_ARMS, 3, fed_pool=False) == 0.0
assert expected_task_seconds(3, (), 3, fed_pool=False) == 0.0
# The barrier can only cost time, never save it.
assert waved >= expected_task_seconds(3, REVIEW_ARMS, 3, fed_pool=True)
def test_a_generation_pays_one_proposer_session_on_top_of_its_tasks():
one = generation_seconds(
task_count=1, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False, unique_shas=1
)
two = generation_seconds(
task_count=2, runs=3, arms=REVIEW_ARMS, workers=3, fed_pool=False, unique_shas=1
)
# Each extra task adds exactly one task's makespan. The proposer and the
# per-SHA sweep overhead are both paid once, not per task.
assert two - one == pytest.approx(
one - PROPOSER_SECONDS - SHA_OVERHEAD_SECONDS, abs=2.0
)
def test_feature_flags_read_the_runner_not_the_wish():
assert graph_pipeline_enabled("def _run_sweep(): pass") == 0
assert graph_pipeline_enabled("graph_prefetch = GraphPrefetch(...)") == 1
assert fed_pool_enabled("def _run_wave(): pass") == 0
assert fed_pool_enabled("def _run_fed_pool(): pass") == 1
@pytest.mark.parametrize("workers", [1, 3, 8])
def test_more_workers_never_lengthen_a_task(workers):
serial = expected_task_seconds(3, REVIEW_ARMS, 1, fed_pool=True)
assert expected_task_seconds(3, REVIEW_ARMS, workers, fed_pool=True) <= serial

View file

@ -79,11 +79,11 @@ def run(index, arm):
assert Path({str(assets)!r}).exists(), 'assets removed while a worker was active'
return {{'resolved': False, 'error_kind': result.state}}
with cancellation_scope(handle_signals=True) as event:
streak, stopped=sweep_task_cells([(i, 'review') for i in range(10)], workers=2, run=run,
streak, tripped=sweep_task_cells([(i, 'review') for i in range(10)], workers=2, run=run,
on_start=lambda *args: None, on_record=lambda i,a,r: rows.append([i,r]),
outage_streak=0, outage_limit=5, cancel_event=event)
Path({str(assets)!r}).unlink()
print(json.dumps({{'rows': rows, 'stopped': stopped}}))
print(json.dumps({{'rows': rows, 'stopped': event.is_set(), 'tripped': tripped}}))
"""
process = subprocess.Popen(
[PYTHON, "-c", script],
@ -104,7 +104,8 @@ print(json.dumps({{'rows': rows, 'stopped': stopped}}))
assert process.returncode == 0, stderr
assert time.monotonic() - started < 15
report = json.loads(stdout)
assert report["stopped"] and [row[0] for row in report["rows"]] == [0, 1]
assert report["stopped"] and not report["tripped"], "cancelled, not an outage"
assert [row[0] for row in report["rows"]] == [0, 1]
assert report["rows"][0][1]["resolved"] is True
assert report["rows"][1][1]["error_kind"] == "cancelled"
with pytest.raises(ProcessLookupError):

View file

@ -33,9 +33,11 @@ from workflow_bench.proposer_sandbox import (
SANDBOX_GIT_EXCLUDES,
VITE_TEMP_DIR,
SANDBOX_PATH,
SANDBOX_REVIEW_OUTPUT,
SANDBOX_PYTHON3,
SANDBOX_SHELL_PREFIX,
SANDBOX_USER_SKILLS,
SANDBOX_WORKSPACE,
ReadOnlyMount,
SandboxError,
_runtime_mount_args,
@ -43,46 +45,65 @@ from workflow_bench.proposer_sandbox import (
build_sandbox_environment,
_force_rmtree,
host_workspace_write_boundary,
prepare_review_workspace,
prepare_sandbox,
preflight_bubblewrap,
sandbox_workspace_write_boundary,
stage_evidence_bundle,
stage_task_assets,
)
from workflow_bench.review_scoring import REVIEW_OUTPUT, parse_review_output
from workflow_bench.task_assets import TaskAssetCache, stage_task_assets as stage_immutable_task_assets
@pytest.mark.parametrize("entry", ["file", "directory", "relative-link", "absolute-link"])
def test_review_preparation_rejects_existing_output_without_touching_target(tmp_path, entry):
@pytest.mark.parametrize("entry", ["directory", "relative-link", "absolute-link"])
def test_review_preparation_rejects_a_reused_artifact_directory(tmp_path, entry):
clone = tmp_path / "clone"
clone.mkdir()
sentinel = tmp_path / "sentinel"
sentinel.write_text("must survive")
output = clone / "review-output.json"
if entry == "file":
output.write_text("existing result")
elif entry == "directory":
output.mkdir()
else:
output.symlink_to(sentinel if entry == "absolute-link" else "../sentinel")
with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox:
stale = proposer_sandbox.review_output_path(sandbox, "review-output.json").parent
if entry == "directory":
stale.mkdir()
(stale / "review-output.json").write_text("a previous cell's verdict")
else:
# relpath, not a hand-written "../sentinel": stale is
# <private_root>/review-output, which is nowhere near tmp_path, so the
# literal produced a dangling link and the assertion below proved nothing.
stale.symlink_to(
sentinel if entry == "absolute-link" else Path(os.path.relpath(sentinel, stale.parent))
)
with pytest.raises(SandboxError, match="already exists"):
proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json")
assert sentinel.read_text() == "must survive"
if entry == "file":
assert output.read_text() == "existing result"
if "link" in entry:
assert output.is_symlink()
def test_review_preparation_creates_a_private_regular_output(tmp_path):
def test_review_preparation_leaves_a_clone_entry_of_the_same_name_alone(tmp_path):
# The artifact no longer lives in the workspace, so a file that happens to
# share its name is just one of the repository's own files.
clone = tmp_path / "clone"
clone.mkdir()
(clone / "review-output.json").write_text("repository content")
with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox:
output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json")
assert (clone / "review-output.json").read_text() == "repository content"
assert clone not in output.parents
def test_review_preparation_creates_a_private_directory_and_not_the_file(tmp_path):
clone = tmp_path / "clone"
clone.mkdir()
with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as sandbox:
output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json")
assert output.read_bytes() == b""
assert stat.S_ISREG(output.lstat().st_mode)
assert stat.S_IMODE(output.stat().st_mode) == 0o600
assert output == proposer_sandbox.review_output_path(sandbox, "review-output.json")
# The DIRECTORY is what has to exist and be writable: the agent writes
# a temp file beside the target and renames it.
assert output.parent.is_dir()
assert stat.S_IMODE(output.parent.stat().st_mode) == 0o700
# The file is deliberately absent — absence is how "never written" is
# told apart from "written badly".
assert not output.exists()
def test_review_preparation_preserves_existing_runtime_files_and_tracks_only_created_paths(tmp_path):
@ -165,6 +186,15 @@ def test_unsafe_host_session_translates_virtual_paths_and_disables_containment(t
assert sandbox.require_pid_namespace is False
assert sandbox.host_path("/workspace/review-output.json") == str(clone / "review-output.json")
assert sandbox.host_path("/evidence/selected-rows.json") == str(evidence / "selected-rows.json")
# The review artifact left the workspace, so the host-unsafe backend has
# to translate its new home too. Untranslated, the review prompt names a
# path that exists on neither backend and the cell writes nothing.
assert sandbox.host_path(
f"{proposer_sandbox.SANDBOX_REVIEW_OUTPUT}/review-output.json"
) == str(proposer_sandbox.review_output_path(sandbox, "review-output.json"))
assert sandbox.host_text(
f"write {proposer_sandbox.SANDBOX_REVIEW_OUTPUT}/review-output.json"
) == f"write {proposer_sandbox.review_output_path(sandbox, 'review-output.json')}"
assert sandbox.host_text("read /evidence and write /workspace/out") == (
f"read {evidence} and write {clone}/out"
)
@ -867,9 +897,8 @@ def test_read_only_review_workspace_exposes_only_one_writable_artifact(tmp_path:
clone.mkdir()
source = clone / "source.ts"
source.write_text("trusted\n")
output = clone / "review-output.json"
output.write_text("")
script = """
import os
from pathlib import Path
try:
Path('/workspace/source.ts').write_text('tampered')
@ -877,15 +906,24 @@ except OSError:
pass
else:
raise SystemExit('review source remained writable')
Path('/workspace/review-output.json').write_text('{"schema_version":1}')
# Write the way the agent's Write tool does: a temp file beside the target,
# then rename. Writing in place would pass against the mount shape that
# shipped every artifact empty, which is the regression this canary exists for.
target = Path('/review-output/review-output.json')
staging = target.with_name(target.name + '.tmp.1.abc')
staging.write_text('{"schema_version":1}')
os.replace(staging, target)
"""
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
output = proposer_sandbox.prepare_review_workspace(sandbox, "review-output.json")
result = run_managed(
[
*sandbox.command_prefix_for(
read_only_workspace=True,
extra_writable_mounts=(
ReadOnlyMount(source=output, target="/workspace/review-output.json"),
ReadOnlyMount(
source=output.parent, target=proposer_sandbox.SANDBOX_REVIEW_OUTPUT
),
),
),
"/usr/bin/python3",
@ -897,9 +935,14 @@ Path('/workspace/review-output.json').write_text('{"schema_version":1}')
require_pid_namespace=True,
)
assert result.ok, result.stderr_tail
assert source.read_text() == "trusted\n"
assert output.read_text() == '{"schema_version":1}'
# Inside the sandbox scope: the artifact now lives under the session's
# private root, which prepare_sandbox removes on exit. run_arm reads it
# here too, while the session is still alive.
assert result.ok, result.stderr_tail
assert source.read_text() == "trusted\n"
assert output.read_text() == '{"schema_version":1}'
# The staging file is gone: the rename landed rather than a copy.
assert list(output.parent.iterdir()) == [output]
@pytest.mark.skipif(os.name == "nt", reason="symlink creation may require elevated Windows privileges")
@ -1170,6 +1213,7 @@ for line in sys.stdin:
review_command = """test -z "${ANTHROPIC_API_KEY:-}" && python3 - <<'PY'
import json
import os
import subprocess
from pathlib import Path
source = Path('/workspace/canary.txt')
@ -1182,7 +1226,10 @@ for operation in (lambda: source.write_text('forbidden'), lambda: source.rename(
pass
else:
raise AssertionError('source mutation was allowed')
Path('/workspace/review-output.json').write_text(json.dumps({'schema_version': 1, 'verdict': 'approve', 'findings': []}))
target = Path('/review-output/review-output.json')
staging = target.with_name(target.name + '.tmp.1.abc')
staging.write_text(json.dumps({'schema_version': 1, 'verdict': 'approve', 'findings': []}))
os.replace(staging, target)
PY"""
if review_layout:
for command in (
@ -1359,7 +1406,11 @@ PY"""
sandbox,
command_prefix=sandbox.command_prefix_for(
read_only_workspace=True,
extra_writable_mounts=(ReadOnlyMount(output, "/workspace/review-output.json"),),
extra_writable_mounts=(
ReadOnlyMount(
output.parent, proposer_sandbox.SANDBOX_REVIEW_OUTPUT
),
),
),
)
result = sandbox.run(
@ -1408,7 +1459,7 @@ PY"""
assert (sandbox.temp / "mcp-called").read_text() == "ok"
if review_layout:
assert output is not None
runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=output)
runner_artifacts.enforce_phase_workspace(clone, before, allowed_artifact=None)
assert json.loads(output.read_text())["verdict"] == "approve"
assert (clone / "canary.txt").read_text() == "hook-readable\nchanged for review\n"
finally:
@ -1418,3 +1469,98 @@ PY"""
if not review_layout:
assert (clone / "bash-called").read_text() == "canary"
def test_review_artifact_binds_a_writable_directory_outside_the_workspace(tmp_path):
"""The bwrap argv, since the mount shape is the whole bug.
bwrap cannot create a mount point inside an already-read-only bind, so a
writable path has to live outside /workspace — and it has to be the
directory, or the agent has nowhere to put the temp file it renames into
place.
"""
clone = tmp_path / "clone"
clone.mkdir()
with prepare_sandbox(clone=clone, claude_bin=sys.executable, backend="host-unsafe") as session:
sandbox = replace(session, backend="bwrap")
output = proposer_sandbox.review_output_path(sandbox, "review-output.json")
output.parent.mkdir(mode=0o700)
argv = sandbox.command_prefix_for(
read_only_workspace=True,
extra_writable_mounts=(
proposer_sandbox.ReadOnlyMount(
source=output.parent,
target=proposer_sandbox.SANDBOX_REVIEW_OUTPUT,
),
),
)
target = proposer_sandbox.SANDBOX_REVIEW_OUTPUT
assert not target.startswith(proposer_sandbox.SANDBOX_WORKSPACE + "/")
# The workspace itself is bound read-only...
workspace_at = argv.index(proposer_sandbox.SANDBOX_WORKSPACE)
assert argv[workspace_at - 2] == "--ro-bind"
# ...and the artifact directory is bound writable, as a directory.
artifact_at = argv.index(target)
assert argv[artifact_at - 2] == "--bind"
assert Path(argv[artifact_at - 1]) == output.parent
assert Path(argv[artifact_at - 1]).is_dir()
assert f"{proposer_sandbox.SANDBOX_WORKSPACE}/review-output.json" not in argv
@pytest.mark.skipif(
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
)
def test_real_bubblewrap_lets_a_review_artifact_be_written_atomically(tmp_path: Path) -> None:
"""The filesystem contract the EROFS defect broke, under a real sandbox.
Argv assertions cannot establish this. The artifact came back empty because
an atomic write - temp file beside the target, then rename - needs a
WRITABLE PARENT DIRECTORY, and only a real bwrap invocation shows whether
the mount grants one. A deterministic writer stands in for the agent: no
model session, no credentials.
Scope: this proves the filesystem and process contract of the production
mount configuration. It does not establish that a particular agent CLI's
own file-access policy permits the same operation - that is a second,
independent gate.
"""
clone = tmp_path / "clone"
clone.mkdir()
(clone / "tracked.txt").write_text("original\n")
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
review_output = prepare_review_workspace(sandbox, REVIEW_OUTPUT)
# The production configuration, not a hand-built mount tuple: the same
# command_prefix_for call run_arm makes for a review cell.
prefix = sandbox.command_prefix_for(
read_only_workspace=True,
extra_writable_mounts=(
ReadOnlyMount(source=review_output.parent, target=SANDBOX_REVIEW_OUTPUT),
),
)
target = f"{SANDBOX_REVIEW_OUTPUT}/{REVIEW_OUTPUT}"
script = (
# 1. temp file beside the destination, then atomic rename over it.
f'printf %s \'{{"schema_version": 1, "verdict": "approve", "findings": []}}\' > {target}.tmp && '
f"mv {target}.tmp {target} && "
# 2. the workspace must refuse the write that the mount forbids.
f"(printf x >> {SANDBOX_WORKSPACE}/tracked.txt 2>/dev/null && echo WORKSPACE-WRITABLE || echo workspace-readonly)"
)
result = subprocess.run(
[*prefix, "/bin/sh", "-c", script],
capture_output=True, text=True, timeout=60, check=False,
)
assert result.returncode == 0, f"atomic write failed inside the sandbox: {result.stderr[-400:]}"
assert "workspace-readonly" in result.stdout, "the workspace must stay read-only"
assert (clone / "tracked.txt").read_text() == "original\n", "the clone was modified"
# Read while the session is alive: the artifact lives under the private
# root that prepare_sandbox removes on exit, which is also why run_arm
# consumes it before leaving the scope.
_verdict, findings = parse_review_output(review_output)
assert findings == ()

View file

@ -0,0 +1,132 @@
"""The two providers' accounting equations, encoded literally.
Adding OpenAI's cache fields to its input_tokens double-counts, because they are
subsets of it. Subtracting Anthropic's under-counts, because they are additional
categories. A single generic struct cannot be right for both, so these tests
pin each equation rather than the field names.
"""
from __future__ import annotations
import pytest
from workflow_bench.provider_usage import (
ANTHROPIC,
OPENAI_RESPONSES,
UsageSemanticsError,
normalize_usage,
)
def _openai(input_tokens: int, cached: int | None = None, cache_write: int | None = None) -> dict:
details: dict[str, int] = {}
if cached is not None:
details["cached_tokens"] = cached
if cache_write is not None:
details["cache_write_tokens"] = cache_write
return {
"input_tokens": input_tokens,
"input_tokens_details": details,
"output_tokens": 300,
"output_tokens_details": {"reasoning_tokens": 250},
}
def test_openai_uncached_request_is_all_ordinary_input() -> None:
usage = normalize_usage(OPENAI_RESPONSES, _openai(1000, cached=0, cache_write=0))
assert usage.ordinary_input_tokens == 1000
assert usage.total_input_tokens == 1000
assert (usage.cache_read_input_tokens, usage.cache_write_input_tokens) == (0, 0)
def test_openai_cache_creation_keeps_the_parts_summing_to_input_tokens() -> None:
"""The subsets must reconstruct the whole, never exceed it."""
usage = normalize_usage(OPENAI_RESPONSES, _openai(1000, cached=0, cache_write=400))
assert usage.ordinary_input_tokens == 600
assert (
usage.ordinary_input_tokens
+ usage.cache_read_input_tokens
+ usage.cache_write_input_tokens
== usage.total_input_tokens
)
def test_openai_cache_hit_plus_new_write_uses_the_documented_subtraction() -> None:
usage = normalize_usage(OPENAI_RESPONSES, _openai(10_000, cached=7_000, cache_write=1_000))
assert usage.ordinary_input_tokens == 2_000
assert usage.total_input_tokens == 10_000, "input_tokens is the whole, not a component"
def test_openai_reasoning_tokens_decompose_output_rather_than_adding_to_it() -> None:
usage = normalize_usage(OPENAI_RESPONSES, _openai(100, cached=0, cache_write=0))
assert usage.output_tokens == 300
assert usage.reasoning_output_tokens == 250
assert usage.reasoning_output_tokens <= usage.output_tokens
def test_anthropic_uncached_total_is_just_input_tokens() -> None:
usage = normalize_usage(
ANTHROPIC,
{"input_tokens": 1000, "cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0, "output_tokens": 200},
)
assert usage.total_input_tokens == 1000
assert usage.ordinary_input_tokens == 1000
def test_anthropic_cached_total_adds_the_cache_categories() -> None:
"""The opposite equation to OpenAI's, on deliberately identical numbers."""
usage = normalize_usage(
ANTHROPIC,
{"input_tokens": 2_000, "cache_creation_input_tokens": 1_000,
"cache_read_input_tokens": 7_000, "output_tokens": 200},
)
assert usage.total_input_tokens == 10_000
assert usage.ordinary_input_tokens == 2_000
def test_the_same_numbers_mean_different_totals_on_the_two_providers() -> None:
"""The whole reason a shared struct is unsafe, in one assertion."""
openai = normalize_usage(OPENAI_RESPONSES, _openai(10_000, cached=7_000, cache_write=1_000))
anthropic = normalize_usage(
ANTHROPIC,
{"input_tokens": 10_000, "cache_creation_input_tokens": 1_000,
"cache_read_input_tokens": 7_000, "output_tokens": 300},
)
assert openai.total_input_tokens == 10_000
assert anthropic.total_input_tokens == 18_000
assert openai.ordinary_input_tokens == 2_000
assert anthropic.ordinary_input_tokens == 10_000
def test_missing_native_cache_fields_are_unknown_and_never_zero() -> None:
"""A zero we invented is indistinguishable from a zero the provider reported."""
usage = normalize_usage(OPENAI_RESPONSES, {"input_tokens": 1000, "output_tokens": 10})
assert usage.cache_read_input_tokens is None
assert usage.cache_write_input_tokens is None
assert usage.ordinary_input_tokens is None, "cannot subtract what was never reported"
assert usage.total_input_tokens == 1000
assert not usage.complete
assert "cache_read_input_tokens" in usage.unknown_fields
def test_an_absent_usage_object_is_entirely_unknown() -> None:
usage = normalize_usage(ANTHROPIC, None)
assert not usage.complete
assert usage.total_input_tokens is None
def test_an_unknown_provider_is_refused_rather_than_guessed() -> None:
with pytest.raises(UsageSemanticsError, match="refusing to guess"):
normalize_usage("some-new-provider", {"input_tokens": 1})
def test_cache_subsets_larger_than_the_whole_are_rejected() -> None:
"""Nonsense arithmetic must surface, not silently produce a negative."""
with pytest.raises(UsageSemanticsError, match="exceed input_tokens"):
normalize_usage(OPENAI_RESPONSES, _openai(100, cached=90, cache_write=50))

View file

@ -0,0 +1,313 @@
"""What the proxy writes must outlive the translation that follows it.
Claude Code receives an Anthropic-shaped response, which has nowhere to put
OpenAI's cached_tokens, cache_write_tokens or reasoning_tokens. If those are not
captured before the translation, the only remaining record of them is a bill.
"""
from __future__ import annotations
import contextlib
import json
from pathlib import Path
from types import SimpleNamespace
import pytest
from workflow_bench import litellm_usage_callback, provider_usage
from workflow_bench.litellm_usage_callback import USAGE_LOG_ENV_VAR, ProviderUsageLogger
from workflow_bench.model_gateway import (
OpenAIGateway,
USAGE_CALLBACK_MODULE,
openai_litellm_config,
write_openai_litellm_config,
)
from workflow_bench.provider_usage import (
ANTHROPIC,
OPENAI_RESPONSES,
USAGE_ENV_VARS,
normalize_usage,
)
class _Usage:
"""Stands in for the provider usage model LiteLLM hands the callback."""
def __init__(self, payload: dict) -> None:
self._payload = payload
def model_dump(self) -> dict:
return self._payload
def _openai_response(usage: dict) -> SimpleNamespace:
return SimpleNamespace(
id="resp_68f2c1",
# The model that actually answered, which is not the role the caller asked for.
model="gpt-5.6-sol-2026-08-01",
usage=_Usage(usage),
)
NATIVE = {
"input_tokens": 48_000,
"input_tokens_details": {"cached_tokens": 44_000, "cache_write_tokens": 1_000},
"output_tokens": 900,
"output_tokens_details": {"reasoning_tokens": 640},
}
@pytest.fixture
def logged(tmp_path: Path, monkeypatch: pytest.MonkeyPatch):
log = tmp_path / "provider_usage.jsonl"
monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(log))
def emit(usage: dict) -> dict:
ProviderUsageLogger()._append(
"success",
{"model": "claude-sonnet-4-5", "custom_llm_provider": "openai", "call_type": "responses"},
_openai_response(usage),
0.0,
1.0,
)
return json.loads(log.read_text().splitlines()[-1])
return emit
def test_native_openai_usage_survives_the_anthropic_translation(logged) -> None:
event = logged(NATIVE)
native = event["native_usage"]
# Verbatim: the fields an Anthropic-shaped response cannot carry.
assert native["input_tokens_details"]["cached_tokens"] == 44_000
assert native["input_tokens_details"]["cache_write_tokens"] == 1_000
assert native["output_tokens_details"]["reasoning_tokens"] == 640
assert event["response_id"] == "resp_68f2c1"
def test_the_actual_model_is_recorded_separately_from_the_requested_role(logged) -> None:
"""Pricing must follow what answered, not what the caller named."""
event = logged(NATIVE)
assert event["requested_model"] == "claude-sonnet-4-5"
assert event["actual_model"] == "gpt-5.6-sol-2026-08-01"
assert "cell_id" not in event, "a proxy-wide variable cannot identify a cell"
def test_the_captured_event_normalizes_with_openai_arithmetic(logged) -> None:
"""Capture and normalization must agree end to end, not just in isolation."""
event = logged(NATIVE)
# The provider the LOG recorded, not one the test supplies - passing
# OPENAI_RESPONSES by hand here is what hid the adapter-key mismatch.
assert event["provider"] == OPENAI_RESPONSES
assert event["provider_label"] == "openai"
usage = normalize_usage(event["provider"], event["native_usage"])
assert usage.total_input_tokens == 48_000
assert usage.ordinary_input_tokens == 3_000
assert usage.cache_read_input_tokens == 44_000
assert usage.complete
def test_usage_without_details_normalizes_to_unknown_rather_than_zero(logged) -> None:
"""The mutation the accounting must not survive: dropped details, silent zeros."""
stripped = {k: v for k, v in NATIVE.items() if k != "input_tokens_details"}
event = logged(stripped)
usage = normalize_usage(event["provider"], event["native_usage"])
assert usage.cache_read_input_tokens is None
assert usage.ordinary_input_tokens is None
assert not usage.complete
def test_a_failed_request_is_still_accounted_for(logged, tmp_path: Path) -> None:
"""The money was spent whether or not the cell produced an artifact."""
import asyncio
logger = ProviderUsageLogger()
args = ({"model": "claude-sonnet-4-5"}, _openai_response(NATIVE), 0.0, 1.0)
# Every hook LiteLLM can call, not the private helper underneath them: the
# sync failure hook was missing entirely and _append could never show that.
logger.log_failure_event(*args)
asyncio.run(logger.async_log_failure_event(*args))
events = [json.loads(line) for line in (tmp_path / "provider_usage.jsonl").read_text().splitlines()]
assert len(events) == 2, "both failure hooks must record"
assert all(e["status"] == "failure" for e in events)
def test_every_public_outcome_hook_records(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""Overriding a subset silently drops whichever path LiteLLM actually uses."""
import asyncio
monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(tmp_path / "usage.jsonl"))
logger = ProviderUsageLogger()
args = ({"model": "m"}, _openai_response(NATIVE), 0.0, 1.0)
logger.log_success_event(*args)
logger.log_failure_event(*args)
asyncio.run(logger.async_log_success_event(*args))
asyncio.run(logger.async_log_failure_event(*args))
events = [json.loads(line) for line in (tmp_path / "usage.jsonl").read_text().splitlines()]
assert [e["status"] for e in events] == ["success", "failure", "success", "failure"]
def test_the_logger_never_raises_into_the_proxy(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
"""Accounting is evidence, not control flow."""
monkeypatch.setenv(USAGE_LOG_ENV_VAR, str(tmp_path / "missing-dir" / "usage.jsonl"))
ProviderUsageLogger()._append("success", {}, object(), 0.0, 1.0)
def test_no_log_is_written_when_the_destination_is_unset(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.delenv(USAGE_LOG_ENV_VAR, raising=False)
ProviderUsageLogger()._append("success", {}, _openai_response(NATIVE), 0.0, 1.0)
assert not list(tmp_path.iterdir())
def test_the_generated_config_loads_the_callback_from_beside_itself(tmp_path: Path) -> None:
"""LiteLLM resolves the dotted path relative to the config directory."""
config = write_openai_litellm_config(tmp_path / "litellm.yaml", ["gpt-5.6-sol"])
assert openai_litellm_config(["gpt-5.6-sol"])["litellm_settings"]["callbacks"] == [
f"{USAGE_CALLBACK_MODULE}.handler"
]
installed = config.parent / f"{USAGE_CALLBACK_MODULE}.py"
assert installed.is_file(), "the proxy cannot import a callback that was never placed"
# Importing it, not grepping it: a text search passes even when the module
# cannot load, which is exactly how a package-relative import survived
# review here. This is the deployment configuration, so load it the way the
# proxy does - by path, as a top-level module.
import importlib.util
spec = importlib.util.spec_from_file_location(USAGE_CALLBACK_MODULE, installed)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
assert isinstance(module.handler, module.ProviderUsageLogger)
def test_the_gateway_forwards_the_usage_environment_into_the_proxy(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""The proxy is a separate process with a constructed environment.
Popen(env=...) replaces the parent environment rather than extending it, so
a variable the callback reads is simply absent unless the gateway forwards
it by name. Without this the accounting looks configured and silently
records nothing on every request - the in-process tests above cannot see
that, because they never cross the subprocess boundary.
"""
for name in USAGE_ENV_VARS:
monkeypatch.setenv(name, f"value-for-{name}")
captured: dict[str, dict[str, str]] = {}
class _Popen:
def __init__(self, *_a, **kwargs):
captured["env"] = kwargs["env"]
raise RuntimeError("stop before launching a real proxy")
# The console-script resolver runs before Popen and is absent in this
# environment (the same reason two gateway tests fail here); the argv it
# builds is not what this test is about.
monkeypatch.setattr(
"workflow_bench.model_gateway.litellm_proxy_argv",
lambda **_k: ["/bin/true"],
)
monkeypatch.setattr("workflow_bench.model_gateway.subprocess.Popen", _Popen)
gateway = OpenAIGateway(
openai_api_key="sk-test",
model_names=["gpt-5.6-sol"],
work_dir=tmp_path,
)
with contextlib.suppress(Exception):
gateway.__enter__()
env = captured.get("env")
assert env is not None, "the proxy was never constructed"
for name in USAGE_ENV_VARS:
assert env.get(name) == f"value-for-{name}", f"{name} never reached the proxy"
# The credential allowlist is still an allowlist, not the parent environment.
assert "PATH" in env and len(env) < 40
def test_an_unresolvable_provider_is_refused_rather_than_guessed() -> None:
"""LiteLLM says "openai" for Chat Completions too, and it counts differently."""
from workflow_bench.provider_usage import canonical_provider
assert canonical_provider("openai", "responses") == OPENAI_RESPONSES
assert canonical_provider("openai", "completion") is None
assert canonical_provider("openai", None) is None
assert canonical_provider("anthropic", "completion") == ANTHROPIC
def test_request_identity_cannot_come_from_the_proxy_environment() -> None:
"""One proxy serves the whole sweep, so its environment identifies the sweep.
attach_openai_gateway wraps all of _run_sweep, and cells run concurrently
under --workers, interleaving requests through that single process. Any
variable forwarded at launch is therefore constant for every event it ever
records. Pinned so a future change does not reintroduce a per-cell
environment variable that would silently stamp one value on every request.
"""
assert USAGE_ENV_VARS == (
"GITNEXUS_BENCH_PROVIDER_USAGE",
"GITNEXUS_BENCH_SWEEP_ID",
), "a per-cell variable here would be constant across concurrent cells"
def test_a_request_records_its_session_so_attribution_stays_possible(logged) -> None:
"""The per-request half of identity, recorded even when the provider omits it."""
event = logged(NATIVE)
assert "session_id" in event, "absent attribution is still a fact about the run"
def test_the_callback_imports_the_way_litellm_actually_loads_it(tmp_path: Path) -> None:
"""By path, as a top-level module, with no parent package and no sys.path entry.
LiteLLM resolves a dotted callback through spec_from_file_location against
the config directory, so the copied file is not part of workflow_bench when
it runs. A relative or sibling import therefore raises ImportError and the
proxy exits before becoming ready - which the in-package tests cannot see,
because they import it as workflow_bench.litellm_usage_callback.
"""
import importlib.util
import shutil
source = Path(litellm_usage_callback.__file__)
installed = tmp_path / f"{USAGE_CALLBACK_MODULE}.py"
shutil.copy(source, installed)
spec = importlib.util.spec_from_file_location(USAGE_CALLBACK_MODULE, installed)
assert spec is not None and spec.loader is not None
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module) # ImportError here is the proxy refusing to start
assert hasattr(module, "handler")
def test_the_callbacks_copied_constants_match_the_canonical_ones() -> None:
"""The copies are deliberate; drifting apart silently is not.
The callback cannot import from the package (see the test above), so it
carries its own literals. These assertions are what keep the duplication
honest.
"""
assert litellm_usage_callback.USAGE_LOG_ENV_VAR == provider_usage.USAGE_LOG_ENV_VAR
assert litellm_usage_callback.SWEEP_ID_ENV_VAR == provider_usage.SWEEP_ID_ENV_VAR
for label, call_type in (
("openai", "responses"),
("openai", "completion"),
("openai", None),
("anthropic", "completion"),
("mystery", "responses"),
):
assert litellm_usage_callback.canonical_provider(label, call_type) == provider_usage.canonical_provider(
label, call_type
), f"resolver drifted for {label!r}/{call_type!r}"

View file

@ -0,0 +1,268 @@
"""A row the runner actually emits must satisfy the reuse reader.
Every existing comparator-reuse test builds its rows by hand. That proves the
predicate's logic and nothing about the producer: a fixture can satisfy
eligibility while a real emitted row never does, and the audit that counts key
names cannot tell the difference. These tests carry one record through the
production path instead:
real run_cell -> production JSONL writer -> load_result_rows
-> row_is_reusable_comparator
Only the expensive dependencies are replaced - the model session, sandbox
launch, repository acquisition, graph preparation. The digest fields the reuse
binding compares are assembled by run_cell itself from its TaskCellContext, so
they stay real: they are the subject of the test, not scaffolding around it.
"""
from __future__ import annotations
import json
from datetime import UTC, datetime, timedelta
from pathlib import Path
from types import SimpleNamespace
from typing import Any
import pytest
from workflow_bench import runner
from workflow_bench.proposer_sandbox import redact_text
from workflow_bench.model_gateway import credential_secrets
from workflow_bench.runner_sessions import PARENT_EVENT_STREAM_SOURCE
from workflow_bench.comparator_reuse import (
ComparatorReuseExpectation,
TaskReuseBinding,
load_result_rows,
row_is_reusable_comparator,
)
TASK_ID = "review-pr-2718-defect"
SHA = "a" * 40
def _snapshot(prefix: str) -> SimpleNamespace:
return SimpleNamespace(
digest=f"{prefix}-content",
manifest_digest=f"{prefix}-manifest",
dependency_content_digest=f"{prefix}-dep-content",
dependency_manifest_digest=f"{prefix}-dep-manifest",
command_digest=f"{prefix}-command",
materialize=lambda *a, **k: None,
)
def _write_like_the_sweep(tmp_path: Path, row: dict[str, Any]) -> Path:
"""Serialize exactly as ``keep`` does in _run_sweep, redaction included.
json.dumps + write_text would skip the redaction the real writer applies,
so a change there could break reusable rows without failing this test - and
redaction is not cosmetic here, since it rewrites the row's own bytes.
"""
results = tmp_path / "results.jsonl"
secrets = credential_secrets(
SimpleNamespace(auth_token="sk-ant-should-never-appear", base_url=None)
)
with results.open("a") as handle:
handle.write(redact_text(json.dumps(row), secrets) + "\n")
return results
@pytest.fixture
def emitted_row(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> dict[str, Any]:
"""One record from the real run_cell, with only expensive work replaced."""
worktree = tmp_path / "clone"
worktree.mkdir()
# The session is what costs money; everything it returns is scripted. The
# record's binding fields are NOT set here - run_cell derives them.
def fake_run_arm(*_a: Any, **_k: Any) -> dict[str, Any]:
return {
"ok": True,
"error_kind": None,
"error_detail": None,
"resolved": True,
"review_evidence_valid": True,
"review_score": {"weighted_f1": 0.5},
"review_weighted_f1": 0.5,
"skill_invoked": True,
"skill_digest": "skill-digest",
"transcript_missing": False,
"transcript_artifacts": [
{
"path": "transcripts/session-1.jsonl",
"sha256": __import__("hashlib").sha256(b'{"type":"ok"}\n').hexdigest(),
"bytes": 14,
"source": PARENT_EVENT_STREAM_SOURCE,
}
],
"session_ids": ["s1"],
"num_turns": 3,
"duration_s": 1.0,
"cost_usd": 0.5,
"input_tokens": 1,
"output_tokens": 1,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
}
for name, value in {
"run_arm": fake_run_arm,
"copy_isolated_tree": lambda *a, **k: worktree,
"make_worktree": lambda *a, **k: worktree,
"sanitize_clone_for_hidden_oracles": lambda *a, **k: SHA,
"stage_task_assets": lambda *a, **k: (),
"isolated_gitnexus_registry_mount": lambda *a, **k: None,
"seed_evaluated_skills": lambda *a, **k: None,
"apply_candidate_overlay": lambda *a, **k: None,
"require_hidden_harness_absent": lambda *a, **k: None,
"require_skill_fingerprint": lambda *a, **k: None,
"enforce_work_evidence": lambda *a, **k: None,
"skill_fingerprint": lambda *a, **k: "skill-digest",
"capture_patch": lambda *a, **k: b"",
"implementation_diff_digest": lambda *a, **k: "",
"diff_churn": lambda *a, **k: {},
"_prepare_untracked_for_diff": lambda *a, **k: None,
"remove_clone": lambda *a, **k: None,
"ce_plugin_dir_for_arm": lambda *a, **k: None,
"ce_plugin_mounts_for_arm": lambda *a, **k: (),
"current_runtime_digest": lambda: "runtime-digest",
"build_sandbox_environment": lambda *a, **k: {},
"credential_secrets": lambda *a, **k: (),
# run_cell requires an immutable base commit before it will record a
# cell; the git plumbing is expensive setup, the SHA it returns is not
# part of the reuse binding under test.
"_sandbox_git": lambda *a, **k: SHA,
# The artifact copy is real; only the read of the agent-written file is
# replaced, since no agent ran to write one.
"_bounded_regular_bytes": lambda *a, **k: b'{"schema_version":1}',
}.items():
monkeypatch.setattr(runner, name, value)
class _Sandbox:
clone = worktree
private_root = tmp_path / "private"
backend = "test-double"
settings_json = "{}"
require_pid_namespace = False
def __enter__(self) -> _Sandbox:
return self
def __exit__(self, *_exc: Any) -> bool:
return False
def command_prefix_for(self, **_k: Any) -> list[str]:
return []
def run(self, *_a: Any, **_k: Any) -> SimpleNamespace:
return SimpleNamespace(ok=True, returncode=0, stdout_tail="", stderr_tail="")
def environment(self, **_k: Any) -> dict[str, str]:
return {}
def host_text(self, value: str) -> str:
return value
monkeypatch.setattr(runner, "prepare_sandbox", lambda **_k: _Sandbox())
ctx = runner.TaskCellContext(
task={"id": TASK_ID, "prompt": "review it", "verify": "true"},
oracle_snapshot=_snapshot("oracle"),
repo=tmp_path / "repo",
task_sha=SHA,
graph_snapshot=_snapshot("graph"),
graph_snapshot_error=None,
asset_snapshot=_snapshot("asset"),
asset_snapshot_error=None,
args=SimpleNamespace(
model="gpt-5.6-sol", effort="xhigh", timeout=60, claude_bin="claude",
base_url=None, auth_token=None, permission_mode=None, arms=["review"],
proposer_model=None, outage_streak=5, runs=1, workers=1,
),
out_dir=tmp_path / "out",
ce_plugin_snapshot=None,
trees_dir=tmp_path / "trees",
bwrap_bin=Path("/bin/true"),
runtime_mounts=(),
candidate_overlay=None,
overlay_digest=None,
sandbox_backend="test-double",
clone_template=None,
sanitized_head=SHA,
)
(tmp_path / "out").mkdir(exist_ok=True)
(tmp_path / "trees").mkdir(exist_ok=True)
(tmp_path / "private").mkdir(exist_ok=True)
# run_cell records review_artifact only when the review source exists, and
# reuse now requires it - a scored review with no artifact is a claim about
# evidence rather than the evidence. Production writes this file; the
# fixture has to as well, or the emitted row is one production never emits.
review_dir = tmp_path / "private" / "review-output"
review_dir.mkdir(exist_ok=True)
(review_dir / "review-output.json").write_text('{"schema_version": 1, "verdict": "approve", "findings": []}')
return runner.run_cell(ctx, 0, "review")
def _expectation(**overrides: Any) -> ComparatorReuseExpectation:
"""Bindings from the sweep's own configuration, not copied out of the row.
Copying the emitted values back in would make producer and consumer agree
because the test arranged it, which is the blind spot being closed.
"""
binding = TaskReuseBinding(
task_base_sha=SHA,
task_prompt_digest=runner.hashlib.sha256(b"review it").hexdigest(),
oracle_digest="oracle-content",
oracle_command_digest="oracle-command",
oracle_manifest_digest="oracle-manifest",
task_asset_manifest_digest="asset-manifest",
sandbox_dependency_manifest_digest="asset-dep-manifest",
)
values: dict[str, Any] = dict(
model="gpt-5.6-sol",
effort="xhigh",
sandbox_backend="test-double",
runtime_digest="runtime-digest",
now=datetime.now(UTC),
max_age=timedelta(days=90),
tasks={TASK_ID: binding},
skill_digests={"review": "skill-digest"},
ce_plugin_version=None,
ce_plugin_manifest_digest=None,
)
values.update(overrides)
return ComparatorReuseExpectation(**values)
def test_a_row_the_runner_emitted_survives_serialization_and_qualifies(
emitted_row: dict[str, Any], tmp_path: Path
) -> None:
"""The producer/consumer contract, end to end through the real writer."""
results = _write_like_the_sweep(tmp_path, emitted_row)
rows = load_result_rows(results)
assert len(rows) == 1, "the production row must survive the reader"
assert row_is_reusable_comparator(rows[0], _expectation()) is True
def test_a_changed_binding_rejects_the_same_emitted_row(
emitted_row: dict[str, Any], tmp_path: Path
) -> None:
"""Fails closed on drift, so the positive case is not vacuous."""
row = load_result_rows(_write_like_the_sweep(tmp_path, emitted_row))[0]
binding = TaskReuseBinding(
task_base_sha=SHA,
task_prompt_digest=runner.hashlib.sha256(b"review it").hexdigest(),
oracle_digest="oracle-content",
oracle_command_digest="oracle-command",
oracle_manifest_digest="oracle-manifest",
task_asset_manifest_digest="asset-manifest",
sandbox_dependency_manifest_digest="DIFFERENT-dependencies",
)
assert row_is_reusable_comparator(row, _expectation(tasks={TASK_ID: binding})) is False

View file

@ -32,6 +32,10 @@ def test_review_corpus_is_immutable_and_task_bound():
assert task["ref"] == case["base_sha"]
assert task["sandbox_copy"] == [f"eval/workflow_bench/review_cases/{patch.name}"]
assert task["setup"] == review_case_setup_command(patch.name)
assert any(
dep.get("source") == "gitnexus-shared/dist" and dep.get("target") == "gitnexus-shared/dist"
for dep in task["sandbox_dependencies"]
)
def test_hidden_labels_are_not_recoverable_from_visible_task_input():

View file

@ -325,3 +325,33 @@ def test_clean_control_rewards_an_empty_approval_and_penalizes_noise():
assert noisy["recall"] is None
assert noisy["clean_pass"] is False
assert noisy["verdict_correct"] is False
def test_parse_review_output_names_the_actual_failure(tmp_path: Path):
"""One message per cause.
Folding empty, malformed and encoding failures together makes a sandbox that
left the artifact at 0 bytes indistinguishable from an encoding fault: every
such cell reports "not valid UTF-8 JSON". A file the agent never created
escaped that fold — lstat sat outside the try, so it raised
FileNotFoundError — but only as a bare OSError, naming no cause at all.
"""
missing = tmp_path / "never-written.json"
with pytest.raises(ValueError, match="was never written"):
parse_review_output(missing)
empty = tmp_path / "empty.json"
empty.touch()
with pytest.raises(ValueError, match="is empty"):
parse_review_output(empty)
not_utf8 = tmp_path / "latin1.json"
not_utf8.write_bytes(b'{"verdict": "\xff\xfe"}')
with pytest.raises(ValueError, match="not valid UTF-8"):
parse_review_output(not_utf8)
prose = tmp_path / "prose.json"
prose.write_text("Here is my review of the changes.", encoding="utf-8")
with pytest.raises(ValueError, match="not valid JSON"):
parse_review_output(prose)

View file

@ -3,13 +3,14 @@
import hashlib
import json
import shutil
import subprocess
from contextlib import nullcontext
from pathlib import Path
from types import SimpleNamespace
import pytest
from workflow_bench import runner, runner_artifacts, runner_sessions
from workflow_bench import proposer_sandbox, runner, runner_artifacts, runner_sessions
from workflow_bench.evolution import skill_fingerprint
from workflow_bench.oracle_assets import review_case_setup_command
from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult
@ -608,6 +609,67 @@ def test_run_cell_reports_a_cleanup_failure_over_its_primary_outcome(monkeypatch
assert "clone is busy" in record["error_detail"]
def _git(repo, *args):
return subprocess.run(["git", "-C", str(repo), *args], check=True, capture_output=True, text=True)
def test_run_cell_runs_the_arm_against_a_copy_of_the_clone_template(monkeypatch, tmp_path):
"""run_cell must copy the template, never re-clone.
run_cell takes the clone-template branch on essentially every multi-cell
sweep: it copies a pre-sanitized template rather than paying `git clone
--no-local` plus repack/prune/fsck per cell. Asserting on a copy the test
makes itself proves nothing about that branch — the clone the arm receives
is what has to come from the template, carrying the template's sanitized
HEAD rather than a recomputed one.
"""
repo = tmp_path / "repo"
repo.mkdir()
_git(repo, "init", "--quiet")
_git(repo, "checkout", "--quiet", "-b", "main")
(repo / "from-template.txt").write_text("sanitized\n")
_git(repo, "add", "-A")
_git(repo, "-c", "user.name=test", "-c", "user.email=test@invalid", "commit", "--quiet", "-m", "base")
sha = _git(repo, "rev-parse", "HEAD").stdout.strip()
trees = tmp_path / "trees"
trees.mkdir()
template = runner.make_worktree(repo, sha, trees)
template_head = _git(template, "rev-parse", "HEAD").stdout.strip()
_stub_cell_dependencies(monkeypatch, tmp_path)
def fail_if_recloned(*_args, **_kwargs):
raise AssertionError("clone template present: run_cell must not re-clone")
monkeypatch.setattr(runner, "make_worktree", fail_if_recloned)
monkeypatch.setattr(runner, "sanitize_clone_for_hidden_oracles", fail_if_recloned)
seen: dict[str, object] = {}
def record_arm(_arm, _task, worktree, _args, **_kwargs):
seen["worktree"] = worktree
seen["head"] = _git(worktree, "rev-parse", "HEAD").stdout.strip()
seen["content"] = (worktree / "from-template.txt").read_text()
# The copy is a private checkout: what the cell writes must not reach
# the template the other cells of this task still copy from.
(worktree / "from-template.txt").write_text("cell-local\n")
return {"resolved": True, "ok": True, "error_kind": None}
monkeypatch.setattr(runner, "run_arm", record_arm)
runner.run_cell(
_cell_context(tmp_path, clone_template=template, sanitized_head=template_head),
0,
"workflow",
)
assert seen["content"] == "sanitized\n"
assert seen["head"] == template_head
assert seen["worktree"] != template
assert (template / "from-template.txt").read_text() == "sanitized\n"
def test_run_cell_does_not_mask_the_staged_review_patch_before_setup(monkeypatch, tmp_path):
"""Review setup applies a patch staged under eval/workflow_bench.
@ -1002,3 +1064,40 @@ def test_progress_line_reports_the_numbers_a_real_run_measured():
assert "cost=$0.5" in line
assert "took=12.0s" in line
assert "error_kind=none" in line
def test_claude_settings_allow_the_review_artifact_directory():
"""The second gate on the artifact path.
The bwrap bind is not the only thing that decides whether the agent can
write: the CLI applies this filesystem policy to its own tools, so a path
missing from allowWrite is unwritable however the mount is shaped. The
artifact lived under /workspace when this list was written, which is why
moving it out needed this entry and nothing caught the omission.
"""
settings = json.loads(proposer_sandbox.build_claude_settings(sandbox_enabled=True))
filesystem = settings["sandbox"]["filesystem"]
assert proposer_sandbox.SANDBOX_REVIEW_OUTPUT in filesystem["allowWrite"]
assert proposer_sandbox.SANDBOX_REVIEW_OUTPUT in filesystem["allowRead"]
assert filesystem["denyRead"] == ["/"]
def test_review_contract_tells_the_agent_the_writable_path():
prompt = runner.REVIEW_PROMPT.format(task="task text")
assert f"{runner.SANDBOX_REVIEW_OUTPUT}/{runner.REVIEW_OUTPUT}" in prompt
assert f"{runner.SANDBOX_WORKSPACE}/{runner.REVIEW_OUTPUT}" not in prompt
# The JSON shape survives .format() with its braces intact.
assert '{"schema_version":1' in prompt
artifact = f"{runner.SANDBOX_REVIEW_OUTPUT}/{runner.REVIEW_OUTPUT}"
assert runner.CE_REVIEW_PROMPT.format(task="task text").count(artifact) == 1
def test_enforce_phase_workspace_can_require_an_untouched_workspace(tmp_path):
(tmp_path / "tracked.py").write_text("original\n")
before = runner_artifacts.workspace_snapshot(tmp_path)
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=None)
(tmp_path / "tracked.py").write_text("the review edited the code it was reviewing\n")
with pytest.raises(ValueError, match="changed the read-only workspace"):
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=None)

View file

@ -212,6 +212,22 @@ def test_prepare_sanitized_graph_builds_once_from_parentless_tree_and_caches_onl
assert removed == [seed]
def test_prepare_sanitized_graph_requires_head_when_given_a_template(tmp_path: Path):
with pytest.raises(SandboxError, match="sanitized HEAD"):
sanitized_graph.prepare_sanitized_graph(
{},
repo=tmp_path,
resolved_sha="b" * 40,
parent=tmp_path,
cache=SimpleNamespace(), # type: ignore[arg-type]
claude_bin="claude",
bwrap_bin="bwrap",
runtime_mounts=(),
clone_template=tmp_path,
sanitized_head=None,
)
def test_graph_snapshot_rejects_arm_sanitization_identity_drift(tmp_path: Path):
assets = SimpleNamespace(
digest="digest",

View file

@ -11,7 +11,7 @@ import io
import json
import time
from workflow_bench.runner_sessions import SessionProgress
from workflow_bench.runner_sessions import SessionProgress, neutralize_ci_log_text
def _drain_lines(stream: io.StringIO) -> list[str]:
@ -325,3 +325,48 @@ def test_cell_failure_detail_line_bounds_a_huge_detail() -> None:
assert line is not None
assert "truncated" in line
assert len(line) < MAX_CELL_DETAIL_CHARS + 200
def test_progress_neutralizes_github_actions_annotation_forms() -> None:
rewritten = neutralize_ci_log_text(
"gitnexus/src/cli/optional-grammars.ts(18,36): error TS2307: Cannot find module "
"'gitnexus-shared'\n::error::Composite projects may not disable incremental compilation.\n"
"##[error]tsc failed"
)
assert "): error TS2307" not in rewritten
assert "): compiler-error TS2307" in rewritten
assert "::error::" not in rewritten
assert "[:]error::" in rewritten
assert "##[error]" not in rewritten
assert "# [error]tsc failed" in rewritten
stream = io.StringIO()
progress = SessionProgress("review-pr-2718-defect-ce_review-run0", stream=stream, heartbeat_s=3600)
events = [
{
"type": "assistant",
"message": {
"content": [{"type": "tool_use", "id": "b1", "name": "Bash", "input": {"command": "npx tsc --noEmit"}}]
},
},
{
"type": "user",
"message": {
"content": [
{
"type": "tool_result",
"tool_use_id": "b1",
"is_error": True,
"content": "gitnexus/src/cli/optional-grammars.ts(18,36): error TS2307: Cannot find module 'gitnexus-shared'",
}
]
},
},
]
for event in events:
_observe(progress, (json.dumps(event) + "\n").encode())
output = stream.getvalue()
assert "): error TS2307" not in output
assert "): compiler-error TS2307" in output
assert "result=error" in output

View file

@ -0,0 +1,279 @@
"""The real sweep must reach the right finalization decision.
`enforce_measurement_health` is unit-tested and the call site is pinned
structurally, but neither shows the guard running inside a sweep. These drive
the real `_run_sweep` with cell execution scripted and everything downstream of
it left alone: folding, aggregation, the artifact writers, the health guard and
the exit selection.
The below-breaker case is the decisive one. A fixture of many unusable cells
aborts through the pre-existing outage breaker instead - `review-evidence-invalid`
is systemic with a limit of 5 - and would pass whether or not the finalization
guard exists. One fresh unusable cell stays under that threshold, so only the
guard can catch it.
"""
from __future__ import annotations
import json
import threading
from collections.abc import Callable
from pathlib import Path
from types import SimpleNamespace
from typing import Any
import pytest
from tests.bench_fixtures import scored_review_row, unusable_review_row
from workflow_bench import runner
TASK = {
"id": "review-pr-2718-defect",
"repo": "~/GitNexus",
"ref": "a" * 40,
"prompt": "review it",
"verify": "true",
"class": "review-defect",
}
def _args(out: Path, **overrides: Any) -> SimpleNamespace:
values: dict[str, Any] = dict(
arms=["review"], claude_bin="claude", effort="xhigh", model="gpt-5.6-sol",
out=out, outage_streak=runner.DEFAULT_OUTAGE_STREAK, promotion_max_task_regression=10.0,
promotion_metric="review_weighted_f1", promotion_min_improvement=1.0,
promotion_min_runs=1, proposer_model=None, reuse_results=None, runs=1, workers=1,
timeout=60, base_url=None, auth_token=None, permission_mode=None,
)
values.update(overrides)
return SimpleNamespace(**values)
def _snapshot(prefix: str) -> SimpleNamespace:
return SimpleNamespace(
digest=f"{prefix}-content", manifest_digest=f"{prefix}-manifest",
dependency_content_digest=f"{prefix}-dep", dependency_manifest_digest=f"{prefix}-depman",
command_digest=f"{prefix}-command", materialize=lambda *a, **k: None,
)
def _sweep(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
record: dict[str, Any] | Callable[[int], dict[str, Any]],
*,
runs: int = 1,
cancel_event: threading.Event | None = None,
candidate_arms: list[str] | None = None,
arms: list[str] | None = None,
after_cell: Callable[[int, str], None] | None = None,
):
"""Drive the real _run_sweep; only cell execution and setup are scripted.
``after_cell`` runs once a cell's record exists, which is how a test sets
cancellation deterministically at a known point instead of racing a sleep.
"""
out = tmp_path / "out"
def scripted_cell(_ctx: Any, run_idx: int, arm: str) -> dict[str, Any]:
row = dict(record(run_idx) if callable(record) else record)
row.update({"task": TASK["id"], "arm": arm, "run": run_idx, "class": TASK["class"]})
if after_cell is not None:
after_cell(run_idx, arm)
return row
monkeypatch.setattr(runner, "run_cell", scripted_cell)
monkeypatch.setattr(runner, "ensure_task_graph", lambda **k: k["env"].graph_snapshots.__setitem__(
k["graph_key"], _snapshot("graph")))
monkeypatch.setattr(runner.TaskAssetCache, "prepare", lambda self, *a, **k: _snapshot("asset"))
# Binding resolution clones the repo and verifies the ref; that is expensive
# setup, and the bindings it would return are supplied directly instead.
monkeypatch.setattr(
runner, "resolve_task_bindings",
lambda tasks, expected, **k: list(expected),
)
return runner._run_sweep(
_args(out, runs=runs, arms=arms or ["review"]),
parser=SimpleNamespace(error=lambda m: (_ for _ in ()).throw(SystemExit(2))),
tasks=[TASK],
skipped_expensive=[],
oracle_snapshots=[_snapshot("oracle")],
expected_task_bindings=[{"repo_identity": str(tmp_path / "repo"), "resolved_sha": "a" * 40}],
ce_plugin_config=None,
bwrap_bin=Path("/bin/true"),
sandbox_backend="test-double",
runtime_mounts=(),
candidate_arms=candidate_arms or [],
candidate_overlay=None,
overlay_digest=None,
promotion_target_bases={},
cancel_event=cancel_event,
), out
def test_one_unusable_cell_below_the_breaker_reaches_the_finalization_guard(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
) -> None:
"""The decisive case: too few failures to trip the breaker, so only the guard can catch it."""
streak = runner.systemic_outage_streak("review-evidence-invalid", 0)
assert streak < runner.DEFAULT_OUTAGE_STREAK, "fixture must stay under the breaker"
unusable = unusable_review_row()
with pytest.raises(SystemExit) as exc:
_sweep(tmp_path, monkeypatch, unusable)
assert exc.value.code == 1
out = capsys.readouterr().out
assert "review: UNUSABLE" in out, "the guard must name the arm and its status"
assert "systemic-outage" not in out, "the breaker must not have tripped"
def test_a_zero_score_stays_a_valid_negative_measurement(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
) -> None:
"""0.0 is a present measurement, not missing evidence.
A truthiness check on the score would misread it as absent and turn a
quality result into an execution-health failure.
"""
zeroed = scored_review_row(
resolved=False, error_kind="oracle-failed",
review_score={"weighted_f1": 0.0}, review_weighted_f1=0.0,
)
_sweep(tmp_path, monkeypatch, zeroed)
out = capsys.readouterr().out
assert "review: OBSERVED_OK" in out
assert "UNUSABLE" not in out
def test_finalization_persists_results_and_report(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""Evidence must survive the sweep, and say the same thing the exit does."""
scored = scored_review_row(resolved=False, error_kind="oracle-failed", review_weighted_f1=0.2)
_result, out = _sweep(tmp_path, monkeypatch, scored)
rows = [json.loads(line) for line in (out / "results.jsonl").read_text().splitlines()]
assert len(rows) == 1 and rows[0]["review_weighted_f1"] == 0.2
assert (out / "report.md").is_file()
def test_cancellation_without_an_outage_exits_130(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
) -> None:
"""An interrupted sweep is interrupted, not aborted.
One admissible cell lands first so the measurement-health guard classifies
the arm DEGRADED rather than UNUSABLE - otherwise the guard would supply
exit 1 and this test would pass without ever exercising exit selection.
"""
cancel_event = threading.Event()
with pytest.raises(SystemExit) as exc:
_sweep(
tmp_path, monkeypatch, lambda _run: scored_review_row(),
runs=3, cancel_event=cancel_event,
after_cell=lambda run_idx, _arm: cancel_event.set() if run_idx == 0 else None,
)
stdout = capsys.readouterr().out
report = (tmp_path / "out" / "report.md").read_text()
assert "Sweep cancelled" in report, "an interruption must be reported as one"
assert "systemic-outage" not in stdout, "no breaker trip in this scenario"
assert exc.value.code == 130
def test_an_outage_keeps_exit_1_even_though_the_breaker_cancels(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str]
) -> None:
"""Precedence: the breaker sets cancel_event, so order decides the exit.
Testing cancellation first would relabel every outage a Ctrl-C. The first
cell is admissible for the same reason as above, and the failures after it
are consecutive and systemic, which is what the breaker actually counts.
"""
def cell(run_idx: int) -> dict[str, Any]:
return scored_review_row() if run_idx == 0 else unusable_review_row()
cancel_event = threading.Event()
with pytest.raises(SystemExit) as exc:
_sweep(tmp_path, monkeypatch, cell,
runs=1 + runner.DEFAULT_OUTAGE_STREAK, cancel_event=cancel_event)
stdout = capsys.readouterr().out
report = (tmp_path / "out" / "report.md").read_text()
assert "systemic-outage" in stdout, "the real breaker must have tripped"
assert cancel_event.is_set(), "the breaker cancels in-flight work"
assert "Sweep aborted" in report
assert exc.value.code == 1, "an outage must not become the 130 of a Ctrl-C"
def test_an_interrupted_sweep_keeps_the_evidence_it_already_paid_for(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""Cancellation must not discard rows that already cost money.
The completed-run persistence test cannot show this: it never interrupts, so
it would pass even if the writer only ran on the clean path.
"""
cancel_event = threading.Event()
with pytest.raises(SystemExit):
_sweep(
tmp_path, monkeypatch, lambda _run: scored_review_row(review_weighted_f1=0.42),
runs=3, cancel_event=cancel_event,
after_cell=lambda run_idx, _arm: cancel_event.set() if run_idx == 0 else None,
)
rows = [
json.loads(line)
for line in (tmp_path / "out" / "results.jsonl").read_text().splitlines()
]
assert len(rows) == 1, "the cell that completed before cancellation must survive"
assert rows[0]["review_weighted_f1"] == 0.42, "its measurement must survive intact"
assert (tmp_path / "out" / "report.md").is_file()
def test_an_interrupted_sweep_emits_nothing_that_authorizes_promotion(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""The semantic condition, not the absence of a file.
promotion.json is still written for an aborted run - it is the record of why
nothing was promoted. What must hold is that nothing in it authorizes a
promotion from partial evidence.
"""
cancel_event = threading.Event()
with pytest.raises(SystemExit):
_sweep(
tmp_path, monkeypatch, lambda _run: scored_review_row(),
runs=3, cancel_event=cancel_event,
# The candidate arm has to RUN, not merely appear in promotion
# metadata: _run_sweep builds cells only from args.arms, so naming it
# in candidate_arms alone left the candidate with no results at all -
# and then "insufficient_evidence" would hold because nothing ran,
# not because partial evidence is barred from promoting.
arms=["review", "candidate_review"],
candidate_arms=["candidate_review"],
after_cell=(
lambda run_idx, arm: cancel_event.set()
if run_idx == 0 and arm == "candidate_review"
else None
),
)
rows = [
json.loads(line)
for line in (tmp_path / "out" / "results.jsonl").read_text().splitlines()
]
assert any(r["arm"] == "candidate_review" for r in rows), (
"the candidate must have produced evidence, or insufficient_evidence "
"would hold merely because nothing ran"
)
promotion = json.loads((tmp_path / "out" / "promotion.json").read_text())
assert promotion["run_status"] == "aborted"
assert promotion["decisions"], "an aborted run still has to say what it decided"
for decision in promotion["decisions"]:
assert decision["decision"] == "insufficient_evidence"
assert any("partial evidence" in reason for reason in decision["reasons"])

View file

@ -5,22 +5,35 @@ import os
import re
import shlex
import subprocess
import threading
from pathlib import Path
import pytest
import yaml
from typing import Any
from workflow_bench import runner
from workflow_bench.evolution import CANDIDATE_ARMS
from workflow_bench.process_control import _CANCELLATION, cancellation_scope
from workflow_bench.runner import (
aggregate,
GraphBuildEnv,
arm_health,
broken_incumbent_arms,
unhealthy_arms,
unmeasured_arms,
build_parser,
infra_error_record,
next_graph_prefetch_target,
normalized_model_identifier,
parse_shortstat,
prefetch_next_graph,
render_report,
savings,
select_tasks,
systemic_outage_streak,
task_has_planned_paid_cells,
)
@ -63,6 +76,15 @@ def test_aggregate_takes_medians_and_counts_resolved():
"diff_deletions": 5,
"class": "demo",
"resolved": 2,
# None of these are reused, so every resolution was measured this sweep.
"resolved_fresh": 2,
# Health is counted separately from resolution: all three executed and
# produced usable evidence, including the one that resolved nothing.
"fresh_attempts": 3,
"admissible": 3,
"execution_failures": 0,
"evidence_failures": 0,
"health_reasons": [],
"runs": 3,
"valid_runs": 3,
"excluded_runs": 0,
@ -245,6 +267,9 @@ def test_shipped_scenarios_opt_out_the_cross_module_cell_and_rebuild_graph_asset
assert skipped == ["cross-module-parse-retry"]
assert all(not task.get("sandbox_copy") for task in tasks)
assert all(task["sandbox_dependencies"] for task in tasks)
assert all(
any(dep.get("source") == "gitnexus-shared/dist" for dep in task["sandbox_dependencies"]) for task in tasks
)
assert all(task["oracle"]["command"] and task["oracle"]["files"] for task in tasks)
assert all("./node_modules/.bin/vitest run" in task["oracle"]["command"] for task in tasks)
assert all("npx vitest" not in task["oracle"]["command"] for task in tasks)
@ -512,3 +537,518 @@ def test_run_evolution_script_is_the_shared_ci_and_local_entrypoint():
assert "--include-expensive" in argv
assert "claude-sonnet-5" not in argv
assert printed.stderr # rewrite notice goes to stderr
def test_planned_paid_cells_treat_missing_reuse_as_paid():
task = {"id": "review-pr-2718-defect"}
assert task_has_planned_paid_cells(
task,
arms=["ce_review", "review", "candidate_review"],
runs=3,
reusable_rows={},
reuse_source=None,
)
reuse_source = Path("/tmp/seed")
rows = {
(task["id"], arm, run_idx): {}
for run_idx in range(3)
for arm in ("ce_review", "review", "candidate_review")
}
assert not task_has_planned_paid_cells(
task,
arms=["ce_review", "review", "candidate_review"],
runs=3,
reusable_rows=rows,
reuse_source=reuse_source,
)
del rows[(task["id"], "candidate_review", 0)]
assert task_has_planned_paid_cells(
task,
arms=["ce_review", "review", "candidate_review"],
runs=3,
reusable_rows=rows,
reuse_source=reuse_source,
)
def test_next_graph_prefetch_skips_ready_shas_and_fully_reused_tasks(tmp_path: Path):
first = {"id": "review-a"}
second = {"id": "review-b"}
third = {"id": "review-c"}
reuse_source = tmp_path / "seed"
reused_second = {
(second["id"], arm, 0): {} for arm in ("ce_review", "review", "candidate_review")
}
target = next_graph_prefetch_target(
[
(first, {"repo_identity": "/repo", "resolved_sha": "aaa"}),
(second, {"repo_identity": "/repo", "resolved_sha": "bbb"}),
(third, {"repo_identity": "/repo", "resolved_sha": "ccc"}),
],
arms=["ce_review", "review", "candidate_review"],
runs=1,
reusable_rows=reused_second,
reuse_source=reuse_source,
ready_keys={("/repo", "aaa")},
)
assert target is not None
task, binding, key = target
assert task["id"] == "review-c"
assert key == ("/repo", "ccc")
assert binding["resolved_sha"] == "ccc"
def test_prefetch_next_graph_runs_ensure_on_a_background_thread(monkeypatch):
started = threading.Event()
seen: list[tuple[str, str]] = []
def fake_ensure(**kwargs):
seen.append(kwargs["graph_key"])
started.set()
monkeypatch.setattr("workflow_bench.runner.ensure_task_graph", fake_ensure)
cancel = threading.Event()
job = prefetch_next_graph(
task={"id": "review-b"},
binding={"repo_identity": "/repo", "resolved_sha": "bbb"},
graph_key=("/repo", "bbb"),
env=GraphBuildEnv(
trees=Path("/tmp"),
task_asset_cache=None,
claude_bin="claude",
bwrap_bin="bwrap",
sandbox_backend="bwrap",
runtime_mounts=(),
clone_templates={},
clone_template_errors={},
graph_snapshots={},
graph_snapshot_errors={},
),
cancel_event=cancel,
)
assert job.key == ("/repo", "bbb")
assert started.wait(timeout=2)
job.join()
assert seen == [("/repo", "bbb")]
def test_a_reused_resolution_does_not_count_as_this_sweeps_health():
"""resolved counts evidence; resolved_fresh counts evidence measured today.
broken_incumbent_arms reads resolved_fresh because a reused row proves last
generation's environment worked. Counting it would make an arm whose cells
were all reused look healthy in exactly the run where a broken environment
should have been caught.
"""
reused = [record(resolved=True, reused=True), record(resolved=True, reused=True)]
agg = aggregate(reused)
assert agg["resolved"] == 2
assert agg["resolved_fresh"] == 0
assert broken_incumbent_arms({"t": {"review": agg}}, {"review"}) == ["review"]
mixed = aggregate([record(resolved=True, reused=True), record(resolved=True)])
assert mixed["resolved_fresh"] == 1
assert broken_incumbent_arms({"t": {"review": mixed}}, {"review"}) == []
def test_graph_build_env_ready_keys_covers_successes_and_failures():
"""A key that failed is attempted, not pending.
next_graph_prefetch_target skips keys already in ready_keys. If a failed
build were omitted, the sweep would prefetch it again every iteration and
pay a full clone and offline index each time for a build that cannot
succeed.
"""
env = GraphBuildEnv(
trees=Path("/tmp"),
task_asset_cache=None,
claude_bin="claude",
bwrap_bin="bwrap",
sandbox_backend="bwrap",
runtime_mounts=(),
clone_templates={("/repo", "aaa"): (Path("/tmp/a"), "aaa")},
clone_template_errors={("/repo", "bbb"): OSError("clone failed")},
graph_snapshots={("/repo", "ccc"): object()},
graph_snapshot_errors={("/repo", "ddd"): OSError("index failed")},
)
assert env.ready_keys() == {
("/repo", "aaa"),
("/repo", "bbb"),
("/repo", "ccc"),
("/repo", "ddd"),
}
def _cell(**overrides) -> dict[str, Any]:
"""One results.jsonl row, healthy unless told otherwise."""
base = record(resolved=True)
base.update({"error_kind": None, "review_evidence_valid": True, "transcript_missing": False})
base.update(overrides)
return base
def _arms(**by_arm) -> dict[str, dict[str, dict[str, Any]]]:
return {"task0": {arm: aggregate(rows) for arm, rows in by_arm.items()}}
def test_a_reviewer_that_scores_badly_is_not_an_unhealthy_harness():
"""Reconstructed from Actions run 33962002890's logged observations.
Every completed cell was resolved=False with error_kind=oracle-failed, at a
median score of 0.212 — the reviews ran, wrote artifacts and were scored.
That is a valid negative for the quality gate to judge. Diagnosing it as a
broken environment is the confusion this classification exists to end.
"""
scored_but_wrong = [_cell(resolved=False, error_kind="oracle-failed") for _ in range(3)]
results = _arms(review=scored_but_wrong, ce_review=list(scored_but_wrong))
assert unhealthy_arms(results, {"review", "ce_review"}) == []
health = arm_health(results, {"review"})["review"]
assert health.admissible == 3 and health.fresh_attempts == 3
assert (health.execution_failures, health.evidence_failures) == (0, 0)
def test_an_all_zero_score_is_still_a_valid_negative():
zeroed = [_cell(resolved=False, error_kind="oracle-failed", review_weighted_f1=0.0) for _ in range(3)]
assert unhealthy_arms(_arms(review=zeroed), {"review"}) == []
def test_artifacts_that_were_never_written_are_an_unhealthy_harness():
"""Reconstructed from Actions run 33912693948.
All 41 artifacts came back 0 bytes because the mount made an atomic write
impossible. The reviews could not produce evidence at all — the opposite of
the case above, and the one a health check must catch. The old caller
excluded review arms entirely, so it could not have.
"""
unwritable = [_cell(resolved=False, ok=False, error_kind="review-evidence-invalid") for _ in range(3)]
flagged = unhealthy_arms(_arms(review=unwritable), {"review"})
assert [h.arm for h in flagged] == ["review"]
assert flagged[0].evidence_failures == 3
assert "review-evidence-invalid" in flagged[0].reasons
def test_one_admissible_cell_leaves_an_arm_degraded_not_healthy():
"""Mixed outcomes are DEGRADED. One usable measurement does not erase two failures.
Not fatal - the sweep still produced evidence - but calling it healthy is
how a partly-broken environment passes review.
"""
mixed = [
_cell(resolved=False, error_kind="oracle-failed"),
_cell(resolved=False, ok=False, error_kind="session-error"),
_cell(resolved=False, ok=False, error_kind="infra-error"),
]
results = _arms(review=mixed)
health = arm_health(results, {"review"})["review"]
assert health.status == "DEGRADED"
assert unhealthy_arms(results, {"review"}) == [], "degraded is diagnostic, not fatal"
assert health.execution_failures == 2, "failures must stay visible, not be erased"
assert health.admissible == 1
def test_a_row_that_fails_both_ways_is_only_subtracted_once():
"""run_arm can produce a row that is an execution AND an evidence failure.
It keeps the first error_kind — a session-error survives — and still sets
review_evidence_valid=False when the artifact will not parse. Counting that
row against admissible twice zeroed an arm that held a real measurement,
which arm_health reports as UNUSABLE and the measurement gate then fails on.
"""
both = _cell(resolved=False, ok=False, error_kind="session-error", review_evidence_valid=False)
results = _arms(review=[both, _cell(resolved=True, error_kind="oracle-failed")])
health = arm_health(results, {"review"})["review"]
assert (health.execution_failures, health.evidence_failures) == (1, 1)
assert health.fresh_attempts == 2
assert health.admissible == 1
assert health.status == "DEGRADED"
assert unhealthy_arms(results, {"review"}) == []
def test_reused_rows_alone_leave_current_health_unknown():
"""Historical success cannot certify this sweep's environment."""
reused = [_cell(reused=True) for _ in range(3)]
results = _arms(review=reused)
assert unmeasured_arms(results, {"review"}) == ["review"]
assert unhealthy_arms(results, {"review"}) == []
assert arm_health(results, {"review"})["review"].measured is False
def test_the_paid_canary_survives_a_prior_run_with_more_run_indices():
"""The canary counts planned cells, not every key reuse selection returned.
Reuse selection accepts any non-negative prior `run`, so a results directory
produced with --runs 5 leaves keys this sweep never plans. Comparing against
those made the "arm is fully reused" test false exactly when it was true,
and the incumbent went a whole sweep without one measured cell.
"""
tasks = [{"id": "task0"}, {"id": "task1"}]
reusable = {(task["id"], "review", run): {} for task in tasks for run in range(5)}
dropped = runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=3)
assert dropped == ("task0", "review", 0)
assert dropped not in reusable
# A second call is a no-op: the arm now has its paid cell.
assert runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=3) is None
def test_an_arm_with_a_planned_paid_cell_keeps_every_reusable_row():
tasks = [{"id": "task0"}]
reusable = {("task0", "review", 0): {}}
assert runner.drop_canary_reuse_key(reusable, arm="review", tasks=tasks, runs=2) is None
assert len(reusable) == 1
def test_reused_successes_do_not_mask_fresh_execution_failures():
rows = [_cell(reused=True), _cell(reused=True), _cell(ok=False, error_kind="session-error")]
flagged = unhealthy_arms(_arms(review=rows), {"review"})
assert [h.arm for h in flagged] == ["review"]
assert flagged[0].fresh_attempts == 1 and flagged[0].execution_failures == 1
def test_a_parseable_artifact_does_not_excuse_a_failed_session():
"""Artifact parseability must not override an execution failure."""
rows = [_cell(ok=False, error_kind="session-error", review_evidence_valid=True) for _ in range(2)]
flagged = unhealthy_arms(_arms(review=rows), {"review"})
assert [h.arm for h in flagged] == ["review"]
assert flagged[0].execution_failures == 2
def test_a_single_unusable_review_is_caught_below_the_breaker_threshold():
"""The decisive regression for the finalization guard.
A fixture of 41 empty artifacts would abort through the outage breaker -
review-evidence-invalid is systemic and the limit is 5 - so it proves
nothing about this path. One fresh unusable cell is under that threshold,
which leaves the finalization check as the only thing that can catch it.
"""
streak = 0
for _ in range(1):
streak = runner.systemic_outage_streak("review-evidence-invalid", streak)
assert streak < runner.DEFAULT_OUTAGE_STREAK, "fixture must not reach the breaker"
results = _arms(review=[_cell(resolved=False, ok=False, error_kind="review-evidence-invalid")])
with pytest.raises(SystemExit) as exc:
runner.enforce_measurement_health(results, {"review"})
assert exc.value.code == 1
def test_finalization_reports_every_arm_and_names_no_cause(capsys):
"""Status for each arm; an empty artifact does not become an EROFS diagnosis."""
results = _arms(
review=[_cell(resolved=False, ok=False, error_kind="review-evidence-invalid")],
ce_review=[_cell(resolved=False, error_kind="oracle-failed")],
)
with pytest.raises(SystemExit):
runner.enforce_measurement_health(results, {"review", "ce_review"})
out = capsys.readouterr().out
assert "review: UNUSABLE" in out
assert "ce_review: OBSERVED_OK" in out
assert "cause=undetermined" in out
assert "EROFS" not in out and "mount" not in out
def test_valid_negatives_do_not_abort_finalization(capsys):
"""The 16h run's shape must survive the real guard, not just the classifier."""
scored_but_wrong = [_cell(resolved=False, error_kind="oracle-failed") for _ in range(3)]
health = runner.enforce_measurement_health(
_arms(review=scored_but_wrong, ce_review=list(scored_but_wrong)), {"review", "ce_review"}
)
assert {h.status for h in health.values()} == {"OBSERVED_OK"}
assert "UNUSABLE" not in capsys.readouterr().out
def test_reused_only_arm_is_reported_unknown_by_finalization(capsys):
runner.enforce_measurement_health(_arms(review=[_cell(reused=True)]), {"review"})
assert "review: UNKNOWN" in capsys.readouterr().out
def test_run_sweep_calls_the_health_guard_and_not_the_legacy_helper():
"""Pins the wiring the caller correction exposed.
Reads the compiled code object's global references rather than the source
text: deleting the call removes the name and fails this test, which is the
mutation check. It does NOT prove the guard runs end to end - _run_sweep
needs bwrap and a sandbox, so no test here drives it.
"""
referenced = runner._run_sweep.__code__.co_names
assert "enforce_measurement_health" in referenced
assert "broken_incumbent_arms" not in referenced
def test_ce_review_is_classified_even_though_it_is_not_a_candidate_arm():
"""ce_review is a comparator, absent from CANDIDATE_ARMS.
Dropping the `- {"review"}` exclusion alone would have left it unchecked.
"""
assert "ce_review" not in set(CANDIDATE_ARMS.values())
health = arm_health(_arms(ce_review=[_cell()]), {"review", "ce_review"})
assert "ce_review" in health
def _packed_cells(tasks: int, runs: int, arms: tuple[str, ...]) -> list[tuple[str, int, str]]:
return [(f"t{t}", r, a) for t in range(tasks) for r in range(runs) for a in arms]
def test_packed_sweep_runs_every_cell_and_folds_in_submission_order():
"""Fold order is the contract the breaker rests on.
Cells finish in whatever order the pool returns them, but the breaker counts
CONSECUTIVE systemic failures, which only means something in a fixed order.
"""
cells = _packed_cells(3, 2, ("review", "candidate_review"))
folded: list[tuple[str, int, str]] = []
streak, tripped = runner.sweep_packed_cells(
cells,
workers=4,
run=lambda task, run_idx, arm: {"error_kind": None, "review_evidence_valid": True},
on_start=lambda *_: None,
on_record=lambda task, run_idx, arm, _rec: folded.append((task, run_idx, arm)),
outage_streak=0,
outage_limit=0,
)
assert folded == cells
assert (streak, tripped) == (0, False)
def test_packed_sweep_trips_the_breaker_on_the_same_cell_waves_would():
"""Packing must not change WHEN a doomed run aborts, only how it is fed."""
cells = _packed_cells(3, 3, ("review",))
fail_from = 2
folded: list[int] = []
def run(task: str, run_idx: int, arm: str) -> dict[str, Any]:
index = cells.index((task, run_idx, arm))
systemic = index >= fail_from
return {
"error_kind": "session-error" if systemic else None,
"review_evidence_valid": not systemic,
}
streak, tripped = runner.sweep_packed_cells(
cells,
workers=2,
run=run,
on_start=lambda *_: None,
on_record=lambda t, r, a, _rec: folded.append(cells.index((t, r, a))),
outage_streak=0,
outage_limit=runner.DEFAULT_OUTAGE_STREAK,
)
assert tripped is True
assert streak == runner.DEFAULT_OUTAGE_STREAK
# Five consecutive systemic failures starting at index 2 -> trips on index 6.
assert folded[-1] == fail_from + runner.DEFAULT_OUTAGE_STREAK - 1
assert folded == sorted(folded), "records must fold in submission order"
def test_packed_sweep_skips_a_task_whose_assets_never_arrive():
"""A task that cannot be prepared is skipped, not run against nothing."""
cells = _packed_cells(3, 2, ("review",))
ran: list[str] = []
runner.sweep_packed_cells(
cells,
workers=3,
run=lambda task, run_idx, arm: ran.append(task)
or {"error_kind": None, "review_evidence_valid": True},
on_start=lambda *_: None,
on_record=lambda *_: None,
outage_streak=0,
outage_limit=0,
await_ready=lambda task: task != "t1",
)
assert set(ran) == {"t0", "t2"}
assert "t1" not in ran
def test_packed_sweep_workers_inherit_the_runs_cancellation_event():
"""A worker that cannot see the event runs on after the sweep is cancelled.
The cells are submitted from a producer THREAD, and a new thread starts with
an empty context - so copying the context at submission copies the wrong one
unless the caller's is captured first. run_managed falls back to
_CANCELLATION when no event is passed, which is how a cell's subprocesses
learn the run was cancelled at all.
"""
seen: list[threading.Event | None] = []
event = threading.Event()
with cancellation_scope(event):
runner.sweep_packed_cells(
_packed_cells(2, 1, ("review",)),
workers=2,
run=lambda *_: seen.append(_CANCELLATION.get()) or {"error_kind": None},
on_start=lambda *_: None,
on_record=lambda *_: None,
outage_streak=0,
outage_limit=0,
)
assert seen and all(observed is event for observed in seen)
def test_packed_sweep_window_must_keep_the_pool_fed():
with pytest.raises(ValueError, match="window must be at least workers"):
runner.sweep_packed_cells(
_packed_cells(1, 1, ("review",)),
workers=4,
run=lambda *_: {"error_kind": None},
on_start=lambda *_: None,
on_record=lambda *_: None,
outage_streak=0,
outage_limit=0,
window=2,
)
def test_a_raising_packed_cell_still_persists_its_settled_siblings():
"""A crash in one cell must not erase the evidence of cells that finished.
run_cell deliberately lets unexpected harness exceptions propagate, and the
wave scheduler answers that by folding every non-failing sibling before it
re-raises. The packed scheduler has to hold the same contract: the later
cells already ran and already cost money, so losing their rows would mean
paying for evidence the sweep then throws away.
"""
folded: list[tuple[int, str]] = []
started = threading.Event()
def run(task_id: str, run_idx: int, arm: str) -> dict[str, Any]:
if run_idx == 0:
# Let the later cell finish first, so there is settled evidence to
# lose at the moment this one raises.
started.wait(timeout=5)
raise RuntimeError("harness bug in cell 0")
started.set()
return {"error_kind": None}
with pytest.raises(RuntimeError, match="harness bug in cell 0"):
runner.sweep_packed_cells(
_packed_cells(1, 2, ("review",)),
workers=2,
run=run,
on_start=lambda *_: None,
on_record=lambda task_id, run_idx, arm, _rec: folded.append((run_idx, arm)),
outage_streak=0,
outage_limit=0,
)
assert (1, "review") in folded, "the sibling that completed was never recorded"

View file

@ -12,9 +12,9 @@ from types import SimpleNamespace
import pytest
from workflow_bench import evolve, runner, runner_sessions, runtime_mounts
from workflow_bench import evolve, runner, runner_artifacts, runner_sessions, runtime_mounts
from workflow_bench.evolution import skill_fingerprint
from workflow_bench.process_control import ManagedProcessResult
from workflow_bench.process_control import ManagedProcessError, ManagedProcessResult
from workflow_bench.proposer_sandbox import SandboxError
from workflow_bench.runner import snapshot_plan_docs
@ -120,11 +120,16 @@ def skill_events(skill_input: dict, *, tool_id: str = "skill-1", is_error: bool
def fake_sandbox(root: Path) -> SimpleNamespace:
# private_root is NOT the clone. Conflating them puts the review artifact
# directory inside the workspace, which the real sandbox never does and
# which hides whether the workspace was left untouched.
private_root = root.parent / f"{root.name}-sandbox-private"
private_root.mkdir(exist_ok=True)
return SimpleNamespace(
backend="test-double",
claude_bin="claude",
clone=root,
private_root=root,
private_root=private_root,
command_prefix=[],
command_prefix_for=lambda **_kwargs: [],
settings_json="{}",
@ -1249,7 +1254,7 @@ def test_planning_cannot_change_source_tests_or_downstream_skill(monkeypatch, tm
@pytest.mark.parametrize(
("attack", "expected_detail"),
[
("workspace", "unauthorized workspace path"),
("workspace", "changed the read-only workspace"),
("skill", "changed the evaluated skill fingerprint"),
],
)
@ -1265,9 +1270,11 @@ def test_review_phase_rejects_workspace_or_skill_mutation(
expected_skill_digest = "expected-skill-fingerprint"
def adversarial_review(prompt, *args, **kwargs):
(tmp_path / "review-output.json").write_text(
'{"schema_version":1,"verdict":"approve","findings":[]}'
)
# Write where the contract now says: the artifact directory outside the
# workspace, which is the only place the agent can write atomically.
artifact = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT)
artifact.parent.mkdir(parents=True, exist_ok=True)
artifact.write_text('{"schema_version":1,"verdict":"approve","findings":[]}')
if attack == "workspace":
source.write_text("review silently changed source")
return session_record()
@ -1298,6 +1305,97 @@ def test_review_phase_rejects_workspace_or_skill_mutation(
assert expected_detail in rec["error_detail"]
def test_a_cancelled_clone_copy_does_not_fall_back_to_an_uncancellable_copytree(monkeypatch, tmp_path):
"""The reflink fallback is for a filesystem, not for a teardown.
run_managed reports cancellation as a non-OK result rather than raising, so
the fallback treated it like an unsupported reflink and started a copytree
that cannot be cancelled — waiting out exactly the full copy the outage
breaker set the cancellation event to avoid.
"""
source = tmp_path / "template"
(source / ".git").mkdir(parents=True)
parent = tmp_path / "clones"
parent.mkdir()
copied: list[object] = []
monkeypatch.setattr(
runner_artifacts,
"run_managed",
lambda *_a, **_k: ManagedProcessResult(
state="cancelled",
returncode=None,
stdout_tail="",
stderr_tail="",
duration_s=0.1,
),
)
monkeypatch.setattr(runner_artifacts.shutil, "copytree", lambda *a, **k: copied.append(a))
with pytest.raises(ManagedProcessError):
runner.copy_isolated_tree(source, parent)
assert copied == []
assert list(parent.iterdir()) == [], "the partial target must be cleaned up"
@pytest.mark.parametrize("arm", ["review", "ce_review"])
def test_run_arm_mounts_the_review_artifact_directory_outside_the_workspace(monkeypatch, tmp_path, arm):
"""A writable FILE inside a read-only directory is not a writable path.
The Write tool creates `<target>.tmp.<n>.<hex>` beside the target and
renames it, so a read-only parent fails the temp create with EROFS and the
artifact stays 0 bytes. The mount target must be the directory, and it must
sit outside the read-only workspace.
Driven through run_arm rather than rebuilt here: an expected tuple assembled
in the test passes whatever run_arm actually mounts, which is the one thing
this needs to prove.
"""
assert not runner.SANDBOX_REVIEW_OUTPUT.startswith(runner.SANDBOX_WORKSPACE + "/")
assert runner.SANDBOX_REVIEW_OUTPUT != runner.SANDBOX_WORKSPACE
verify_calls: list[dict] = []
sandbox = fake_sandbox(tmp_path)
sandbox.command_prefix_for = lambda **kwargs: verify_calls.append(kwargs) or []
def review_session(prompt, *args, **kwargs):
artifact = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT)
artifact.parent.mkdir(parents=True, exist_ok=True)
artifact.write_text('{"schema_version":1,"verdict":"approve","findings":[]}')
return session_record()
monkeypatch.setattr(runner, "run_claude", review_session)
monkeypatch.setattr(runner, "skill_fingerprint", lambda *_a, **_k: "skill-digest")
monkeypatch.setattr(runner, "run_verify", lambda *a, **k: (True, "ok"))
runner.run_arm(
arm,
{"prompt": "p", "verify": "true"},
tmp_path,
bench_args(),
sandbox=sandbox,
expected_skill_digest="skill-digest",
)
review_output = runner.review_output_path(sandbox, runner.REVIEW_OUTPUT)
expected = (runner.ReadOnlyMount(source=review_output.parent, target=runner.SANDBOX_REVIEW_OUTPUT),)
# The EROFS bug is about the AGENT's write, so the mount that has to be the
# directory is the writable one on the review session — not the read-only
# exposure the verify command gets afterwards. Assert both: they are
# separate arguments to separate command prefixes.
writable = [call["extra_writable_mounts"] for call in verify_calls if "extra_writable_mounts" in call]
assert writable, "the review session must be given a writable artifact mount"
assert writable[-1] == expected, "mount the directory, not the file"
read_only = [call["extra_read_only_mounts"] for call in verify_calls if "extra_read_only_mounts" in call]
assert read_only, "the verify invocation must be given the artifact mount"
assert read_only[-1] == expected, "mount the directory, not the file"
assert not expected[0].target.startswith(f"{runner.SANDBOX_WORKSPACE}/")
# The artifact the harness later reads is the one inside that mount.
assert review_output.parent in review_output.parents
def _git(repo, *args, check=True):
return subprocess.run(["git", "-C", str(repo), *args], check=check, capture_output=True, text=True)
@ -1348,3 +1446,34 @@ def test_make_worktree_clone_has_no_tags_but_keeps_all_branches(tmp_path):
current = _git(target, "rev-parse", "HEAD").stdout.strip()
assert current == other_sha
def test_copy_isolated_tree_does_not_share_git_objects_or_refs(tmp_path):
repo = tmp_path / "repo"
repo.mkdir()
_git(repo, "init", "--quiet")
_git(repo, "checkout", "--quiet", "-b", "main")
sha = _git_commit(repo, "base")
clones = tmp_path / "clones"
clones.mkdir()
template = runner.make_worktree(repo, sha, clones)
(template / "marker.txt").write_text("template\n")
copy = runner.copy_isolated_tree(template, clones)
assert copy != template
assert (copy / "marker.txt").read_text() == "template\n"
(copy / "marker.txt").write_text("copy\n")
assert (template / "marker.txt").read_text() == "template\n"
copy_head = _git(copy, "rev-parse", "HEAD").stdout.strip()
template_head = _git(template, "rev-parse", "HEAD").stdout.strip()
assert copy_head == template_head == sha
# An equal initial HEAD is also what a shared ref namespace looks like, so
# write a ref and prove the template cannot see it. A linked worktree would
# pass every assertion above, including the alternates check — its `.git` is
# a file, so the directory inspected below simply does not exist.
_git(copy, "branch", "copy-only")
assert _git(copy, "show-ref", "--verify", "refs/heads/copy-only").returncode == 0
assert _git(template, "show-ref", "--verify", "refs/heads/copy-only", check=False).returncode != 0
assert (copy / ".git").is_dir()
alternates = copy / ".git" / "objects" / "info" / "alternates"
assert not alternates.exists()

View file

@ -149,8 +149,10 @@ UNSAFE_NO_BWRAP=1 RUNS=1 ./workflow_bench/run-evolution.sh
This mode runs review sessions directly in disposable host worktrees and is
**not** a security boundary: it does not isolate the network or create a PID
namespace, and a session that can `chmod` can undo the workspace lock. The
harness still drops write bits on the clone except `review-output.json` so
accidental `npm install` / analyze writes cannot invalidate review evidence.
harness drops write bits on the whole clone, with no carve-out, so accidental
`npm install` / analyze writes cannot invalidate review evidence. The review
artifact is not in the clone at all: it lives in a writable directory bound at
`/review-output`, outside the workspace.
Sandbox cleanup restores owner write bits before deleting the private TMPDIR,
because a session that `copytree`s the locked clone would otherwise leave
non-empty 0555 directories that `rmtree` cannot remove. Historical review
@ -168,6 +170,36 @@ router thresholds as an incumbent policy, not permanent truth. Candidate
changes run offline in the same throwaway clones as the incumbent; production
skills never rewrite themselves from a live task.
On the self-hosted evolution box, `run-evolution.sh` passes
`--max-runtime-from-instance-window` and the CLI derives its own cap from
`/proc/uptime` at startup (24h EventBridge window minus a 90-minute upload
reserve), in the same breath as it starts the clock that cap is measured
against — a budget computed anywhere earlier is spent by the seconds between. A `workflow_dispatch` that lands on an
already-running instance therefore exits in-process instead of vanishing when
the box stops — a cancelled GitHub job skips even `if: always()`, which is
how run 33962002890 lost 51 finished sessions. Local runs are uncapped.
A review generation is 6 tasks × 3 arms × 3 runs. Serial workers=1 at ~19
minutes per session is a 16-hour job (run 33962002890). Two harness changes
cut that without shrinking the gate:
- **Comparator reuse.** `evolve.py` forwards the seed / prior generation as
`--reuse-results`. Incumbent `review` and `ce_review` rows are copied into
the new `results.jsonl` when model, effort, task SHA, prompt digest, oracle
bytes, incumbent skill digest, CE plugin digest, and sandbox backend still
match. Candidate arms always run. A weekly generation with an unchanged
incumbent therefore pays 18 sessions, not 54. A promotion, model change,
task-corpus change, or harness `RUNTIME_DIGEST` change invalidates the
lock and re-runs the comparators.
- **Sanitized clone templates.** Each unique task SHA is cloned and
sanitized once. Cells copy that parentless snapshot (reflink when the
filesystem allows) instead of `git clone --no-local` plus repack/prune/fsck
54 times. Isolation is a private `.git`, not a second copy of full history.
Dispatch defaults to `--workers 3` so those 18 paid cells can overlap. Size
workers to the host: a cell that loses CPU and hits the session ceiling is
an excluded run the gate refuses.
Build an overlay that mirrors only the canonical repo-local skill paths:
```text
@ -276,9 +308,12 @@ without weakening today's deterministic promotion boundary.
The evolution workflow runs an offline containment preflight with the pinned
Claude Code 2.1.214 binary before starting a paid proposer or benchmark. The
review canary seals the workspace read-only and exposes only the pre-created
`review-output.json` as writable. Runtime mount placeholders are prepared in
the disposable clone before sealing it; existing config bytes are preserved.
review canary seals the workspace read-only and writes nothing into it: the
artifact directory is bound at `/review-output` outside the workspace, and the
file itself is deliberately absent until the session creates it, so its absence
distinguishes "never written" from "written badly". Runtime mount placeholders
are prepared in the disposable clone before sealing it; existing config bytes
are preserved.
Any pre-existing result entry, including a symlink, is rejected. Required
canaries fail when their runtime or Bubblewrap is unavailable.
@ -368,10 +403,10 @@ paired benchmark as any other candidate.
For ad-hoc use, run the driver on the existing re-evaluation triggers
(model/harness change or 90-day staleness). The repository workflow runs a
deliberate weekly drift check: scheduled concurrency stays serial unless
`GITNEXUS_EVOLUTION_WORKERS` is raised after a funded host-sized proof, and
`--workers` is bounded to 1–8 before paid work starts. `--generations` remains
the only loop bound.
deliberate weekly drift check: dispatch defaults to three concurrent cells
of one task; scheduled concurrency still requires
`GITNEXUS_EVOLUTION_WORKERS=3` after a clean proof. `--workers` is bounded
to 1–8 before paid work starts. `--generations` remains the only loop bound.
## Free-model setup (no paid tokens)

View file

@ -0,0 +1,604 @@
"""Reuse frozen comparator cells when the current sweep is still the same experiment.
Weekly skill evolution re-runs incumbent ``review`` / ``ce_review`` (and the
implementation incumbents) even when the model, effort, tasks, oracles,
incumbent skill bytes, and CE plugin have not changed. Those arms are the
baseline the gate compares a *new* candidate against — they are not the
thing being evolved. Replaying them burns two-thirds of a generation.
This module selects prior ``results.jsonl`` rows that are safe to carry
forward. Candidate arms are never reused. A mismatch on any bound field
falls through to a paid cell. Missing artifacts also fall through: a reused
row that the proposer cannot read is worse than spending the tokens again.
"""
from __future__ import annotations
import hashlib
import json
import os
import re
import stat
from collections.abc import Iterator, Mapping, Sequence
from contextlib import contextmanager
from dataclasses import dataclass
from datetime import UTC, datetime, timedelta
from pathlib import Path, PurePosixPath
from typing import Any
from .evolution import CANDIDATE_ARMS, EVIDENCE_MAX_AGE_DAYS
from .proposer_sandbox import SandboxError
from .runner_sessions import MAX_TRANSCRIPT_BYTES, PARENT_EVENT_STREAM_SOURCE
from .runtime_mounts import CE_ARMS
from .task_assets import COPY_CHUNK_BYTES, _write_all
REUSABLE_COMPARATOR_ARMS = frozenset(
{
"review",
"ce_review",
"workflow",
"workflow_direct",
"ce_workflow",
"ce_workflow_direct",
"baseline",
"baseline_nomcp",
}
)
# Must stay aligned with runner.EXCLUDED_ERROR_KINDS plus review-invalid.
# A reused row becomes promotion evidence; excluded kinds cannot enter that set.
REUSE_EXCLUDED_ERROR_KINDS = frozenset(
{
"session-error",
"infra-error",
"evidence-unverified",
"cleanup-failure",
"review-evidence-invalid",
"cancelled",
}
)
_TRANSCRIPT_NAME = re.compile(r"[A-Za-z0-9._-]{1,200}")
CellKey = tuple[str, str, int]
@dataclass(frozen=True)
class TaskReuseBinding:
"""Per-task identity the prior row must still match."""
task_base_sha: str
task_prompt_digest: str
oracle_digest: str
oracle_command_digest: str
oracle_manifest_digest: str
# The cell's environment is part of its identity: a comparator measured
# against different task assets or different sandbox dependencies is a
# measurement of a different machine, not a baseline for this sweep.
task_asset_manifest_digest: str | None = None
sandbox_dependency_manifest_digest: str | None = None
@dataclass(frozen=True)
class ComparatorReuseExpectation:
"""Sweep-wide lock for comparator reuse. Any drift pays for a fresh cell."""
model: str
effort: str
sandbox_backend: str
runtime_digest: str | None
now: datetime
max_age: timedelta
tasks: Mapping[str, TaskReuseBinding]
skill_digests: Mapping[str, str | None]
ce_plugin_version: str | None
ce_plugin_manifest_digest: str | None
def load_result_rows(path: Path) -> list[dict[str, Any]]:
"""Load ``results.jsonl``; skip malformed lines the same way evolve does."""
rows: list[dict[str, Any]] = []
for line in path.read_text().splitlines():
if not line.strip():
continue
try:
row = json.loads(line)
except json.JSONDecodeError:
continue
if isinstance(row, dict):
rows.append(row)
return rows
def current_runtime_digest() -> str | None:
"""Harness lockfile digest exported by ``run-evolution.sh``, if present."""
value = os.environ.get("RUNTIME_DIGEST", "").strip()
return value or None
def row_is_reusable_comparator(row: Mapping[str, Any], expected: ComparatorReuseExpectation) -> bool:
"""True when ``row`` is a complete, still-valid comparator measurement."""
arm = row.get("arm")
if not isinstance(arm, str) or arm in CANDIDATE_ARMS or arm not in REUSABLE_COMPARATOR_ARMS:
return False
if row.get("error_kind") in REUSE_EXCLUDED_ERROR_KINDS:
return False
if row.get("error_kind") not in (None, ""):
return False
if row.get("ok") is not True:
return False
if row.get("transcript_missing") is True:
return False
if row.get("candidate_overlay_digest") not in (None, ""):
return False
# Age against the ORIGINAL measurement, not the copy time: materialize_reused_row
# restamps recorded_at, so a chained row would otherwise refresh its own clock
# and never expire. Bound both directions - a future stamp is corrupt, not fresh.
recorded = _parse_recorded_at(row.get("reused_from_recorded_at") or row.get("recorded_at"))
if recorded is None:
return False
age = expected.now - recorded
if age > expected.max_age or age < timedelta(0):
return False
if row.get("model") != expected.model and row.get("benchmark_model") != expected.model:
return False
if row.get("effort") != expected.effort:
return False
if row.get("sandbox_backend") != expected.sandbox_backend:
return False
# Fail closed. A row with no runtime_digest was measured by a harness that
# did not record one, which is exactly the drift this lock exists to catch;
# treating the absence as agreement made every legacy row reusable forever.
prior_runtime = row.get("runtime_digest")
if not isinstance(prior_runtime, str) or not prior_runtime:
return False
if not expected.runtime_digest or prior_runtime != expected.runtime_digest:
return False
task_id = row.get("task")
binding = expected.tasks.get(task_id) if isinstance(task_id, str) else None
if binding is None:
return False
if row.get("task_base_sha") != binding.task_base_sha:
return False
if row.get("task_prompt_digest") != binding.task_prompt_digest:
return False
if row.get("oracle_digest") != binding.oracle_digest:
return False
if row.get("oracle_command_digest") != binding.oracle_command_digest:
return False
if row.get("oracle_manifest_digest") != binding.oracle_manifest_digest:
return False
# Fail closed on both sides, as the runtime digest does: an unbound
# expectation means this sweep could not determine its own environment, and
# a row without the field was measured before it was recorded.
for field, bound in (
("task_asset_manifest_digest", binding.task_asset_manifest_digest),
("sandbox_dependency_manifest_digest", binding.sandbox_dependency_manifest_digest),
):
prior = row.get(field)
if not isinstance(prior, str) or not prior or not bound or prior != bound:
return False
if arm in CE_ARMS:
if row.get("ce_plugin_version") != expected.ce_plugin_version:
return False
if row.get("ce_plugin_manifest_digest") != expected.ce_plugin_manifest_digest:
return False
else:
expected_skill = expected.skill_digests.get(arm)
if not expected_skill or row.get("skill_digest") != expected_skill:
return False
if arm in {"review", "ce_review"}:
if row.get("review_evidence_valid") is not True:
return False
# The artifact, not just the score derived from it. materialize_reused_row
# copies it only when the name is present, so without this a row whose
# artifact copy never happened could be carried forward as a scored
# review that a proposer then cannot read - evidence by assertion.
review_artifact = row.get("review_artifact")
if not isinstance(review_artifact, str) or not review_artifact:
return False
if not isinstance(row.get("review_score"), dict):
return False
if row.get("review_weighted_f1") is None:
return False
artifacts = row.get("transcript_artifacts")
if not isinstance(artifacts, list) or not artifacts:
return False
try:
for artifact in artifacts:
_transcript_metadata(artifact)
except SandboxError:
return False
return True
def select_reusable_comparator_rows(
rows: Sequence[Mapping[str, Any]],
*,
expected: ComparatorReuseExpectation,
) -> dict[CellKey, dict[str, Any]]:
"""Index reusable rows by ``(task, arm, run)``. Conflicting duplicates drop the key."""
chosen: dict[CellKey, dict[str, Any]] = {}
blocked: set[CellKey] = set()
for row in rows:
if not row_is_reusable_comparator(row, expected):
continue
task_id = row["task"]
arm = row["arm"]
run = row.get("run")
if not isinstance(run, int) or isinstance(run, bool) or run < 0:
continue
key = (str(task_id), str(arm), run)
if key in blocked:
continue
previous = chosen.get(key)
if previous is None:
chosen[key] = dict(row)
continue
if _row_identity(previous) != _row_identity(row):
blocked.add(key)
chosen.pop(key, None)
return chosen
def materialize_reused_row(
row: Mapping[str, Any],
*,
source_dir: Path,
dest_dir: Path,
) -> dict[str, Any]:
"""Copy digest-bound artifacts into this sweep's evidence dir and stamp reuse."""
source, _ = _resolved_directory(source_dir, label="reuse source")
dest, _ = _resolved_directory(dest_dir, label="reuse destination")
if source == dest:
raise SandboxError("comparator reuse cannot read and write the same results directory")
materialized = dict(row)
materialized["reused"] = True
# Keep the FIRST measurement time across a chain. Overwriting it with the
# previous copy's stamp let a row refresh its own clock every generation and
# outlive the max_age bound entirely.
materialized["reused_from_recorded_at"] = row.get("reused_from_recorded_at") or row.get("recorded_at")
materialized["recorded_at"] = datetime.now(UTC).isoformat()
artifacts = row.get("transcript_artifacts")
if not isinstance(artifacts, list) or not artifacts:
raise SandboxError("reused row is missing transcript_artifacts")
# Every path below is resolved against a held descriptor, never re-walked
# from a name. Both roots are already symlink-free (_resolved_directory
# resolved them), and pinning them here means the components under them
# cannot be swapped out from under a check that already passed.
with (
_open_pinned_root(source_dir, label="reuse source") as source_fd,
_open_pinned_root(dest_dir, label="reuse destination") as dest_fd,
):
copied_artifacts: list[dict[str, Any]] = []
for artifact in artifacts:
copied_artifacts.append(_copy_transcript_artifact(source_fd, dest_fd, artifact))
materialized["transcript_artifacts"] = copied_artifacts
review_name = row.get("review_artifact")
if isinstance(review_name, str) and review_name:
_copy_named_artifact(source_fd, dest_fd, review_name, label="review artifact")
task = row.get("task")
arm = row.get("arm")
run = row.get("run")
if isinstance(task, str) and isinstance(arm, str) and isinstance(run, int) and not isinstance(run, bool):
patch_name = f"{task}-{arm}-run{run}.patch"
if _is_regular_at(patch_name, dir_fd=source_fd):
_copy_named_artifact(source_fd, dest_fd, patch_name, label="patch artifact")
return materialized
def default_reuse_max_age() -> timedelta:
return timedelta(days=EVIDENCE_MAX_AGE_DAYS)
def _row_identity(row: Mapping[str, Any]) -> tuple[Any, ...]:
return (
row.get("skill_digest"),
row.get("oracle_digest"),
row.get("review_weighted_f1"),
row.get("ce_plugin_manifest_digest"),
row.get("recorded_at"),
)
def _parse_recorded_at(value: Any) -> datetime | None:
if not isinstance(value, str) or not value:
return None
try:
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
except ValueError:
return None
if parsed.tzinfo is None:
parsed = parsed.replace(tzinfo=UTC)
return parsed.astimezone(UTC)
def _transcript_metadata(metadata: Any) -> tuple[str, str, int]:
if not isinstance(metadata, dict) or set(metadata) != {"path", "sha256", "bytes", "source"}:
raise SandboxError("transcript artifact metadata must contain only path, sha256, bytes, and source")
relative = metadata["path"]
digest = metadata["sha256"]
size = metadata["bytes"]
if metadata["source"] != PARENT_EVENT_STREAM_SOURCE:
raise SandboxError("transcript artifact source is not the parent event stream")
if not isinstance(relative, str) or not isinstance(digest, str) or not re.fullmatch(r"[0-9a-f]{64}", digest):
raise SandboxError("transcript artifact metadata is malformed")
if not isinstance(size, int) or isinstance(size, bool) or size < 0 or size > MAX_TRANSCRIPT_BYTES:
raise SandboxError("transcript artifact byte count is out of range")
relative_path = PurePosixPath(relative)
if (
relative_path.is_absolute()
or len(relative_path.parts) != 2
or relative_path.parts[0] != "transcripts"
or any(part in {"", ".", ".."} for part in relative_path.parts)
or _TRANSCRIPT_NAME.fullmatch(relative_path.parts[1]) is None
):
raise SandboxError(f"unsafe transcript artifact path: {relative!r}")
return relative, digest, size
def _resolved_directory(path: Path, *, label: str) -> tuple[Path, tuple[int, int]]:
"""An existing, non-symlink directory, resolved through its parents.
Deliberately weaker than proposer_sandbox's same-shaped helper, which
refuses every symlink hop in the path. That one guards a MOUNT ROOT, where
a hop changes what an untrusted session is handed. This one guards a DATA
directory whose contents are validated individually anyway - every file
read goes through ``_regular_file`` (lstat, symlinks rejected) and every
write through ``O_NOFOLLOW`` - so a symlinked parent grants nothing those
guards do not already cover, while refusing one would reject ordinary
setups such as a symlinked artifacts directory or macOS's /var.
Separately named because they make different promises. Do not merge them
without first deciding which promise the reuse path should make.
"""
resolved = path.expanduser()
try:
metadata = resolved.lstat()
except OSError as exc:
raise SandboxError(f"{label} is unavailable: {resolved}: {exc}") from exc
if stat.S_ISLNK(metadata.st_mode) or not stat.S_ISDIR(metadata.st_mode):
raise SandboxError(f"{label} must be a real directory: {resolved}")
return resolved.resolve(), (metadata.st_dev, metadata.st_ino)
@contextmanager
def _open_pinned_root(path: Path, *, label: str) -> Iterator[int]:
"""Open a checked root and prove it is still the directory that was checked.
The symlink POLICY above is deliberate and unchanged: parent hops stay
allowed, so a symlinked artifacts directory or macOS's /var still works.
What is closed here is separate from that policy - the gap between checking
a name and using it. lstat names one directory and resolve() re-walks the
same name afterwards, so a prior sweep that renames its results root and
drops a symlink in its place is resolved to somewhere else entirely, and
O_NOFOLLOW on the open cannot see a link that resolve() already followed.
Comparing the opened descriptor's identity to the checked one costs an
fstat and rejects nothing that holds still: a stable directory always
matches itself. It matters for reuse specifically because the failure is
silent - rows would be copied out of the wrong directory and folded into a
comparator baseline as though they were this sweep's own evidence.
"""
resolved, expected = _resolved_directory(path, label=label)
with _open_real_directory(resolved, label=label) as fd:
opened = os.fstat(fd)
if (opened.st_dev, opened.st_ino) != expected:
raise SandboxError(f"{label} was replaced between the check and the open: {resolved}")
yield fd
def _copy_transcript_artifact(source_fd: int, dest_fd: int, metadata: Mapping[str, Any]) -> dict[str, Any]:
relative, expected_digest, expected_size = _transcript_metadata(metadata)
name = PurePosixPath(relative).name
# Both `transcripts` components are opened as descriptors, not checked as
# names. An lstat that passes and a pathname that is used afterwards are two
# different directories whenever a concurrent writer renames the first one
# away — which the reuse directory, written by a prior sweep, invites.
with (
_open_real_directory("transcripts", dir_fd=dest_fd, label="transcript destination", create=True) as dest_dir_fd,
_open_real_directory("transcripts", dir_fd=source_fd, label="transcript source") as source_dir_fd,
):
os.fchmod(dest_dir_fd, 0o700)
# One descriptor for the whole transfer, and ONE read of it. Hashing the
# source and then reading it again to copy leaves the recorded digest
# describing bytes that are not the bytes written: the descriptor stops
# the pathname being substituted, not the inode being rewritten, and
# this directory belongs to a sweep that may still be writing. Digest
# what is copied, then judge it.
with _open_regular(name, dir_fd=source_dir_fd, label="transcript") as artifact_fd:
digest, copied_bytes = _copy_owner_only(
artifact_fd, name, dir_fd=dest_dir_fd, max_bytes=expected_size
)
if copied_bytes != expected_size or digest != expected_digest:
# The destination now holds bytes no expectation vouches for.
os.unlink(name, dir_fd=dest_dir_fd)
drift = "size" if copied_bytes != expected_size else "digest"
raise SandboxError(f"reused transcript {drift} drifted: {relative}")
return {"path": relative, "sha256": digest, "bytes": expected_size, "source": PARENT_EVENT_STREAM_SOURCE}
def _copy_named_artifact(source_fd: int, dest_fd: int, name: str, *, label: str) -> None:
relative = PurePosixPath(name)
if relative.is_absolute() or len(relative.parts) != 1 or relative.parts[0] in {"", ".", ".."}:
raise SandboxError(f"unsafe {label} path: {name!r}")
with _open_regular(name, dir_fd=source_fd, label=label) as artifact_fd:
# No expectation is recorded for these, so the digest is discarded - but
# "no recorded size" is not "no limit". The source is a prior sweep
# directory that can change between sweeps, so a replaced artifact could
# be arbitrarily large; MAX_TRANSCRIPT_BYTES is the ceiling the capture
# path already enforces on evidence of this kind.
_, copied = _copy_owner_only(artifact_fd, name, dir_fd=dest_fd, max_bytes=MAX_TRANSCRIPT_BYTES)
if copied > MAX_TRANSCRIPT_BYTES:
os.unlink(name, dir_fd=dest_fd)
raise SandboxError(f"reused {label} exceeds {MAX_TRANSCRIPT_BYTES} bytes: {name}")
def _require_openat() -> None:
"""openat is what makes a checked directory and a used directory the same one.
Without it the only alternative is to re-walk the name after the check,
which is exactly the race this module is guarding. Refusing is safe: the
caller in runner treats a SandboxError from reuse as "run a paid cell", so
a platform without openat pays for the cells rather than copying through a
directory nobody verified. The sweep itself is Linux-only anyway (bwrap,
/proc/uptime); this is about the unit tests and about failing loudly.
"""
if os.open not in os.supports_dir_fd or os.lstat not in os.supports_dir_fd:
raise SandboxError("comparator reuse requires POSIX openat support (os.supports_dir_fd)")
def _is_regular_at(name: str, *, dir_fd: int) -> bool:
"""True when `name` under the pinned directory is a regular non-symlink file."""
try:
metadata = os.lstat(name, dir_fd=dir_fd)
except OSError:
return False
return stat.S_ISREG(metadata.st_mode)
@contextmanager
def _open_real_directory(
path: Path | str,
*,
dir_fd: int | None = None,
label: str,
create: bool = False,
) -> Iterator[int]:
"""Open one directory that is not a symlink, and hold it for every use below.
``O_DIRECTORY | O_NOFOLLOW`` makes the check and the open a single syscall,
so unlike an ``lstat`` followed by a path, there is no window in which the
directory can be replaced. ``_resolved_directory`` still tolerates a
symlinked reuse ROOT — it hands this function the already-resolved path —
but every component below it is pinned.
"""
_require_openat()
if create:
try:
os.mkdir(path, 0o700, dir_fd=dir_fd)
except FileExistsError:
# Already there is the ordinary case — a second artifact from the
# same row. What it already IS still has to be proven, and the
# O_DIRECTORY|O_NOFOLLOW open below is what proves it, so there is
# nothing to do here.
pass
except OSError as exc:
raise SandboxError(f"{label} cannot be created: {path}: {exc}") from exc
try:
descriptor = os.open(
path,
os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0),
dir_fd=dir_fd,
)
except FileNotFoundError as exc:
# Absent is a different fact from present-but-not-a-real-directory, and
# the caller falls through to a paid cell on either.
raise SandboxError(f"{label} is missing: {path}") from exc
except OSError as exc:
raise SandboxError(f"{label} must be a real directory: {path}: {exc}") from exc
try:
# O_DIRECTORY is the check on Linux; the fstat covers a platform whose
# os module does not define it, where the flag degrades to 0.
if not stat.S_ISDIR(os.fstat(descriptor).st_mode):
raise SandboxError(f"{label} must be a real directory: {path}")
yield descriptor
finally:
os.close(descriptor)
@contextmanager
def _open_regular(name: str, *, dir_fd: int, label: str) -> Iterator[int]:
"""Open a regular non-symlink file under a pinned directory, and hold it.
Checking a name and then re-opening it is a race the reuse directory is
exposed to: it is written by a previous sweep and read by this one, so a
concurrent writer can replace a validated file with a symlink in between.
Resolving against ``dir_fd`` removes the directory half, ``O_NOFOLLOW``
refuses the leaf link, and the fstat comparison proves the open descriptor
is the inode that was checked — the same guarantee
evolution._bounded_regular_bytes makes for evidence files.
"""
_require_openat()
try:
before = os.lstat(name, dir_fd=dir_fd)
except OSError as exc:
raise SandboxError(f"{label} is missing: {name}: {exc}") from exc
if stat.S_ISLNK(before.st_mode) or not stat.S_ISREG(before.st_mode):
raise SandboxError(f"{label} must be a regular non-symlink file: {name}")
try:
descriptor = os.open(name, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0), dir_fd=dir_fd)
except OSError as exc:
raise SandboxError(f"{label} is unreadable: {name}: {exc}") from exc
try:
opened = os.fstat(descriptor)
if not stat.S_ISREG(opened.st_mode) or (opened.st_dev, opened.st_ino) != (before.st_dev, before.st_ino):
raise SandboxError(f"{label} changed while opening: {name}")
yield descriptor
finally:
os.close(descriptor)
def _copy_owner_only(source: int, name: str, *, dir_fd: int, max_bytes: int | None = None) -> tuple[str, int]:
"""Copy one open file into the pinned directory; return what was written.
The digest is taken from the same buffers that are written, so it describes
the copy rather than a state the source was in at some earlier read.
``max_bytes`` bounds the copy itself. The source is a prior sweep directory
this module already treats as concurrently writable, so a transcript
appended to after its metadata was recorded would otherwise be streamed to
EOF and only then compared against its declared size - filling the
destination, or never reaching EOF at all, long before the drift check could
reject it. Stopping one byte past the ceiling keeps that comparison
meaningful while bounding the work.
"""
# O_CREAT|O_EXCL is the existence check, and unlike a stat beforehand it is
# atomic: a file appearing between check and open cannot slip through.
try:
descriptor = os.open(
name,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0),
0o600,
dir_fd=dir_fd,
)
except FileExistsError as exc:
raise SandboxError(f"reuse destination already exists: {name}") from exc
try:
os.fchmod(descriptor, 0o600)
os.lseek(source, 0, os.SEEK_SET)
digest = hashlib.sha256()
written = 0
limit = None if max_bytes is None else max_bytes + 1
while True:
want = COPY_CHUNK_BYTES if limit is None else min(COPY_CHUNK_BYTES, limit - written)
if want <= 0:
break
chunk = os.read(source, want)
if not chunk:
break
digest.update(chunk)
written += len(chunk)
_write_all(descriptor, chunk)
os.fsync(descriptor)
return digest.hexdigest(), written
finally:
os.close(descriptor)

View file

@ -48,6 +48,7 @@ import yaml
from . import runner
from . import runner_sessions
from .comparator_reuse import current_runtime_digest
from .model_gateway import (
ANTHROPIC_API_KEY_ENV,
attach_openai_gateway,
@ -698,8 +699,17 @@ def run_proposer(
bwrap_bin: Path,
sandbox_backend: str = "bwrap",
progress_label: str | None = None,
started_monotonic: float | None = None,
) -> dict[str, Any]:
"""Run one proposer in confinement and copy only validated outputs out."""
"""Run one proposer in confinement and copy only validated outputs out.
``started_monotonic`` is the sweep clock, not a precomputed budget. The
per-session ``--timeout`` is sized for a whole generation, so a proposer
started with only the sweep minimum left would otherwise run far past the
instance window; the clock is passed rather than the leftover because the
clone, the sanitize pass and the sandbox setup below all happen before the
session starts, and a number sampled by the caller is already stale by then.
"""
with tempfile.TemporaryDirectory(prefix="wfevolve-") as tmp:
clone = runner.make_worktree(REPO_ROOT, "HEAD", Path(tmp))
@ -732,11 +742,45 @@ def run_proposer(
host_text = getattr(sandbox, "host_text", lambda value: value)
environment_builder = getattr(sandbox, "environment", build_sandbox_environment)
backend = getattr(sandbox, "backend", "bwrap")
# Sampled here, after the setup above: this is the last
# moment before the session starts, so it is the only reading
# the session's own timeout can honestly be clamped to.
remaining_seconds = (
None
if started_monotonic is None
else remaining_runtime_seconds(
max_runtime_seconds=args.max_runtime_seconds,
started_monotonic=started_monotonic,
)
)
# An exhausted cap must stop the run, not buy one more second.
# remaining_runtime_seconds floors at 0, and max(1, ...) turned
# that 0 into a one-second paid session: the admission check
# happens before cloning, sanitizing and sandbox setup, so those
# unbounded steps can spend the rest of the window and leave
# nothing for the upload reserve this cap exists to protect.
if remaining_seconds is not None and remaining_seconds < 1:
# The caller stops the run on a not-ok record, which is the
# right outcome: an exhausted cap should end the generation,
# not start a session it cannot afford to finish.
return {
"ok": False,
"error_kind": "runtime-cap-exhausted",
"error_detail": (
"the wall-clock cap elapsed during proposer setup "
"(clone, sanitize, sandbox), before the session started"
),
"duration_s": 0.0,
"num_turns": 0,
"cost_usd": None,
}
record = runner.run_claude(
host_text(prompt),
clone,
claude_bin=sandbox.claude_bin,
timeout=args.timeout,
timeout=(
args.timeout if remaining_seconds is None else min(args.timeout, remaining_seconds)
),
model=args.proposer_model,
effort=args.effort,
env=model_session_environment(
@ -823,6 +867,119 @@ def _timeout_arm_key(arm: str) -> str:
return CANDIDATE_ARMS.get(arm, arm)
EVENTBRIDGE_INSTANCE_WINDOW_SECONDS = 86_400
EVENTBRIDGE_STOP_RESERVE_SECONDS = 5_400
MIN_INSTANCE_SWEEP_SECONDS = 600
def instance_window_budget_seconds(
uptime_seconds: float,
*,
window_seconds: int = EVENTBRIDGE_INSTANCE_WINDOW_SECONDS,
reserve_seconds: int = EVENTBRIDGE_STOP_RESERVE_SECONDS,
min_seconds: int = MIN_INSTANCE_SWEEP_SECONDS,
) -> int:
"""Seconds a sweep may run before an EventBridge 24h instance stop.
The dedicated evolution box is started ~15 minutes before the Saturday
cron and stopped 24h later. A ``workflow_dispatch`` that lands on an
already-running box inherits the leftover uptime, not a fresh day.
Run 33962002890 dispatched Friday 10:57 UTC and was still on its last
review cell when the Saturday 03:00 stop cancelled the runner — 51
finished sessions never uploaded because a cancelled job skips even
``if: always()``. Capping the in-process sweep so it *fails* (instead
of vanishing) leaves the reserve for the upload step.
"""
if window_seconds < 1 or reserve_seconds < 0 or min_seconds < 1:
raise ValueError("instance window and minimum must be positive; reserve must be non-negative")
if not math.isfinite(uptime_seconds) or uptime_seconds < 0:
raise ValueError("uptime must be a finite non-negative number")
leftover = int(window_seconds - uptime_seconds - reserve_seconds)
if leftover < min_seconds:
raise ValueError(
f"instance window has only {leftover}s left after a {reserve_seconds}s "
f"upload reserve (uptime {uptime_seconds:.0f}s of {window_seconds}s); "
f"need at least {min_seconds}s"
)
return leftover
def _instance_uptime_or_none() -> float | None:
"""The uptime read main() takes before it knows whether it needs it.
Deferring the read until after argument parsing would put the parse back
inside the interval the cap is supposed to cover, so it happens first and
an unreadable /proc/uptime is only an error if the flag turns out to be set.
"""
try:
return read_instance_uptime_seconds()
except ValueError:
return None
def read_instance_uptime_seconds(uptime_path: Path = Path("/proc/uptime")) -> float:
"""Host uptime, the clock the EventBridge stop is scheduled against."""
try:
return float(uptime_path.read_text().split()[0])
except (OSError, IndexError, ValueError) as exc:
raise ValueError(f"cannot read instance uptime from {uptime_path}: {exc}") from exc
def instance_window_budget_from_uptime(
uptime_seconds: float,
*,
window_seconds: int | None = None,
reserve_seconds: int | None = None,
) -> int:
"""Apply the EventBridge window env overrides to an already-read uptime.
Separate from the read so ``main`` can take the uptime in the same breath
as its own clock: the budget and the clock it is measured against have to
describe one instant, or the interval between them is spent by nobody and
charged to the sweep.
"""
window = (
window_seconds
if window_seconds is not None
else int(os.environ.get("EVENTBRIDGE_INSTANCE_WINDOW_SECONDS", str(EVENTBRIDGE_INSTANCE_WINDOW_SECONDS)))
)
reserve = (
reserve_seconds
if reserve_seconds is not None
else int(os.environ.get("EVENTBRIDGE_STOP_RESERVE_SECONDS", str(EVENTBRIDGE_STOP_RESERVE_SECONDS)))
)
return instance_window_budget_seconds(uptime_seconds, window_seconds=window, reserve_seconds=reserve)
def remaining_runtime_seconds(*, max_runtime_seconds: int | None, started_monotonic: float) -> int | None:
"""Seconds left in an optional wall-clock cap, or None when uncapped."""
if max_runtime_seconds is None:
return None
if max_runtime_seconds < 1:
raise ValueError("max runtime must be positive")
leftover = max_runtime_seconds - (time.monotonic() - started_monotonic)
return max(0, int(leftover))
def capped_timeout_seconds(requested: int, remaining: int | None) -> int:
"""Clamp one managed-process timeout to the leftover instance window."""
if requested < 1:
raise ValueError("requested timeout must be positive")
if remaining is None:
return requested
if remaining < 1:
raise ValueError("no time remains in the instance window")
return min(requested, remaining)
def generation_timeout_seconds(
*,
task_count: int,
@ -877,6 +1034,7 @@ def runner_argv(
task_bindings: list[dict[str, Any]],
target_base_digests: dict[str, str],
proposer_model: str | None = None,
reuse_results: Path | None = None,
) -> list[str]:
incumbent_arms = resolve_incumbent_arms(overlay_dir, args.arms)
paired_arms = executed_benchmark_arms(incumbent_arms)
@ -927,6 +1085,8 @@ def runner_argv(
argv += ["--ce-plugin-dir", str(args.ce_plugin_dir), "--ce-plugin-version", args.ce_plugin_version]
if args.unsafe_no_bwrap:
argv.append("--unsafe-no-bwrap")
if reuse_results is not None:
argv += ["--reuse-results", str(reuse_results)]
return argv
@ -944,6 +1104,12 @@ def runner_environment(args: argparse.Namespace) -> dict[str, str]:
# actually show progress rather than a burst at the end.
"PYTHONUNBUFFERED": "1",
}
# process_control replaces the child environment wholesale, so a digest the
# workflow exported reaches the runner only if it is forwarded here. Without
# this the runner stamps no runtime_digest and the reuse lock never engages.
runtime_digest = current_runtime_digest()
if runtime_digest:
env["RUNTIME_DIGEST"] = runtime_digest
if args.auth_token:
env[ANTHROPIC_API_KEY_ENV] = args.auth_token
return env
@ -1199,6 +1365,13 @@ def _require_finite_metric(value: Any, name: str, *, nullable: bool = False, max
raise ValueError(f"promotion has invalid {name}")
def _positive_int(value: str) -> int:
parsed = int(value)
if parsed < 1:
raise argparse.ArgumentTypeError(f"{value} is not a positive integer")
return parsed
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--tasks", required=True, type=Path)
@ -1237,7 +1410,8 @@ def build_parser() -> argparse.ArgumentParser:
"--seed-results",
type=Path,
default=None,
help="prior wfbench results dir used as generation-0 proposer evidence",
help="prior wfbench results dir used as generation-0 proposer evidence "
"and as --reuse-results for unchanged incumbent/CE cells",
)
parser.add_argument(
"--initial-overlay",
@ -1271,6 +1445,19 @@ def build_parser() -> argparse.ArgumentParser:
default=runner_sessions.SESSION_TIMEOUT_SECONDS,
help="per session, seconds",
)
parser.add_argument(
"--max-runtime-seconds",
type=_positive_int,
default=None,
help="wall-clock cap for the whole evolve process (CI derives this from "
"instance uptime so the sweep exits before EventBridge stops the box)",
)
parser.add_argument(
"--max-runtime-from-instance-window",
action="store_true",
help="derive --max-runtime-seconds from /proc/uptime at startup, so the "
"budget and the clock it is measured against describe one instant",
)
parser.add_argument("--base-url", default=None)
parser.add_argument(
"--anthropic-api-key",
@ -1308,8 +1495,26 @@ def build_parser() -> argparse.ArgumentParser:
def main() -> int:
# These two lines are the cap, and they are adjacent on purpose: the clock
# the sweep is measured against, and the uptime the budget is derived from.
# run-evolution.sh used to compute the budget in its own `uv run python -c`
# and pass a number, so the script's remaining work and this interpreter's
# startup were spent by nobody and charged to the sweep — out of the upload
# reserve the cap exists to protect. Nothing can be spent between them now.
started_monotonic = time.monotonic()
instance_uptime = _instance_uptime_or_none()
parser = build_parser()
args = parser.parse_args()
if args.max_runtime_from_instance_window:
if args.max_runtime_seconds is not None:
parser.error("--max-runtime-from-instance-window and --max-runtime-seconds are mutually exclusive")
if instance_uptime is None:
parser.error("--max-runtime-from-instance-window needs a readable /proc/uptime")
try:
args.max_runtime_seconds = instance_window_budget_from_uptime(instance_uptime)
except ValueError as exc:
parser.error(str(exc))
print(f"capping the sweep to {args.max_runtime_seconds}s so the instance-window reserve can upload evidence")
if args.generations < 1:
parser.error("--generations must be positive")
if args.runs < 1 or args.timeout < 1:
@ -1379,6 +1584,7 @@ def main() -> int:
try:
return _run_generations(
args,
started_monotonic=started_monotonic,
selected_task_rows=selected_task_rows,
skipped_expensive=skipped_expensive,
selected_tasks=selected_tasks,
@ -1394,6 +1600,7 @@ def main() -> int:
def _run_generations(
args: argparse.Namespace,
*,
started_monotonic: float,
selected_task_rows: list[dict[str, Any]],
skipped_expensive: list[str],
selected_tasks: list[dict[str, Any]],
@ -1465,6 +1672,19 @@ def _run_generations(
incumbent_arms=requested_arms,
prior_proposal=staged_prior_included,
)
# Check the window before the paid session, not after it. A
# generation that cannot fit its sweep should not buy a proposal
# first and discover the deadline on the way out.
before_proposer = remaining_runtime_seconds(
max_runtime_seconds=args.max_runtime_seconds,
started_monotonic=started_monotonic,
)
if before_proposer is not None and before_proposer < MIN_INSTANCE_SWEEP_SECONDS:
print(
f"[gen {generation}] stopping with {before_proposer}s left before the "
f"instance window ends; not starting a proposer session"
)
return 1
print(f"[gen {generation}] proposing…")
record = run_proposer(
prompt,
@ -1475,6 +1695,11 @@ def _run_generations(
bwrap_bin=bwrap_bin,
sandbox_backend=sandbox_backend,
progress_label=f"gen {generation} proposer",
# The clock, not the reading taken above: run_proposer clones,
# sanitizes and builds a sandbox before the session starts, so
# before_proposer is stale by then. It still decides whether to
# start at all — it just cannot decide how long to allow.
started_monotonic=started_monotonic,
)
# Redact any API token echoed into the session record (e.g. an
# error_detail stderr_tail) before it enters the uploaded artifact.
@ -1513,6 +1738,16 @@ def _run_generations(
print(f"[gen {generation}] promotion targets contain uncommitted or drifted bytes")
return 1
print(f"[gen {generation}] benchmarking candidate…")
leftover = remaining_runtime_seconds(
max_runtime_seconds=args.max_runtime_seconds,
started_monotonic=started_monotonic,
)
if leftover is not None and leftover < MIN_INSTANCE_SWEEP_SECONDS:
print(
f"[gen {generation}] stopping with {leftover}s left before the "
f"instance window ends; partial evidence is in {out_root}/"
)
return 1
benchmark_argv = runner_argv(
args,
bench_dir,
@ -1520,20 +1755,30 @@ def _run_generations(
task_bindings=selected_tasks,
target_base_digests=target_base_digests,
proposer_model=generation_proposer_model,
reuse_results=evidence_dir,
)
benchmark_command = (
benchmark_argv
if sandbox_backend == "host-unsafe"
else pid_namespace_command(benchmark_argv, bwrap_bin=bwrap_bin)
)
bench = run_managed(
benchmark_command,
timeout=generation_timeout_seconds(
sweep_timeout = capped_timeout_seconds(
generation_timeout_seconds(
task_count=len(selected_task_rows),
runs=args.runs,
session_timeout=args.timeout,
incumbent_arms=incumbent_arms,
),
leftover,
)
if leftover is not None:
print(
f"[gen {generation}] sweep timeout {sweep_timeout}s "
f"(instance window leftover {leftover}s)"
)
bench = run_managed(
benchmark_command,
timeout=sweep_timeout,
env=runner_environment(args),
require_pid_namespace=sandbox_backend == "bwrap",
# The sweep is the multi-hour phase; without this its per-run
@ -1545,7 +1790,13 @@ def _run_generations(
# The sweep runs with GITNEXUS_BENCH_ANTHROPIC_API_KEY in its environment,
# so its detail/stderr tail is a token-bearing sink like any other.
detail = redacted_failure(args, str(bench.detail or bench.stderr_tail[-1000:]))
print(f"[gen {generation}] benchmark run failed ({bench.state}, exit {bench.returncode}): {detail}")
if leftover is not None and bench.state == "timeout":
print(
f"[gen {generation}] benchmark hit the instance-window budget "
f"({sweep_timeout}s); partial evidence is in {bench_dir}: {detail}"
)
else:
print(f"[gen {generation}] benchmark run failed ({bench.state}, exit {bench.returncode}): {detail}")
return 1
promotion = json.loads((bench_dir / "promotion.json").read_text())
for line in summarize_gate(promotion):

View file

@ -0,0 +1,136 @@
"""Append each upstream request's usage exactly as the provider reported it.
This runs INSIDE the LiteLLM proxy, on the far side of the translation that
turns an OpenAI response into the Anthropic shape Claude Code expects. That is
the only point that still knows which provider served the request, what model
actually answered, and what the native usage object said before its fields were
renamed into someone else's semantics.
Deliberately self-contained: the proxy loads this file by path from the config
directory, so it cannot assume ``workflow_bench`` is importable. Normalization
lives in workflow_bench.provider_usage and runs offline over what this writes -
the native object is the evidence, and deriving from it here would mean the
derivation could not be revisited without re-running a paid sweep.
Never raises. A cell that fails still spent money upstream, and losing the
accounting because the log write failed would be the worse outcome.
"""
from __future__ import annotations
import json
import os
import threading
from typing import Any
from litellm.integrations.custom_logger import CustomLogger
# Literals, not imports. LiteLLM loads this file BY PATH from the config
# directory via spec_from_file_location, so it has no parent package and the
# directory is not on sys.path - a relative or sibling import raises
# ImportError and the proxy refuses to start. workflow_bench.provider_usage
# holds the canonical copies and a test asserts these agree with them, which
# catches drift without coupling at import time.
USAGE_LOG_ENV_VAR = "GITNEXUS_BENCH_PROVIDER_USAGE"
SWEEP_ID_ENV_VAR = "GITNEXUS_BENCH_SWEEP_ID"
def canonical_provider(label, call_type): # noqa: ANN001, ANN201
"""Adapter key for the usage shape, or None when it cannot be resolved.
Mirrors workflow_bench.provider_usage.canonical_provider; see the note
above for why this is a copy rather than an import.
"""
if label == "openai" and call_type and "responses" in call_type:
return "openai-responses"
if label == "anthropic":
return "anthropic"
return None
SCHEMA_VERSION = 1
_LOCK = threading.Lock()
def _plain(value: Any) -> Any:
"""Provider usage arrives as pydantic models; keep the shape, drop the class."""
for attr in ("model_dump", "dict"):
method = getattr(value, attr, None)
if callable(method):
try:
return method()
except Exception:
pass
if isinstance(value, dict):
return value
return None
class ProviderUsageLogger(CustomLogger):
async def async_log_success_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001
self._append("success", kwargs, response_obj, start_time, end_time)
async def async_log_failure_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001
# Failed requests are billed too, and a sweep that only accounts for
# successes understates what it spent.
self._append("failure", kwargs, response_obj, start_time, end_time)
def log_success_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001
self._append("success", kwargs, response_obj, start_time, end_time)
def log_failure_event(self, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001
# The synchronous counterpart. Overriding only the success hook here
# recorded successes and let failures fall through to the base class,
# which accounts for nothing - and a failed request is still billed, so
# a sweep missing them understates what it spent.
self._append("failure", kwargs, response_obj, start_time, end_time)
def _append(self, status, kwargs, response_obj, start_time, end_time) -> None: # noqa: ANN001
path = os.environ.get(USAGE_LOG_ENV_VAR)
if not path:
return
try:
params = kwargs.get("litellm_params") or {}
call_type = kwargs.get("call_type")
provider_label = kwargs.get("custom_llm_provider") or params.get("custom_llm_provider")
metadata = params.get("metadata") or {}
event = {
"schema_version": SCHEMA_VERSION,
"status": status,
# Identity. The REQUESTED model is the caller's role name and the
# ACTUAL model is what answered; pricing must follow the second,
# because several roles map onto one upstream model here.
"requested_model": kwargs.get("model"),
"actual_model": getattr(response_obj, "model", None),
# Two fields, because they answer different questions. The raw
# label is what LiteLLM said; "provider" is the adapter key,
# which needs the call type too - LiteLLM reports "openai" for
# both Chat Completions and Responses and those report usage
# differently. Unresolvable stays None so normalize_usage
# refuses rather than guessing token semantics.
"provider_label": provider_label,
"provider": canonical_provider(provider_label, call_type),
"response_id": getattr(response_obj, "id", None),
"call_type": call_type,
"sweep_id": os.environ.get(SWEEP_ID_ENV_VAR),
# The per-request half of identity, and the only thing that can
# attribute a request to a cell: one proxy serves the whole
# sweep, so anything read from the environment is the same for
# every event. Recorded even when absent, because knowing the
# attribution is unavailable is itself a fact about the run.
"session_id": metadata.get("litellm_session_id") or metadata.get("session_id"),
"started_at": str(start_time),
"completed_at": str(end_time),
# Verbatim. Not flattened, not renamed, not summed.
"native_usage": _plain(getattr(response_obj, "usage", None)),
}
line = json.dumps(event, default=str) + "\n"
with _LOCK, open(path, "a", encoding="utf-8") as handle:
handle.write(line)
except Exception:
# Accounting is evidence, not control flow: never take the sweep down.
return
handler = ProviderUsageLogger()

View file

@ -0,0 +1,311 @@
#!/usr/bin/env python3
"""Cheap cost model for the skill-evolution review generation.
This is the ce-optimize measurement harness. It does not start Claude and it
does not replay a run. It reads the review corpus, the evolve defaults and the
workflow's workers default, then schedules the measured cell durations in
``session_durations.json`` the way ``sweep_task_cells`` schedules real cells.
Everything priced here is measured. Cell durations and the proposer session
come from a real artifact, and the work outside the agent sessions comes from
that run's own step wall minus the time its sessions and proposer account for.
Weekly assumes a matching seed, so every reusable comparator cell is skipped
and only the candidate arm is paid. Cold assumes an empty seed.
"""
from __future__ import annotations
import json
import math
import re
import statistics as st
import subprocess
import sys
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[2]
EVAL_ROOT = REPO_ROOT / "eval"
REVIEW_TASKS = EVAL_ROOT / "workflow_bench" / "tasks.review.scenarios.yaml"
EVOLVE_PY = EVAL_ROOT / "workflow_bench" / "evolve.py"
RUNNER_PY = EVAL_ROOT / "workflow_bench" / "runner.py"
ARTIFACTS_PY = EVAL_ROOT / "workflow_bench" / "runner_artifacts.py"
REUSE_PY = EVAL_ROOT / "workflow_bench" / "comparator_reuse.py"
WORKFLOW = REPO_ROOT / ".github" / "workflows" / "gitnexus-skill-evolution.yml"
MEASURED = json.loads(
(Path(__file__).resolve().parent / "session_durations.json").read_text(encoding="utf-8")
)
# Per arm, because the arms are not interchangeable and the weekly lane pays
# only the candidate one. Cells are submitted run-major and arm-minor
# (runner.py ``planned``), so at workers=3 every wave holds one cell of each
# arm and the slowest arm sets the wave.
DURATIONS_BY_ARM: dict[str, tuple[float, ...]] = {
arm: tuple(values) for arm, values in MEASURED["cell_duration_s_by_arm"].items()
}
PROPOSER_SECONDS: float = MEASURED["proposer_duration_s"]
_RESIDUAL = MEASURED["residual"]
# Clone, graph build, sandbox, teardown: the sweep's own time, taken as that
# run's step wall minus what its sessions and proposer account for. Charged
# SERIALLY, outside the pool, and charged PER SHA rather than per cell. The
# residual mixes per-cell work with per-SHA graph setup and the artifact cannot
# separate them; per-SHA is the direction that refuses to credit a run for
# shrinking work it still performs, which per-cell did - a weekly generation
# pays one arm instead of three but builds exactly the same graphs. See
# session_durations.json residual._split_assumption.
SHA_OVERHEAD_SECONDS: float = _RESIDUAL["sha_overhead_s"]
# runner.py CANDIDATE_ARMS derives the candidate arm from its incumbent, and
# only an incumbent row can be reused from a prior generation.
CANDIDATE_ARM = "candidate_review"
REVIEW_ARMS = ("ce_review", "review", CANDIDATE_ARM)
SUITE_FILES = (
"tests/test_measure_evolution_cost.py",
"tests/test_comparator_reuse.py",
"tests/test_evolve.py",
"tests/test_sanitized_graph.py",
"tests/test_workflow_bench.py",
"tests/test_workflow_bench_sessions.py",
"tests/test_session_progress.py",
)
def _read(path: Path) -> str:
return path.read_text(encoding="utf-8")
def review_tasks(text: str) -> list[dict[str, str]]:
tasks: list[dict[str, str]] = []
current: dict[str, str] | None = None
for raw in text.splitlines():
line = raw.strip()
if line.startswith("id:"):
if current is not None:
tasks.append(current)
current = {"id": line.split(":", 1)[1].strip()}
elif line.startswith("ref:") and current is not None:
current["ref"] = line.split(":", 1)[1].strip()
if current is not None:
tasks.append(current)
return tasks
def evolve_default(name: str, text: str) -> int:
match = re.search(rf'add_argument\("--{re.escape(name)}".*?default=(\d+)', text, flags=re.S)
if match is None:
raise ValueError(f"evolve.py is missing --{name} default")
return int(match.group(1))
def workflow_dispatch_workers(text: str) -> int:
match = re.search(r"^\s+workers:\n(?:.*\n)*?^\s+default: '(\d+)'", text, flags=re.M)
if match is None:
raise ValueError("workflow_dispatch workers default is missing")
return int(match.group(1))
def feature_enabled() -> tuple[int, int]:
evolve = _read(EVOLVE_PY)
runner = _read(RUNNER_PY)
artifacts = _read(ARTIFACTS_PY)
reuse = int(
REUSE_PY.is_file()
and "--reuse-results" in evolve
and "select_reusable_comparator_rows" in runner
and "CANDIDATE" in _read(REUSE_PY)
)
templates = int("def copy_isolated_tree" in artifacts and "clone_templates" in runner)
return reuse, templates
def graph_pipeline_enabled(runner_text: str) -> int:
"""True when the runner prefetches the next SHA during paid sessions."""
return int("prefetch_next_graph" in runner_text or "GraphPrefetch" in runner_text)
def fed_pool_enabled(runner_text: str) -> int:
"""True when the sweep feeds a live pool instead of waiting on waves."""
return int("def _run_fed_pool" in runner_text)
def paid_arms(weekly: bool, reuse_enabled: bool) -> tuple[str, ...]:
"""Arms a generation actually pays for."""
if weekly and reuse_enabled:
return (CANDIDATE_ARM,)
return REVIEW_ARMS
def task_cells(runs: int, arms: tuple[str, ...], offset: int) -> list[float]:
"""One task's cell durations in submission order: run-major, arm-minor.
Each arm draws from its own measured sample, cycled from ``offset`` so the
caller can average over every alignment instead of trusting one.
"""
cells: list[float] = []
for run_idx in range(runs):
for arm in arms:
sample = DURATIONS_BY_ARM[arm]
cells.append(sample[(offset + run_idx) % len(sample)])
return cells
def wave_makespan(durations: list[float], workers: int) -> float:
"""Today's scheduler: fixed waves of ``workers``, with a barrier between."""
return sum(
max(durations[start : start + workers]) for start in range(0, len(durations), workers)
)
def fed_makespan(durations: list[float], workers: int) -> float:
"""Continuously fed pool: a free worker takes the next cell immediately."""
busy_until = [0.0] * workers
for duration in durations:
first = min(range(workers), key=busy_until.__getitem__)
busy_until[first] += duration
return max(busy_until)
def expected_task_seconds(
runs: int, arms: tuple[str, ...], workers: int, *, fed_pool: bool
) -> float:
"""Mean makespan of one task over every alignment of the measured samples.
One fixed alignment would let an accident of the source run - its slowest
cells happen to come first - decide the answer. Averaging keeps the real
multiset and the real ordering effects without that artifact, and stays
deterministic.
"""
if runs < 1 or not arms:
return 0.0
makespan = fed_makespan if fed_pool else wave_makespan
# lcm, not max: with samples of 13 and 14, max would wrap the shorter one
# and count its first entry twice.
alignments = math.lcm(*(len(DURATIONS_BY_ARM[arm]) for arm in arms))
return (
sum(makespan(task_cells(runs, arms, offset), workers) for offset in range(alignments))
/ alignments
)
def generation_seconds(
*,
task_count: int,
runs: int,
arms: tuple[str, ...],
workers: int,
fed_pool: bool,
unique_shas: int,
) -> int:
"""Whole generation: proposer, then the tasks back to back, plus overhead.
Prices a HEALTHY sweep. A run whose cells return unusable evidence does not
reach this wall at all: the outage breaker aborts after
``DEFAULT_OUTAGE_STREAK`` consecutive systemic failures, which for the
sample's own error sequence is cell 5 of 41.
Sweep overhead is charged per SHA, so it does not shrink with the arm count.
Weekly pays one arm instead of three but builds the same graphs, and billing
that per cell credited it for a saving the real run never makes.
"""
return round(
PROPOSER_SECONDS
+ task_count * expected_task_seconds(runs, arms, workers, fed_pool=fed_pool)
+ unique_shas * SHA_OVERHEAD_SECONDS
)
def _pytest_python() -> list[str]:
venv_python = EVAL_ROOT / ".venv" / "bin" / "python"
if venv_python.is_file():
return [str(venv_python)]
if (EVAL_ROOT / "uv.lock").is_file():
return ["uv", "run", "--locked", "--extra", "dev", "python"]
return [sys.executable]
def suite_passed() -> int:
files = [name for name in SUITE_FILES if (EVAL_ROOT / name).is_file()]
if not files:
return 0
cmd = [*_pytest_python(), "-m", "pytest", *files, "-q", "--tb=no", "--no-header"]
try:
completed = subprocess.run(
cmd, cwd=EVAL_ROOT, check=False, capture_output=True, text=True, timeout=240
)
except (OSError, subprocess.TimeoutExpired):
return 0
return int(completed.returncode == 0)
def main() -> int:
tasks = review_tasks(_read(REVIEW_TASKS))
evolve = _read(EVOLVE_PY)
runner = _read(RUNNER_PY)
runs = evolve_default("runs", evolve)
workers = workflow_dispatch_workers(_read(WORKFLOW))
reuse_enabled, clone_templates_enabled = feature_enabled()
fed_pool = fed_pool_enabled(runner)
# Both walls build the same graphs; the arm count does not change that.
unique_shas = len({t.get("ref", "") for t in tasks if t.get("ref")})
payload: dict[str, object] = {}
for label, weekly in (("weekly", True), ("cold", False)):
arms = paid_arms(weekly, bool(reuse_enabled))
payload[f"estimated_{label}_wall_seconds"] = generation_seconds(
task_count=len(tasks),
runs=runs,
arms=arms,
workers=workers,
fed_pool=bool(fed_pool),
unique_shas=unique_shas,
)
payload[f"paid_{label}_cells"] = len(tasks) * runs * len(arms)
# What the wave barrier costs: the same cells, continuously fed.
payload[f"fed_pool_{label}_wall_seconds"] = generation_seconds(
task_count=len(tasks),
runs=runs,
arms=arms,
workers=workers,
fed_pool=True,
unique_shas=unique_shas,
)
all_durations = [d for sample in DURATIONS_BY_ARM.values() for d in sample]
payload.update(
{
"suite_passed": suite_passed(),
"promotion_min_runs": evolve_default("promotion-min-runs", evolve),
"review_task_count": len(tasks),
"candidate_cells": len(tasks) * runs,
"workers": workers,
"unique_task_shas": len({t.get("ref", "") for t in tasks if t.get("ref")}),
"reuse_enabled": reuse_enabled,
"clone_templates_enabled": clone_templates_enabled,
"graph_pipeline_enabled": graph_pipeline_enabled(runner),
"fed_pool_enabled": fed_pool,
"measured_cell_count": len(all_durations),
"median_cell_seconds": round(st.median(all_durations)),
"mean_cell_seconds": round(st.mean(all_durations)),
"max_cell_seconds": round(max(all_durations)),
"median_candidate_cell_seconds": round(st.median(DURATIONS_BY_ARM[CANDIDATE_ARM])),
"mean_candidate_cell_seconds": round(st.mean(DURATIONS_BY_ARM[CANDIDATE_ARM])),
"proposer_seconds": round(PROPOSER_SECONDS),
"sha_overhead_seconds": round(SHA_OVERHEAD_SECONDS, 1),
}
)
json.dump(payload, sys.stdout, sort_keys=True)
sys.stdout.write("\n")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View file

@ -31,6 +31,8 @@ from typing import Any
import yaml
from .provider_usage import USAGE_ENV_VARS
ANTHROPIC_API_KEY_ENV = "GITNEXUS_BENCH_ANTHROPIC_API_KEY"
LEGACY_ANTHROPIC_API_KEY_ENV = "GITNEXUS_BENCH_AUTH_TOKEN"
OPENAI_API_KEY_ENV = "GITNEXUS_BENCH_OPENAI_API_KEY"
@ -173,6 +175,9 @@ def resolve_model_access(
return ModelAccess(start_proxy=False)
USAGE_CALLBACK_MODULE = "provider_usage_callback"
def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]:
seen: list[str] = []
for name in model_names:
@ -196,7 +201,15 @@ def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]:
}
for name in seen
],
"litellm_settings": {"request_timeout": GATEWAY_REQUEST_TIMEOUT_S},
"litellm_settings": {
"request_timeout": GATEWAY_REQUEST_TIMEOUT_S,
# Captures each upstream request's usage as the provider reported
# it, before translation renames OpenAI's fields into Anthropic's
# shape and loses which arithmetic applies. Resolved by LiteLLM
# relative to the config directory, which is why the module is
# copied next to the config rather than imported from the package.
"callbacks": [f"{USAGE_CALLBACK_MODULE}.handler"],
},
"general_settings": {"master_key": "os.environ/LITELLM_MASTER_KEY"},
}
@ -204,9 +217,26 @@ def openai_litellm_config(model_names: Sequence[str]) -> dict[str, Any]:
def write_openai_litellm_config(path: Path, model_names: Sequence[str]) -> Path:
path.write_text(yaml.safe_dump(openai_litellm_config(model_names), sort_keys=False))
path.chmod(0o600)
_install_usage_callback(path.parent)
return path
def _install_usage_callback(config_dir: Path) -> Path:
"""Place the usage logger where LiteLLM resolves callbacks from.
LiteLLM loads a dotted callback path as a file relative to the config
directory before falling back to a package import, and the proxy runs as
its own process that need not have this package on sys.path. Copying the
one module is what makes the callback resolvable in both cases.
"""
source = Path(__file__).with_name("litellm_usage_callback.py")
destination = config_dir / f"{USAGE_CALLBACK_MODULE}.py"
destination.write_text(source.read_text())
destination.chmod(0o600)
return destination
def _free_loopback_port() -> int:
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
sock.bind(("127.0.0.1", 0))
@ -302,6 +332,17 @@ class OpenAIGateway(AbstractContextManager["OpenAIGateway"]):
"OPENAI_API_KEY": self.openai_api_key,
"LITELLM_MASTER_KEY": self.auth_token,
}
# The proxy is a separate process and Popen(env=...) REPLACES the
# parent environment rather than extending it, so anything the usage
# callback reads has to be forwarded by name. Without this the callback
# loads, finds no destination, and returns silently on every request -
# the accounting looks configured and records nothing. Forwarded
# individually rather than by inheriting the environment, because the
# allowlist above is the gateway's credential boundary.
for name in USAGE_ENV_VARS:
value = os.environ.get(name)
if value:
env[name] = value
if os.name == "nt":
# Windows subprocess DLL/socket initialization needs SystemRoot.
# Keep the rest of the gateway's credential boundary explicit.

View file

@ -22,6 +22,15 @@ from .process_control import ManagedProcessResult, run_managed
MAX_EVIDENCE_FILE_BYTES = 256 * 1024
MAX_BUNDLE_BYTES = 2 * 1024 * 1024
SANDBOX_WORKSPACE = "/workspace"
# The review artifact lives OUTSIDE the workspace, in its own writable
# directory. A writable FILE inside a read-only directory is not writable to
# anything that writes atomically: the Write tool creates
# `<target>.tmp.<n>.<hex>` beside the target and renames it, so a read-only
# parent fails the temp create with EROFS and the artifact is never written.
# Binding a writable directory outside /workspace lets the rename land while
# the workspace itself stays entirely read-only.
SANDBOX_REVIEW_OUTPUT = "/review-output"
REVIEW_OUTPUT_DIRNAME = "review-output"
SANDBOX_HOME = "/home/agent"
SANDBOX_TMP = "/tmp"
SANDBOX_CLAUDE = "/opt/claude/claude"
@ -118,36 +127,41 @@ REVIEW_RUNTIME_DIRECTORIES = (
)
def prepare_review_workspace(sandbox: SandboxSession, artifact_name: str) -> Path:
"""Prepare disposable mount targets; never truncate a pre-existing entry."""
def review_output_path(sandbox: SandboxSession, artifact_name: str) -> Path:
"""Host path of the review artifact: a private directory, not the clone.
One source of truth for the location, so the mount, the parse and the
artifact copy cannot drift apart.
"""
clone = _real_directory(sandbox.clone, label="review clone")
if PurePosixPath(artifact_name).name != artifact_name or "\\" in artifact_name or artifact_name in ("", ".", ".."):
raise SandboxError("review artifact must be a root filename")
output = clone / artifact_name
# No agent runs while this private clone is being prepared. On POSIX the
# directory descriptor additionally binds the exclusive create to its owner.
directory_fd = None
try:
if os.name != "nt":
directory_fd = os.open(clone, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
fd = os.open(
artifact_name if directory_fd is not None else output,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_NOFOLLOW", 0),
0o600,
dir_fd=directory_fd,
)
try:
if not stat.S_ISREG(os.fstat(fd).st_mode):
raise SandboxError("review artifact must be a regular file")
finally:
os.close(fd)
except FileExistsError as exc:
raise SandboxError("review artifact already exists") from exc
finally:
if directory_fd is not None:
os.close(directory_fd)
return Path(sandbox.private_root) / REVIEW_OUTPUT_DIRNAME / artifact_name
def prepare_review_workspace(sandbox: SandboxSession, artifact_name: str) -> Path:
"""Prepare disposable mount targets; never truncate a pre-existing entry.
Creates the artifact's own directory and returns the path the agent is
expected to write. The file itself is deliberately NOT pre-created: the
agent writes it atomically (temp file beside the target, then rename), so
the directory is what has to be writable, and an existing empty file would
only be something for the write to trip over. Absence is meaningful — it is
how ``parse_review_output`` tells "never written" from "written badly".
"""
output = review_output_path(sandbox, artifact_name)
# No agent runs while this private root is being prepared, and the
# exclusive create is what proves the directory is ours rather than
# something a previous cell left behind.
try:
output.parent.mkdir(mode=0o700, parents=False, exist_ok=False)
except FileExistsError as exc:
raise SandboxError("review artifact directory already exists") from exc
except OSError as exc:
raise SandboxError(f"review artifact directory is unavailable: {output.parent}") from exc
clone = _real_directory(sandbox.clone, label="review clone")
if sandbox.backend != "bwrap":
return output
created: list[str] = []
@ -214,6 +228,10 @@ class SandboxSession:
ReadOnlyMount(self.clone, SANDBOX_WORKSPACE),
ReadOnlyMount(self.home, SANDBOX_HOME),
ReadOnlyMount(self.temp, SANDBOX_TMP),
# The review artifact directory is a real mount on bwrap, so the
# host-unsafe backend has to translate it too. Without this the
# review prompt names a path that exists on neither backend.
ReadOnlyMount(Path(self.private_root) / REVIEW_OUTPUT_DIRNAME, SANDBOX_REVIEW_OUTPUT),
]
for mount in sorted(mappings, key=lambda item: len(item.target), reverse=True):
target = mount.target.rstrip("/")
@ -234,9 +252,16 @@ class SandboxSession:
SANDBOX_WORKSPACE,
SANDBOX_HOME,
SANDBOX_TMP,
SANDBOX_REVIEW_OUTPUT,
]
ordered = sorted(set(targets), key=len, reverse=True)
pattern = re.compile("|".join(re.escape(target) for target in ordered))
# Only translate at a path boundary. "/review-output" occurs twice in
# "/review-output/review-output.json" - once as the directory and once
# inside the filename - and rewriting the second turned the artifact path
# into nonsense. A target must be followed by "/", whitespace, a quote or
# end of string to be a path rather than a prefix of a longer name.
boundary = r"""(?=[/\s"']|$)"""
pattern = re.compile("(?:" + "|".join(re.escape(target) for target in ordered) + ")" + boundary)
return pattern.sub(lambda match: self.host_path(match.group(0)), value)
@property
@ -549,12 +574,17 @@ def build_claude_settings(*, sandbox_enabled: bool = True) -> str:
"allowLocalBinding": False,
},
"filesystem": {
"allowWrite": [SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME],
# SANDBOX_REVIEW_OUTPUT is the review artifact directory. The
# bwrap bind alone is not enough: this policy is a second,
# independent gate the CLI applies to its own tools, and a path
# missing here is unwritable however the mount is shaped.
"allowWrite": [SANDBOX_WORKSPACE, SANDBOX_TMP, SANDBOX_HOME, SANDBOX_REVIEW_OUTPUT],
"denyRead": ["/"],
"allowRead": [
SANDBOX_WORKSPACE,
SANDBOX_TMP,
SANDBOX_HOME,
SANDBOX_REVIEW_OUTPUT,
"/usr",
"/bin",
"/lib",

View file

@ -0,0 +1,188 @@
"""Per-request usage as the provider reported it, plus a derived cross-provider view.
The benchmark has been reading token counts out of Claude Code's session output,
which is Anthropic-shaped whatever actually served the request. That works until
the upstream is OpenAI, because the two providers do not merely name their fields
differently - they mean opposite things by them:
Anthropic: total_input = input_tokens
+ cache_creation_input_tokens
+ cache_read_input_tokens
(input_tokens is only the UNCACHED remainder; cache fields ADD)
OpenAI: total_input = input_tokens
ordinary = input_tokens - cached_tokens - cache_write_tokens
(input_tokens is the WHOLE; cache fields are SUBSETS)
Adding OpenAI's three together double-counts; subtracting Anthropic's
under-counts. So the native object is authoritative and is stored verbatim, and
the normalized view is derived from it per provider.
The second rule is that a field nobody reported is UNKNOWN, not zero. A stored
``cache_read = 0`` previously could mean either "the provider said zero" or "our
adapter never looked", and those two must never be written identically again:
the first says caching is not working, the second says we cannot tell.
"""
from __future__ import annotations
from collections.abc import Mapping
from dataclasses import dataclass
from typing import Any
SCHEMA_VERSION = 1
# Read by the in-proxy callback and forwarded by the gateway that launches it.
# Defined here because this module is pure stdlib: model_gateway can import the
# name without importing litellm, which only the callback needs.
#
# Both are SWEEP-scoped, and that is a constraint rather than an oversight.
# attach_openai_gateway wraps the whole sweep (runner.py), so one proxy serves
# every cell and its environment is fixed for that proxy's lifetime - while
# cells run concurrently under --workers and interleave requests through it. An
# environment variable therefore cannot carry a per-cell identity: it would
# record one constant against every event. Attributing a request to a cell
# needs an identifier that travels WITH the request; see the session fields the
# callback records for the intended hook.
USAGE_LOG_ENV_VAR = "GITNEXUS_BENCH_PROVIDER_USAGE"
SWEEP_ID_ENV_VAR = "GITNEXUS_BENCH_SWEEP_ID"
USAGE_ENV_VARS = (USAGE_LOG_ENV_VAR, SWEEP_ID_ENV_VAR)
ANTHROPIC = "anthropic"
OPENAI_RESPONSES = "openai-responses"
class UsageSemanticsError(ValueError):
"""The native usage object does not satisfy its own provider's arithmetic."""
@dataclass(frozen=True)
class NormalizedUsage:
"""Cross-provider view. ``None`` means the provider did not report it.
Deliberately not defaulted to 0: see the module docstring. Every consumer
that sums these has to decide what to do about unknown, and making it None
forces that decision to be explicit instead of silently counting zero.
"""
ordinary_input_tokens: int | None
cache_read_input_tokens: int | None
cache_write_input_tokens: int | None
total_input_tokens: int | None
output_tokens: int | None
reasoning_output_tokens: int | None
@property
def complete(self) -> bool:
return all(
value is not None
for value in (
self.ordinary_input_tokens,
self.cache_read_input_tokens,
self.cache_write_input_tokens,
self.total_input_tokens,
self.output_tokens,
)
)
@property
def unknown_fields(self) -> tuple[str, ...]:
return tuple(
name for name, value in sorted(vars(self).items()) if value is None
)
def _int_or_none(source: Mapping[str, Any] | None, key: str) -> int | None:
"""Absent, null, or non-numeric all read as unknown rather than zero."""
if not isinstance(source, Mapping):
return None
value = source.get(key)
if isinstance(value, bool) or not isinstance(value, int):
return None
return value
def _normalize_openai_responses(usage: Mapping[str, Any]) -> NormalizedUsage:
"""input_tokens is the WHOLE; cached and cache-write are subsets of it."""
total = _int_or_none(usage, "input_tokens")
details = usage.get("input_tokens_details")
cache_read = _int_or_none(details, "cached_tokens")
cache_write = _int_or_none(details, "cache_write_tokens")
output_details = usage.get("output_tokens_details")
ordinary: int | None = None
if total is not None and cache_read is not None and cache_write is not None:
ordinary = total - cache_read - cache_write
if ordinary < 0:
raise UsageSemanticsError(
f"OpenAI cached ({cache_read}) + cache_write ({cache_write}) "
f"exceed input_tokens ({total})"
)
return NormalizedUsage(
ordinary_input_tokens=ordinary,
cache_read_input_tokens=cache_read,
cache_write_input_tokens=cache_write,
total_input_tokens=total,
output_tokens=_int_or_none(usage, "output_tokens"),
# A decomposition of output_tokens, not an addition to it.
reasoning_output_tokens=_int_or_none(output_details, "reasoning_tokens"),
)
def _normalize_anthropic(usage: Mapping[str, Any]) -> NormalizedUsage:
"""input_tokens is the uncached REMAINDER; the cache fields add to it."""
ordinary = _int_or_none(usage, "input_tokens")
cache_read = _int_or_none(usage, "cache_read_input_tokens")
cache_write = _int_or_none(usage, "cache_creation_input_tokens")
total: int | None = None
if ordinary is not None and cache_read is not None and cache_write is not None:
total = ordinary + cache_read + cache_write
return NormalizedUsage(
ordinary_input_tokens=ordinary,
cache_read_input_tokens=cache_read,
cache_write_input_tokens=cache_write,
total_input_tokens=total,
output_tokens=_int_or_none(usage, "output_tokens"),
reasoning_output_tokens=None,
)
def canonical_provider(label: str | None, call_type: str | None) -> str | None:
"""Map LiteLLM's provider label onto an adapter key, or None if unsure.
LiteLLM reports ``custom_llm_provider`` as "openai" for both Chat
Completions and Responses, and those two report usage differently, so the
label alone cannot pick an adapter. The call type is what distinguishes
them. Returning None when it does not is deliberate: normalize_usage
refuses an unknown provider rather than guessing token semantics, which is
the whole point of keeping the native object authoritative.
"""
if label == "openai" and call_type and "responses" in call_type:
return OPENAI_RESPONSES
if label in _ADAPTERS:
return label
return None
_ADAPTERS = {
ANTHROPIC: _normalize_anthropic,
OPENAI_RESPONSES: _normalize_openai_responses,
}
def normalize_usage(provider: str, native_usage: Mapping[str, Any] | None) -> NormalizedUsage:
"""Derive the cross-provider view. Never mutates or replaces the native object."""
adapter = _ADAPTERS.get(provider)
if adapter is None:
raise UsageSemanticsError(
f"no usage adapter for provider {provider!r}; refusing to guess its token semantics"
)
if not isinstance(native_usage, Mapping):
return NormalizedUsage(None, None, None, None, None, None)
return adapter(native_usage)

View file

@ -12,6 +12,9 @@ from typing import Any, Mapping, Sequence
from .oracle_assets import TaskOracleSnapshot
REVIEW_OUTPUT = "review-output.json"
# Task verify/oracle commands read the artifact location from here rather
# than hardcoding a path, so one command works under bwrap and host-unsafe.
REVIEW_OUTPUT_ENV_VAR = "GITNEXUS_BENCH_REVIEW_OUTPUT"
REVIEW_SCHEMA_VERSION = 1
MAX_REVIEW_BYTES = 256 * 1024
MAX_FINDINGS = 100
@ -112,13 +115,33 @@ def _parse_review_finding(raw: Any, index: int) -> ReviewFinding:
def parse_review_output(path: Path) -> tuple[str, tuple[ReviewFinding, ...]]:
metadata = path.lstat()
# Distinguish these. Folding empty, malformed and encoding failures into one
# message is how a sandbox that left the artifact at 0 bytes read for 15
# runs as an encoding fault: json.loads("") raises, and every such cell
# reported "not valid UTF-8 JSON". A path the agent never created was not in
# that fold — the lstat below sat outside the try and raised
# FileNotFoundError — but it reached the caller as a bare OSError rather
# than saying what was wrong, which is why it is named here too.
try:
metadata = path.lstat()
except FileNotFoundError as exc:
raise ValueError("review output was never written") from exc
except OSError as exc:
raise ValueError(f"review output is unreadable: {exc.strerror}") from exc
if path.is_symlink() or not path.is_file() or metadata.st_size > MAX_REVIEW_BYTES:
raise ValueError("review output must be a bounded regular non-symlink file")
if metadata.st_size == 0:
raise ValueError("review output is empty")
try:
raw = json.loads(path.read_text())
except (OSError, UnicodeError, json.JSONDecodeError) as exc:
raise ValueError("review output is not valid UTF-8 JSON") from exc
text = path.read_text(encoding="utf-8")
except OSError as exc:
raise ValueError(f"review output is unreadable: {exc.strerror}") from exc
except UnicodeError as exc:
raise ValueError("review output is not valid UTF-8") from exc
try:
raw = json.loads(text)
except json.JSONDecodeError as exc:
raise ValueError(f"review output is not valid JSON: {exc.msg} at line {exc.lineno}") from exc
if not isinstance(raw, Mapping) or set(raw) != {"schema_version", "verdict", "findings"}:
raise ValueError("review output requires exactly schema_version, verdict, and findings")
if raw["schema_version"] != REVIEW_SCHEMA_VERSION:

View file

@ -15,6 +15,8 @@
# MODEL PROPOSER_MODEL EFFORT GENERATIONS RUNS WORKERS PROVIDER
# EVOLUTION_PROFILE CE_PLUGIN_DIR CE_PLUGIN_VERSION
# INCLUDE_EXPENSIVE SEED_RESULTS CLAUDE_BIN OUT_ROOT
# CI (passes --max-runtime-from-instance-window; the CLI reads /proc/uptime)
# EVENTBRIDGE_INSTANCE_WINDOW_SECONDS EVENTBRIDGE_STOP_RESERVE_SECONDS
# UNSAFE_NO_BWRAP=1 (local review diagnostics only)
# GITNEXUS_BENCH_ANTHROPIC_API_KEY (legacy GITNEXUS_BENCH_AUTH_TOKEN)
# GITNEXUS_BENCH_OPENAI_API_KEY
@ -193,6 +195,18 @@ if ((${#passthrough[@]})); then
cmd+=("${passthrough[@]}")
fi
# A cancelled GitHub job skips even `if: always()`, so evidence dies with the
# runner. The evolution box is EventBridge-stopped 24h after boot; a Friday
# dispatch inherits leftover uptime. Cap the sweep so it fails in-process and
# the upload step still runs (run 33962002890). The CLI reads /proc/uptime
# itself, in the same breath as it starts the clock the cap is measured
# against; computing a number here — in a separate interpreter, before the
# provenance work and the exec below — charged the sweep for every second
# this script spent afterwards.
if [[ -n "${CI:-}" && -r /proc/uptime ]]; then
cmd+=(--max-runtime-from-instance-window)
fi
if ((dry_run)); then
printf '%q ' "${cmd[@]}"
printf '\n'
@ -220,5 +234,9 @@ SOURCE_SHA="${source_sha}" RUNTIME_DIGEST="${runtime_digest}" SANDBOX_BACKEND="$
}, null, 2) + "\n")' "${out_root}/runtime-provenance.json"
export PYTHONUNBUFFERED=1
# The runner stamps this on every results.jsonl row and refuses to reuse a
# comparator cell when a prior row's digest disagrees. Keep it on the evolve
# process, not only in the provenance JSON sidecar.
export RUNTIME_DIGEST="${runtime_digest}"
cd "${eval_dir}"
exec "${cmd[@]}"

File diff suppressed because it is too large Load diff

View file

@ -217,11 +217,25 @@ def enforce_phase_workspace(
worktree: Path,
before: dict[str, str],
*,
allowed_artifact: Path,
allowed_artifact: Path | None,
) -> None:
"""Require a phase to change only its one explicit workspace artifact."""
"""Require a phase to change only its one explicit workspace artifact.
``allowed_artifact=None`` is the stricter contract: the phase must leave
the workspace byte-identical. That is what a review phase whose artifact
lives outside the workspace has to satisfy — there is nothing in there it
is entitled to touch.
"""
root = worktree.expanduser().absolute()
if allowed_artifact is None:
after = workspace_snapshot(root)
changed = sorted(
path for path in before.keys() | after.keys() if before.get(path) != after.get(path)
)
if changed:
raise ValueError(f"phase changed the read-only workspace: {', '.join(changed[:5])}")
return
artifact = allowed_artifact.expanduser().absolute()
try:
relative = PurePosixPath(artifact.relative_to(root).as_posix())
@ -325,6 +339,65 @@ def new_plan_doc(worktree: Path, before: dict[Path, str]) -> Path:
return changed[0]
def _assert_self_contained_git_objects(clone: Path) -> None:
"""Refuse clones that share pack/object bytes with another repository."""
alternates = clone / ".git" / "objects" / "info" / "alternates"
if alternates.exists():
raise RuntimeError(f"clone unexpectedly has an external object alternate: {alternates}")
objects = clone / ".git" / "objects"
if not objects.is_dir():
raise RuntimeError(f"clone is missing a git object store: {clone}")
for obj in objects.rglob("*"):
if obj.is_file() and obj.stat().st_nlink > 1:
raise RuntimeError(f"clone object is hardlinked to host storage: {obj}")
def copy_isolated_tree(source: Path, parent: Path) -> Path:
"""Copy a sanitized clone without sharing git objects or a ref namespace.
``git clone --no-local`` of GitNexus plus ``sanitize_clone_for_hidden_oracles``
(repack/prune/fsck) is minutes per cell. After sanitization the snapshot is
one parentless commit; copying that tree is the isolation boundary the
contamination bug actually required (a private ``.git``), not a second
fetch of full history. Prefer ``cp --reflink=auto`` so XFS/btrfs pay COW;
fall back to a full copy on filesystems that cannot reflink.
"""
try:
source_meta = source.expanduser().lstat()
except OSError as exc:
raise RuntimeError(f"clone template is unavailable: {source}: {exc}") from exc
if stat.S_ISLNK(source_meta.st_mode) or not stat.S_ISDIR(source_meta.st_mode):
raise RuntimeError(f"clone template must be a real directory: {source}")
source = source.expanduser().resolve()
target = Path(tempfile.mkdtemp(prefix="wfbench-", dir=parent))
target.rmdir()
try:
copied = run_managed(
["cp", "-a", "--reflink=auto", str(source), str(target)],
timeout=600,
)
if not copied.ok:
# The fallback is for a filesystem that cannot reflink, which shows
# up as a normal nonzero exit. A cancellation or timeout is reported
# the same way (run_managed returns it rather than raising), and
# copytree cannot be cancelled — so falling back there makes the
# outage breaker wait out the full copy it set the event to avoid.
if copied.state != "exited":
raise ManagedProcessError(["cp", "-a", "--reflink=auto", str(source), str(target)], copied)
shutil.copytree(source, target, symlinks=True, copy_function=shutil.copy2)
_assert_self_contained_git_objects(target)
return target
except BaseException as primary:
if target.exists():
try:
shutil.rmtree(target)
except OSError as cleanup:
primary.add_note(f"clone copy cleanup also failed: {type(cleanup).__name__}: {cleanup}")
raise
def make_worktree(repo: Path, ref: str, parent: Path) -> Path:
"""Create a self-contained clone per benchmark arm."""
@ -344,12 +417,7 @@ def make_worktree(repo: Path, ref: str, parent: Path) -> Path:
],
timeout=600,
)
alternates = target / ".git" / "objects" / "info" / "alternates"
if alternates.exists():
raise RuntimeError(f"clone unexpectedly has an external object alternate: {alternates}")
for obj in (target / ".git" / "objects").rglob("*"):
if obj.is_file() and obj.stat().st_nlink > 1:
raise RuntimeError(f"clone object is hardlinked to host storage: {obj}")
_assert_self_contained_git_objects(target)
for candidate in (ref, f"origin/{ref}"):
proc = run_managed(
["git", "-C", str(target), "checkout", "--detach", "--quiet", candidate],

View file

@ -57,6 +57,24 @@ MAX_PROGRESS_PENDING = 256
MAX_PROGRESS_TOOL_ID_CHARS = 256
MAX_TOOL_PREVIEW_CHARS = 800
_SAFE_TOOL_NAME = re.compile(r"[A-Za-z0-9._:-]{1,64}")
_GHA_WORKFLOW_COMMAND = re.compile(r"(^|[\n\r])::")
_GHA_HASH_COMMAND = re.compile(r"##\[")
_GHA_COMPILER_ANNOTATION = re.compile(r"\((\d+),(\d+)\):\s+error\b", re.IGNORECASE)
def neutralize_ci_log_text(text: str) -> str:
"""Stop GitHub Actions from promoting tool output into check annotations.
Run 33962002890 logged in-sandbox ``tsc`` failures as
``file.ts(line,col): error TS2307``, which Actions parsed as workflow
annotations on ``.github``. The same parser treats ``::error::`` and
``##[error]`` as commands. Progress previews are evidence, not CI
signaling, so rewrite those forms before they hit the job log.
"""
text = _GHA_WORKFLOW_COMMAND.sub(r"\1[:]", text)
text = _GHA_HASH_COMMAND.sub("# [", text)
return _GHA_COMPILER_ANNOTATION.sub(r"(\1,\2): compiler-error", text)
def _safe_tool_name(value: Any) -> str:
@ -177,7 +195,9 @@ class SessionProgress:
def _say(self, message: str) -> None:
# Queue only: the stdout drain thread calls observe() and must not
# block on a full log pipe (process_control.stdout_observer contract).
self._pending_messages.append(f"[{self.label} {self._elapsed()}] {message}")
self._pending_messages.append(
neutralize_ci_log_text(f"[{self.label} {self._elapsed()}] {message}")
)
self._last_spoke = time.monotonic()
def _emit_pending(self) -> None:

View file

@ -24,7 +24,7 @@ from .proposer_sandbox import (
build_sandbox_environment,
prepare_sandbox,
)
from .runner_artifacts import make_worktree, remove_clone
from .runner_artifacts import copy_isolated_tree, make_worktree, remove_clone
from .task_assets import TaskAssetCache, TaskAssetSnapshot, _is_harness_sandbox_copy
GRAPH_ASSET_PATHS = (
@ -360,14 +360,29 @@ def prepare_sanitized_graph(
bwrap_bin: Path | str,
runtime_mounts: Sequence[ReadOnlyMount],
sandbox_backend: str = "bwrap",
clone_template: Path | None = None,
sanitized_head: str | None = None,
) -> SanitizedGraphSnapshot:
"""Sanitize, index offline once, scrub, and freeze graph assets for all arms."""
"""Sanitize, index offline once, scrub, and freeze graph assets for all arms.
When ``clone_template`` is an already-sanitized snapshot, this copies it
(the copy is scrubbed and indexed) so the template stays a clean cell
seed. Callers that already paid for ``make_worktree`` + sanitization
should pass that template rather than cloning GitNexus again.
"""
validate_no_prebuilt_graph_assets(task)
seed = make_worktree(repo, resolved_sha, parent)
if clone_template is not None:
if not isinstance(sanitized_head, str) or not sanitized_head:
raise SandboxError("clone template requires the sanitized HEAD")
seed = copy_isolated_tree(clone_template, parent)
else:
seed = make_worktree(repo, resolved_sha, parent)
sanitized_head = None
primary: BaseException | None = None
try:
sanitized_head = sanitize_clone_for_hidden_oracles(seed)
if sanitized_head is None:
sanitized_head = sanitize_clone_for_hidden_oracles(seed)
_scrub_source_references(seed)
_neutralize_target_index_inputs(seed)
with prepare_sandbox(

View file

@ -0,0 +1,33 @@
{
"_provenance": "Actions run 33912693948 (2026-09-04), review profile, gen-0, workers=1. Artifact gitnexus-evolution-33912693948-1: gen-0/bench/results.jsonl and gen-0/proposer-session.json. Step wall from the Actions API.",
"_caveat": "Every cell in that run returned unusable evidence (32 review-evidence-invalid, 6 session-error, 3 skill-not-invoked); two hit the 5400s ceiling and it cost 653. Durations are real, but a run that resolves cleanly may sit lower. It is the only live artifact - the 2026-07-22 green run's has expired.",
"_order": "Submission order, deliberately unsorted: the model cycles these, so sorting would hand each task a uniform block and hide the variance being measured.",
"_duration_scope": "duration_s is the sum of the cell's Claude session durations (runner_sessions.py). It excludes the clone, graph materialize, asset staging, sandbox setup and teardown - those live in the residual below.",
"session_ceiling_s": 5400,
"proposer_duration_s": 344.7,
"cell_duration_s_by_arm": {
"candidate_review": [
2485.6, 1338.0, 3075.2, 702.5, 1240.4, 5400.0, 762.1, 342.9, 826.3, 489.7, 675.4, 337.8, 734.3
],
"ce_review": [
3744.6, 2140.4, 1418.4, 436.1, 653.3, 1022.3, 851.2, 1191.1, 902.1, 847.2, 502.6, 991.3,
1222.7, 544.0
],
"review": [
5400.0, 2976.4, 1162.8, 627.1, 901.3, 436.9, 704.3, 963.4, 631.8, 627.9, 663.5, 662.4, 361.3,
741.1
]
},
"residual": {
"benchmark_step_wall_s": 54623,
"session_seconds": 51737.7,
"proposer_seconds": 344.7,
"unaccounted_s": 2540.6,
"cells": 41,
"unique_shas": 5,
"_note": "Everything the sweep spent outside the agent sessions: per-SHA sanitize and `analyze --pdg --index-only`, plus each cell's clone, materialize, staging, sandbox and teardown. That run predates clone templates and graph prefetch, so this is an upper bound for the current code. The split between per-SHA and per-cell is not recoverable from the artifact, so the model charges it per cell and serially, outside the pool - the pessimistic reading of an already-small term.",
"_split_assumption": "The residual mixes per-SHA graph setup with per-cell clone/sandbox/teardown and the artifact cannot separate them. The model charges it per SHA, not per cell, because only that direction refuses to credit a run for shrinking work it still performs: a weekly generation pays one arm instead of three but builds the same graphs. This overstates cold slightly and refuses to understate weekly. Replace with measured per-SHA and per-cell times when a run records them separately.",
"sha_overhead_s": 508.1
},
"_breaker": "Replaying this sample's error_kind sequence through today's systemic_outage_streak trips the outage breaker at cell 5 of 41 (DEFAULT_OUTAGE_STREAK=5). The source run executed all 41, so its runner did not break on this sequence. The durations stay valid as per-cell timings; what they cannot describe is a 54-cell sweep with this failure profile, because the current code would never run one."
}

View file

@ -0,0 +1,747 @@
#!/usr/bin/env python3
"""Run the real sweep scheduler against stub sessions and time it.
``measure_evolution_cost`` is arithmetic: it predicts wall clock from a model of
what ``sweep_task_cells`` does. This runs the actual function - real threads,
the real wave barrier, the real outage breaker - and replaces only the paid
agent session with a sleep. If the two disagree, the model is wrong.
Durations are the measured per-arm samples from ``session_durations.json``
divided by ``--scale``, so a cell that really took 1416s takes ~0.28s here. The
shape is preserved deliberately: the median cell is 826s against a 5400s
ceiling, and that spread is the whole reason a barrier costs anything. Uniform
random sleeps would erase the effect under test.
Schedulers, all consuming one identical seeded plan:
``wave`` the shipped ``sweep_task_cells`` - fixed waves of ``workers``, a
barrier between them, one task at a time.
``fed`` a continuously fed pool per task (H1). Naive: no breaker, no graph
gating. Present to price the barrier alone.
``packed`` one pool across every task (H2). Naive, same caveat.
``faithful``H2 carrying the invariants the shipped scheduler actually holds:
a global submission order, in-order folding, the outage breaker, and
per-task graph readiness gating. This is the one to believe.
python3 -m workflow_bench.simulate_sweep --compare --repeat 5
python3 -m workflow_bench.simulate_sweep --breaker-fidelity
"""
from __future__ import annotations
import argparse
import json
import math
import random
import statistics
import subprocess
import sys
import threading
import time
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass, field
from typing import Any
from . import runner
from .measure_evolution_cost import (
CANDIDATE_ARM,
DURATIONS_BY_ARM,
REVIEW_ARMS,
REVIEW_TASKS,
SHA_OVERHEAD_SECONDS,
_read,
expected_task_seconds,
review_tasks,
)
DEFAULT_SCALE = 5000.0
# --contention-sweep measures both of these regardless of --workers, so the
# window has to be valid for the LARGEST of them, not for the parsed value.
CONTENTION_WORKERS = (3, 6)
SYSTEMIC_KIND = "session-error"
# A cell is mostly a model session waiting on the network, but its tool calls -
# git, vitest, analyze - burn real CPU in real subprocesses. Sleeping threads
# model the wait and nothing else, so every speedup measured that way is an
# upper bound. This burns WORK, not wall clock: a fixed number of sha256 rounds
# in a subprocess, which takes longer when cores are contended. That is the
# effect under test, and it has to be a subprocess - Python threads burning
# Python would measure the GIL rather than the machine.
_BURN_SRC = (
"import hashlib,sys\n"
"n=int(sys.argv[1]); b=b'x'*4096; h=hashlib.sha256()\n"
"for _ in range(n): h.update(b)\n"
"sys.stdout.write(h.hexdigest()[:8])\n"
)
def calibrate_burn(probe_rounds: int = 400_000) -> float:
"""sha256 rounds per second, one uncontended subprocess. Measured, not assumed."""
started = time.monotonic()
subprocess.run(
[sys.executable, "-c", _BURN_SRC, str(probe_rounds)],
check=True,
capture_output=True,
)
return probe_rounds / (time.monotonic() - started)
def _execute_cell(cell: Cell, cpu_fraction: float, burn_rate: float) -> None:
"""The stub session: wait for the API, then do the tool-call work."""
if cpu_fraction <= 0:
time.sleep(cell.seconds)
return
time.sleep(cell.seconds * (1.0 - cpu_fraction))
rounds = int(cell.seconds * cpu_fraction * burn_rate)
if rounds > 0:
subprocess.run(
[sys.executable, "-c", _BURN_SRC, str(rounds)], check=True, capture_output=True
)
@dataclass(frozen=True)
class Cell:
task: int
run: int
arm: str
seconds: float
systemic: bool = False
@dataclass
class Outcome:
wall_s: float
executed: int
tripped_at: int | None = None
folded: list[int] = field(default_factory=list)
def build_plan(
*,
task_count: int,
runs: int,
arms: tuple[str, ...],
scale: float,
seed: int,
fail_from: int | None = None,
) -> list[list[Cell]]:
"""Per-task cells in submission order, with durations drawn once.
Shared by every scheduler so a comparison cannot be an artifact of one of
them drawing luckier cells. ``fail_from`` marks every cell at or after that
global index systemic, which is what the breaker-fidelity mode needs.
"""
rng = random.Random(seed)
plan: list[list[Cell]] = []
index = 0
for task in range(task_count):
cells: list[Cell] = []
for run_idx in range(runs):
for arm in arms:
sample = DURATIONS_BY_ARM[arm]
cells.append(
Cell(
task=task,
run=run_idx,
arm=arm,
seconds=sample[rng.randrange(len(sample))] / scale,
systemic=fail_from is not None and index >= fail_from,
)
)
index += 1
plan.append(cells)
return plan
def _flatten(plan: list[list[Cell]]) -> list[Cell]:
return [cell for cells in plan for cell in cells]
def _record(cell: Cell) -> dict[str, Any]:
kind = SYSTEMIC_KIND if cell.systemic else None
return {
"run": cell.run,
"arm": cell.arm,
"ok": not cell.systemic,
"resolved": not cell.systemic,
"error_kind": kind,
"review_evidence_valid": not cell.systemic,
}
def _graph_builder(
ready: list[threading.Event], graph_seconds: float, stop: threading.Event
) -> threading.Thread:
"""One graph at a time, in task order - they are CPU and IO heavy."""
def build() -> None:
for event in ready:
if stop.is_set():
return
time.sleep(graph_seconds)
event.set()
thread = threading.Thread(target=build, name="graph-builder", daemon=True)
thread.start()
return thread
def run_wave(
plan: list[list[Cell]],
workers: int,
*,
outage_limit: int,
graph_seconds: float,
cpu_fraction: float = 0.0,
burn_rate: float = 0.0,
) -> Outcome:
"""The shipped scheduler, driven for real, task after task."""
ready = [threading.Event() for _ in plan]
stop = threading.Event()
_graph_builder(ready, graph_seconds, stop)
executed = 0
lock = threading.Lock()
streak = 0
tripped_at: int | None = None
folded: list[int] = []
base = 0
started = time.monotonic()
for task, cells in enumerate(plan):
ready[task].wait()
by_key = {(c.run, c.arm): c for c in cells}
def fake_run(run_idx: int, arm: str) -> dict[str, Any]:
nonlocal executed
cell = by_key[(run_idx, arm)]
_execute_cell(cell, cpu_fraction, burn_rate)
with lock:
executed += 1
return _record(cell)
order = {(c.run, c.arm): base + i for i, c in enumerate(cells)}
def on_record(run_idx: int, arm: str, rec: dict[str, Any]) -> None:
# Mirror the breaker's own evaluation so the reported trip point is
# the cell that crossed the limit, not merely the last one folded -
# sweep_task_cells folds a whole wave before it evaluates.
nonlocal streak, tripped_at
index = order[(run_idx, arm)]
folded.append(index)
streak = runner.systemic_outage_streak(rec["error_kind"], streak)
if outage_limit and streak >= outage_limit and tripped_at is None:
tripped_at = index
streak, tripped = runner.sweep_task_cells(
[(c.run, c.arm) for c in cells],
workers=workers,
run=fake_run,
on_start=lambda *_: None,
on_record=on_record,
outage_streak=streak,
outage_limit=outage_limit,
)
base += len(cells)
if tripped:
break
stop.set()
return Outcome(wall_s=time.monotonic() - started, executed=executed, tripped_at=tripped_at, folded=folded)
def _drain_naive(cells: list[Cell], workers: int) -> int:
with ThreadPoolExecutor(max_workers=workers) as pool:
list(pool.map(lambda c: time.sleep(c.seconds), cells))
return len(cells)
def run_fed(plan: list[list[Cell]], workers: int, *, outage_limit: int, graph_seconds: float) -> Outcome:
"""H1 without invariants: fed pool per task. Prices the barrier alone.
Graph building is deliberately identical to ``run_wave`` - the same
background builder, started before the clock - because that is what makes
the claim in the first line true. Sleeping ``graph_seconds`` serially before
each task instead, as this did, charged fed for overlap that wave gets for
free: the wave builder prepares task N+1 while task N's cells run. The
fed-versus-wave delta then mixed the loss of that overlap into what was
reported as the price of the barrier.
"""
ready = [threading.Event() for _ in plan]
stop = threading.Event()
_graph_builder(ready, graph_seconds, stop)
executed = 0
started = time.monotonic()
for task, cells in enumerate(plan):
ready[task].wait()
executed += _drain_naive(cells, workers)
return Outcome(wall_s=time.monotonic() - started, executed=executed)
def run_packed(plan: list[list[Cell]], workers: int, *, outage_limit: int, graph_seconds: float) -> Outcome:
"""H2 without invariants. Upper bound, not a design."""
started = time.monotonic()
time.sleep(graph_seconds)
executed = _drain_naive(_flatten(plan), workers)
return Outcome(wall_s=time.monotonic() - started, executed=executed)
def run_faithful(
plan: list[list[Cell]],
workers: int,
*,
outage_limit: int,
graph_seconds: float,
window: int | None = None,
cpu_fraction: float = 0.0,
burn_rate: float = 0.0,
) -> Outcome:
"""H2 carrying the invariants the shipped scheduler holds.
Global submission order is task-major, run-major, arm-minor - the same total
order the wave scheduler folds in, just continued across task boundaries. A
folder walks results in exactly that order, so "consecutive systemic
failures" keeps its meaning; the breaker trips on the same logical cell it
would have in waves. Cells already in flight when it trips are the overrun,
bounded by ``workers - 1`` exactly as the wave docstring promises.
A task's cells are not submitted until its graph is ready, which is what
makes this a schedule rather than a wish: the graph builder is serial, so
packing cannot outrun it.
``window`` is the design question. Queue every cell at once and workers race
far ahead of the fold pointer, so a breaker trip has already paid for cells
nobody has looked at - measured at 5 against a bound of 2. Holding
submission to ``window`` cells beyond the fold point caps the overrun at
``window - 1``, which is the wave's own ``workers - 1`` bound when the two
are equal, while still packing across task boundaries. Defaults to whatever
``runner.sweep_packed_cells`` defaults to, so a run that names no window
compares the shipped policy rather than a more tightly queued prototype.
"""
if window is None:
window = max(workers * runner.PACKED_WINDOW_MULTIPLIER, workers)
if window < workers:
# Same rule sweep_packed_cells enforces. Without it a window below 1
# never lets the producer past its own gate and the run hangs.
raise ValueError("window must be at least workers, or the pool starves")
cells = _flatten(plan)
ready = [threading.Event() for _ in plan]
stop = threading.Event()
_graph_builder(ready, graph_seconds, stop)
results: list[dict[str, Any] | None] = [None] * len(cells)
executed = 0
lock = threading.Lock()
halt = threading.Event()
def work(index: int) -> None:
nonlocal executed
if halt.is_set():
return
cell = cells[index]
_execute_cell(cell, cpu_fraction, burn_rate)
with lock:
executed += 1
results[index] = _record(cell)
gate = threading.Condition()
fold_pointer = 0
futures: list[Any] = []
producer_done = threading.Event()
started = time.monotonic()
pool = ThreadPoolExecutor(max_workers=workers)
def produce() -> None:
submitted = 0
for task, task_cells in enumerate(plan):
ready[task].wait()
for _ in task_cells:
with gate:
while submitted - fold_pointer >= window and not halt.is_set():
gate.wait(timeout=0.5)
if halt.is_set():
producer_done.set()
return
futures.append(pool.submit(work, submitted))
submitted += 1
gate.notify_all()
producer_done.set()
producer = threading.Thread(target=produce, name="cell-producer", daemon=True)
producer.start()
streak = 0
tripped_at: int | None = None
folded: list[int] = []
try:
index = 0
while True:
with gate:
while index >= len(futures) and not producer_done.is_set():
gate.wait(timeout=0.5)
if index >= len(futures):
break
future = futures[index]
future.result()
record = results[index]
if record is not None:
folded.append(index)
streak = runner.systemic_outage_streak(record["error_kind"], streak)
if outage_limit and streak >= outage_limit:
tripped_at = index
halt.set()
with gate:
gate.notify_all()
for pending in futures[index + 1 :]:
pending.cancel()
break
index += 1
with gate:
fold_pointer = index
gate.notify_all()
finally:
halt.set()
with gate:
gate.notify_all()
stop.set()
producer.join(timeout=5)
pool.shutdown(wait=True)
return Outcome(
wall_s=time.monotonic() - started, executed=executed, tripped_at=tripped_at, folded=folded
)
def run_production_packed(
plan: list[list[Cell]],
workers: int,
*,
outage_limit: int,
graph_seconds: float,
cpu_fraction: float = 0.0,
burn_rate: float = 0.0,
window: int | None = None,
) -> Outcome:
"""Drive the REAL runner.sweep_packed_cells, not a prototype of it.
Same relationship run_wave has to sweep_task_cells: only the paid session is
stubbed. If this disagrees with the faithful prototype, the shipped function
is what is wrong.
"""
cells = _flatten(plan)
by_key = {(f"t{c.task}", c.run, c.arm): c for c in cells}
order = {(f"t{c.task}", c.run, c.arm): i for i, c in enumerate(cells)}
ready = [threading.Event() for _ in plan]
stop = threading.Event()
_graph_builder(ready, graph_seconds, stop)
executed = 0
lock = threading.Lock()
folded: list[int] = []
tripped_at: int | None = None
streak_seen = {"streak": 0}
def run_cell(task_id: str, run_idx: int, arm: str) -> dict[str, Any]:
nonlocal executed
cell = by_key[(task_id, run_idx, arm)]
_execute_cell(cell, cpu_fraction, burn_rate)
with lock:
executed += 1
return _record(cell)
def on_record(task_id: str, run_idx: int, arm: str, rec: dict[str, Any]) -> None:
nonlocal tripped_at
index = order[(task_id, run_idx, arm)]
folded.append(index)
streak_seen["streak"] = runner.systemic_outage_streak(rec["error_kind"], streak_seen["streak"])
if outage_limit and streak_seen["streak"] >= outage_limit and tripped_at is None:
tripped_at = index
def await_ready(task_id: str) -> bool:
ready[int(task_id[1:])].wait()
return True
started = time.monotonic()
runner.sweep_packed_cells(
[(f"t{c.task}", c.run, c.arm) for c in cells],
workers=workers,
run=run_cell,
on_start=lambda *_: None,
on_record=on_record,
outage_streak=0,
outage_limit=outage_limit,
window=window,
await_ready=await_ready,
)
wall = time.monotonic() - started
stop.set()
return Outcome(wall_s=wall, executed=executed, tripped_at=tripped_at, folded=folded)
SCHEDULERS = {
"wave": run_wave,
"fed": run_fed,
"packed": run_packed,
"faithful": run_faithful,
"production": run_production_packed,
}
def _window_kwargs(name: str, window: int | None) -> dict[str, int]:
"""``--window`` only means anything to the two schedulers that hold one."""
return {"window": window} if window is not None and name in ("faithful", "production") else {}
def _plan_args(args: argparse.Namespace, weekly: bool, seed: int, fail_from: int | None = None):
arms = (CANDIDATE_ARM,) if weekly else REVIEW_ARMS
return {
"task_count": len(review_tasks(_read(REVIEW_TASKS))),
"runs": args.runs,
"arms": arms,
"scale": args.scale,
"seed": seed,
"fail_from": fail_from,
}, arms
def breaker_fidelity(args: argparse.Namespace) -> list[dict[str, Any]]:
"""Does packing still trip where waves trip, and overrun no further?"""
rows: list[dict[str, Any]] = []
limit = runner.DEFAULT_OUTAGE_STREAK
window = args.window if args.window is not None else max(
args.workers * runner.PACKED_WINDOW_MULTIPLIER, args.workers
)
for fail_from in (0, 4, 12):
kwargs, _arms = _plan_args(args, weekly=False, seed=args.seed, fail_from=fail_from)
plan = build_plan(**kwargs)
total = sum(len(c) for c in plan)
row: dict[str, Any] = {
"fail_from": fail_from, "limit": limit, "total_cells": total, "window": window
}
for name in ("wave", "faithful", "production"):
out = SCHEDULERS[name](
plan, args.workers, outage_limit=limit, graph_seconds=args.graph_seconds,
**_window_kwargs(name, window),
)
row[name] = {
"tripped_at": out.tripped_at,
"executed": out.executed,
"overrun": out.executed - (out.tripped_at + 1) if out.tripped_at is not None else None,
}
row["same_trip_point"] = (
row["wave"]["tripped_at"] == row["faithful"]["tripped_at"] == row["production"]["tripped_at"]
)
# The producer holds submission to ``window`` cells beyond the fold
# pointer, so at most ``window - 1`` cells past the tripping one can
# already be in flight. At ``window == workers`` that is exactly the
# wave scheduler's own ``workers - 1`` bound.
row["overrun_within_bound"] = (
row["production"]["overrun"] is not None
and row["production"]["overrun"] <= window - 1
)
rows.append(row)
return rows
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--workers", type=int, default=3)
parser.add_argument("--scale", type=float, default=DEFAULT_SCALE)
parser.add_argument("--seed", type=int, default=1729)
parser.add_argument("--repeat", type=int, default=1)
parser.add_argument("--runs", type=int, default=3)
parser.add_argument("--scheduler", choices=sorted(SCHEDULERS), default="wave")
parser.add_argument("--compare", action="store_true")
parser.add_argument("--breaker-fidelity", action="store_true")
parser.add_argument("--window-sweep", action="store_true", help="wall clock vs breaker overrun")
parser.add_argument("--contention-sweep", action="store_true", help="does the gain survive real CPU?")
parser.add_argument(
"--window",
type=int,
default=None,
help="submission window for the packed schedulers; defaults to the shipped policy",
)
parser.add_argument(
"--graph-seconds",
type=float,
default=None,
help="per-task graph build; defaults to the measured per-SHA overhead, scaled",
)
args = parser.parse_args()
# Both are checked here rather than where they are used: a bad --scale
# divides by zero before anything runs, and a negative --graph-seconds
# kills the graph-builder thread, after which every scheduler waits on a
# readiness event nobody will ever set.
# NaN defeats every comparison it appears in, so "> 0" and ">= 0" both admit
# it and the failure surfaces far from the flag: NaN durations reach
# time.sleep in a worker or the graph thread and raise there, after which the
# schedulers wait forever on a readiness event nobody will set. Infinity is
# worse than a crash - it silently scales every duration to zero and the run
# reports a sweep that took no time.
if not math.isfinite(args.scale) or args.scale <= 0:
parser.error("--scale must be a finite positive number")
if args.graph_seconds is not None and (not math.isfinite(args.graph_seconds) or args.graph_seconds < 0):
parser.error("--graph-seconds must be a finite non-negative number")
# Counts are indexed or handed to a thread pool without further checking, so
# a zero turns into an IndexError on plans[0], a median over an empty
# sequence, or ThreadPoolExecutor's own error - none of which name the flag
# that caused them.
if args.workers < 1:
parser.error("--workers must be at least 1")
if args.repeat < 1:
parser.error("--repeat must be at least 1")
if args.runs < 1:
parser.error("--runs must be at least 1")
# run_faithful and sweep_packed_cells both refuse a window below the worker
# count - a smaller one starves the pool, because the producer waits for a
# fold pointer to pass a cell it was never allowed to submit. Enforcing it
# here turns an uncaught ValueError partway through a measurement into an
# argument error before anything runs. Checked against the largest worker
# count this invocation will actually use: --contention-sweep runs its own
# counts irrespective of --workers, so validating against --workers alone
# let the 3-worker measurements finish and then raised on the 6-worker one.
window_workers = args.workers
if args.contention_sweep:
window_workers = max(window_workers, max(CONTENTION_WORKERS))
if args.window is not None and args.window < window_workers:
parser.error(f"--window must be at least the worker count ({window_workers}); a smaller window starves the pool")
if args.graph_seconds is None:
args.graph_seconds = SHA_OVERHEAD_SECONDS / args.scale
if args.contention_sweep:
burn_rate = statistics.median(calibrate_burn() for _ in range(3))
rows = []
for cpu_fraction in (0.0, 0.25, 0.5):
for workers in CONTENTION_WORKERS:
plans = [
build_plan(**_plan_args(args, False, args.seed + i)[0])
for i in range(args.repeat)
]
measured = {}
for name in ("wave", "faithful", "production"):
fn = SCHEDULERS[name]
measured[name] = statistics.median(
fn(
plan,
workers,
outage_limit=0,
graph_seconds=args.graph_seconds,
cpu_fraction=cpu_fraction,
burn_rate=burn_rate,
**_window_kwargs(name, args.window),
).wall_s
for plan in plans
)
serial = statistics.median(
sum(c.seconds for c in _flatten(plan)) for plan in plans
)
rows.append(
{
"cpu_fraction": cpu_fraction,
"workers": workers,
"wave_s": round(measured["wave"], 2),
"faithful_s": round(measured["faithful"], 2),
"production_s": round(measured["production"], 2),
"packing_gain_pct": round(
(measured["faithful"] - measured["wave"]) / measured["wave"] * 100, 1
),
"wave_speedup": round(serial / measured["wave"], 2),
"faithful_speedup": round(serial / measured["faithful"], 2),
}
)
print(json.dumps({"burn_rate": round(burn_rate), "nproc": __import__("os").cpu_count(), "rows": rows}, indent=2))
return 0
if args.window_sweep:
total = len(review_tasks(_read(REVIEW_TASKS))) * args.runs * len(REVIEW_ARMS)
rows = []
for window in (args.workers, args.workers * 2, args.workers * 4, total):
clean = [build_plan(**_plan_args(args, False, args.seed + i)[0]) for i in range(args.repeat)]
wall = statistics.median(
run_faithful(
p, args.workers, outage_limit=0, graph_seconds=args.graph_seconds, window=window
).wall_s
for p in clean
)
failing = build_plan(**_plan_args(args, weekly=False, seed=args.seed, fail_from=12)[0])
trip = run_faithful(
failing,
args.workers,
outage_limit=runner.DEFAULT_OUTAGE_STREAK,
graph_seconds=args.graph_seconds,
window=window,
)
rows.append(
{
"window": window,
"cold_wall_s": round(wall, 3),
"tripped_at": trip.tripped_at,
"executed": trip.executed,
"overrun_cells": trip.executed - (trip.tripped_at + 1)
if trip.tripped_at is not None
else None,
}
)
print(json.dumps({"workers": args.workers, "rows": rows}, indent=2))
return 0
if args.breaker_fidelity:
print(
json.dumps(
{"workers": args.workers, "graph_seconds": round(args.graph_seconds, 4),
"rows": breaker_fidelity(args)},
indent=2,
)
)
return 0
names = sorted(SCHEDULERS) if args.compare else [args.scheduler]
rows: list[dict[str, Any]] = []
for label, weekly in (("weekly", True), ("cold", False)):
plans = []
for i in range(args.repeat):
kwargs, arms = _plan_args(args, weekly, args.seed + i)
plans.append(build_plan(**kwargs))
serial = statistics.median(sum(c.seconds for c in _flatten(p)) for p in plans)
predicted = (
len(plans[0])
* expected_task_seconds(args.runs, arms, args.workers, fed_pool=False)
/ args.scale
)
for name in names:
observed = statistics.median(
SCHEDULERS[name](
p,
args.workers,
outage_limit=0,
graph_seconds=args.graph_seconds,
**_window_kwargs(name, args.window),
).wall_s
for p in plans
)
rows.append(
{
"profile": label,
"scheduler": name,
"workers": args.workers,
"observed_s": round(observed, 3),
"wave_model_s": round(predicted, 3),
"serial_s": round(serial, 3),
"speedup_vs_serial": round(serial / observed, 3) if observed else None,
}
)
print(json.dumps({"scale": args.scale, "repeat": args.repeat, "rows": rows}, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View file

@ -9,14 +9,17 @@ tasks:
sandbox_copy: [eval/workflow_bench/review_cases/pr-2718.patch]
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2718.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2718. Report only actionable defects introduced by the local diff.
verify: test -s review-output.json
verify: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
oracle:
command: test -s review-output.json
command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
files: [{ source: review-pr-2718-defect.labels.json, target: review-labels.json }]
sandbox_dependencies: &deps
- { source: node_modules, target: node_modules }
- { source: gitnexus/node_modules, target: gitnexus/node_modules }
- { source: gitnexus-shared/node_modules, target: gitnexus-shared/node_modules }
# Host-built types/JS. Historical clones have no dist/, so `tsc` in the
# read-only workspace otherwise reports TS2307/TS6379 (run 33962002890).
- { source: gitnexus-shared/dist, target: gitnexus-shared/dist }
- <<: *review_case
id: review-pr-2794-defect
@ -25,7 +28,7 @@ tasks:
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2794.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2794. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
files: [{ source: review-pr-2794-defect.labels.json, target: review-labels.json }]
sandbox_dependencies: *deps
@ -36,7 +39,7 @@ tasks:
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2108.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2108. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
files: [{ source: review-pr-2108-defect.labels.json, target: review-labels.json }]
sandbox_dependencies: *deps
@ -47,7 +50,7 @@ tasks:
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258.patch && rm -rf eval/workflow_bench
prompt: Review the historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
files: [{ source: review-pr-2258-defect.labels.json, target: review-labels.json }]
sandbox_dependencies: *deps
@ -59,7 +62,7 @@ tasks:
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2258b.patch && rm -rf eval/workflow_bench
prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2258. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
files: [{ source: review-pr-2258-clean.labels.json, target: review-labels.json }]
sandbox_dependencies: *deps
@ -71,6 +74,6 @@ tasks:
setup: git apply --exclude='eval/workflow_bench/*' eval/workflow_bench/review_cases/pr-2773.patch && rm -rf eval/workflow_bench
prompt: Review this historical snapshot of https://github.com/abhigyanpatwari/GitNexus/pull/2773. Report only actionable defects introduced by the local diff.
oracle:
command: test -s review-output.json
command: test -s "$GITNEXUS_BENCH_REVIEW_OUTPUT"
files: [{ source: review-pr-2773-clean.labels.json, target: review-labels.json }]
sandbox_dependencies: *deps

View file

@ -44,6 +44,11 @@ tasks:
target: gitnexus/node_modules
- source: gitnexus-shared/node_modules
target: gitnexus-shared/node_modules
# Host-built types/JS. The clone has no dist/, and node_modules/gitnexus-shared
# is a relative symlink into that unbuilt tree — without this mount, in-sandbox
# `tsc --noEmit` / vitest fail with TS2307 / TS6379 (run 33962002890).
- source: gitnexus-shared/dist
target: gitnexus-shared/dist
prompt: >
Add -j as a short alias for --json on the gitnexus status command
(gitnexus/src/cli/index.ts), and cover the alias with a unit test in

View file

@ -1010,6 +1010,13 @@ export const startAnalyze = async (request: {
force?: boolean;
embeddings?: boolean;
token?: string;
/**
* Index-branch selector. Omitted: a `url` with no existing clone takes the
* remote's default branch, an existing clone updates whichever branch it
* already has checked out, and a `path` request is not cloned at all and
* indexes that working tree as it stands.
*/
branch?: string;
}): Promise<{ jobId: string; status: string }> => {
const response = await fetchWithTimeout(
`${_backendUrl}/api/analyze`,

View file

@ -0,0 +1,285 @@
# Analyze phase breakdown — where the time actually goes
Measured 2026-09-06/07 while landing #3194, #3196 and #3200. Records what the
`analyze` pipeline costs per phase, and — as importantly — the optimizations
that were measured and **rejected**, so the next person does not re-derive them.
Unlike `parse-throughput.md` (a synthetic-fixture scaffold), these numbers come
from a real repo corpus. They are not a CI gate; the gate for the parse
dispatch path is `bench/parse-dispatch-rounds/`.
---
## Method, and one trap that invalidates everything
**The corpus must be a git repository.** A non-git checkout cannot record a
schema fingerprint, so the tool forces a full rebuild on every run and prints:
> `non-git repositories never record a schema fingerprint, so this run rebuilds regardless`
A "warm" run measured that way is a forced cold rebuild wearing a warm label. A
37.7s figure was recorded that way during this work and was meaningless. On a
real git repo an unchanged re-analyze short-circuits to `Already up to date`.
**And the analyzer binary must not change between runs.** Rebuilding or
re-copying `dist` changes the runner identity, and the tool then prints:
> `analyzer runner identity changed ...; forcing a full rebuild so the index
provenance matches the analyzer`
Every run after a rebuild is a full rebuild. To measure the incremental path,
run once to stamp the identity, then edit and run again **without touching
`dist`**. An earlier revision of this document reported full-rebuild numbers as
if they were incremental for exactly this reason; they are corrected below.
**And a mark on an `await` is not a mark on that call.** The post-pipeline tail
was timed by injecting timestamp marks into a disposable `dist`. An `await` that
directly follows native work also absorbs whatever libuv still had queued, so
the cost lands on the wrong line — that is how 845ms ended up attributed to a
dynamic import that actually costs 0.035ms. Sanity-check any mark that lands on
a call with no plausible work in it.
That is three separate "silently fall back to full work" guards — non-git
corpus, runner identity, and the escalation gate. Read the banner on every run
before trusting a number.
Corpus: this repository, `git archive` of HEAD into a scratch dir, then
`git init && git add -A && git commit`. 5350 paths, 2234 parseable, ~30MB.
16 workers, `dist` on a local overlay filesystem (see "Filesystem" below).
Phase numbers come from the `✓ Phase: <name> (<ms>)` lines under
`NODE_ENV=development`; the post-pipeline tail has no such lines and was
measured by injecting timestamp marks into a disposable copy of the built
`dist`.
---
## Cold analyze
| | before #3194 | after #3196 |
| ---------------- | -----------: | ----------: |
| total | 110.3s | **63.6s** |
| parse | 74.0s | 36.0s |
| scopeResolution | 17.0s | 16.0s |
| all other phases | 1.2s | 1.2s |
| dispatches | 221 | 15 |
Graph output identical throughout: 51,286 nodes / 163,092 edges / 2106 clusters
/ 759 flows.
## Re-analyze after a one-file edit — the developer loop
Leaf file (`cli/update-notice.ts`, 2 importers), stable runner identity, so the
incremental path is genuinely taken. **31.7s total.**
| step | ms | share |
| ----------------------------------------------------- | -------: | ------: |
| scopeResolution | ~13900 | 44% |
| **FTS index rebuild** (`buildSearchIndexesOrDegrade`) | **7525** | **24%** |
| graph write (`loadGraphToLbug`, subgraph) | 3566 | 11% |
| parse | ~2800 | 9% |
| post-FTS event-loop drain (see below) | 845 | 3% |
| everything else | ~3000 | 9% |
The 845ms was originally recorded against `import('./platform/capabilities.js')`.
That import is not the cost: the CLI already imports the module statically, so a
cached dynamic import measures 0.035ms and `getRuntimeCapabilities()` 0.1ms. The
`await` there is the first yield after the native FTS build, so it absorbs
whatever libuv work was still queued. Any mark placed on an `await` immediately
after native work charges that work to the wrong line.
The incremental machinery works: the graph write was a 3,980-node subgraph, not
the full 51,288. **Everything the #3194/#3196 parse work optimized is ~9% of
this.**
A FULL-rebuild run of the same repo is 36-39s, with the graph write at ~6.3s and
FTS at ~9.7s. Do not quote those as edit-loop numbers.
## FTS index rebuild — the largest non-resolution cost, and it is a floor
A true incremental leaf edit rebuilds **8** of the 20 configured indexes — the
tables the writeback actually DMLs. Per-index cost (measured against a copy of
the corpus index, `bench/` probe, reproduced within 2% of the in-analyze number):
```
3541ms File.file_fts <- 47% of the incremental FTS cost on its own
1696ms Function.function_fts
521ms Method 506ms Const 486ms Property
403ms Interface 324ms Class 291ms TypeAlias
------
7769ms 8 indexes (in-analyze: 7525ms)
```
A FULL rebuild does all 20 and costs ~10.2s; the extra 12 indexes are only
~2.5s, so **the narrowing is already doing its job**. An earlier revision of
this document called the narrowing "inconsistent" because one run rebuilt 8 and
another 20 — the 20-index run was a forced full rebuild (the runner-identity
trap above), not a leaf edit. There is nothing to fix there.
`File.file_fts` dominates because File rows carry whole-file content: 2442 rows,
32.9 MB, and Ladybug tokenizes at ~9.8 MB/s.
### Four ways out, all measured, all closed
**Narrow further — no.** The 8 tables are exactly the ones holding rows for the
6 files in the write set (1 changed + 5 importers). There is no fat.
**Build the indexes concurrently — impossible.** A second connection issuing
`CREATE_FTS_INDEX` fails immediately:
> `Cannot start a new write transaction in the system. Only one write transaction at a time is allowed in the system.`
Eight builds on one connection serialize exactly (7886ms concurrent vs 7769ms
serial).
**Raise the connection's thread count — no effect.** min-of-3 wall time at
4 / 8 / 16 / default(24) threads: 7298 / 7133 / 7109 / 7345 ms, inside the ~400ms
per-config spread. CPU burned does move — 8.8s / 9.7s / 11.8s / 13.6s — so the
default over-subscribes ~60% for nothing, but wall time is flat.
**Skip `File.file_fts` when content did not change — cannot happen.** Any file
edit changes a File row, and Ladybug's FTS is not incremental: an index built
before an insert does not see the new row, so a changed row forces a whole-table
rebuild. Dropping `content` from the index takes it 3541ms → **241ms**, but that
is deleting full-file keyword search (#2317/#2323), not optimizing it. Capping
the indexed content is a bad trade — the size distribution is flat, so a 64 KB
cap still indexes 90% of the bytes while truncating the 72 largest files.
### The one lever left: overlap
The FTS build runs on a libuv thread, not the main thread, and fully overlaps
blocking JS:
```
index alone 3859ms
index + 3000ms of JS burn 3337ms (serial would be ~6859ms)
```
So the 3.5s File index could hide entirely behind the pipeline's ~17s of
main-thread JS. File rows are the only ones that make this possible: they are
`{ name, filePath }` from `processStructure`, with content lazy-read from disk at
CSV time, so they are fully determined by the file scan — before parsing, before
resolution.
Two things block it today, and neither is small:
1. The DB is **closed** for the whole pipeline (`closeLbug` before
`runPipelineFromRepo`, `initLbug` after). An early File write means holding a
write handle across the pipeline.
2. It moves `liveIndexMutationStarted` before the pipeline. A pipeline failure
would then leave the live index with fresh File content and stale symbols,
instead of untouched.
## scopeResolution — 14.7s, and it is two different problems
`PROF_SCOPE_RESOLUTION=1` already exists and reports the internal split. Marks
injected around the phase supply the rest. On a one-file edit:
| step | ms | share |
| ---------------------------- | -------: | ------: |
| ParsedFile store rehydration | 4426 | 30% |
| **`emit`** | **7161** | **49%** |
| `resolve` | 966 | 7% |
| `finalize` | 552 | 4% |
| `extract` | 369 | 3% |
| per-language teardown, misc | ~750 | 5% |
`extract` is small because the parse cache works: 2121/2121 pre-extracted hits
on the TypeScript pass. Nothing here re-parses.
### Rehydration: three full store scans, one per language
The corpus resolves three languages — python (54 files), typescript (2121),
javascript (59) — and `loadParsedFilesForPaths` walks **every** shard on each
pass. The store is 413 shards / 301 MB. A pass that wants a single file costs
335 ms, because the skip decision needs the envelope's path listing and that
listing is only trustworthy after the payload digest is checked.
So the fixed cost is paid per language, and **it scales with language count**,
not with how many files that language has. Measured, min-of-5, three passes:
```
before python 411ms typescript 2872ms javascript 507ms = 3834ms
after python 415ms typescript 2805ms javascript 248ms = 3484ms
```
Memoizing each shard's authenticated listing for the run removes the repeat
scans (−350 ms here; roughly −250 ms per additional language elsewhere). The
first pass is unchanged by construction — it is what populates the memo.
What is left is a floor. The digest is **not** the cost: SHA-256 over all
301 MB takes 134 ms (2.25 GB/s, hardware-accelerated), so swapping it for
CRC-32 would buy ~90 ms and cost an envelope-format bump. The remainder is
`v8.deserialize` (~1240 ms) plus the cross-shard string intern walk
(~1085 ms), and the intern walk is not optional — dropping it regresses
retained heap ~59%, which is #2649's constraint.
### `emit` is the real target
7.2s, 49% of the phase and ~21% of the whole edit loop. It is a fan of passes,
each walking every reference site in the repo:
```
1828ms emitCallableValueFlow 480ms emitFreeCallFallback
1590ms emitReceiverBoundCalls 426ms emitReferencesViaLookup
681ms resolveDefGraphId 313ms emitUniqueNamePropertyAccesses
529ms lookupCore 280ms emitReturnShapeMemberAccesses
472ms tryEmitEdge 222ms emitPropertyDispatchCalls
449ms getScope
```
No hot inner loop, nothing quadratic, no single pass worth more than 12% of the
phase. Micro-optimizing any of them is not the win.
The win is not running them. A one-file edit re-emits all 163,094 edges to
write a 3,980-node subgraph. Every pass above runs over all 2121 TypeScript
files because the pipeline rebuilds the full in-memory graph every run and only
the DB _write_ is incremental. Making `emit` incremental means knowing which
files' edges can change when one file's registry contribution changes — which
is a design, not a patch, and #2649 rules out "just cache the emitted edges".
---
## Measured and rejected
Recorded because each cost real time to establish and each looks attractive
from the armchair.
**More workers buys nothing.** Isolated-harness wall time at 16 / 20 / 24
workers: 44.1s / 44.8s / 43.3s — a 1.5s spread against a 3.7s within-size
spread. A full-analyze sweep appeared to show 20 beating 16 by 6.3s; it was
noise, and the sweep was invalid anyway because `GITNEXUS_WORKER_POOL_SIZE` was
silently clamped at the time (fixed in #3200). `DEFAULT_POOL_SIZE_CAP = 16`
stands.
**Bundling the worker entry buys ~250ms.** Worker boot profiled at 12.2s per
worker, of which `getPackageScopeConfig` 5.86s + `internalModuleStat` 3.14s +
`lstat`/`open` ~2.2s — ESM module resolution, not native grammars (all 11
`tree-sitter-*` imports together are 33ms) and not V8 compile (53ms;
`NODE_COMPILE_CACHE` gives zero gain). An esbuild bundle takes 16-worker boot
8.6s → 0.37s.
**That was a filesystem artifact.** On a normal overlay filesystem the same
boot is 515ms stock vs 263ms bundled — 0.2% of a 110s analyze. The 8.6s only
reproduces with the repo on a 9p mount (WSL2 `D:\`). Dropped.
Caveat carried by every number here: `dist` on a 9p mount costs ~3.5s of a 73s
run (73.0s vs 69.6s on overlay). Measure on a local filesystem.
---
## Open
The post-pipeline tail is fully accounted (99.7%): FTS rebuild, the graph write,
and 0.8s of post-FTS event-loop drain between them.
FTS is closed as an optimization target except for the overlap above, which is a
scheduling change to `run-analyze.ts`'s open/close discipline rather than
anything about FTS.
`scopeResolution` is now broken down. Its rehydration half has a floor and one
repeat-scan win that is taken. Its other half — `emit`, 7.2s — is the largest
remaining target in the whole edit loop, and the only way at it is incremental
resolution.
Every optimization in this document that looked compelling from the armchair
died under measurement. Measure first, and check the banner.

View file

@ -0,0 +1,31 @@
{
"_what": "Baselines for bench/parse-dispatch-rounds/measure.mjs --check. Guards parse-cache pack layout and dispatch-round cadence. Neither is visible in graph output — batching that changed output would be a bug — so nothing else in the repo can see these regress. Four of the five arms are deterministic; only pack_scaling_ratio is a timing signal.",
"_triage": "READ THIS BEFORE RE-RUNNING. layout_fingerprint, packs, single_file_packs, rounds, cjk_rounds and ascii_rounds are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. pack_scaling_ratio is the only timing arm; runner contention dominates it, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is that one, suspect the machine.",
"layout_fingerprint": "cc875fd264498964b463aef55cec0166d57468a092303e94f1ed7f09fe141a44",
"_layout_fingerprint_note": "sha256 over the sorted pack membership — which files share a pack, and their order within it. Every parse-cache key derives from a pack's file set, so a change here invalidates every cached chunk for every user. This is a CORRECTNESS gate: drift needs a SCHEMA_BUMP in src/storage/parse-cache.ts alongside a new fingerprint, never a lone re-baseline.",
"packs": 774,
"single_file_packs": 251,
"_shape_note": "THE FLOOR. Without these two, every arm below is a ceiling over nothing. `rounds` only asserts something while the corpus OVER-SPLITS — 774 packs where the byte budget alone needs 5, 251 of them holding a single file. Shrink the corpus until packing stops over-splitting and rounds still reads 5 and still passes, asserting a property the corpus no longer has. bench/import-target learned this the hard way: four heap arms read 0 B and passed every ceiling, because a ceiling says 'not too big' and nothing said 'still measuring something'.",
"rounds": 5,
"_rounds_note": "Exact round count for the fixed corpus at the 2MB budget, folded through the production accumulator in pipeline-phases/parse-round-budget.ts. HIGHER (toward packs=774) means dispatch went back to one barrier per cache pack — the #3196 regression, measured at ~1.5x on a cold analyze with no visible symptom. LOWER (toward 1) means the close condition stopped firing, so an open round retains the whole repo until the tail drain (#2649 heap shape). Both directions verified to fail this arm before it was recorded: forcing close-every-chunk reads 774, disabling the close reads 1.",
"cjk_rounds": 8,
"ascii_rounds": 3,
"_encoding_note": "The round budget bounds what the MAIN THREAD HOLDS, so it must count UTF-8 bytes. String.length returns UTF-16 code units: a CJK character is one unit but three UTF-8 bytes, so reverting the unit would let a CJK-heavy repo hold ~3x its nominal budget before draining. The two corpora are constructed to have IDENTICAL UTF-16 length and differ only in encoded size, so under String.length both close 3 rounds and the arm collapses. Verified: reverting roundFileBytes to content.length takes cjk_rounds 8 -> 3. This is the arm that pins the change no unit test could — round cadence changes no graph output, so a test asserting output passes either way.",
"pack_scaling_budget": 1.6,
"_pack_scaling_note": "(t_4n / t_n) / 4 for packParseCacheChunks; ~1.0 is linear. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget — bench/callable-value-flow's widening_overhead gate failed twice on a shared runner at 2.07 and 1.975 against a 1.9 budget while the code was correct, on a sub-11ms measurement. A ratio divides the machine out. Measured over 5 runs on a NON-idle box: 0.940, 0.946, 0.977, 0.998, 1.085 (peak-to-peak 1.154). Budget is 1.6, i.e. 1.47x the measured maximum — this file's siblings use ~1.5x on ratios. It catches packParseCacheChunks going superlinear (it sorts within each bucket, so a global sort or a nested scan lands here) and is not tight enough to police drift. min-of-15 estimator, matching bench/import-target's finding that N=5 tripped its own budget ~1 run in 20 while N=15 held every language inside a 1.13-1.26x swing.",
"_measured": {
"pack_scaling_ratio": 1.085,
"pack_scaling_ratio_samples": [0.94, 0.946, 0.977, 0.998, 1.085],
"small_ms": 1.91,
"large_ms_4x": 7.46,
"reps": 15
},
"_measured_note": "Maxima over 5 runs on a box that was NOT idle, so the ratio spread is an upper bound on its real noise. small_ms/large_ms_4x are recorded for context only — nothing gates on them, because an absolute millisecond is exactly the gate this file avoids."
}

View file

@ -0,0 +1,279 @@
/**
* Build-free bench for parse-cache pack layout and dispatch-round cadence.
*
* WHY THIS EXISTS. `WorkerPool.dispatch` is a barrier: it resolves only once
* every job it created has committed. Packs are keyed `(language,
* sha256(path) % 128)`, so the byte budget almost never binds and most packs
* land far below the pool size — this repo produced 1285 packs where the
* budget alone needed 16, and 549 held a single file. Dispatching one pack at
* a time therefore left most workers idle for every round-trip. #3194 fixed
* fan-out WITHIN a pack; #3196 batched packs into bounded rounds and took a
* cold analyze from 110.3s to 70.5s (221 dispatches -> 15).
*
* Nothing guarded that. Round boundaries are deliberately invisible to the
* graph — batching that changed output would be a bug — so no test can see the
* regression, and it would come back as a silent 1.5x on every cold analyze.
* Two earlier attempts to pin this as a unit test failed for exactly that
* reason: one scraped a logger line the progress stream does not carry, the
* other asserted graph content that is identical either way.
*
* FOUR ARMS, and only the last is a timing arm:
*
* - `rounds` — EXACT. The regression signal. A fixed corpus and budget must
* produce a fixed number of rounds. Per-pack dispatch coming back sends this
* to `packs`; a broken close condition sends it to 1.
*
* - `cjk_rounds` vs `ascii_rounds` — EXACT. The round budget bounds what the
* MAIN THREAD HOLDS, so it must count UTF-8 bytes. `String.length` returns
* UTF-16 code units: a CJK character is one unit but three UTF-8 bytes, so
* reverting the unit would let a CJK-heavy repo hold ~3x its nominal budget
* before draining — the #2649 heap-failure shape. The two corpora are
* identical in UTF-16 length and differ only in encoded size, so under
* `String.length` they would close the SAME number of rounds. Only a UTF-8
* count separates them.
*
* - `packs` / `single_file_packs` — EXACT, and they are the FLOOR. `rounds`
* only asserts something while the corpus over-splits (774 packs where the
* byte budget alone needs 5). Shrink the corpus past that and `rounds` still
* reads 5 and still passes, gating a property the corpus no longer has.
* bench/import-target learned this when four heap arms read 0 B and passed.
*
* - `pack_scaling_ratio` — the only timing arm, and a RATIO not a millisecond
* ceiling. (t_4n/t_n)/4 divides the machine out; ~1.0 is linear. A fixed ms
* budget on a shared runner is a coin flip, and this repo has the scar:
* bench/callable-value-flow's gate failed twice at 2.07 and 1.975 against a
* 1.9 budget with correct code, on a sub-11ms measurement. Catches
* `packParseCacheChunks` going superlinear; not tight enough to police drift.
*
* Usage:
* node --import tsx bench/parse-dispatch-rounds/measure.mjs # report
* node --import tsx bench/parse-dispatch-rounds/measure.mjs --check # CI gate
*/
import { performance } from 'node:perf_hooks';
import { createHash } from 'node:crypto';
import { readFileSync } from 'node:fs';
import { packParseCacheChunks } from '../../src/storage/parse-cache.js';
import { createRoundBudget } from '../../src/core/ingestion/pipeline-phases/parse-round-budget.js';
const baselines = JSON.parse(
new URL('./baselines.json', import.meta.url).pathname
? readFileSync(new URL('./baselines.json', import.meta.url), 'utf8')
: '{}',
);
/** Matches DEFAULT_CHUNK_BYTE_BUDGET / the round budget's default in parse-impl.ts. */
const BUDGET = 2 * 1024 * 1024;
/**
* A repo shaped like a real one: many languages, so `(language, bucket)`
* packing over-splits well past what the byte budget alone would need. Sizes
* are deliberately uneven — a uniform corpus hides an off-by-one in the fold.
*/
function mixedCorpus(scale = 1) {
const langs = [
['ts', 900],
['py', 400],
['java', 260],
['go', 240],
['rb', 120],
['rs', 180],
['php', 90],
['cs', 140],
];
const files = [];
for (const [ext, count] of langs) {
for (let i = 0; i < count * scale; i++) {
files.push({
path: `src/${ext}/mod${i}.${ext}`,
// 400B - 8KB, varying by index so packs are not uniform.
size: 400 + ((i * 977) % 7700),
language: ext,
});
}
}
return files;
}
/** Feed chunks through the real accumulator and count the rounds it closes. */
function roundsFor(chunks, contentsByPath, budgetBytes) {
const budget = createRoundBudget(budgetBytes);
let rounds = 0;
for (const chunk of chunks) {
if (budget.addChunk(chunk.map((p) => contentsByPath.get(p)))) rounds++;
}
// The tail drain closes a partially-filled round when anything is left.
if (budget.bufferedBytes > 0) rounds++;
return rounds;
}
/**
* Two corpora with IDENTICAL UTF-16 length and different UTF-8 size. Under
* `String.length` both close the same number of rounds; under UTF-8 the CJK
* one closes strictly more.
*/
function encodingCorpora() {
// 1 UTF-16 unit / 3 UTF-8 bytes each, vs 1 unit / 1 byte each.
const cjkLine = '説'.repeat(240);
const asciiLine = 'a'.repeat(240);
const count = 260;
const files = Array.from({ length: count }, (_, i) => ({
path: `src/enc/mod${i}.ts`,
size: 240,
language: 'ts',
}));
const chunks = packParseCacheChunks(files, BUDGET);
const cjk = new Map(files.map((f) => [f.path, cjkLine]));
const ascii = new Map(files.map((f) => [f.path, asciiLine]));
// A budget small enough that both corpora close several rounds.
const encBudget = 24 * 1024;
return {
utf16Length: cjkLine.length === asciiLine.length,
cjkRounds: roundsFor(chunks, cjk, encBudget),
asciiRounds: roundsFor(chunks, ascii, encBudget),
};
}
/**
* Min-of-N estimator. `fastest` rather than a mean because the minimum is the
* least contaminated sample on a shared runner — the same choice, and the same
* reason, as bench/import-target's `fastest()`.
*/
function fastest(fn, reps) {
fn(); // warm
let best = Infinity;
for (let r = 0; r < reps; r++) {
const t0 = performance.now();
fn();
best = Math.min(best, performance.now() - t0);
}
return best;
}
const REPS = 15;
const corpus = mixedCorpus();
const corpus4x = mixedCorpus(4);
const packs = packParseCacheChunks(corpus, BUDGET);
// A RATIO, not a millisecond ceiling. Wall-clock is runner-speed-dependent and
// a fixed ms budget on a shared runner is a coin flip — this file's sibling
// benches record exactly that failure. (t_4n / t_n) / 4 divides the machine
// out: ~1.0 is linear, and packParseCacheChunks going superlinear (it sorts
// within each bucket) shows up here regardless of how fast the box is.
const smallMs = fastest(() => packParseCacheChunks(corpus, BUDGET), REPS);
const largeMs = fastest(() => packParseCacheChunks(corpus4x, BUDGET), REPS);
const packScaling = largeMs / smallMs / 4;
const contents = new Map(corpus.map((f) => [f.path, 'x'.repeat(f.size)]));
const rounds = roundsFor(packs, contents, BUDGET);
const enc = encodingCorpora();
/**
* Order-independent hash of the pack layout: which files share a pack, and in
* what order within it. Catches a packing change that leaves the counts intact
* but moves files between packs — which would silently change every cache key.
*/
const layoutFingerprint = createHash('sha256')
.update(
packs
.map((chunk) => chunk.join(','))
.sort()
.join('\n'),
)
.digest('hex');
const singleFilePacks = packs.filter((c) => c.length === 1).length;
const totalBytes = corpus.reduce((sum, f) => sum + f.size, 0);
const budgetFloor = Math.ceil(totalBytes / BUDGET);
console.log(`files : ${corpus.length}`);
console.log(
`packs : ${packs.length} (expect ${baselines.packs}; byte budget alone needs ${budgetFloor})`,
);
console.log(`rounds : ${rounds} (expect ${baselines.rounds})`);
console.log(`single_file_packs : ${singleFilePacks} (expect ${baselines.single_file_packs})`);
console.log(`cjk_rounds : ${enc.cjkRounds} (UTF-8 bytes)`);
console.log(`ascii_rounds : ${enc.asciiRounds} (same UTF-16 length)`);
console.log(`layout_fingerprint : ${layoutFingerprint.slice(0, 16)}`);
console.log(
`pack_scaling_ratio : ${packScaling.toFixed(3)} (budget <= ${baselines.pack_scaling_budget}; ~1.0 is linear)`,
);
console.log(
`reps : ${REPS} small ${smallMs.toFixed(2)}ms / 4x ${largeMs.toFixed(2)}ms`,
);
if (process.argv.includes('--check')) {
let failed = false;
if (layoutFingerprint !== baselines.layout_fingerprint) {
failed = true;
console.error(
`\nFAIL layout_fingerprint: ${layoutFingerprint}\n` +
` expected ${baselines.layout_fingerprint}\n` +
` Pack membership moved. Every parse-cache key is derived from a pack's\n` +
` file set, so this invalidates every cached chunk for every user. If the\n` +
` change is intended, it needs a SCHEMA_BUMP in src/storage/parse-cache.ts\n` +
` alongside a new fingerprint here — never re-baseline it alone.`,
);
}
if (rounds !== baselines.rounds) {
failed = true;
console.error(
`\nFAIL rounds: ${rounds}, expected exactly ${baselines.rounds}.\n` +
` HIGHER (toward packs=${packs.length}) means rounds stopped batching and\n` +
` dispatch went back to one barrier per cache pack — the #3196 regression,\n` +
` worth ~1.5x on a cold analyze with no visible symptom.\n` +
` LOWER (toward 1) means the close condition stopped firing, so an open\n` +
` round retains the whole repo until the tail drain (#2649 heap shape).\n` +
` Check createRoundBudget in pipeline-phases/parse-round-budget.ts.`,
);
}
if (!enc.utf16Length) {
failed = true;
console.error(
`\nFAIL encoding arm is broken: its two corpora no longer share a UTF-16 length.`,
);
} else if (enc.cjkRounds <= enc.asciiRounds) {
failed = true;
console.error(
`\nFAIL cjk_rounds ${enc.cjkRounds} <= ascii_rounds ${enc.asciiRounds}.\n` +
` These corpora have identical UTF-16 length and differ only in encoded\n` +
` size, so equal round counts mean the budget is counting String.length\n` +
` again instead of Buffer.byteLength. A CJK-heavy repo would then hold\n` +
` ~3x its nominal budget on the main thread before draining.\n` +
` See roundFileBytes in pipeline-phases/parse-round-budget.ts.`,
);
}
// SHAPE — the floor. Without it every arm below is a ceiling over nothing:
// shrink the corpus until packing stops over-splitting and `rounds` still
// reads 5 and still passes, asserting a property the corpus no longer has.
if (packs.length !== baselines.packs || singleFilePacks !== baselines.single_file_packs) {
failed = true;
console.error(
`\nFAIL shape: packs ${packs.length} (expected ${baselines.packs}), ` +
`single_file_packs ${singleFilePacks} (expected ${baselines.single_file_packs}).\n` +
` The corpus must stay one that OVER-SPLITS — ${packs.length} packs where the\n` +
` byte budget alone needs ${budgetFloor}. That gap is the entire reason rounds\n` +
` exist, so if it closes, the rounds arm below asserts nothing.`,
);
}
if (packScaling > baselines.pack_scaling_budget) {
failed = true;
console.error(
`\nFAIL pack_scaling_ratio: ${packScaling.toFixed(3)} exceeds ` +
`${baselines.pack_scaling_budget} (~1.0 is linear).\n` +
` packParseCacheChunks grew superlinearly in file count — it sorts within\n` +
` each bucket, so a global sort or a nested scan lands here.\n` +
` This is the ONLY timing arm in this file: re-run on an idle machine\n` +
` before investigating, and check \`reps\` in the report first.`,
);
}
if (failed) process.exit(1);
console.log('\nOK — within budget.');
}

View file

@ -0,0 +1,16 @@
{
"_comment": "Correctness counts are exact. Timing budgets are deliberately loose and only guard large regressions in the post-extraction Zig workspace pass.",
"small": {
"modules": 40,
"calls_per_module": 12,
"gated_calls": 480,
"ms_budget": 1000
},
"large": {
"modules": 160,
"calls_per_module": 12,
"gated_calls": 1920,
"ms_budget": 4000
},
"linear_scaling_slack": 1.375
}

View file

@ -0,0 +1,138 @@
#!/usr/bin/env node
/**
* Build-free scaling and correctness guard for Zig cross-file static gates.
*
* The workspace pass parses every indexed Zig file, resolves direct @import
* aliases, then applies sibling boolean constants to call sites. This bench
* makes both relevant axes explicit: file count and calls per importer. It
* also fingerprints the number of calls classified dead, so a fast no-op
* implementation cannot pass the timing gate.
*
* Usage:
* node --import tsx bench/zig-cross-file-resolution/measure.mjs
* node --import tsx bench/zig-cross-file-resolution/measure.mjs --check
*/
import { readFileSync } from 'node:fs';
import { dirname, join } from 'node:path';
import { fileURLToPath } from 'node:url';
import { performance } from 'node:perf_hooks';
import { populateZigWorkspaceStaticGating } from '../../src/core/ingestion/languages/zig/workspace-static-gating.ts';
const HERE = dirname(fileURLToPath(import.meta.url));
const SMALL_MODULES = 40;
const LARGE_MODULES = 160;
const CALLS_PER_MODULE = 12;
const REPS = 7;
const range = (line, col) => ({ startLine: line, startCol: col, endLine: line, endCol: col + 4 });
function corpus(modules) {
const parsedFiles = [];
const fileContents = new Map();
for (let i = 0; i < modules; i++) {
const cfgPath = `bench/cfg${i}.zig`;
const appPath = `bench/app${i}.zig`;
fileContents.set(cfgPath, 'pub const ENABLED = false;\n');
fileContents.set(
appPath,
`const cfg = @import(\"./cfg${i}.zig\");\n` +
Array.from(
{ length: CALLS_PER_MODULE },
(_, j) => `pub fn run${j}() void { if (cfg.ENABLED) dead${j}(); }`,
).join('\n'),
);
parsedFiles.push(Object.freeze({ filePath: cfgPath, referenceSites: Object.freeze([]) }));
parsedFiles.push(
Object.freeze({
filePath: appPath,
referenceSites: Object.freeze(
Array.from({ length: CALLS_PER_MODULE }, (_, j) => ({
kind: 'call',
name: `dead${j}`,
atRange: range(j + 2, 44),
})),
),
}),
);
}
return { parsedFiles, fileContents };
}
function run(modules) {
const { parsedFiles, fileContents } = corpus(modules);
populateZigWorkspaceStaticGating(parsedFiles, { fileContents });
let gatedCalls = 0;
for (const file of parsedFiles) {
for (const site of file.referenceSites) if (site.staticGated === true) gatedCalls++;
}
return gatedCalls;
}
function measure(modules) {
run(modules);
let bestMs = Infinity;
let gatedCalls = 0;
for (let i = 0; i < REPS; i++) {
const start = performance.now();
gatedCalls = run(modules);
bestMs = Math.min(bestMs, performance.now() - start);
}
return {
modules,
calls_per_module: CALLS_PER_MODULE,
gated_calls: gatedCalls,
min_ms: Number(bestMs.toFixed(2)),
};
}
const report = { small: measure(SMALL_MODULES), large: measure(LARGE_MODULES) };
report.workload_ratio = LARGE_MODULES / SMALL_MODULES;
report.scaling_ratio = Number(
(report.large.min_ms / Math.max(report.small.min_ms, 0.01)).toFixed(3),
);
report.linear_factor = Number((report.scaling_ratio / report.workload_ratio).toFixed(3));
if (!process.argv.includes('--check')) {
console.log(JSON.stringify(report, null, 2));
process.exit(0);
}
const baseline = JSON.parse(readFileSync(join(HERE, 'baseline.json'), 'utf8'));
const failures = [];
const requirePositiveNumber = (path, value) => {
if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) {
failures.push(`${path}: expected a finite positive number, got ${JSON.stringify(value)}`);
return false;
}
return true;
};
for (const arm of ['small', 'large']) {
for (const key of ['modules', 'calls_per_module', 'gated_calls']) {
if (report[arm][key] !== baseline[arm][key]) {
failures.push(`${arm}.${key}: expected ${baseline[arm][key]}, got ${report[arm][key]}`);
}
}
if (
requirePositiveNumber(`${arm}.ms_budget`, baseline[arm].ms_budget) &&
report[arm].min_ms > baseline[arm].ms_budget
) {
failures.push(`${arm}.min_ms ${report[arm].min_ms} exceeds budget ${baseline[arm].ms_budget}`);
}
}
if (
requirePositiveNumber('linear_scaling_slack', baseline.linear_scaling_slack) &&
report.linear_factor > baseline.linear_scaling_slack
) {
failures.push(
`linear_factor ${report.linear_factor} exceeds slack ${baseline.linear_scaling_slack} ` +
`(runtime ${report.scaling_ratio}x for ${report.workload_ratio}x work)`,
);
}
console.log(JSON.stringify(report, null, 2));
if (failures.length > 0) {
console.error('[zig-cross-file-resolution --check] FAIL');
for (const failure of failures) console.error(` - ${failure}`);
process.exit(1);
}
console.log('[zig-cross-file-resolution --check] PASS');

View file

@ -1888,9 +1888,9 @@
"license": "MIT"
},
"node_modules/@types/node": {
"version": "26.4.0",
"resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.0.tgz",
"integrity": "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ==",
"version": "26.4.1",
"resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.1.tgz",
"integrity": "sha512-k97ENvZWtvA6yqz5/FS6a7duDgOPEeOQOc2iKS/nY6mX6qJUKtLnWzQS+Xj6tXweyj6ZcTAK2Qecetnvi9nCLA==",
"devOptional": true,
"license": "MIT",
"dependencies": {
@ -2994,9 +2994,9 @@
}
},
"node_modules/express-rate-limit": {
"version": "8.6.2",
"resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.6.2.tgz",
"integrity": "sha512-YH4ru+eOJxQABscKFfRCy9R7x9QFGdezclVMwwgFFndzS2Xnm0uo6B0ABZsLhcpeptGv2qvuJVWlQr9gQZoC3A==",
"version": "8.7.0",
"resolved": "https://registry.npmjs.org/express-rate-limit/-/express-rate-limit-8.7.0.tgz",
"integrity": "sha512-hOwV7WOxXfjRpAM1DSJWZDXx3GhplwD8IfwuwvogD8i1Qnkgosw/H45s4ZnFAUHDAhPjlY9hLBvJhKmGMyY26g==",
"license": "MIT",
"dependencies": {
"debug": "^4.4.3",

View file

@ -31,6 +31,10 @@
import fs from 'node:fs';
import path from 'node:path';
import { readRepoControlFile } from '../config/repo-control-file.js';
import {
InvalidBranchError,
validateBranchName as validateBranchNameCore,
} from '../core/git-ref.js';
import type { AnalyzeOptions } from './analyze-options.js';
export const GITNEXUS_RC_FILENAME = '.gitnexusrc';
@ -38,9 +42,6 @@ export const GITNEXUS_RC_FILENAME = '.gitnexusrc';
/** Final fallback when no branch is configured or detectable. */
export const DEFAULT_BRANCH_FALLBACK = 'main';
/** Git refs longer than this are almost certainly a mistake / injection attempt. */
const BRANCH_MAX_LENGTH = 255;
/**
* Thrown for any `.gitnexusrc` problem (missing-file is NOT an error — it
* returns `undefined`). The message is user-facing and names the file so the
@ -157,45 +158,18 @@ const assertNoHiddenChars = (value: string, source: string): void => {
/**
* Validate a user-supplied branch name (from CLI or `.gitnexusrc`). Returns the
* trimmed name or throws {@link GitNexusRcError}. Conservative but accepts the
* shapes real branches use (`feature/foo-bar`, `release/1.2`, `develop`).
* trimmed name or throws {@link GitNexusRcError}. Rules live in
* `core/git-ref.ts`; this wrapper keeps the CLI / `.gitnexusrc` error type.
*/
export function validateBranchName(value: string, source: string): string {
const trimmed = value.trim();
if (!trimmed) {
throw new GitNexusRcError(`${source}: branch name must not be empty.`);
try {
return validateBranchNameCore(value, source);
} catch (err) {
if (err instanceof InvalidBranchError) {
throw new GitNexusRcError(err.message);
}
throw err;
}
if (trimmed.length > BRANCH_MAX_LENGTH) {
throw new GitNexusRcError(`${source}: branch name is too long (max ${BRANCH_MAX_LENGTH}).`);
}
assertNoHiddenChars(trimmed, source);
if (/\s/.test(trimmed)) {
throw new GitNexusRcError(`${source}: branch name must not contain whitespace.`);
}
// git ref-name rules (subset): reject characters git itself forbids in refs.
if (/[~^:?*[\\]/.test(trimmed)) {
throw new GitNexusRcError(
`${source}: branch name contains characters not allowed in a git ref (~ ^ : ? * [ \\).`,
);
}
if (trimmed.startsWith('-')) {
throw new GitNexusRcError(`${source}: branch name must not start with "-".`);
}
if (trimmed.includes('..')) {
throw new GitNexusRcError(`${source}: branch name must not contain "..".`);
}
// Git permits a backtick in a ref, but the branch is embedded inside a
// Markdown inline-code span in the generated AGENTS.md/CLAUDE.md regression
// example, where a backtick would close the span early and let the rest of
// the template render as instruction text. Reject it at this single
// chokepoint so all three tiers (CLI flag, .gitnexusrc, auto-detect via
// sanitizeDetectedBranch) are covered (#1996 tri-review P1).
if (trimmed.includes('`')) {
throw new GitNexusRcError(
`${source}: branch name must not contain a backtick (it would break the generated Markdown).`,
);
}
return trimmed;
}
/**

View file

@ -4,7 +4,7 @@ import path from 'node:path';
import { execFileSync } from 'node:child_process';
import { acquireFileLock, FileLockBusyError } from '../../storage/file-lock.js';
import { getGlobalDir } from '../../storage/repo-manager.js';
import { isProcessAlive, readProcessStartTime } from '../../utils/process-identity.js';
import { isProcessAlive, readProcessStartTimeCached } from '../../utils/process-identity.js';
import { loadAutoSyncConfig } from './config.js';
import { runAutoSyncOnce } from './runner.js';
import { getAutoSyncMutexPath, getAutoSyncWatchDir } from './state.js';
@ -632,7 +632,7 @@ function resolveWatchDeps(deps: Partial<AutoSyncWatchControlDeps> = {}): AutoSyn
return undefined;
}
}),
readProcessStartTime: deps.readProcessStartTime ?? readProcessStartTime,
readProcessStartTime: deps.readProcessStartTime ?? readProcessStartTimeCached,
sleep:
deps.sleep ??
((ms) =>

View file

@ -0,0 +1,128 @@
/**
* Git ref-name validation used by both the CLI and the HTTP analyze route.
*
* Lives in `core/` so `server/api.ts` does not import `cli/analyze-config`
* (that import closed a cli → server → cli cycle: `cli/serve.ts` already
* imports `createServer`). The CLI keeps a thin wrapper that rethrows
* {@link InvalidBranchError} as `GitNexusRcError`.
*/
/** Git refs longer than this are almost certainly a mistake / injection attempt. */
const BRANCH_MAX_LENGTH = 255;
/**
* Thrown when a user-supplied branch name fails {@link validateBranchName}.
* Callers at a product boundary map this to their own error type (CLI:
* `GitNexusRcError`; HTTP: 400).
*/
export class InvalidBranchError extends Error {
constructor(message: string) {
super(message);
this.name = 'InvalidBranchError';
}
}
/**
* Reject control characters and hidden / bidirectional Unicode in a string
* value. These have no legitimate place in a branch name and would otherwise
* let a committed config or HTTP body smuggle invisible controls into
* generated AGENTS.md / CLAUDE.md content.
*/
const isHiddenOrControl = (codePoint: number): boolean =>
codePoint < 0x20 ||
codePoint === 0x7f ||
(codePoint >= 0x200b && codePoint <= 0x200f) || // zero-width + LRM/RLM
(codePoint >= 0x202a && codePoint <= 0x202e) || // bidi embeddings/overrides
(codePoint >= 0x2060 && codePoint <= 0x2064) || // word-joiner + invisible math
(codePoint >= 0x2066 && codePoint <= 0x206f) || // bidi isolates + deprecated
codePoint === 0xfeff; // BOM / zero-width no-break space
const assertNoHiddenChars = (value: string, source: string): void => {
for (const ch of value) {
const cp = ch.codePointAt(0);
if (cp !== undefined && isHiddenOrControl(cp)) {
throw new InvalidBranchError(
`${source}: value contains control or hidden/bidirectional characters, which are not allowed.`,
);
}
}
};
/**
* Validate a user-supplied branch name. Returns the trimmed name or throws
* {@link InvalidBranchError}. Conservative but accepts the shapes real
* branches use (`feature/foo-bar`, `release/1.2`, `develop`).
*/
export function validateBranchName(value: string, source: string): string {
const trimmed = value.trim();
if (!trimmed) {
throw new InvalidBranchError(`${source}: branch name must not be empty.`);
}
if (trimmed.length > BRANCH_MAX_LENGTH) {
throw new InvalidBranchError(`${source}: branch name is too long (max ${BRANCH_MAX_LENGTH}).`);
}
assertNoHiddenChars(trimmed, source);
if (/\s/.test(trimmed)) {
throw new InvalidBranchError(`${source}: branch name must not contain whitespace.`);
}
// git ref-name rules (subset): reject characters git itself forbids in refs.
if (/[~^:?*[\\]/.test(trimmed)) {
throw new InvalidBranchError(
`${source}: branch name contains characters not allowed in a git ref (~ ^ : ? * [ \\).`,
);
}
if (trimmed.startsWith('-')) {
throw new InvalidBranchError(`${source}: branch name must not start with "-".`);
}
// Force-refspec prefix (`git fetch origin +main` / `+refs/heads/main:…`).
// Rejected here so neither the CLI nor HTTP can pass a force-update refspec
// through as a "branch" (#3199 review, defense in depth).
if (trimmed.startsWith('+')) {
throw new InvalidBranchError(`${source}: branch name must not start with "+".`);
}
// The symbolic ref HEAD (case-sensitive). A repo can have a branch named
// `head`; git itself treats only `HEAD` as the current-commit alias.
if (trimmed === 'HEAD') {
throw new InvalidBranchError(`${source}: branch name must not be "HEAD".`);
}
if (trimmed.includes('..')) {
throw new InvalidBranchError(`${source}: branch name must not contain "..".`);
}
// The remaining `git check-ref-format` rules. Without these the validator
// accepted refs git itself refuses (`feature.lock`, `/feature`, `feature/`,
// `feature//next`, `@`, `.hidden`), so the failure surfaced later from the
// git subprocess instead of here. No real branch can violate them — git
// could not have created one — so nothing that works today starts failing.
if (trimmed.endsWith('.lock') || trimmed.split('/').some((part) => part.endsWith('.lock'))) {
throw new InvalidBranchError(`${source}: branch name must not end with ".lock".`);
}
if (trimmed.startsWith('/') || trimmed.endsWith('/')) {
throw new InvalidBranchError(`${source}: branch name must not start or end with "/".`);
}
if (trimmed.includes('//')) {
throw new InvalidBranchError(`${source}: branch name must not contain consecutive slashes.`);
}
if (trimmed === '@') {
throw new InvalidBranchError(`${source}: branch name must not be the single character "@".`);
}
if (trimmed.includes('@{')) {
throw new InvalidBranchError(`${source}: branch name must not contain "@{".`);
}
if (trimmed.endsWith('.') || trimmed.split('/').some((part) => part.startsWith('.'))) {
throw new InvalidBranchError(
`${source}: branch name must not end with "." or have a path component starting with ".".`,
);
}
// Git permits a backtick in a ref, but the branch is embedded inside a
// Markdown inline-code span in the generated AGENTS.md/CLAUDE.md regression
// example, where a backtick would close the span early and let the rest of
// the template render as instruction text. Reject it at this single
// chokepoint so all three tiers (CLI flag, .gitnexusrc, auto-detect via
// sanitizeDetectedBranch) are covered (#1996 tri-review P1).
if (trimmed.includes('`')) {
throw new InvalidBranchError(
`${source}: branch name must not contain a backtick (it would break the generated Markdown).`,
);
}
return trimmed;
}

View file

@ -13,17 +13,14 @@
* Conservative by design: we only tag an edge when we can prove the
* gating expression evaluates to `false`. Anything ambiguous → live.
*
* Scope of v1:
* Supported scope:
*
* (a) **File-local** consts (`pub const FOO = false;`, plus const-to-const
* aliases up to 5 hops), built once per file by `buildZigBoolConstMap`.
* (b) **Cross-file** (`const cfg = @import("./cfg.zig"); if (cfg.FOO)`) is
* NOT resolved yet. The evaluator keeps the seam for it (`importAliases`
* + `lookupBoolsForPath`, consumed by the `field_expression` case), but
* the only caller passes an empty alias map and a lookup that always
* returns `undefined`, because the capture emitter runs in the parse
* worker and sees only the current file. Tracked in #3162. Until then
* every `cfg.FOO` condition folds to unknown, i.e. live.
* (b) **Cross-file** direct imports (`const cfg = @import("./cfg.zig");
* if (cfg.FOO)`) are enriched after per-file extraction. The workspace
* caller supplies `importAliases` and `lookupBoolsForPath`; the parse
* worker still uses empty/undefined inputs and remains file-local.
*
* Also out of scope: multi-hop member access (`cfg.sub.FOO`), re-exported
* consts, runtime-evaluated bools (`const FOO = computeIt();`), and

View file

@ -19,6 +19,7 @@ import { resolveZigImportInternal } from '../../import-resolvers/zig.js';
import { zigProvider } from '../zig.js';
import { expandZigWildcardNames, zigArityCompatibility, zigMergeBindings } from './index.js';
import { populateZigRangeBindings } from './range-binding.js';
import { populateZigWorkspaceStaticGating } from './workspace-static-gating.js';
export const zigScopeResolver: ScopeResolver = {
language: SupportedLanguages.Zig,
@ -67,6 +68,8 @@ export const zigScopeResolver: ScopeResolver = {
populateOwners: (parsed: ParsedFile) => populateClassOwnedMembers(parsed),
populateWorkspaceReferences: populateZigWorkspaceStaticGating,
// Payload captures — `for (items) |it|`, `if (opt) |v|`, `while (it.next())
// |x|` — typed from the subject's binding after finalize (F6).
populateRangeBindings: populateZigRangeBindings,

View file

@ -0,0 +1,106 @@
import type { ParsedFile, ReferenceSite } from 'gitnexus-shared';
import { getTreeSitterBufferSize } from '../../constants.js';
import type { ZigBuildZonConfig } from '../../language-config.js';
import { resolveZigImportInternal } from '../../import-resolvers/zig.js';
import {
buildZigBoolConstMap,
collectZigStaticGatedRanges,
isPositionStaticGated,
type ZigImportAliasMap,
} from '../../call-extractors/zig-static-gating.js';
import { parseSourceSafe, ParseTimeoutError } from '../../../tree-sitter/safe-parse.js';
import { getZigParser } from './query.js';
type ZigTree = ReturnType<ReturnType<typeof getZigParser>['parse']>;
export function populateZigWorkspaceStaticGating(
parsedFiles: ParsedFile[],
ctx: {
readonly fileContents: ReadonlyMap<string, string>;
readonly treeCache?: { get(filePath: string): unknown };
readonly resolutionConfig?: unknown;
},
): void {
const parser = getZigParser();
const trees = new Map<string, ZigTree>();
const bools = new Map<string, ReturnType<typeof buildZigBoolConstMap>>();
for (const parsed of parsedFiles) {
const source = ctx.fileContents.get(parsed.filePath);
if (source === undefined) continue;
let tree = ctx.treeCache?.get(parsed.filePath) as ZigTree | undefined;
if (tree === undefined) {
try {
tree = parseSourceSafe(parser, source, undefined, {
bufferSize: getTreeSitterBufferSize(source),
});
} catch (err) {
if (err instanceof ParseTimeoutError) continue;
throw err;
}
}
trees.set(parsed.filePath, tree);
bools.set(parsed.filePath, buildZigBoolConstMap(tree.rootNode));
}
const knownPaths = new Set(trees.keys());
for (const [index, parsed] of parsedFiles.entries()) {
const tree = trees.get(parsed.filePath);
if (tree === undefined) continue;
const aliases = collectImportAliases(
tree,
parsed.filePath,
knownPaths,
ctx.resolutionConfig as ZigBuildZonConfig | null | undefined,
);
if (aliases.size === 0) continue;
const ranges = collectZigStaticGatedRanges(
tree.rootNode,
bools.get(parsed.filePath) ?? new Map(),
aliases,
(filePath) => bools.get(filePath),
);
if (ranges.length === 0) continue;
const next = parsed.referenceSites.map((site) =>
site.kind === 'call' &&
site.staticGated !== true &&
isPositionStaticGated(site.atRange.startLine, site.atRange.startCol, ranges)
? ({ ...site, staticGated: true } satisfies ReferenceSite)
: site,
);
parsedFiles[index] = Object.freeze({ ...parsed, referenceSites: Object.freeze(next) });
}
}
function collectImportAliases(
tree: ZigTree,
fromFile: string,
knownPaths: ReadonlySet<string>,
resolutionConfig?: ZigBuildZonConfig | null,
): ZigImportAliasMap {
const candidates = new Map<string, string>();
const declarationCounts = new Map<string, number>();
for (const decl of tree.rootNode.descendantsOfType('variable_declaration')) {
const names = decl.namedChildren.filter((node) => node.type === 'identifier');
const binding = names[0]?.text;
if (binding === undefined) continue;
declarationCounts.set(binding, (declarationCounts.get(binding) ?? 0) + 1);
const builtin = decl.namedChildren.find(
(node) => node.type === 'builtin_function' && node.text.startsWith('@import('),
);
const raw = builtin?.descendantsOfType('string').at(0)?.text;
if (raw === undefined) continue;
const specifier = raw.replace(/^['"]|['"]$/g, '');
const target = resolveZigImportInternal(fromFile, specifier, knownPaths, resolutionConfig);
if (target !== null) candidates.set(binding, target);
}
const aliases = new Map<string, string>();
for (const [binding, target] of candidates) {
// Alias lookup below is name-based rather than position-aware. If a name
// is redeclared in another lexical scope, fail open instead of applying
// either module's constants to every use of that spelling.
if (declarationCounts.get(binding) === 1) aliases.set(binding, target);
}
return aliases;
}

View file

@ -7,6 +7,7 @@ import { accumulateExportedTypesFromParsedNode, type ExportedTypeMap } from './c
import type { ParsedFile } from 'gitnexus-shared';
import { WorkerPool } from './workers/worker-pool.js';
import type { DispatchGroup } from './workers/worker-pool.js';
import type { SkippedPath } from './workers/clone-safety.js';
import type { CfgSkipCounts } from './cfg/collect.js';
import { logger } from '../logger.js';
@ -206,12 +207,12 @@ export const mergeChunkResults = (
};
/**
* Dispatch a chunk's files to the worker pool and return the RAW per-worker
* results, WITHOUT merging them into the graph. Split out from
* {@link processParsing} so the parse loop can overlap one chunk's
* merge (main-thread, via {@link mergeChunkResults}) with the NEXT chunk's
* worker parse — the merge is the only remaining serial main-thread step once
* ParsedFile serialization moved into the workers (#worker-idle pipelining).
* Dispatch ONE chunk's files to the worker pool and return the RAW per-worker
* results, WITHOUT merging them into the graph. A thin single-group wrapper
* over {@link dispatchChunkParseRound}, used by {@link processParsing}'s
* one-shot path. The chunk-to-chunk overlap this once described now lives in
* `parse-impl.ts` at ROUND granularity (`startRound` / `drainRound` /
* `closeRound`), which batches several chunks into one dispatch.
* Returns `[]` for an all-unparseable chunk (the caller merges `[]` → empty).
*/
export const dispatchChunkParse = async (
@ -227,26 +228,53 @@ export const dispatchChunkParse = async (
*/
chunkHash?: string,
): Promise<ParseWorkerResult[]> => {
const parseableFiles: ParseWorkerInput[] = [];
for (const file of files) {
const lang = getLanguageFromFilename(file.path);
if (lang) parseableFiles.push({ path: file.path, content: file.content });
}
if (parseableFiles.length === 0) return [];
const total = files.length;
const chunkResults = await workerPool.dispatch<ParseWorkerInput, ParseWorkerResult>(
parseableFiles,
(filesProcessed) => {
onFileProgress?.(Math.min(filesProcessed, total), total, 'Parsing...');
},
chunkHash,
const [chunkResults = []] = await dispatchChunkParseRound(
[{ items: files, chunkHash }],
workerPool,
onFileProgress,
);
// Capture raw results for the incremental parse cache before merging.
if (outRawResults) {
for (const r of chunkResults) outRawResults.push(r);
}
return chunkResults;
};
/**
* Dispatch SEVERAL parse-cache chunks as one pool round and return their raw
* results, one array per input group in input order.
*
* `WorkerPool.dispatch` is a barrier, so one round-trip per chunk leaves most
* slots idle whenever a chunk is smaller than the pool — which stable
* `(language, hash(path) % 128)` packs usually are. Batching chunks into one
* `dispatchGroups` call removes those barriers; jobs are still cut at chunk
* boundaries, so every result stays attributable to the chunk whose cache key
* owns it.
*/
export const dispatchChunkParseRound = async (
groups: ReadonlyArray<DispatchGroup<{ path: string; content: string }>>,
workerPool: WorkerPool,
onFileProgress?: FileProgressCallback,
): Promise<ParseWorkerResult[][]> => {
const dispatchGroups: DispatchGroup<ParseWorkerInput>[] = groups.map((group) => {
const items: ParseWorkerInput[] = [];
for (const file of group.items) {
const lang = getLanguageFromFilename(file.path);
if (lang) items.push({ path: file.path, content: file.content });
}
return { items, chunkHash: group.chunkHash };
});
const total = groups.reduce((sum, group) => sum + group.items.length, 0);
if (dispatchGroups.every((group) => group.items.length === 0)) return groups.map(() => []);
const perGroup = await workerPool.dispatchGroups<ParseWorkerInput, ParseWorkerResult>(
dispatchGroups,
(filesProcessed) => {
onFileProgress?.(Math.min(filesProcessed, total), total, 'Parsing...');
},
);
const chunkResults = perGroup.flat();
// Skipped-language telemetry (worker output, independent of the merge).
const skippedLanguages = new Map<string, number>();
@ -311,7 +339,7 @@ export const dispatchChunkParse = async (
}
onFileProgress?.(total, total, 'done');
return chunkResults;
return perGroup;
};
// ============================================================================

View file

@ -20,7 +20,7 @@ import {
enrichExportedTypeMap,
type BindingEntry,
} from '../binding-accumulator.js';
import { mergeChunkResults, dispatchChunkParse } from '../parsing-processor.js';
import { mergeChunkResults, dispatchChunkParseRound } from '../parsing-processor.js';
import {
fileContentHash,
computeChunkHash,
@ -69,6 +69,8 @@ import {
createWorkerPool,
workerPoolDisabledByEnv,
resolveAutoPoolSize,
envWorkerPoolSize,
resolveHostParallelism,
WorkerPoolInitializationError,
WorkerPoolDisabledError,
} from '../workers/worker-pool.js';
@ -115,6 +117,8 @@ import {
import { isDebugHeapEnabled, logHeapProbe } from '../utils/heap-probe.js';
import { logger } from '../../logger.js';
import { mapConcurrent } from '../../../lib/utils.js';
import { createRoundBudget } from './parse-round-budget.js';
// ── Constants ──────────────────────────────────────────────────────────────
/**
@ -213,9 +217,38 @@ const CHUNK_BYTES_PER_WORKER = DEFAULT_CHUNK_BYTE_BUDGET;
*/
const TARGET_JOBS_PER_WORKER = 3;
/**
* Concurrent durable ParsedFile directory resets per round. Matches the file
* reader's `READ_CONCURRENCY`, because both compete for the same descriptors.
*/
const DURABLE_RESET_CONCURRENCY = 32;
/** Floor for a derived sub-batch so jobs don't shrink to per-file IPC churn. */
const MIN_SUB_BATCH_BYTES = 256 * 1024;
/**
* Source bytes an open round may HOLD — cache hits and misses alike — before
* it is dispatched and drained.
*
* A `dispatch` is a barrier, so one round-trip per cache pack leaves most slots
* idle: packs are keyed by `(language, hash(path) % 128)` and routinely land far
* under {@link DEFAULT_CHUNK_BYTE_BUDGET} (this repo: 1285 packs where the byte
* budget alone needs 16, 549 of them holding a single file). Rounds batch packs
* into one `dispatchGroups` call without touching pack identity.
*
* This is the in-flight cap, the same role Piscina's `maxQueue` plays: bigger
* rounds remove more barriers but hold more file content and more un-merged
* worker output on the main thread at once. Defaulting to one chunk budget
* keeps in-flight source bytes at the magnitude the loop already prefetched
* (`parseChunkConcurrency`, 2 chunks ahead). Override via
* `GITNEXUS_PARSE_ROUND_BYTES`.
*/
function resolveParseRoundByteBudget(options?: PipelineOptions): number {
const env = Number(process.env.GITNEXUS_PARSE_ROUND_BYTES);
if (Number.isFinite(env) && env > 0) return env;
return resolveChunkByteBudget(options);
}
function resolveChunkByteBudget(options?: PipelineOptions): number {
const opt = options?.chunkByteBudget;
if (typeof opt === 'number' && Number.isFinite(opt) && opt > 0) return opt;
@ -557,12 +590,42 @@ export async function runChunkedParseAndResolve(
// cores-based auto size is capped by source bytes / CHUNK_BYTES_PER_WORKER
// so a tiny repo does not spawn a full idle pool. Cache pack membership
// is independent of this number (#3088).
const explicitPoolSize = options?.workerPoolSize;
// `--workers <N>` and `GITNEXUS_WORKER_POOL_SIZE` are both deliberate
// operator input, so both bypass the work-proportional cap below. Only the
// env path used to be clamped by it, which made the documented escape hatch
// silently do nothing: on a 30MB repo the cap resolves to 16, so an operator
// asking for 24 still got 16 with no warning, while `--workers 24` got 24.
const explicitPoolSize = options?.workerPoolSize ?? envWorkerPoolSize();
// Cores-based auto size, bounded by source bytes so a tiny repo does not
// spawn a full idle pool.
const workProportionalCap = Math.max(1, Math.ceil(totalBytes / CHUNK_BYTES_PER_WORKER));
// An operator's number is honored, but never exceeds the number of files
// there are to parse — `GITNEXUS_WORKER_POOL_SIZE=100000` on a five-file repo
// should not become the literal thread count. This bounds `--workers` and the
// env var identically, keeping the parity above intact. Note it does NOT
// shrink an incremental re-analyze: `totalParseable` counts every parseable
// file in the scan, not the changed ones, so a warm run of a large repo still
// spawns the full requested pool.
const effectivePoolSize =
explicitPoolSize && explicitPoolSize > 0
? explicitPoolSize
? Math.min(explicitPoolSize, Math.max(1, totalParseable))
: Math.min(resolveAutoPoolSize(), workProportionalCap);
// Deliberate over-subscription is the operator's call, so this warns rather
// than caps — silently capping is what the override exists to stop. But an
// exported `GITNEXUS_WORKER_POOL_SIZE` applies to EVERY analyze in a
// long-lived caller (watch auto-sync, the MCP server), including small
// incremental ones, and that is easy to set once and forget.
if (explicitPoolSize && explicitPoolSize > 0) {
const hostParallelism = resolveHostParallelism();
if (effectivePoolSize > hostParallelism) {
logger.warn(
{ requested: explicitPoolSize, spawning: effectivePoolSize, hostParallelism },
`Worker pool size ${effectivePoolSize} exceeds this host's ${hostParallelism} usable core(s); ` +
`parsing is CPU-bound, so the extra workers add memory pressure without throughput. ` +
`This applies to every analyze while the override is set.`,
);
}
}
// Cache packs: stable (language, hash(path) mod 128) buckets, then the
// per-call byte budget inside each bucket (#3088). Pool size is used only
// for worker count and sub-batch fan-out, not membership.
@ -793,25 +856,84 @@ export async function runChunkedParseAndResolve(
const verboseThroughputLog = isDev || isVerboseIngestionEnabled();
const heapProbeEveryN = isDebugHeapEnabled() ? 25 : 0;
// ── Merge pipelining (#worker-idle) ──────────────────────────────────────
// Merging a chunk's worker results into the graph is the only remaining
// serial main-thread step (ParsedFile serialization now runs in workers).
// To stop the whole pool idling during that merge, we OVERLAP it with the
// NEXT chunk's worker parse: a freshly-dispatched worker chunk is parked in
// `pendingWorkerChunk`, and we merge+finalize it only AFTER starting the
// following chunk's dispatch — so the workers parse chunk N+1 while the
// main thread merges chunk N. Chunk ORDER is preserved (N finalized before
// N+1), which keeps the deferred aggregation deterministic. Cache-hit
// chunks drain any pending chunk first, then finalize inline (no worker
// dispatch to overlap).
interface PendingWorkerChunk {
readonly rawResults: ParseWorkerResult[];
readonly chunkIdx: number;
readonly chunkHash: string | null;
readonly chunkFiles: Array<{ path: string; content: string }>;
readonly chunkStartMs: number | null;
}
let pendingWorkerChunk: PendingWorkerChunk | null = null;
// ── Dispatch rounds + merge pipelining (#worker-idle) ────────────────────
// Two separate idle sources, handled together here.
//
// 1. Barrier per chunk. `dispatch` resolves only when every job it created
// has committed, so dispatching one cache pack at a time strands the
// pool whenever a pack is smaller than it — which stable packs usually
// are. Chunks accumulate into a ROUND (bounded by `roundByteBudget` of
// cache-missing source) and go out in one `dispatchGroups` call.
// 2. Serial merge. Merging worker results into the graph is the only
// remaining serial main-thread step (ParsedFile serialization now runs
// in workers). A dispatched round is parked in `pendingRound` and
// merged only AFTER the following round's dispatch has started, so the
// workers parse round N+1 while the main thread merges round N.
//
// Chunk ORDER is preserved throughout — rounds drain in order and entries
// inside a round finalize by `chunkIdx` — which keeps deferred aggregation
// deterministic regardless of how chunks were batched. Cache hits ride
// along as round entries so they observe the same ordering without forcing
// a dispatch.
/**
* One chunk queued into the current round. A `hit` already has its worker
* output (from the parse cache); a `miss` gets it from the round's single
* `dispatchGroups` call. Both are finalized in `chunkIdx` order when the
* round drains, which is what keeps deferred aggregation deterministic
* regardless of how chunks were batched.
*/
type RoundEntry =
| {
readonly kind: 'hit';
readonly chunkIdx: number;
// A hit never reaches a worker, so it needs the file COUNT (progress,
// throughput log) but never the source strings. Holding those would
// pin the whole repo's text for a warm run, which is what the
// buffered budget below exists to bound.
readonly fileCount: number;
readonly chunkStartMs: number | null;
readonly cachedRaw: ParseWorkerResult[];
}
| {
readonly kind: 'miss';
readonly chunkIdx: number;
readonly chunkHash: string | null;
readonly chunkFiles: Array<{ path: string; content: string }>;
readonly chunkStartMs: number | null;
};
/**
* Chunk hashes whose durable ParsedFile directory could not be reset. The
* old generation's shards are still on disk, so a warm hit would union
* stale shards with the new ones. Treated exactly like a quarantined chunk:
* skip the parse-cache write so the next run re-dispatches into a clean
* directory rather than trusting a generation we could not clear.
*/
const durablePrepareFailures = new Set<string>();
const roundByteBudget = resolveParseRoundByteBudget(options);
let roundEntries: RoundEntry[] = [];
/**
* Bytes an open round is HOLDING, counting hits as well as misses.
*
* Counting only the cache-MISSING bytes would bound just what the workers
* are asked to do, so a warm run — where nothing misses — would never reach
* the close condition and would buffer every chunk's cached output until
* the tail drain. That is the #2649 heap failure on a large repo. Counting
* both keeps a hits-only run draining at the same cadence as a cold one;
* `startRound` already supports a round with no misses.
*
* Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool,
* so the cap means the same thing here as it does for a job's payload.
*/
const roundBudget = createRoundBudget(roundByteBudget);
/**
* Files QUEUED into rounds so far. `filesParsedSoFar` only advances when a
* round drains, so it is the right number for the throughput log but would
* pin a warm run's progress bar at the phase floor for the whole loop.
*/
let queuedFilesSoFar = 0;
let pendingRound: { entries: RoundEntry[]; missResults: ParseWorkerResult[][] } | null = null;
// Apply one chunk's merged worker data: per-chunk aggregation into the
// run-level accumulators + the throughput log. Shared by the cache-hit
@ -821,7 +943,7 @@ export async function runChunkedParseAndResolve(
const applyChunkResults = async (
chunkWorkerData: WorkerExtractedData | null,
chunkIdx: number,
chunkFiles: Array<{ path: string; content: string }>,
fileCount: number,
chunkStartMs: number | null,
): Promise<void> => {
if (chunkWorkerData) {
@ -898,18 +1020,18 @@ export async function runChunkedParseAndResolve(
}
}
filesParsedSoFar += chunkFiles.length;
filesParsedSoFar += fileCount;
if (verboseThroughputLog && chunkStartMs !== null) {
const elapsedMs = Date.now() - chunkStartMs;
const filesPerSec = elapsedMs > 0 ? (chunkFiles.length * 1000) / elapsedMs : 0;
const filesPerSec = elapsedMs > 0 ? (fileCount * 1000) / elapsedMs : 0;
const stats = workerPool?.getStats?.();
const poolFrag = stats
? ` pool: ${stats.activeSlots}/${stats.size} active, ` +
`${stats.quarantined} quarantined${stats.poolBroken ? ', BROKEN' : ''}`
: ' (cache replay)';
logger.info(
`📊 chunk ${chunkIdx + 1}/${numChunks}: ${chunkFiles.length} files in ${elapsedMs}ms ` +
`📊 chunk ${chunkIdx + 1}/${numChunks}: ${fileCount} files in ${elapsedMs}ms ` +
`(${filesPerSec.toFixed(1)} files/s)${poolFrag}`,
);
}
@ -917,15 +1039,25 @@ export async function runChunkedParseAndResolve(
// Merge + finalize a parked worker chunk: graph merge (the overlapped
// main-thread step) → parse-cache write-guard → run-level aggregation.
const finalizeWorkerChunk = async (p: PendingWorkerChunk): Promise<void> => {
const chunkWorkerData = mergeChunkResults(graph, symbolTable, p.rawResults, exportedTypeMap);
const finalizeWorkerChunk = async (
p: Extract<RoundEntry, { kind: 'miss' }>,
rawResults: ParseWorkerResult[],
): Promise<void> => {
const chunkWorkerData = mergeChunkResults(graph, symbolTable, rawResults, exportedTypeMap);
// Persist raw results for this chunk hash (skipping when any chunk file
// was worker-quarantined, so the narrower rawResults isn't cached under
// the full-chunk key — see the original inline note / U20.U2).
if (parseCache && p.chunkHash && p.rawResults.length > 0) {
if (parseCache && p.chunkHash && rawResults.length > 0) {
const quarantineSet = new Set(workerPool?.getQuarantinedPaths?.() ?? []);
const chunkHadQuarantine = p.chunkFiles.some((f) => quarantineSet.has(f.path));
if (chunkHadQuarantine) {
const durableGenerationStale = durablePrepareFailures.has(p.chunkHash);
if (durableGenerationStale) {
logger.warn(
{ chunkHash: p.chunkHash.slice(0, 8) },
'parse-cache SKIP: durable generation for this chunk could not be reset, ' +
'so its shards may be stale; next run will re-dispatch it',
);
} else if (chunkHadQuarantine) {
if (isDev) {
const quarantinedInChunk = p.chunkFiles.filter((f) => quarantineSet.has(f.path)).length;
logger.info(
@ -935,7 +1067,7 @@ export async function runChunkedParseAndResolve(
);
}
} else {
await persistParseCacheChunk(parseCache, p.chunkHash, p.rawResults);
await persistParseCacheChunk(parseCache, p.chunkHash, rawResults);
if (isDev) {
logger.info(
`📦 parse-cache MISS+store: chunk ${p.chunkIdx + 1}/${numChunks} (${p.chunkFiles.length} files, ${p.chunkHash.slice(0, 8)})`,
@ -943,7 +1075,185 @@ export async function runChunkedParseAndResolve(
}
}
}
await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles, p.chunkStartMs);
await applyChunkResults(chunkWorkerData, p.chunkIdx, p.chunkFiles.length, p.chunkStartMs);
};
/**
* Dispatch a round's cache misses as ONE pool round. Returns the parked
* round; the caller drains it after starting the next one so the workers
* parse round N+1 while the main thread merges round N (the same overlap
* the per-chunk loop had, at round granularity).
*/
const startRound = async (
entries: RoundEntry[],
): Promise<{ entries: RoundEntry[]; results: Promise<ParseWorkerResult[][]> } | null> => {
if (entries.length === 0) return null;
const misses = entries.filter((entry) => entry.kind === 'miss');
if (misses.length === 0) {
return { entries, results: Promise.resolve([]) };
}
// Each chunk resets its own directory, so these are independent and run
// concurrently: serially they would sit on the critical path this round
// exists to shorten, with the pool idle and the previous round's merge
// waiting, once per miss.
//
// BOUNDED, though. A round can hold hundreds of small packs, and each
// reset is a recursive rm + mkdir. Firing all of them at once competes
// for descriptors with the chunk prefetch this loop already has in
// flight, and `readFileContents` degrades a losing read SILENTLY by
// contract — a dropped file would vanish from the chunk, from the graph,
// and from the chunk hash, shipping a narrowed index with exit 0. Same
// helper and width the file reads use.
await mapConcurrent(
misses,
async (miss) => {
if (durableParsedFileDir === undefined || miss.chunkHash === null) return;
try {
await prepareDurableParsedFileChunk(durableParsedFileDir, miss.chunkHash);
} catch (err) {
// The durable store is an optimization — degrade like the restore
// path does instead of failing the analyze. Workers recreate the
// directory on write, so at worst the old generation lingers.
// Caught per chunk so one failure cannot abort the others.
durablePrepareFailures.add(miss.chunkHash);
logger.warn(
{ err, chunkHash: miss.chunkHash.slice(0, 8) },
'parsedfile-cache: could not reset durable chunk generation; ' +
'continuing without caching this chunk',
);
}
},
{ concurrency: DURABLE_RESET_CONCURRENCY },
);
const roundFiles = misses.reduce((sum, miss) => sum + miss.chunkFiles.length, 0);
const firstIdx = misses[0].chunkIdx;
const lastIdx = misses[misses.length - 1].chunkIdx;
const progressForRound = (current: number, _total: number, filePath: string) => {
// Rounds queued before this one are already counted in
// `queuedFilesSoFar`; `current` is this round's own worker progress.
const globalCurrent = queuedFilesSoFar - roundFiles + current;
// Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
onProgress({
phase: 'parsing',
percent: Math.round(parsingProgress),
message:
firstIdx === lastIdx
? `Parsing chunk ${firstIdx + 1}/${numChunks}...`
: `Parsing chunks ${firstIdx + 1}-${lastIdx + 1}/${numChunks}...`,
detail: filePath,
stats: {
filesProcessed: globalCurrent,
totalFiles: totalParseable,
nodesCreated: graph.nodeCount,
},
});
};
const activeWorkerPool = getOrCreateWorkerPool();
if (verboseThroughputLog) {
logger.info(
`🚚 round: ${misses.length} chunk(s) ${firstIdx + 1}-${lastIdx + 1}/${numChunks}, ` +
`${roundFiles} files in one dispatch`,
);
}
const results = dispatchChunkParseRound(
misses.map((miss) => ({
items: miss.chunkFiles,
chunkHash: miss.chunkHash ?? undefined,
})),
activeWorkerPool,
progressForRound,
);
// Mark handled so a rejection during the overlap drain below isn't
// flagged as unhandled; the `await` in drainRound re-throws it for real
// handling.
results.catch(() => {});
return { entries, results };
};
/**
* Merge + finalize every chunk of a parked round, in `chunkIdx` order.
* Takes RESOLVED worker output: the round's dispatch must already have
* settled before this runs, because the pool allows only one dispatch in
* flight at a time (see `closeRound`).
*/
const drainRound = async (round: {
entries: RoundEntry[];
missResults: ParseWorkerResult[][];
}): Promise<void> => {
const missResults = round.missResults;
const missCount = round.entries.filter((entry) => entry.kind === 'miss').length;
// `dispatchGroups` returns one array per input group. If that contract
// ever breaks, every later entry in this round would silently merge the
// wrong chunk's results and skip its cache write, with a clean exit.
if (missResults.length !== missCount) {
throw new Error(
`Parse round result mismatch: ${missResults.length} result group(s) for ${missCount} dispatched chunk(s).`,
);
}
let missIdx = 0;
for (const entry of round.entries) {
if (entry.kind === 'hit') {
const chunkWorkerData = mergeChunkResults(
graph,
symbolTable,
entry.cachedRaw,
exportedTypeMap,
);
await applyChunkResults(
chunkWorkerData,
entry.chunkIdx,
entry.fileCount,
entry.chunkStartMs,
);
continue;
}
await finalizeWorkerChunk(entry, missResults[missIdx++]);
}
};
/**
* Close the accumulated round.
*
* `WorkerPool.dispatch`/`dispatchGroups` is NOT reentrant — concurrent
* calls race on the shared per-slot busy/in-flight state and wedge the
* pool until every worker idle-times out. So exactly one dispatch is in
* flight here: start this round, merge the PREVIOUS round (whose results
* are already resolved) while these workers run, then await this round and
* park it resolved for the next close to merge.
*/
const closeRound = async (): Promise<void> => {
const started = await startRound(roundEntries);
roundEntries = [];
roundBudget.reset();
const previous = pendingRound;
pendingRound = null;
if (previous) {
try {
await drainRound(previous);
} catch (err) {
// The round started above is still on the workers. Unwinding now
// reaches this function's `finally`, which calls `terminate()` — and
// terminate kills busy workers outright, which is the #2432
// mid-N-API SIGABRT hazard. Let the in-flight round settle first so
// the pool is idle, then propagate the original failure.
await started?.results.catch(() => undefined);
throw err;
}
}
if (!started) return;
let missResults: ParseWorkerResult[][];
try {
missResults = await started.results;
} catch (err) {
if (!(err instanceof WorkerPoolInitializationError)) throw err;
// Every worker crashed during startup and the pool's bounded self-heal
// was exhausted. Fail fast (#1741) — there is no sequential parser to
// degrade to. `handleWorkerStartupFailure` always throws, so
// `missResults` stays definitely assigned for the parked round below.
handleWorkerStartupFailure(err);
}
pendingRound = { entries: started.entries, missResults };
};
for (let chunkIdx = 0; chunkIdx < numChunks; chunkIdx++) {
@ -1038,17 +1348,14 @@ export async function runChunkedParseAndResolve(
durableExpectedPaths !== undefined &&
(await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths));
// Set by whichever branch queues this chunk; drives the close below.
let roundIsFull = false;
if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) {
// Cache hit: replay cached worker output. Finalize any parked worker
// chunk FIRST so deferred aggregation stays in chunk order, then merge
// + apply this hit inline (no worker dispatch to overlap).
if (pendingWorkerChunk) {
await finalizeWorkerChunk(pendingWorkerChunk);
pendingWorkerChunk = null;
}
chunkCacheHits++;
parseCacheHitFileCount += chunkFiles.length;
const chunkWorkerData = mergeChunkResults(graph, symbolTable, cachedRaw, exportedTypeMap);
if (isDev) {
logger.info(
`📦 parse-cache HIT: chunk ${chunkIdx + 1}/${numChunks} (${chunkFiles.length} files, ${chunkHash?.slice(0, 8) ?? 'unknown'})`,
@ -1062,92 +1369,41 @@ export async function runChunkedParseAndResolve(
// takes 70-95 so the UI advances through the (potentially long)
// resolution stages instead of holding at 82 (M2 from PR #1693
// review).
percent: Math.round(20 + ((filesParsedSoFar + cachedFiles) / totalParseable) * 50),
percent: Math.round(20 + ((queuedFilesSoFar + cachedFiles) / totalParseable) * 50),
message: `Parsing chunk ${chunkIdx + 1}/${numChunks} (cache)...`,
stats: {
filesProcessed: filesParsedSoFar + cachedFiles,
filesProcessed: queuedFilesSoFar + cachedFiles,
totalFiles: totalParseable,
nodesCreated: graph.nodeCount,
},
});
// The durable gate already snapshotted warm `.v8` shards into the
// run-scoped store for scope resolution.
await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs);
// run-scoped store for scope resolution. Queue into the round so this
// hit still finalizes in `chunkIdx` order relative to its neighbours.
roundEntries.push({
kind: 'hit',
chunkIdx,
fileCount: chunkFiles.length,
chunkStartMs,
cachedRaw,
});
roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content));
queuedFilesSoFar += chunkFiles.length;
} else {
// Cache miss: dispatch to workers, capture the raw results, store
// them under the chunk hash for the next run.
// Cache miss: queue for the round's single dispatch; the raw results
// are stored under the chunk hash when the round drains.
chunkCacheMisses++;
reparsedFileCount += chunkFiles.length;
if (durableParsedFileDir !== undefined && chunkHash !== null) {
try {
await prepareDurableParsedFileChunk(durableParsedFileDir, chunkHash);
} catch (err) {
// The durable store is an optimization — degrade like the restore
// path does instead of failing the analyze. Workers recreate the
// directory on write, so at worst the old generation lingers.
logger.warn(
{ err, chunkHash: chunkHash.slice(0, 8) },
'parsedfile-cache: could not reset durable chunk generation; continuing',
);
}
}
const progressForChunk = (current: number, _total: number, filePath: string) => {
const globalCurrent = filesParsedSoFar + current;
// Parse phase covers 20-70 (M2). Deferred extraction handles 70-95.
const parsingProgress = 20 + (globalCurrent / totalParseable) * 50;
onProgress({
phase: 'parsing',
percent: Math.round(parsingProgress),
message: `Parsing chunk ${chunkIdx + 1}/${numChunks}...`,
detail: filePath,
stats: {
filesProcessed: globalCurrent,
totalFiles: totalParseable,
nodesCreated: graph.nodeCount,
},
});
};
const activeWorkerPool = getOrCreateWorkerPool();
// Worker path — PIPELINE: kick off this chunk's dispatch, merge the
// PREVIOUS chunk while these workers parse, then park this chunk for
// the next iteration to merge (overlapping its parse). The deferred
// merge + parse-cache write-guard + aggregation all run in
// `finalizeWorkerChunk`, in chunk order. The pool is the sole parse
// path — `getOrCreateWorkerPool` returns a pool or throws.
const dispatchPromise = dispatchChunkParse(
chunkFiles,
activeWorkerPool,
progressForChunk,
undefined,
chunkHash ?? undefined,
);
// Mark handled so a rejection during the overlap drain below isn't
// flagged as unhandled; the `await` re-throws it for real handling.
dispatchPromise.catch(() => {});
if (pendingWorkerChunk) {
await finalizeWorkerChunk(pendingWorkerChunk);
pendingWorkerChunk = null;
}
let chunkResults: ParseWorkerResult[];
try {
chunkResults = await dispatchPromise;
} catch (err) {
if (!(err instanceof WorkerPoolInitializationError)) throw err;
// Every worker crashed during startup and the pool's bounded
// self-heal was exhausted. Fail fast (#1741) — there is no sequential
// parser to degrade to. `handleWorkerStartupFailure` always throws, so
// `chunkResults` stays definitely assigned for the parked chunk below.
handleWorkerStartupFailure(err);
}
pendingWorkerChunk = {
rawResults: chunkResults,
chunkIdx,
chunkHash,
chunkFiles,
chunkStartMs,
};
roundEntries.push({ kind: 'miss', chunkIdx, chunkHash, chunkFiles, chunkStartMs });
roundIsFull = roundBudget.addChunk(chunkFiles.map((file) => file.content));
queuedFilesSoFar += chunkFiles.length;
}
// One cap, on what the main thread is holding. That bounds the worker
// round too, since a round's dispatched bytes are a subset of its
// buffered bytes.
if (roundIsFull) await closeRound();
// (Per-chunk aggregation + parse-cache write + throughput log now run in
// `applyChunkResults` / `finalizeWorkerChunk` — see the merge-pipelining
// block above. Route/import/inheritance edges are emitted later: route
@ -1155,11 +1411,13 @@ export async function runChunkedParseAndResolve(
// scope-resolution phase, RING4-2 #943.)
}
// Drain the final parked worker chunk — the last pipelined chunk has no
// successor to overlap its merge with, so merge + finalize it here.
if (pendingWorkerChunk) {
await finalizeWorkerChunk(pendingWorkerChunk);
pendingWorkerChunk = null;
// Drain the tail: close the partially-filled round, then drain the round
// it parked — the last round has no successor to overlap its merge with.
if (roundEntries.length > 0) await closeRound();
if (pendingRound) {
const last = pendingRound;
pendingRound = null;
await drainRound(last);
}
if (isDev && parseCache && (chunkCacheHits > 0 || chunkCacheMisses > 0)) {

View file

@ -0,0 +1,63 @@
/**
* The fold that decides when an open dispatch round closes.
*
* Extracted so the decision is a shared, inspectable unit rather than four
* loose statements inside `runChunkedParseAndResolve`. The parse loop is
* STREAMING — it reads chunk contents lazily, so it cannot know every chunk's
* size up front and cannot "plan" rounds ahead. That makes an accumulator, not
* a planner, the honest shape: feed it each chunk as it is queued and it tells
* you whether the round is now full.
*
* Being a real unit is what makes round cadence observable. Round boundaries
* are otherwise invisible from outside the parse phase: they change no graph
* output (that is the point of batching) and surface only in a log line, which
* is why `bench/parse-dispatch-rounds` measures this directly rather than
* inferring cadence from a full analyze.
*/
/** Bytes a file contributes to the open round's retained total. */
export const roundFileBytes = (content: string): number => Buffer.byteLength(content, 'utf8');
export interface RoundBudget {
/**
* Add one queued chunk's files. Returns true when the round is now full and
* the caller should close it. Closing resets the accumulator.
*/
addChunk(contents: readonly string[]): boolean;
/** Bytes currently held by the open round. */
readonly bufferedBytes: number;
/** Reset without closing — used when the caller closes for another reason. */
reset(): void;
}
/**
* `budgetBytes` bounds what the main thread HOLDS, counting cache hits as well
* as misses. Counting only cache-missing bytes would bound just the work sent
* to workers, so a warm run — where nothing misses — would never reach the
* close condition and would buffer every chunk's cached output until the tail
* drain. That is the #2649 heap failure on a large repo.
*
* Measured in UTF-8 bytes, matching `estimateItemBytes` in the worker pool.
* `String.length` would return UTF-16 code units, undercounting non-ASCII
* source by up to 3x and letting a CJK-heavy repo hold well past its nominal
* budget before draining.
*/
export const createRoundBudget = (budgetBytes: number): RoundBudget => {
let bufferedBytes = 0;
return {
addChunk(contents) {
for (const content of contents) bufferedBytes += roundFileBytes(content);
if (bufferedBytes >= budgetBytes) {
bufferedBytes = 0;
return true;
}
return false;
},
get bufferedBytes() {
return bufferedBytes;
},
reset() {
bufferedBytes = 0;
},
};
};

View file

@ -694,6 +694,21 @@ export interface ScopeResolver {
ctx: { readonly fileContents: ReadonlyMap<string, string> },
) => void;
/**
* Optional workspace-wide enrichment of extracted reference sites. Runs
* after all files have been extracted and before reference finalization.
* Use this when a per-file capture needs conservative facts from an
* imported sibling (for example a compile-time branch constant).
*/
readonly populateWorkspaceReferences?: (
parsedFiles: ParsedFile[],
ctx: {
readonly fileContents: ReadonlyMap<string, string>;
readonly treeCache?: { get(filePath: string): unknown };
readonly resolutionConfig?: unknown;
},
) => void;
/**
* Recognize a `super(...)`-style receiver text. Python returns
* `/^super\s*\(/.test(t)`. Java returns `t === 'super'`. C++ may

View file

@ -145,6 +145,7 @@ export function selectScopeSourcePathsToRead(
): string[] {
const hasPostExtractHooks =
provider.populateWorkspaceOwners !== undefined ||
provider.populateWorkspaceReferences !== undefined ||
provider.populateNamespaceSiblings !== undefined ||
provider.populateRangeBindings !== undefined ||
provider.emitPostResolutionEdges !== undefined;

View file

@ -639,6 +639,11 @@ export function runScopeResolution(
`lang=${provider.language} parsedFiles=${parsedFiles.length} preExtractedHits=${preExtractedHits} skipped=${filesSkipped}`,
);
provider.populateWorkspaceOwners?.(parsedFiles, { fileContents: getFileContents() });
provider.populateWorkspaceReferences?.(parsedFiles, {
fileContents: getFileContents(),
treeCache,
resolutionConfig: input.resolutionConfig,
});
// A callable-flow-only provider has no reason to build the whole-graph
// lookup or finalize ordinary references when none of its files emitted a

View file

@ -1543,6 +1543,10 @@ function reportWarning(message: string): void {
}
}
// Keep compiled queries across jobs in this worker. A language can select
// multiple native grammars, so both grammar identity and query text matter.
const compiledQueries = new WeakMap<object, Map<string, Parser.Query>>();
const processFileGroup = (
files: ParseWorkerInput[],
language: SupportedLanguages,
@ -1553,7 +1557,14 @@ const processFileGroup = (
let query: Parser.Query;
try {
const lang = parser.getLanguage();
query = new Parser.Query(lang, queryString);
let queries = compiledQueries.get(lang);
if (!queries) {
queries = new Map();
compiledQueries.set(lang, queries);
}
const cached = queries.get(queryString);
query = cached ?? new Parser.Query(lang, queryString);
if (!cached) queries.set(queryString, query);
} catch (err) {
reportWarning(
`Query compilation failed for ${language}: ${err instanceof Error ? err.message : String(err)}`,

View file

@ -96,10 +96,51 @@ export function buildDispatchMessage<T>(items: readonly T[]): {
transferList,
};
}
/**
* One content-addressed parse-cache chunk's worth of work inside a pool round.
* See {@link WorkerPool.dispatchGroups}.
*/
export interface DispatchGroup<TInput> {
readonly items: readonly TInput[];
/**
* Chunk hash tagged onto every job derived from `items`, exactly as the
* `chunkHash` argument of {@link WorkerPool.dispatch} does for a lone chunk.
*/
readonly chunkHash?: string;
}
export interface WorkerPool {
/**
* Dispatch items across workers. Items are split into bounded jobs, each job
* is committed independently, and stalled jobs are split/retried locally.
* Dispatch several content-addressed chunks in ONE pool round.
*
* `dispatch` is a barrier: it resolves only once every job it created has
* committed, so dispatching one small parse-cache pack at a time leaves most
* slots idle for the whole round-trip. Stable packs are keyed by
* `(language, hash(path) % 128)`, which routinely yields packs far below the
* byte budget — on this repo, 1285 packs where the budget alone needs 16, and
* 549 of them hold a single file. Batching packs into one round removes those
* barriers without touching pack identity: jobs are still cut at group
* boundaries, so each job carries exactly one `chunkHash` and every result
* stays attributable to the pack that owns its cache key.
*
* Returns one result array per input group, in input order. A group whose
* items were all quarantined yields an empty array.
*
* Required, not optional. `getQuarantinedPaths?` and `getStats?` below are
* marked optional as a compatibility accommodation for `WorkerPool` shapes
* that predate them — not as a convention for new members. Making this one
* optional would force a `?.` plus a fallback branch at its only production
* call site, and that branch could never run.
*/
dispatchGroups<TInput, TResult>(
groups: readonly DispatchGroup<TInput>[],
onProgress?: (filesProcessed: number) => void,
): Promise<TResult[][]>;
/**
* Dispatch ONE chunk across workers — {@link WorkerPool.dispatchGroups} with
* a single group. Items are split into bounded jobs, each job is committed
* independently, and stalled jobs are split/retried locally.
*
* Files in {@link WorkerPool.getQuarantinedPaths} are filtered out before
* dispatch — they have already caused a worker death this pool lifetime and
@ -434,7 +475,9 @@ const DEFAULT_WORKER_READY_TIMEOUT_MS = 5_000;
* extraction / structured-clone overhead, and the marginal worker adds
* memory pressure (tree-sitter state + sub-batch buffer) without much
* throughput gain. Operators on bigger machines override via
* `GITNEXUS_WORKER_POOL_SIZE` or `--workers <N>`.
* `GITNEXUS_WORKER_POOL_SIZE` or `--workers <N>`; both are deliberate
* operator input and bypass the work-proportional sizing in `parse-impl`,
* which only bounds the AUTO default.
*/
const DEFAULT_POOL_SIZE_CAP = 16;
@ -612,7 +655,7 @@ export function resolveWorkerPoolOptions(
* GITNEXUS_WORKER_POOL_SIZE=`) is an accident, not a request for zero workers;
* only a literal `0` disables the pool.
*/
function envWorkerPoolSize(): number | undefined {
export function envWorkerPoolSize(): number | undefined {
const raw = process.env.GITNEXUS_WORKER_POOL_SIZE;
if (raw === undefined || raw.trim() === '') return undefined;
return nonNegativeInteger(raw);
@ -656,9 +699,19 @@ export function resolveAutoPoolSize(): number {
// pool cap exists to prevent. Falls back to os.cpus().length on
// older Node versions. Mirrors `capabilities.ts:85`
// (`defaultEmbeddingThreads`).
const cores =
typeof os.availableParallelism === 'function' ? os.availableParallelism() : os.cpus().length;
return Math.min(DEFAULT_POOL_SIZE_CAP, Math.max(1, cores - 1));
return Math.min(DEFAULT_POOL_SIZE_CAP, Math.max(1, resolveHostParallelism() - 1));
}
/**
* Usable parallelism for this process. Prefers `os.availableParallelism` so
* cgroup CPU limits are honored, falling back to `os.cpus().length` on older
* Node. Exported so callers that size work against the host (rather than
* against the pool default) do not re-derive the fallback.
*/
export function resolveHostParallelism(): number {
return typeof os.availableParallelism === 'function'
? os.availableParallelism()
: os.cpus().length;
}
/**
@ -863,15 +916,21 @@ function inFlightExcludePath<TInput>(job: WorkerJob<TInput>, lastProgress: numbe
return path ? [path] : [];
}
/**
* Cut `items` into bounded jobs. `startIndexOffset` places those jobs on a
* shared index space so several groups can be laid out end to end in one
* dispatch round and every result still sorts back into global input order.
*/
function createJobs<TInput>(
items: TInput[],
items: readonly TInput[],
maxItems: number,
maxBytes: number,
timeoutMs: number,
chunkHash?: string,
startIndexOffset = 0,
): WorkerJob<TInput>[] {
const jobs: WorkerJob<TInput>[] = [];
let startIndex = 0;
let startIndex = startIndexOffset;
let batch: TInput[] = [];
let batchBytes = 0;
@ -1263,11 +1322,44 @@ export const createWorkerPool = (
workers.map((_, i) => bringSlotReady(i)),
).then(() => undefined);
const dispatch = async <TInput, TResult>(
items: TInput[],
/**
* Guards the one-dispatch-at-a-time contract. The dispatch machinery keeps
* its jobs/busy-slot/in-flight state per call, so two concurrent dispatches
* hand the same slots out twice: both stall, and the failure surfaces only
* when every worker hits its idle timeout (10s+ of a wedged pool with no
* indication of the cause). Fail loudly at the call instead.
*/
let dispatchInFlight = false;
/**
* Claim the pool synchronously, then run the dispatch. The claim CANNOT be
* taken inside `dispatchGroupsInner`: its first statement awaits the
* readiness gate, so two calls made in the same tick would both get past the
* check before either set the flag.
*/
const dispatchGroups = <TInput, TResult>(
groups: readonly DispatchGroup<TInput>[],
onProgress?: (filesProcessed: number) => void,
chunkHash?: string,
): Promise<TResult[]> => {
): Promise<TResult[][]> => {
if (dispatchInFlight) {
return Promise.reject(
new WorkerPoolDispatchError(
'Worker pool dispatch is already in flight. `dispatch`/`dispatchGroups` is not ' +
'reentrant — await the previous call before starting another on the same pool.',
[],
),
);
}
dispatchInFlight = true;
return dispatchGroupsInner<TInput, TResult>(groups, onProgress).finally(() => {
dispatchInFlight = false;
});
};
const dispatchGroupsInner = async <TInput, TResult>(
groups: readonly DispatchGroup<TInput>[],
onProgress?: (filesProcessed: number) => void,
): Promise<TResult[][]> => {
// Await the initial-spawn readiness gate (F13). On first dispatch
// this blocks for up to poolOptions.workerReadyTimeoutMs while every initial
// worker's `{type:'ready'}` handshake is checked; on subsequent
@ -1285,7 +1377,8 @@ export const createWorkerPool = (
[],
);
}
if (items.length === 0) return [];
const emptyPerGroup = (): TResult[][] => groups.map(() => []);
if (groups.every((group) => group.items.length === 0)) return emptyPerGroup();
if (activeSlots.size === 0) {
const detail =
initialReadinessFailures.length > 0
@ -1308,23 +1401,56 @@ export const createWorkerPool = (
// Layer 3: filter out quarantined paths so a known-bad file never reaches
// a worker again this pool lifetime. The caller queries
// `getQuarantinedPaths` after dispatch to route filtered items.
const dispatchableItems: TInput[] = [];
for (const item of items) {
const path = itemPath(item);
if (path !== undefined && quarantine.has(path)) continue;
dispatchableItems.push(item);
}
if (dispatchableItems.length === 0) return [];
const jobs = createJobs(
dispatchableItems,
poolOptions.subBatchSize,
poolOptions.subBatchMaxBytes,
poolOptions.subBatchIdleTimeoutMs,
chunkHash,
// Quarantine is empty on every run that has not had a worker die, so the
// filter below would be an identity copy of every group's items. Skip it.
const dispatchableGroups =
quarantine.size === 0
? groups
: groups.map((group) => {
const items: TInput[] = [];
for (const item of group.items) {
const path = itemPath(item);
if (path !== undefined && quarantine.has(path)) continue;
items.push(item);
}
return { items, chunkHash: group.chunkHash };
});
const dispatchableCount = dispatchableGroups.reduce(
(sum, group) => sum + group.items.length,
0,
);
if (dispatchableCount === 0) return emptyPerGroup();
return new Promise<TResult[]>((resolve, reject) => {
// Stable cache packs can be much smaller than either job ceiling. Split
// those packs across the live slots too, otherwise each serial dispatch
// feeds only one worker. Keep both configured ceilings as upper bounds.
const maxItemsPerJob = Math.min(
poolOptions.subBatchSize,
Math.max(1, Math.floor(dispatchableCount / activeSlots.size)),
);
// Lay the groups end to end on one index space and cut jobs at every group
// boundary. A job therefore belongs to exactly one group, which is what
// lets a result be attributed back to the parse-cache chunk that owns it
// (and what keeps `chunkHash` a per-job constant through splits/requeues).
const jobs: WorkerJob<TInput>[] = [];
const groupEnds: number[] = [];
let groupStart = 0;
for (const group of dispatchableGroups) {
for (const job of createJobs(
group.items,
maxItemsPerJob,
poolOptions.subBatchMaxBytes,
poolOptions.subBatchIdleTimeoutMs,
group.chunkHash,
groupStart,
)) {
jobs.push(job);
}
groupStart += group.items.length;
groupEnds.push(groupStart);
}
return await new Promise<TResult[][]>((resolve, reject) => {
const results: WorkerJobResult<TResult>[] = [];
const inFlightProgress = new Array(size).fill(0);
// Tracks which slots are currently mid-job so the "wake idle slots"
@ -1352,10 +1478,7 @@ export const createWorkerPool = (
const reportProgress = () => {
if (!onProgress) return;
const inFlight = inFlightProgress.reduce((sum, value) => sum + value, 0);
const next = Math.min(
dispatchableItems.length,
Math.max(maxReported, completedFiles + inFlight),
);
const next = Math.min(dispatchableCount, Math.max(maxReported, completedFiles + inFlight));
if (next === maxReported) return;
maxReported = next;
onProgress(next);
@ -1456,7 +1579,19 @@ export const createWorkerPool = (
retireWorkerAfterTimeout(existing, workerIndex, reason);
return;
}
await existing.terminate().catch(() => undefined);
// Recovery must settle before dispatch returns, but a failed thread
// may never acknowledge termination. Bound that wait as in shutdown.
const termination = existing.terminate().then(
() => undefined,
() => undefined,
);
if (!(await settledWithin(termination, poolOptions.shutdownDrainMs))) {
existing.unref?.();
logger.warn(
{ workerIndex, drainMs: poolOptions.shutdownDrainMs, reason },
`Worker ${workerIndex} did not finish terminating within the shutdown drain; continuing recovery.`,
);
}
};
const replaceWorker = async (
@ -1532,9 +1667,19 @@ export const createWorkerPool = (
if (jobs.length === 0 && activeWorkers === 0) {
stopped = true;
results.sort((a, b) => a.startIndex - b.startIndex);
if (onProgress && maxReported < dispatchableItems.length)
onProgress(dispatchableItems.length);
resolve(results.map((result) => result.data));
if (onProgress && maxReported < dispatchableCount) onProgress(dispatchableCount);
// Partition back per group. Job (and split sub-job) start indices
// stay inside their group's span, so a single forward walk over the
// sorted results assigns every result to exactly one group.
const perGroup: TResult[][] = groupEnds.map(() => []);
let groupIdx = 0;
for (const result of results) {
while (groupIdx < groupEnds.length - 1 && result.startIndex >= groupEnds[groupIdx]) {
groupIdx++;
}
perGroup[groupIdx].push(result.data);
}
resolve(perGroup);
}
};
@ -1898,11 +2043,13 @@ export const createWorkerPool = (
// (`error`, `exit`, msg-channel error). Bridges the per-job teardown
// into the pool-level handleWorkerDeath recovery + breaker logic.
const recoverAndResume = async (reason: string, excludePaths: readonly string[]) => {
activeWorkers--;
busySlots.delete(workerIndex);
inFlightProgress[workerIndex] = 0;
requeueRemainder(job, excludePaths);
// Keep recovery in flight so another slot finishing cannot settle
// this dispatch before the replacement is ready for the next one.
await handleWorkerDeath(workerIndex, reason, excludePaths);
activeWorkers--;
if (stopped) return;
// Slot may have been dropped or respawned. Kick the current slot
// if still active, then wake any other idle live slots so the
@ -1949,7 +2096,6 @@ export const createWorkerPool = (
// is respawned (or dropped) and can dispatch the next
// job deterministically.
void (async () => {
activeWorkers--;
busySlots.delete(workerIndex);
requeueRemainder(job, decision.excludePaths);
await handleWorkerDeath(
@ -1958,6 +2104,7 @@ export const createWorkerPool = (
decision.excludePaths,
'retire',
);
activeWorkers--;
if (stopped) return;
if (activeSlots.has(workerIndex)) runWorker(workerIndex);
wakeIdleSlots();
@ -2273,8 +2420,18 @@ export const createWorkerPool = (
activeSlots.clear();
};
const dispatch = async <TInput, TResult>(
items: TInput[],
onProgress?: (filesProcessed: number) => void,
chunkHash?: string,
): Promise<TResult[]> => {
const [result] = await dispatchGroups<TInput, TResult>([{ items, chunkHash }], onProgress);
return result ?? [];
};
return {
dispatch,
dispatchGroups,
terminate,
size,
getQuarantinedPaths: () => quarantine.snapshot(),

View file

@ -335,10 +335,19 @@ export async function createSearchFTSIndexes(
// the old name+content index would silently persist. `dropFTSIndex` no-ops
// when the index is absent (first-ever analyze) and clears the per-connection
// memo so the create below actually runs.
// ponytail: this rebuilds every FTS index on every analyze instead of
// skipping when present; FTS build is proportional to symbol-table size and
// runs inside the existing FTS phase. Gate on a stored schema fingerprint if
// this rebuild cost ever shows up in analyze profiles.
// The cost DID show up in analyze profiles — 7.5s of a 31.7s edit loop on a
// 5350-file repo, `bench/analyze-phase-breakdown.md` — and a "skip when the
// index is already present" gate is NOT the answer, so don't reach for it.
// Every caller that reaches this loop has already dropped the indexes it
// passes in `tables`: the incremental writeback drops them because Ladybug
// cannot DML a table with a live FTS index (#2589), and a full rebuild
// builds into a fresh staging DB that never had one. A presence gate would
// therefore never fire. The cost is inherent — Ladybug's FTS is not
// incremental, so one changed row means re-tokenizing the whole table, and
// `File` alone is ~33MB of file content at ~10MB/s. The measured floor and
// the four exits that were tried and closed (narrow further, build
// concurrently, raise the connection thread count, drop `content`) are in
// that document.
try {
await dropFTSIndex(table, indexName);
await createFTSIndex(table, indexName, [...properties], stemmer);

View file

@ -68,6 +68,13 @@ export interface AnalyzeJob {
repoUrl?: string;
repoPath?: string;
repoName?: string;
/**
* Index-branch selector this job was started with, part of the job's dedup
* identity. A repo is not "the same repo" for reuse purposes when a different
* branch was asked for — reusing across branches would hand the caller a 202
* for a job indexing something else.
*/
branch?: string;
progress: AnalyzeJobProgress;
error?: string;
/** Set only when a terminal `failed` job still persisted usable work. */
@ -94,15 +101,25 @@ export class JobManager {
this.cleanupTimer = setInterval(() => this.cleanup(), CLEANUP_INTERVAL_MS);
}
/** Create a new job, or return existing active job for the same repo. */
createJob(params: { repoUrl?: string; repoPath?: string }): AnalyzeJob {
// Dedup: return existing active job for the same repo (by URL or path)
/**
* Create a new job, or return the existing active job for the same repo AND
* the same branch.
*
* Branch is part of the identity deliberately. Deduping on repo alone would
* return the in-flight job for branch A to a caller that asked for branch B,
* and that caller would read the resulting 202/`complete` as "B is indexed"
* — the same silent wrong-branch outcome that made `branch` worth honoring in
* the first place. Falling through instead lets the single-slot guard below
* reject the request outright, which is a truthful answer.
*/
createJob(params: { repoUrl?: string; repoPath?: string; branch?: string }): AnalyzeJob {
// Dedup: return existing active job for the same repo (by URL or path) and branch
for (const job of this.jobs.values()) {
if (!this.isTerminal(job.status)) {
const isSameRepo =
(params.repoUrl && job.repoUrl === params.repoUrl) ||
(params.repoPath && job.repoPath === params.repoPath);
if (isSameRepo) {
if (isSameRepo && job.branch === params.branch) {
return job;
}
}
@ -120,6 +137,7 @@ export class JobManager {
status: 'queued',
repoUrl: params.repoUrl,
repoPath: params.repoPath,
branch: params.branch,
progress: { phase: 'queued', percent: 0, message: 'Waiting to start...' },
startedAt: Date.now(),
retryCount: 0,

View file

@ -22,6 +22,7 @@ import {
listRegisteredRepos,
registryPathEquals,
} from '../storage/repo-manager.js';
import { BRANCHES_DIR, branchSlug } from '../storage/branch-index.js';
import { logger } from '../core/logger.js';
import { autoHeapCapMb } from '../core/ingestion/utils/effective-ram.js';
import { isTerminalJobStatus, type JobManager } from './analyze-job.js';
@ -49,6 +50,20 @@ export interface LaunchOptions {
springActuatorPath?: string;
asyncApiSpecPath?: string;
registryName?: string;
/**
* Index-branch selector, forwarded to `AnalyzeOptions.branch`.
*
* Setting it does not by itself mean a `branches/<slug>/` sub-directory:
* `resolveBranchPlacement` (storage/branch-index.ts) keeps the run on the flat
* slot when that slot has no recorded owner, or when its owner already IS this
* label. Only a label that differs from the flat slot's owner gets its own
* sub-directory.
*
* The caller is responsible for having the branch checked out —
* `resolveWriteTarget` in core refuses a label that disagrees with the working
* tree, which is what keeps one branch's content out of another's slot (#2106).
*/
branch?: string;
}
const MAX_WORKER_RETRIES = 2;
@ -66,17 +81,6 @@ const MAX_WORKER_RETRIES = 2;
const FINALIZE_SETTLE_TIMEOUT_MS = 60_000;
const FINALIZE_SETTLE_POLL_MS = 200;
/**
* Resolve once the analyzed repo's index is settled at `storagePath`: the
* LadybugDB file and metadata both exist AND were (re)written by THIS job
* (mtime >= jobStartMs — bare existence is not enough, a re-analysis leaves
* the previous index in place while it works), and no transient WAL/shadow/
* checkpoint sidecars remain (the worker's native close has finished).
*
* Never rejects. Timing out logs and proceeds (pre-gate behavior) rather
* than failing a job whose analysis genuinely succeeded — e.g. a no-op
* non-force analyze legitimately rewrites nothing.
*/
/**
* Look up the analyzed repo's registered storage path. The request's
* user-provided path is used only as a comparison key; the filesystem probes
@ -91,7 +95,47 @@ const registeredStoragePath = async (targetPath: string): Promise<string | null>
return entry?.storagePath ?? null;
};
const waitForSettledIndex = async (targetPath: string, jobStartMs: number): Promise<void> => {
/**
* Resolve the directory this run's index actually landed in.
*
* `registerRepo` always records the FLAT `.gitnexus` as `entry.storagePath`,
* but a pinned `--branch` run whose label differs from the flat slot's owner
* writes `lbug`/`gitnexus.json` under `branches/<slug>/` instead. Probing the
* flat path for such a run watches files it never rewrote, so the gate below
* would spin to its timeout on a perfectly successful analysis (#3199 review).
*
* `isPrimaryBranch` is the worker's own report of `!placement.branch`, so this
* follows the placement core actually chose rather than recomputing it here
* (the flat slot's recorded owner can be adopted mid-run, which would make a
* recomputation race the thing it is trying to observe).
*/
const settleDirFor = (
registryStoragePath: string,
branch: string | undefined,
isPrimaryBranch: boolean | undefined,
): string =>
branch && isPrimaryBranch === false
? path.join(registryStoragePath, BRANCHES_DIR, branchSlug(branch))
: registryStoragePath;
/**
* Resolve once the analyzed repo's index is settled at `storagePath`: the
* LadybugDB file and metadata both exist AND were (re)written by THIS job
* (mtime >= jobStartMs — bare existence is not enough, a re-analysis leaves
* the previous index in place while it works), and no transient WAL/shadow/
* checkpoint sidecars remain (the worker's native close has finished).
*
* Never rejects. Timing out logs and proceeds (pre-gate behavior) rather
* than failing a job whose analysis genuinely succeeded. The `alreadyUpToDate`
* fast path never rewrites `lbug` (see `run-analyze.ts`) and skips this wait
* at the `complete` handler so it does not hold the analyze slot for 60s.
*/
const waitForSettledIndex = async (
targetPath: string,
jobStartMs: number,
branch?: string,
isPrimaryBranch?: boolean,
): Promise<void> => {
const settled = (storagePath: string): boolean => {
try {
const lbugStat = statSync(path.join(storagePath, 'lbug'));
@ -112,7 +156,7 @@ const waitForSettledIndex = async (targetPath: string, jobStartMs: number): Prom
// Re-resolved each round: the worker registers the repo as part of the
// finalization this gate is waiting out.
const storagePath = await registeredStoragePath(targetPath);
if (storagePath && settled(storagePath)) return;
if (storagePath && settled(settleDirFor(storagePath, branch, isPrimaryBranch))) return;
if (Date.now() > deadline) {
logger.warn(
{ targetPath },
@ -173,6 +217,13 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) {
// Capture stderr for crash diagnostics
let stderrChunks = '';
// A terminal IPC message (`complete`/`error`) means the worker finished
// and is now winding down — it calls process.exit(0) ~500ms later. The
// job is deliberately still non-terminal at that point because the
// finalization gate is running, so without this flag the exit handler
// below reads that clean exit as a crash and retries a SUCCESSFUL
// analysis, three times, before failing it (#3199 review).
let terminalIpcSeen = false;
child.stderr?.on('data', (chunk: Buffer) => {
stderrChunks += chunk.toString();
if (stderrChunks.length > 4096) stderrChunks = stderrChunks.slice(-4096);
@ -186,6 +237,8 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) {
const current = jobManager.getJob(job.id);
if (!current || isTerminalJobStatus(current.status)) return;
if (msg.type === 'complete' || msg.type === 'error') terminalIpcSeen = true;
if (msg.type === 'progress') {
jobManager.updateJob(job.id, {
status: 'analyzing',
@ -202,7 +255,16 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) {
// below true in practice: the repo is actually queryable when the
// client receives the SSE complete event, and an index this run knows
// to be incomplete is never published at all.
waitForSettledIndex(targetPath, jobStartMs)
//
// alreadyUpToDate never opens LadybugDB and never rewrites `lbug`
// (run-analyze.ts early-return; CLI notes the same). The mtime gate
// would spin the full 60s and hold the single global analyze slot.
// ftsRepairedOnly DOES rewrite `lbug` (initLbug + createSearchFTSIndexes)
// so it still waits.
const settle = msg.result.alreadyUpToDate
? Promise.resolve()
: waitForSettledIndex(targetPath, jobStartMs, opts.branch, msg.result.isPrimaryBranch);
settle
.then(() => closeDbHandle())
.catch(() => {}) // best-effort: eviction failure must not fail the job
.then(() => {
@ -296,6 +358,13 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) {
const j = jobManager.getJob(job.id);
if (!j || isTerminalJobStatus(j.status)) return;
// The worker already reported a terminal outcome; this exit is it
// winding down, not dying. The job is still non-terminal only because
// the finalization gate above has not resolved yet, and that gate owns
// the outcome — retrying here would fork a second worker over a
// finished, successful analysis.
if (terminalIpcSeen) return;
// Worker crashed — attempt retry if under the limit
if (j.retryCount < MAX_WORKER_RETRIES) {
j.retryCount++;
@ -339,6 +408,7 @@ export function createLaunchAnalysisWorker(deps: LaunchDeps) {
...(opts.springActuatorPath ? { springActuatorPath: opts.springActuatorPath } : {}),
...(opts.asyncApiSpecPath ? { asyncApiSpecPath: opts.asyncApiSpecPath } : {}),
...(opts.registryName ? { registryName: opts.registryName } : {}),
...(opts.branch ? { branch: opts.branch } : {}),
},
});
};

View file

@ -44,10 +44,11 @@ import type { AnalyzeResult } from '../core/run-analyze.js';
* ones (e.g. `isPrimaryBranch?`), so an optional non-serializable field could be
* advertised by the type yet silently dropped by the runtime allowlist.
*
* `isPrimaryBranch` is intentionally excluded: the parent (`api.ts`) reads only
* `repoName`, and nothing consumes `isPrimaryBranch` across this fork (its CLI
* consumer calls `runFullAnalysis` in-process). Add a field here only when a
* server-side IPC consumer actually needs it — and only if it is JSON-safe.
* `isPrimaryBranch` IS on the wire, under exactly the rule this comment used to
* cite for excluding it: a server-side consumer now needs it. `analyze-launch.ts`
* settles the index the run actually wrote, and only the worker knows whether
* core chose the flat slot or a `branches/<slug>/` sub-slot. It is a boolean, so
* it is JSON-safe by construction.
*/
export type AnalyzeResultIpc = Pick<
AnalyzeResult,
@ -58,6 +59,7 @@ export type AnalyzeResultIpc = Pick<
| 'ftsRepairedOnly'
| 'ftsSkipped'
| 'graphWriteCollapsed'
| 'isPrimaryBranch'
>;
/**
@ -78,5 +80,9 @@ export function projectAnalyzeResultForIpc(result: AnalyzeResult): AnalyzeResult
// outcome the CLI does; without it the worker reports a clean `complete`
// for a run whose edges are mostly missing.
graphWriteCollapsed: result.graphWriteCollapsed,
// Tells the parent which slot this run wrote — the flat `.gitnexus` or a
// `branches/<slug>/` sub-slot — so its finalization gate watches the files
// this job actually rewrote (#3199 review).
isPrimaryBranch: result.isPrimaryBranch,
};
}

View file

@ -57,6 +57,7 @@ import { assertString, BadRequestError, createRouteLimiter } from './validation.
import { parseGrepQuery, GREP_TIME_BUDGET_MS } from './grep-params.js';
import { runGrepScanInWorker } from './grep-scan.js';
import {
analyzeCloneOptions,
extractWebRepoName,
getCloneDir,
cloneOrPull,
@ -64,6 +65,10 @@ import {
GITHUB_TOKEN_HOSTS,
} from './git-clone.js';
import { createAnalyzeUploadHandler } from './analyze-upload.js';
// Shared with the CLI's `--branch` (via the analyze-config wrapper) so both
// entry points accept the same refs. Imported from core — not cli/ — so
// createServer does not close a cycle with cli/serve.ts.
import { InvalidBranchError, validateBranchName } from '../core/git-ref.js';
import {
assertServeAuthForPublicOrigin,
createPublicOriginMatcher,
@ -1519,6 +1524,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
springActuatorPath,
asyncApiSpecPath,
token: repoToken,
branch: repoBranch,
} = req.body;
// Input type validation
@ -1550,6 +1556,27 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
return;
}
// Branch: optional index-branch selector, validated with the same rules
// as the CLI's `--branch` so both entry points accept the same refs.
// Rejecting here (rather than letting the clone fail) keeps a malformed
// ref from ever reaching `git`.
if (repoBranch !== undefined && typeof repoBranch !== 'string') {
res.status(400).json({ error: '"branch" must be a string' });
return;
}
let analyzeBranch: string | undefined;
if (repoBranch !== undefined) {
try {
analyzeBranch = validateBranchName(repoBranch, '"branch"');
} catch (err) {
if (err instanceof InvalidBranchError) {
res.status(400).json({ error: err.message });
return;
}
throw err;
}
}
// Token: optional, restricted charset to prevent header smuggling
// (CRLF), bound length, and bound to github.com (see validateAnalyzeToken).
const tokenError = validateAnalyzeToken(repoToken, repoUrl);
@ -1575,7 +1602,11 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
return;
}
const job = jobManager.createJob({ repoUrl, repoPath: repoLocalPath });
const job = jobManager.createJob({
repoUrl,
repoPath: repoLocalPath,
branch: analyzeBranch,
});
// If job was already running (dedup), just return its id. The token is
// not part of the dedup identity and is never stored on the job, so a
@ -1603,11 +1634,15 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
// Clone if URL provided
if (repoUrl && !repoLocalPath) {
const repoName = extractWebRepoName(repoUrl);
targetPath = getCloneDir(repoName);
// Branch-pinned runs get their own clone dir, so they never share
// a working tree with the unpinned one (see getCloneDir).
targetPath = getCloneDir(repoName, analyzeBranch);
jobManager.updateJob(job.id, {
status: 'cloning',
repoName,
// url+branch: same value as registryName (dir basename), not
// the extractWebRepoName stem used only as getCloneDir's first arg.
repoName: analyzeBranch ? path.basename(targetPath) : repoName,
progress: { phase: 'cloning', percent: 0, message: `Cloning ${repoUrl}...` },
});
@ -1619,7 +1654,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
progress: { phase: progress.phase, percent: 5, message: progress.message },
});
},
repoToken ? { token: repoToken } : undefined,
analyzeCloneOptions(repoToken, analyzeBranch),
);
}
@ -1633,6 +1668,20 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
dropEmbeddings,
springActuatorPath,
asyncApiSpecPath,
branch: analyzeBranch,
// Both clone dirs share an `origin`, so the name `registerRepo`
// infers from the remote would be identical and the second one
// would fail with RegistryNameCollisionError. Register the pinned
// clone under its directory name instead: unique per branch, and
// it re-derives through getCloneDir for DELETE /api/repo.
//
// Gated on the SAME condition as the clone above: when a caller
// supplies both `url` and `path` nothing is cloned, and renaming
// the operator's own local repo after its directory would be a
// surprise unrelated to branch pinning.
...(analyzeBranch && repoUrl && !repoLocalPath
? { registryName: path.basename(targetPath) }
: {}),
});
} catch (err: any) {
if (targetPath) releaseRepoLock(getStoragePath(targetPath));

View file

@ -11,6 +11,7 @@ import fs from 'fs/promises';
import os from 'node:os';
import { logger } from '../core/logger.js';
import { getGlobalDir } from '../storage/repo-manager.js';
import { branchSlug } from '../storage/branch-index.js';
import { sanitizeRepoName, stripUrlCredentials } from '../storage/git.js';
import { validateGitUrl } from '../core/net/url-guard.js';
import {
@ -84,14 +85,65 @@ export function extractWebRepoName(url: string): string {
return safeName;
}
/** Get the clone target directory for a repo name. */
export function getCloneDir(repoName: string): string {
/**
* Longest single path component the supported filesystems accept (ext4, APFS,
* NTFS all cap at 255). Compared against `.length`, which equals the byte
* count here because every name this guards is ASCII by construction
* (REPO_NAME_PATTERN and sanitizeRepoName both restrict to `[a-zA-Z0-9._-]`).
*/
const MAX_PATH_COMPONENT_BYTES = 255;
/**
* `branchSlug` for a clone-directory name, trimmed to fit one path component.
*
* `validateBranchName` allows a ref up to 255 characters and `branchSlug`
* appends `-` plus 8 hash characters, so `<repo>__<slug>` can reach 267 — past
* the filesystem limit, and the clone would then fail to create its target
* directory (#3199 review).
*
* Only the READABLE half is trimmed; the 8-character hash is always kept, and
* it is a digest of the full ref, so two long branches that share a prefix
* still get different directories. The slug is not trimmed inside
* `branchSlug` itself because the per-branch *index* slots already use those
* names on disk — shortening them there would orphan existing indexes.
*/
const boundedBranchSegment = (repoName: string, branch: string): string => {
const slug = branchSlug(branch);
if (`${repoName}__${slug}`.length <= MAX_PATH_COMPONENT_BYTES) return slug;
const hash = slug.slice(slug.lastIndexOf('-')); // "-" + 8 hex
const budget = MAX_PATH_COMPONENT_BYTES - repoName.length - '__'.length - hash.length;
// A repo name long enough to leave no budget falls through to the caller's
// length check, which rejects it rather than building an unusable path.
return `${slug.slice(0, Math.max(0, budget))}${hash}`;
};
/** Get the clone target directory for a repo name, optionally pinned to a branch. */
export function getCloneDir(repoName: string, branch?: string): string {
// Re-validate at the boundary even though extractRepoName already checked —
// callers may pass a repoName from another source (test fixtures, scripts).
if (!repoName || repoName === '.' || repoName === '..' || !REPO_NAME_PATTERN.test(repoName)) {
throw new Error('Invalid repository name');
}
return path.join(CLONE_ROOT, repoName);
// A branch-pinned analyze gets its OWN working tree.
//
// Sharing one checkout per repo made `branch` unusable in practice: the tree
// is dirty after any analyze (generated AGENTS.md / CLAUDE.md / .claude/), so
// a pinned request hit `cloneOrPull`'s porcelain refusal; and a later request
// that OMITTED `branch` would pull whatever branch the last pin left checked
// out and index it as the default (#3199 review). Separate directories remove
// both, because the two requests no longer share a tree.
//
// `branchSlug` is the same helper the per-branch index slots use, so the two
// layouts agree on how a ref becomes a path segment. It emits only
// `[a-zA-Z0-9._-]`, so the composed name still satisfies REPO_NAME_PATTERN and
// round-trips through this function — which is how DELETE /api/repo re-derives
// the directory from the registry name.
const dirName = branch ? `${repoName}__${boundedBranchSegment(repoName, branch)}` : repoName;
if (!REPO_NAME_PATTERN.test(dirName) || dirName.length > MAX_PATH_COMPONENT_BYTES) {
throw new Error('Invalid repository name');
}
return path.join(CLONE_ROOT, dirName);
}
export interface CloneProgress {
@ -99,6 +151,29 @@ export interface CloneProgress {
message: string;
}
/**
* Build the `cloneOrPull` options for an `/api/analyze` request.
*
* Extracted from the route so the token/branch combination is unit-testable.
* Inline, the branch-only case was the one nothing asserted: every existing
* test still passed if `branch` were dropped whenever no token was supplied —
* i.e. silently cloning the default branch for every public URL, which is the
* exact behavior #3198 is about (#3199 review).
*
* Returns `undefined` rather than `{}` when neither is set, because that is
* what `cloneOrPull` treats as "no options" at its own call sites.
*/
export function analyzeCloneOptions(
token?: string,
branch?: string,
): Pick<CloneOrPullOptions, 'token' | 'branch'> | undefined {
if (!token && !branch) return undefined;
return {
...(token ? { token } : {}),
...(branch ? { branch } : {}),
};
}
export interface CloneOrPullOptions {
token?: string;
allowedCloneRoot?: string;
@ -294,11 +369,127 @@ export async function assertRemoteMatchesRequestedUrl(
}
}
/**
* Fetch refspec that updates `origin/<branch>` from `refs/heads/<branch>`.
*
* The leading `+` is git's dest-update prefix (`+refs/heads/*:refs/remotes/origin/*`
* is what `git clone` writes into `.git/config`). Without it, `fetch --depth 1`
* refuses to move `origin/<branch>` when the shallow history cannot prove a
* fast-forward — so a same-branch re-index stays stuck on the old tip.
*
* The user string is interpolated inside `refs/heads/…`, never as a raw pull
* dest. A branch named `+develop` becomes `+refs/heads/+develop:…`, not a
* force-update of `develop`.
*/
function branchFetchRefspec(branch: string): string {
return `+refs/heads/${branch}:refs/remotes/origin/${branch}`;
}
/** Overlays `analyze` writes into a clone; they must not block a same-ref update. */
const GITNEXUS_GENERATED_OVERLAYS = ['./AGENTS.md', './CLAUDE.md', './.claude'] as const;
/**
* Restore only GitNexus-generated overlays so a same-ref update is not
* blocked by analyze dirt. Path-limited and root-anchored (`./`): tracked
* files are checked out from HEAD; untracked overlays (including gitignored
* ones — `AGENTS.md` / `.claude/` are commonly ignored) are `git clean -fdx`'d.
* A slash-free `AGENTS.md` would also hit `docs/AGENTS.md`. Never a
* whole-clone `git clean --force -d`.
*/
async function restoreGitNexusGeneratedOverlays(
runGitImpl: typeof runGit,
cwd: string,
gitOpts: RunGitOptions,
): Promise<void> {
for (const overlay of GITNEXUS_GENERATED_OVERLAYS) {
const listed = (await runGitImpl(['ls-files', '--', overlay], cwd, gitOpts)).trim();
if (!listed) continue;
await runGitImpl(['checkout', 'HEAD', '--', overlay], cwd, gitOpts);
}
// Path-limited: untracked analyze output still blocks checkout when the
// incoming tree has the same path, and otherwise leaves a dirty tree to
// index. `-x` is required because these overlays are often gitignored.
// Never a whole-clone `git clean --force -d`.
await runGitImpl(['clean', '-fdx', '--', ...GITNEXUS_GENERATED_OVERLAYS], cwd, gitOpts);
}
/**
* True when the working tree is already at the requested pin: either HEAD is
* that named branch, or HEAD is detached at the same SHA as `branch` /
* `origin/<branch>`. A missing ref falls through to the switch path.
*/
async function matchRequestedRef(
runGitImpl: typeof runGit,
cwd: string,
branch: string,
gitOpts: RunGitOptions,
): Promise<'branch' | 'sha' | undefined> {
const abbrev = (await runGitImpl(['rev-parse', '--abbrev-ref', 'HEAD'], cwd, gitOpts)).trim();
if (abbrev === branch) return 'branch';
// Detached HEAD reports `HEAD`; compare SHAs so a tag/SHA pin is not a switch.
if (abbrev !== 'HEAD') return undefined;
let headSha: string;
try {
headSha = (await runGitImpl(['rev-parse', 'HEAD'], cwd, gitOpts)).trim();
} catch {
return undefined;
}
for (const candidate of [branch, `origin/${branch}`] as const) {
try {
// Peel annotated tags (`v1.0` is a tag object; HEAD is the commit).
const requestedSha = (
await runGitImpl(['rev-parse', `${candidate}^{commit}`], cwd, gitOpts)
).trim();
if (requestedSha && requestedSha === headSha) return 'sha';
} catch {
// Ref missing — try origin/<branch>, then the switch path.
}
}
return undefined;
}
async function fetchAndCheckoutRequestedBranch(
runGitImpl: typeof runGit,
cwd: string,
branch: string,
gitOpts: RunGitOptions,
): Promise<void> {
// Analyze clones are `--depth 1`. `merge --ff-only` cannot walk O→N when
// the remote moved 2+ commits (the merge-base is not in the shallow
// history). `checkout -B` points the local branch at the fetched tip —
// same as the switch path, no ancestry walk. No `--force`: leftover
// non-overlay dirt still refuses.
await runGitImpl(['fetch', '--depth', '1', 'origin', branchFetchRefspec(branch)], cwd, gitOpts);
await runGitImpl(['checkout', '-B', branch, `origin/${branch}`], cwd, gitOpts);
}
/**
* Clone or pull a git repository.
* If targetDir doesn't exist: git clone --depth 1
* If targetDir exists with .git: git pull --ff-only (after verifying the
* existing clone's remote.origin matches the requested URL).
*
* If targetDir doesn't exist: git clone --depth 1, adding `--branch <branch>`
* when one is requested.
*
* If targetDir exists with .git, its remote.origin is verified against the
* requested URL first, and then the branch decides the update:
* - no `options.branch`: git pull --ff-only, which updates the current
* branch in place via its configured upstream. Nothing moves, so no
* dirty-tree check applies.
* - a `options.branch` that is ALREADY the current named branch: restore
* GitNexus overlays, then fetch via
* `+refs/heads/<branch>:refs/remotes/origin/<branch>` and
* `checkout -B <branch> origin/<branch>` (shallow clones cannot
* `merge --ff-only` across a 2+ commit move). Never a raw
* `origin <branch>` pull refspec. No porcelain refuse.
* - a detached HEAD whose SHA already matches the requested ref (tag /
* SHA pin): restore overlays only. Do not fetch/merge — a same-named
* branch could otherwise fast-forward the pin past the tag.
* - a `options.branch` that DIFFERS from the current pin: fetch that ref,
* then `checkout -B <branch> origin/<branch>` — so the requested branch,
* not the one already checked out, is what ends up in the working tree.
* This is the switching case, and it refuses a dirty tree unless
* `overwriteLocalChanges` is set.
*
* Security:
* - targetDir must resolve inside CLONE_ROOT (~/.gitnexus/repos/). The
@ -382,13 +573,36 @@ export async function cloneOrPull(
await assertRemoteMatchesRequestedUrl(safeTarget, url, options?.timeoutMs);
onProgress?.({ phase: 'pulling', message: 'Pulling latest changes...' });
const runGitImpl = options?.runGitForTest ?? runGit;
if (options?.branch) {
const gitOpts = {
token: options?.token,
url,
timeoutMs: options?.timeoutMs,
};
// Already at the requested pin? Then there is no switch to make, so do
// not take the checkout path below — that would run the porcelain check
// against a tree ANALYZE ITSELF dirtied (it writes AGENTS.md / CLAUDE.md /
// .claude/ into the clone), which made a pinned RE-index impossible: the
// first pin succeeded and every later one failed asking for
// `overwrite_local_changes`, a flag this route deliberately does not pass
// because it would `git clean --force -d` the directory (#3199 review).
//
// "Already there" is a named-branch match OR a detached HEAD whose SHA
// equals `branch` / `origin/<branch>` (tag / SHA pin). A missing ref
// falls through to the switch path, which still refuses a dirty tree.
//
// Same-named-branch update uses the heads/ → remotes/ fetch refspec plus
// `checkout -B <branch> origin/<branch>`, never `pull origin <user-string>`
// (a leading `+` would otherwise be a force-fetch). Only
// `remote.origin.url` is verified above; `branch.<name>.remote` /
// `.merge` are not, so an implicit-upstream pull can update a different
// ref while the job still carries this branch (#3199 review).
const requestedRefMatch = options?.branch
? await matchRequestedRef(runGitImpl, safeTarget, options.branch, gitOpts)
: undefined;
if (options?.branch && !requestedRefMatch) {
if (!options.overwriteLocalChanges) {
const status = await runGitImpl(['status', '--porcelain'], safeTarget, {
token: options?.token,
url,
timeoutMs: options?.timeoutMs,
});
const status = await runGitImpl(['status', '--porcelain'], safeTarget, gitOpts);
if (status.trim()) {
throw new Error(
`Refusing to update ${safeTarget}: local changes detected. Set overwrite_local_changes: true to overwrite them.`,
@ -396,19 +610,9 @@ export async function cloneOrPull(
}
}
await runGitImpl(
[
'fetch',
'--depth',
'1',
'origin',
`refs/heads/${options.branch}:refs/remotes/origin/${options.branch}`,
],
['fetch', '--depth', '1', 'origin', branchFetchRefspec(options.branch)],
safeTarget,
{
token: options?.token,
url,
timeoutMs: options?.timeoutMs,
},
gitOpts,
);
await runGitImpl(
[
@ -419,11 +623,7 @@ export async function cloneOrPull(
`origin/${options.branch}`,
],
safeTarget,
{
token: options?.token,
url,
timeoutMs: options?.timeoutMs,
},
gitOpts,
);
if (options.overwriteLocalChanges) {
// `checkout --force` rewrites tracked files only, so untracked sources
@ -432,18 +632,17 @@ export async function cloneOrPull(
// ignored paths must survive, and `-e /.gitnexus` is belt-and-braces
// because `.git/info/exclude` is skipped on a read-only storage mount
// and a freshly cloned repo may not have been analyzed yet at all.
await runGitImpl(['clean', '--force', '-d', '-e', '/.gitnexus'], safeTarget, {
token: options?.token,
url,
timeoutMs: options?.timeoutMs,
});
await runGitImpl(['clean', '--force', '-d', '-e', '/.gitnexus'], safeTarget, gitOpts);
}
} else if (options?.branch && requestedRefMatch === 'branch') {
await restoreGitNexusGeneratedOverlays(runGitImpl, safeTarget, gitOpts);
await fetchAndCheckoutRequestedBranch(runGitImpl, safeTarget, options.branch, gitOpts);
} else if (options?.branch && requestedRefMatch === 'sha') {
// Tag / SHA pin: already at the requested commit. Fetching
// `refs/heads/<name>` would follow a same-named branch past the pin.
await restoreGitNexusGeneratedOverlays(runGitImpl, safeTarget, gitOpts);
} else {
await runGitImpl(['pull', '--ff-only'], safeTarget, {
token: options?.token,
url,
timeoutMs: options?.timeoutMs,
});
await runGitImpl(['pull', '--ff-only'], safeTarget, gitOpts);
}
} else {
if (targetExists && (await fs.readdir(safeTarget)).length > 0) {

View file

@ -3,7 +3,7 @@ import fs from 'node:fs/promises';
import os from 'node:os';
import path from 'node:path';
import { setTimeout as sleep } from 'node:timers/promises';
import { isProcessAlive, readProcessStartTime } from '../utils/process-identity.js';
import { isProcessAlive, readProcessStartTimeCached } from '../utils/process-identity.js';
const HOSTNAME = os.hostname();
@ -46,7 +46,9 @@ export async function acquireFileLock(
pid,
ownerId: crypto.randomUUID(),
processStartTime:
options.processStartTime ?? (options.readProcessStartTime ?? readProcessStartTime)(pid) ?? '',
options.processStartTime ??
(options.readProcessStartTime ?? readProcessStartTimeCached)(pid) ??
'',
hostname: options.hostname ?? HOSTNAME,
};
if (!owner.processStartTime) {
@ -69,7 +71,7 @@ export async function acquireFileLock(
resolvedPath,
owner,
options.isProcessAlive ?? isProcessAlive,
options.readProcessStartTime ?? readProcessStartTime,
options.readProcessStartTime ?? readProcessStartTimeCached,
)
) {
continue;

View file

@ -169,6 +169,7 @@ export const getParsedFileStoreDir = (storagePath: string): string =>
/** Remove any prior run's shards so a fresh parse starts clean. Idempotent. */
export const clearParsedFileStore = async (storagePath: string): Promise<void> => {
await fs.rm(getParsedFileStoreDir(storagePath), { recursive: true, force: true });
forgetShardListings();
};
const isV8ShardName = (name: string): boolean => name.endsWith('.v8') && !name.includes('.v8.');
@ -245,13 +246,52 @@ const listV8Shards = async (dir: string): Promise<string[]> => {
}
};
/**
* Per-run memo of each shard's authenticated path listing, so the SECOND and
* later passes over the store can decide "this shard holds nothing I want"
* without reopening it.
*
* Scope resolution calls `loadParsedFilesForPaths` once per language, and each
* call walks every shard in the store. The skip decision needs the envelope's
* path listing, and that listing is only trustworthy once the payload digest
* has been checked — so today a pass that wants 50 Python files still reads and
* SHA-256s all ~300MB of a TypeScript-dominated store to prove it can skip it.
* Measured on a 2234-file repo: a pass wanting a single file costs 335ms, and
* the three language passes together spend 4426ms in here. The cost scales with
* LANGUAGE COUNT, so a polyglot repo pays it worst.
*
* Keyed by size+mtime as well as name. Shard names are content-addressed
* (parse-chunk hash + worker id), so a name collision across different content
* should be impossible — but that invariant lives in the parse-cache keying,
* not here, and one `stat` per shard is a few ms against the hundreds this
* saves. The memo holds one store directory at a time: a different `dir` (a new
* repo in a long-lived MCP process, or a wiped store) drops the previous set
* rather than accumulating.
*/
let shardListingMemo: { dir: string; byIdentity: Map<string, readonly string[]> } | undefined;
const shardIdentity = (file: string, size: number, mtimeMs: number): string =>
`${file}\0${size}\0${mtimeMs}`;
const memoFor = (dir: string): Map<string, readonly string[]> => {
if (shardListingMemo?.dir !== dir) shardListingMemo = { dir, byIdentity: new Map() };
return shardListingMemo.byIdentity;
};
/** Drop the memo when the store it describes is removed. */
const forgetShardListings = (): void => {
shardListingMemo = undefined;
};
export const loadParsedFilesForPaths = async (
storagePath: string,
wantPaths: ReadonlySet<string>,
): Promise<Map<string, ParsedFile>> => {
const out = new Map<string, ParsedFile>();
if (wantPaths.size === 0) return out;
const shardPaths = await listV8Shards(getParsedFileStoreDir(storagePath));
const storeDir = getParsedFileStoreDir(storagePath);
const shardPaths = await listV8Shards(storeDir);
const listings = memoFor(storeDir);
const pool = new Map<string, string>();
let droppedSites = 0;
let filesWithDroppedSites = 0;
@ -274,11 +314,28 @@ export const loadParsedFilesForPaths = async (
}
};
for (const shardFull of shardPaths) {
// Re-stat rather than trusting the name alone, then reuse this run's
// authenticated listing to skip without opening the file. A shard with no
// memoized listing — first pass, or an envelope whose listing did not
// validate — falls through to the full read, so this only ever removes
// work that a later pass had already proved unnecessary.
const st = await fs.stat(shardFull).catch(() => undefined);
const identity = st ? shardIdentity(shardFull, st.size, st.mtimeMs) : undefined;
const memoized = identity === undefined ? undefined : listings.get(identity);
if (memoized !== undefined && !memoized.some((p) => wantPaths.has(p))) {
bytesSinceGc += st?.size ?? 0;
await maybeYieldAndGc(bytesSinceGc >= parsedFileLoadGc.byteBudget);
continue;
}
const loaded = await tryLoadV8Cache(shardFull, pool, wantPaths);
if (loaded === undefined) {
await maybeYieldAndGc(false);
continue;
}
if (identity !== undefined && loaded.paths !== undefined) {
listings.set(identity, loaded.paths);
}
if (loaded.kind === 'skip') {
bytesSinceGc += loaded.bytes;
await maybeYieldAndGc(bytesSinceGc >= parsedFileLoadGc.byteBudget);

View file

@ -185,8 +185,22 @@ const decodePrefix = (buf: Buffer): EnvelopeMeta | undefined => {
const runtimeCompatible = (meta: EnvelopeMeta): boolean =>
meta.recordedNodeMajor === nodeMajor() && meta.recordedV8 === process.versions.v8;
export type V8CacheHit = { kind: 'hit'; value: unknown; bytes: number };
export type V8CacheSkip = { kind: 'skip'; bytes: number };
/**
* `paths` is the envelope's digest-authenticated path listing, present only
* when it parsed and its entry count matched the recorded `pathCount` — i.e.
* exactly when the skip decision below is willing to trust it. A caller that
* loads the same immutable shard more than once per run can memoize it and
* make its own skip decision without reopening the file (see
* `parsedfile-store.ts`). Absent means "this envelope has no listing worth
* trusting", which is fail-closed: load, never skip.
*/
export type V8CacheHit = {
kind: 'hit';
value: unknown;
bytes: number;
paths?: readonly string[];
};
export type V8CacheSkip = { kind: 'skip'; bytes: number; paths: readonly string[] };
export type V8CacheLoad = V8CacheHit | V8CacheSkip;
export type V8CacheInspection = { paths: readonly string[] };
@ -294,20 +308,29 @@ export const tryLoadV8Cache = async (
const payload = await readVerifiedPayload(fh, meta, pathRaw);
if (!payload) return undefined;
if (wantPaths && wantPaths.size > 0 && meta.pathBytes > 0) {
// Parsed once and reused for both the skip decision and the returned
// listing, so a caller that memoizes it is trusting exactly the bytes this
// function was already willing to skip on. Anything that fails these checks
// stays `undefined` and falls through to the load.
let listedPaths: readonly string[] | undefined;
if (meta.pathBytes > 0) {
const listed = parseCachePathListing(pathRaw);
if (
listed !== null &&
listed.length === meta.pathCount &&
listed.length > 0 &&
!listed.some((p) => wantPaths.has(p))
) {
return { kind: 'skip', bytes: st.size };
if (listed !== null && listed.length === meta.pathCount && listed.length > 0) {
listedPaths = listed;
}
}
if (
wantPaths &&
wantPaths.size > 0 &&
listedPaths &&
!listedPaths.some((p) => wantPaths.has(p))
) {
return { kind: 'skip', bytes: st.size, paths: listedPaths };
}
const value = v8.deserialize(payload);
if (internPool) internGraphStrings(value, internPool);
return { kind: 'hit', value, bytes: st.size };
return { kind: 'hit', value, bytes: st.size, paths: listedPaths };
} catch (err) {
if (!isEnoent(err)) {
logger.debug({ err, filePath }, 'v8 cache: load failed; treating as miss');

View file

@ -38,3 +38,22 @@ export function readProcessStartTime(pid: number): string | undefined {
return undefined;
}
}
let ownStartTime: string | undefined;
/**
* `readProcessStartTime`, except this process's own start time is probed once.
* It cannot change while we are running, and every `acquireFileLock` — plus
* each retry attempt and each stale-lock reclaim guard — stamps the owner file
* with it. On Windows that probe is a `powershell.exe` spawn and a WMI query,
* so a process taking several locks pays it several times for one constant.
*
* A foreign pid is never cached: that process can exit and its pid can be
* reused, which is the very thing the stamp exists to detect. A failed probe
* is not cached either — one transient failure would otherwise leave the
* process unable to take a lock for its whole lifetime.
*/
export function readProcessStartTimeCached(pid: number): string | undefined {
if (pid !== process.pid) return readProcessStartTime(pid);
return (ownStartTime ??= readProcessStartTime(pid));
}

View file

@ -9,6 +9,9 @@
// Cross-file alias — should resolve `cfg.FOO`, `cfg.BAR` against cfg.zig.
const cfg = @import("./cfg.zig");
const cfg_no_ext = @import("./cfg");
const wrapped_cfg = wrap(@import("./cfg.zig"));
const shadowed_cfg = @import("./cfg.zig");
pub const UPGRADERS_ENABLED: bool = false;
pub const DEBUG: bool = true;
@ -30,6 +33,7 @@ pub const CYCLE_B = CYCLE_A;
pub const ALIAS_TO_VAR = IS_RUNTIME_FLAG_FALSE;
pub fn run() void {
const local_cfg = @import("./cfg.zig");
// Live: not under any if-gate.
live_unconditional();
@ -162,6 +166,18 @@ pub fn run() void {
if (cfg.NOT_A_BOOL != 0) {
live_cross_file_not_bool();
}
if (cfg_no_ext.FOO) {
gated_extensionless_cross_file_foo();
}
// Function-local imports use the same workspace constants.
if (local_cfg.FOO) {
gated_local_cross_file_foo();
}
// An import nested inside another initializer does not bind the variable
// directly to that module, so its members must remain unknown/fail-open.
if (wrapped_cfg.FOO) {
live_wrapped_cross_file_foo();
}
// Bare literal gate: no constant table involved, but it must still be gated.
if (false) {
@ -205,10 +221,37 @@ pub fn run() void {
_ = e3;
}
pub fn run_shadowed_alias() void {
const shadowed_cfg = @import("./other.zig");
if (shadowed_cfg.FOO) {
live_shadowed_cross_file_foo();
}
}
fn live_unconditional() void {
_ = 1;
}
fn wrap(value: anytype) @TypeOf(value) {
return value;
}
fn gated_local_cross_file_foo() void {
_ = 1;
}
fn gated_extensionless_cross_file_foo() void {
_ = 1;
}
fn live_wrapped_cross_file_foo() void {
_ = 1;
}
fn live_shadowed_cross_file_foo() void {
_ = 1;
}
fn gated_simple() void {
_ = 1;
}

View file

@ -0,0 +1 @@
pub const FOO: bool = true;

View file

@ -127,12 +127,26 @@ describe('CLI update notice subprocess behavior', () => {
'dir',
);
// The refresh child parks inside fetch() until this test releases it. That
// orders the parent's exit against work that is provably still in flight,
// instead of racing it against a wall-clock budget: the child's real cost
// (node boot, tsx transpile, a lock acquisition that shells out to
// ps/powershell) has no bounded upper limit on a loaded CI runner.
const started = path.join(home, 'refresh-started');
const release = path.join(home, 'refresh-release');
const preload = path.join(home, 'mock-refresh.mjs');
fs.writeFileSync(
preload,
`Object.defineProperty(process.stderr, 'isTTY', { value: true, configurable: true });
`import fs from 'node:fs';
Object.defineProperty(process.stderr, 'isTTY', { value: true, configurable: true });
globalThis.fetch = async () => {
await new Promise((resolve) => setTimeout(resolve, 750));
fs.writeFileSync(${JSON.stringify(started)}, '');
// Bounded so an abandoned child (test failed before releasing, temp home
// already deleted) still exits instead of spinning forever.
const deadline = Date.now() + 60_000;
while (!fs.existsSync(${JSON.stringify(release)}) && Date.now() < deadline) {
await new Promise((resolve) => setTimeout(resolve, 25));
}
return new Response(JSON.stringify({ version: '99.0.0' }), {
status: 200,
headers: { 'content-type': 'application/json' },
@ -141,7 +155,6 @@ globalThis.fetch = async () => {
`,
);
const startedAt = Date.now();
await new Promise<void>((resolve, reject) => {
const parent = spawn(
process.execPath,
@ -162,17 +175,20 @@ globalThis.fetch = async () => {
else reject(new Error(`notifier parent exited ${String(code)}`));
});
});
const elapsed = Date.now() - startedAt;
expect(elapsed).toBeLessThan(1_800);
// The parent already exited above, so reaching a still-parked child proves
// the refresh outlived it and was never awaited.
const cache = path.join(home, 'update-check.json');
await expect.poll(() => fs.existsSync(started), { timeout: 30_000, interval: 50 }).toBe(true);
expect(fs.existsSync(cache)).toBe(false);
await expect.poll(() => fs.existsSync(cache), { timeout: 15_000, interval: 100 }).toBe(true);
fs.writeFileSync(release, '');
await expect.poll(() => fs.existsSync(cache), { timeout: 30_000, interval: 50 }).toBe(true);
expect(JSON.parse(fs.readFileSync(cache, 'utf8'))).toMatchObject({
latestVersion: '99.0.0',
registry: 'https://registry.npmjs.org',
});
}, 20_000);
}, 90_000);
it('prints the localized notice on a forced-TTY stderr and keeps stdout clean', () => {
const home = tempHome();

View file

@ -0,0 +1,225 @@
/**
* Dispatch rounds — batching cache packs into one pool round.
*
* `WorkerPool.dispatch` is a barrier, so one round-trip per parse-cache pack
* strands the pool whenever a pack is smaller than it. Packs are keyed by
* `(language, hash(path) % 128)`, so on a real repo most of them are: this
* repository produces 1285 packs where the byte budget alone needs 16, and 549
* of those hold a single file. Chunks now accumulate into a round bounded by
* `GITNEXUS_PARSE_ROUND_BYTES` and go out in one `dispatchGroups` call.
*
* Batching must be invisible to the graph. These tests pin the two ways it
* could stop being invisible:
* 1. Ordering — deferred aggregation runs in `chunkIdx` order, so the graph
* must not depend on how chunks were grouped into rounds.
* 2. Attribution — a round returns one result array per pack, so a pack's
* parse-cache entry must hold ITS OWN worker output. Mis-attribution would
* survive a cold run and only surface as a corrupted warm replay, which is
* what the second test exercises.
*/
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
import { runChunkedParseAndResolve } from '../../src/core/ingestion/pipeline-phases/parse-impl.js';
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
import { PARSE_CACHE_VERSION, packParseCacheChunks } from '../../src/storage/parse-cache.js';
import type { ParseWorkerResult } from '../../src/core/ingestion/workers/parse-worker.js';
const ORIGINAL_ROUND_BYTES = process.env.GITNEXUS_PARSE_ROUND_BYTES;
/**
* Enough files, across enough languages, that `(language, bucket)` packing
* yields many more packs than the byte budget would — the shape that makes
* per-pack dispatch a barrier problem in the first place.
*/
const FIXTURE: ReadonlyArray<[string, string]> = [
...Array.from({ length: 12 }, (_, i): [string, string] => [
`src/mod${i}.ts`,
`export function ts${i}() { return ${i}; }\n`,
]),
...Array.from({ length: 8 }, (_, i): [string, string] => [
`src/mod${i}.py`,
`def py${i}():\n return ${i}\n`,
]),
...Array.from({ length: 6 }, (_, i): [string, string] => [
`src/Mod${i}.java`,
`public class Mod${i} { public int go() { return ${i}; } }\n`,
]),
...Array.from({ length: 6 }, (_, i): [string, string] => [
`src/mod${i}.go`,
`package main\n\nfunc Go${i}() int { return ${i} }\n`,
]),
];
describe('parse-impl dispatch rounds', () => {
let repoPath = '';
let storageDir = '';
beforeEach(() => {
repoPath = fs.mkdtempSync(path.join(os.tmpdir(), 'parse-impl-dispatch-rounds-'));
storageDir = fs.mkdtempSync(path.join(os.tmpdir(), 'parse-impl-rounds-storage-'));
for (const [rel, content] of FIXTURE) {
const full = path.join(repoPath, rel);
fs.mkdirSync(path.dirname(full), { recursive: true });
fs.writeFileSync(full, content);
}
});
afterEach(() => {
for (const dir of [repoPath, storageDir]) {
if (dir && fs.existsSync(dir)) fs.rmSync(dir, { recursive: true, force: true });
}
if (ORIGINAL_ROUND_BYTES === undefined) delete process.env.GITNEXUS_PARSE_ROUND_BYTES;
else process.env.GITNEXUS_PARSE_ROUND_BYTES = ORIGINAL_ROUND_BYTES;
});
const files = () =>
FIXTURE.map(([rel]) => ({ path: rel, size: fs.statSync(path.join(repoPath, rel)).size }));
/**
* Order-independent fingerprint of the graph. Counts alone would let a
* mis-attributed chunk (right totals, wrong contents) pass.
*/
const fingerprint = (graph: ReturnType<typeof createKnowledgeGraph>): string =>
Array.from(graph.nodes.values())
.map((node) => {
const props = node.properties as { name?: string; filePath?: string } | undefined;
return `${node.label}|${props?.name ?? ''}|${props?.filePath ?? ''}`;
})
.sort()
.join('\n');
const run = async (parseCache?: {
version: string;
entries: Map<string, ParseWorkerResult[]>;
usedKeys: Set<string>;
storagePath: string;
onDiskKeys: Set<string>;
}) => {
const scan = files();
const rels = scan.map((f) => f.path);
const graph = createKnowledgeGraph();
await runChunkedParseAndResolve(
graph,
scan,
rels,
scan.length,
repoPath,
Date.now(),
() => {},
parseCache ? { parseCache } : {},
);
return graph;
};
it('the fixture really does split into more packs than the byte budget needs', () => {
// Guards the premise: if packing ever stopped over-splitting, the tests
// below would still pass while measuring nothing.
const packs = packParseCacheChunks(
files().map((f) => ({
path: f.path,
size: f.size,
language: f.path.slice(f.path.lastIndexOf('.') + 1),
})),
2 * 1024 * 1024,
);
const totalBytes = files().reduce((sum, f) => sum + f.size, 0);
expect(totalBytes).toBeLessThan(2 * 1024 * 1024);
expect(packs.length).toBeGreaterThan(1);
});
it('produces the same graph whether chunks are batched into rounds or dispatched one by one', async () => {
// 1 byte closes a round after every cache-missing chunk — the pre-round
// behaviour, and the control arm for the batched default.
process.env.GITNEXUS_PARSE_ROUND_BYTES = '1';
const perChunk = await run();
delete process.env.GITNEXUS_PARSE_ROUND_BYTES;
const batched = await run();
expect(batched.nodeCount).toBe(perChunk.nodeCount);
expect(batched.relationshipCount).toBe(perChunk.relationshipCount);
expect(fingerprint(batched)).toBe(fingerprint(perChunk));
// Pin real symbols so an empty-graph regression cannot satisfy the above.
const names = fingerprint(batched);
expect(names).toContain('ts0');
expect(names).toContain('py0');
expect(names).toContain('Mod0');
expect(names).toContain('Go0');
});
it('keeps hit and miss chunks attributed to their own files inside one round', async () => {
// The realistic incremental shape: some packs warm, some cold, batched into
// the SAME round. `drainRound` walks the round's entries in `chunkIdx`
// order but pulls worker output with a separate `missIdx` cursor, so a
// hit sitting between two misses is exactly where that cursor can slip.
// Cold-then-warm alone never exercises it -- every entry is the same kind.
const cache = {
version: PARSE_CACHE_VERSION,
entries: new Map<string, ParseWorkerResult[]>(),
usedKeys: new Set<string>(),
storagePath: storageDir,
onDiskKeys: new Set<string>(),
};
const cold = await run(cache);
const cachedPacks = cache.onDiskKeys.size + cache.entries.size;
expect(cachedPacks).toBeGreaterThan(1);
// Edit ONE file. Its pack now misses; every other pack still hits, so the
// next run's rounds carry both kinds together.
fs.writeFileSync(
path.join(repoPath, 'src/mod0.ts'),
'export function ts0() { return 999; }\nexport function ts0Extra() { return 1; }\n',
);
const mixed = await run(cache);
// The edited file's NEW symbol must be present, proving the miss chunk's
// fresh worker output landed under its own file...
const mixedPrint = fingerprint(mixed);
expect(mixedPrint).toContain('ts0Extra');
// ...and every untouched file's symbols must still be present and attached
// to their own paths, proving no hit chunk was overwritten by, or swapped
// with, a neighbouring miss chunk's results.
const coldPrint = fingerprint(cold);
const untouched = coldPrint
.split('\n')
.filter((entry) => !entry.endsWith('|src/mod0.ts'))
.sort();
const mixedUntouched = mixedPrint
.split('\n')
.filter((entry) => !entry.endsWith('|src/mod0.ts'))
.sort();
expect(mixedUntouched).toEqual(untouched);
});
it('stores each pack’s own worker output, so a warm replay reproduces the cold graph', async () => {
const cache = {
version: PARSE_CACHE_VERSION,
entries: new Map<string, ParseWorkerResult[]>(),
usedKeys: new Set<string>(),
storagePath: storageDir,
onDiskKeys: new Set<string>(),
};
// Cold: every pack misses, and the round writes each pack's results under
// that pack's own hash.
const cold = await run(cache);
// With a storagePath the chunk bodies land on disk and the hash is tracked
// in `onDiskKeys`; without one they stay in `entries`. Count both so the
// assertion pins "more than one pack was cached", not the storage route.
expect(cache.onDiskKeys.size + cache.entries.size).toBeGreaterThan(1);
expect(cache.usedKeys.size).toBe(cache.onDiskKeys.size + cache.entries.size);
// Warm: every pack replays from its cache entry with no worker dispatch.
// If a round had attributed pack A's results to pack B's key, the replayed
// graph would differ here even though the cold run looked correct.
const warm = await run(cache);
expect(fingerprint(warm)).toBe(fingerprint(cold));
expect(warm.nodeCount).toBe(cold.nodeCount);
expect(warm.relationshipCount).toBe(cold.relationshipCount);
});
});

View file

@ -50,11 +50,13 @@ function scanned(repo: string, files: string[]) {
}
/**
* Capture every per-chunk progress message emitted during a run.
* parse-impl emits one per chunk in the "Parsing chunk X/Y" form, so
* counting unique chunk indices in the captured stream is a stable
* proxy for the number of chunks the loop actually produced. Avoids
* exposing internal counter state from parse-impl.
* Read the chunk count out of the progress stream. parse-impl reports progress
* as "Parsing chunk X/Y" for a single chunk and "Parsing chunks X-Z/Y" when a
* dispatch round batches several — so the DENOMINATOR, not the number of
* distinct messages, is the count of packs the loop produced. Reading `Y`
* keeps this independent of how chunks are grouped into rounds while still
* exercising the real budget-resolution path inside
* `runChunkedParseAndResolve`, rather than re-deriving packs in the test.
*/
async function countChunksFromProgress(
repoPath: string,
@ -63,7 +65,7 @@ async function countChunksFromProgress(
): Promise<number> {
const scan = scanned(repoPath, files);
const graph = createKnowledgeGraph();
const chunkIndices = new Set<string>();
const totals = new Set<number>();
await runChunkedParseAndResolve(
graph,
scan,
@ -73,8 +75,8 @@ async function countChunksFromProgress(
Date.now(),
(p) => {
if (typeof p.message !== 'string') return;
const m = /Parsing chunk (\d+)\/(\d+)/.exec(p.message);
if (m !== null) chunkIndices.add(`${m[1]}/${m[2]}`);
const m = /Parsing chunks? \d+(?:-\d+)?\/(\d+)/.exec(p.message);
if (m !== null) totals.add(Number(m[1]));
},
// Chunk count is byte-budget-driven and emitted before the pool runs, so it
// is independent of worker vs sequential. Sequential parsing was removed, so
@ -82,7 +84,10 @@ async function countChunksFromProgress(
// integration tier.
{ ...options },
);
return chunkIndices.size;
// Every message in a run carries the same denominator; more than one value
// would mean the loop changed its chunk count mid-run.
expect(totals.size).toBeLessThanOrEqual(1);
return totals.values().next().value ?? 0;
}
describe('parse-impl chunkByteBudget resolution (U14 / F7)', () => {

View file

@ -7,8 +7,11 @@
* such branches keep `staticGated` falsy.
*/
import { describe, it, expect, beforeAll } from 'vitest';
import fs from 'node:fs';
import path from 'path';
import { FIXTURES, getRelationships, runPipelineFromRepo, type PipelineResult } from './helpers.js';
import { populateZigWorkspaceStaticGating } from '../../../src/core/ingestion/languages/zig/workspace-static-gating.js';
import type { ParsedFile } from 'gitnexus-shared';
describe('Zig static-gated edges', () => {
let result: PipelineResult;
@ -174,12 +177,9 @@ describe('Zig static-gated edges', () => {
expect(isGated('gated_chain_tail')).toBe(true);
});
// Cross-file positive cases: the gating module resolves `alias.NAME` through
// `lookupBoolsForPath`, but the scope-capture emitter runs per file in the
// parse worker with only `{ path, content }` in hand — no sibling sources —
// so v1 stamps file-local constants only. Re-enable once the emitter can
// see imported files (see PR description, "Cross-file constants").
it.skip('tags `if (cfg.FOO)` cross-file when FOO is false in cfg.zig (tracked: #3162)', () => {
// Cross-file cases are enriched after per-file extraction, once sibling
// source facts are available but before reference finalization.
it('tags `if (cfg.FOO)` cross-file when FOO is false in cfg.zig', () => {
expect(isGated('gated_cross_file_foo')).toBe(true);
});
@ -187,7 +187,7 @@ describe('Zig static-gated edges', () => {
expect(isGated('live_cross_file_bar')).toBe(false);
});
it.skip('tags the ELSE branch of `if (cfg.BAR)` when BAR is true (tracked: #3162)', () => {
it('tags the ELSE branch of `if (cfg.BAR)` when BAR is true', () => {
expect(isGated('gated_cross_file_else')).toBe(true);
});
@ -198,4 +198,57 @@ describe('Zig static-gated edges', () => {
it('does NOT tag `cfg.NOT_A_BOOL != 0` (imported decl is not a bool literal)', () => {
expect(isGated('live_cross_file_not_bool')).toBe(false);
});
it('resolves a relative cross-file import with an omitted .zig extension', () => {
expect(isGated('gated_extensionless_cross_file_foo')).toBe(true);
});
it('tags a cross-file bool accessed through a function-local import alias', () => {
expect(isGated('gated_local_cross_file_foo')).toBe(true);
});
it('does NOT treat a nested @import as the declaration direct module alias', () => {
expect(isGated('live_wrapped_cross_file_foo')).toBe(false);
});
it('fails open when an import alias is shadowed in another lexical scope', () => {
expect(isGated('live_shadowed_cross_file_foo')).toBe(false);
});
it('replaces a frozen parsed file instead of mutating it', () => {
const site = Object.freeze({
kind: 'call',
atRange: { startLine: 153, startCol: 8, endLine: 153, endCol: 30 },
staticGated: false,
});
const original = Object.freeze({
filePath: 'src/main.zig',
referenceSites: Object.freeze([site]),
}) as unknown as ParsedFile;
const parsedFiles = [
original,
Object.freeze({
filePath: 'src/cfg.zig',
referenceSites: Object.freeze([]),
}) as unknown as ParsedFile,
];
expect(() =>
populateZigWorkspaceStaticGating(parsedFiles, {
fileContents: new Map([
[
'src/main.zig',
fs.readFileSync(path.join(FIXTURES, 'zig-static-gating', 'src', 'main.zig'), 'utf8'),
],
[
'src/cfg.zig',
fs.readFileSync(path.join(FIXTURES, 'zig-static-gating', 'src', 'cfg.zig'), 'utf8'),
],
]),
}),
).not.toThrow();
expect(parsedFiles[0]).not.toBe(original);
expect(Object.isFrozen(parsedFiles[0])).toBe(true);
expect(parsedFiles[0]?.referenceSites[0]?.staticGated).toBe(true);
});
});

View file

@ -0,0 +1,210 @@
/**
* End-to-end HTTP test of POST /api/analyze `branch` validation.
*
* `branch` is handed to `git` as a ref (`clone --branch`, `checkout -B`), so the
* route validates it with the same `validateBranchName` the CLI's `--branch`
* uses. This proves the REAL production route wires that in — express.json body
* parsing, the requireTrustedOrigin guard, the route handler invoking the
* validator, and the 400 status/error shape on the wire.
*
* Only rejection paths are asserted: each returns 400 BEFORE any clone, so the
* test is hermetic (no network, no background git, no real repo). The accepted
* path would spawn a background clone; the parent→worker half of it is covered
* in test/unit/analyze-launch-collapse.test.ts, which asserts `branch` reaches
* the worker's `AnalyzeOptions`.
*
* This lives in its own file rather than alongside the token cases because
* /api/analyze is rate-limited to 10 requests/minute per IP — one spawned server
* per concern keeps each suite clear of that ceiling.
*
* Mirrors the spawn+health-poll harness in server-analyze-token-validation.test.ts;
* the integration suite always builds dist first (pretest:integration).
*/
import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process';
import fs from 'node:fs';
import http from 'node:http';
import os from 'node:os';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { afterAll, beforeAll, describe, expect, it } from 'vitest';
const __dirname = path.dirname(fileURLToPath(import.meta.url));
const REPO_ROOT = path.resolve(__dirname, '..', '..');
const DIST_CLI = path.join(REPO_ROOT, 'dist', 'cli', 'index.js');
const STARTUP_BUDGET_MS = process.env.CI ? 30_000 : 15_000;
const allocateFreePort = (): Promise<number> =>
new Promise((resolve, reject) => {
const probe = http.createServer();
probe.once('error', reject);
probe.listen(0, '127.0.0.1', () => {
const addr = probe.address();
if (typeof addr !== 'object' || !addr) {
probe.close();
reject(new Error('could not allocate ephemeral port'));
return;
}
const port = addr.port;
probe.close((err) => (err ? reject(err) : resolve(port)));
});
});
const httpJson = (
port: number,
method: string,
reqPath: string,
body?: unknown,
): Promise<{ status: number; body: string }> =>
new Promise((resolve, reject) => {
const payload = body === undefined ? undefined : JSON.stringify(body);
const req = http.request(
{
host: '127.0.0.1',
port,
path: reqPath,
method,
headers: payload
? { 'content-type': 'application/json', 'content-length': Buffer.byteLength(payload) }
: {},
},
(res) => {
const chunks: Buffer[] = [];
res.on('data', (c) => chunks.push(c));
res.on('end', () =>
resolve({ status: res.statusCode ?? 0, body: Buffer.concat(chunks).toString('utf8') }),
);
},
);
req.on('error', reject);
req.setTimeout(5_000, () => {
req.destroy();
reject(new Error(`${method} ${reqPath} timed out`));
});
if (payload) req.write(payload);
req.end();
});
const postAnalyze = (port: number, body: unknown) => httpJson(port, 'POST', '/api/analyze', body);
// Spawned `serve` on Windows can report ready before the socket is reachable
// from the parent (see server-http-startup.test.ts); validateBranchName's own
// unit coverage runs on every platform.
const describeBlock = process.platform === 'win32' ? describe.skip : describe;
describeBlock('POST /api/analyze branch validation (real server)', () => {
let proc: ChildProcessWithoutNullStreams | undefined;
let homeDir: string | undefined;
let port = 0;
beforeAll(async () => {
if (!fs.existsSync(DIST_CLI)) {
throw new Error(`Missing ${DIST_CLI} — run npm run build before integration tests`);
}
port = await allocateFreePort();
homeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-analyze-branch-'));
proc = spawn(
process.execPath,
[DIST_CLI, 'serve', '--port', String(port), '--host', '127.0.0.1'],
{
cwd: REPO_ROOT,
env: { ...process.env, GITNEXUS_HOME: homeDir, NODE_OPTIONS: '' },
stdio: ['ignore', 'pipe', 'pipe'],
},
);
let stderr = '';
proc.stderr.on('data', (buf) => {
stderr += buf.toString();
});
const startedAt = Date.now();
while (Date.now() - startedAt < STARTUP_BUDGET_MS) {
if (proc.exitCode !== null) {
throw new Error(`serve exited ${proc.exitCode} before ready.\nstderr:\n${stderr}`);
}
try {
const { status } = await httpJson(port, 'GET', '/api/health');
if (status === 200) return;
} catch {
// Server still starting — retry until budget expires.
}
await new Promise((r) => setTimeout(r, 100));
}
throw new Error(
`serve did not become ready within ${STARTUP_BUDGET_MS}ms.\nstderr:\n${stderr}`,
);
}, 60_000);
afterAll(async () => {
if (proc && !proc.killed) {
proc.kill('SIGTERM');
await new Promise<void>((resolve) => {
const timer = setTimeout(() => {
proc?.kill('SIGKILL');
resolve();
}, 3_000);
proc?.on('exit', () => {
clearTimeout(timer);
resolve();
});
});
}
proc = undefined;
if (homeDir) {
fs.rmSync(homeDir, { recursive: true, force: true });
homeDir = undefined;
}
});
it('rejects a non-string branch', async () => {
const { status, body } = await postAnalyze(port, {
url: 'https://github.com/owner/repo',
branch: 42,
});
expect(status).toBe(400);
expect(JSON.parse(body).error).toContain('"branch" must be a string');
});
it('rejects a branch containing whitespace', async () => {
const { status, body } = await postAnalyze(port, {
url: 'https://github.com/owner/repo',
branch: 'feature branch',
});
expect(status).toBe(400);
expect(JSON.parse(body).error).toContain('whitespace');
});
it('rejects a branch using characters git forbids in a ref', async () => {
const { status, body } = await postAnalyze(port, {
url: 'https://github.com/owner/repo',
branch: 'feature^bad',
});
expect(status).toBe(400);
expect(JSON.parse(body).error).toContain('not allowed in a git ref');
});
it('rejects a branch that would read as a git option', async () => {
// `git clone --branch --upload-pack=evil` would otherwise let a ref choose
// the subprocess git runs; buildBranchCloneArgs keeps the `--` separator,
// and this closes the same shape one layer earlier.
const { status, body } = await postAnalyze(port, {
url: 'https://github.com/owner/repo',
branch: '--upload-pack=evil',
});
expect(status).toBe(400);
expect(JSON.parse(body).error).toContain('must not start with "-"');
});
it('rejects a whitespace-only branch rather than silently indexing the default', async () => {
// The bug this feature fixes was a silent fallback to the default branch;
// an unusable selector must fail loudly, never degrade into that behavior.
const { status, body } = await postAnalyze(port, {
url: 'https://github.com/owner/repo',
branch: ' ',
});
expect(status).toBe(400);
expect(JSON.parse(body).error).toContain('must not be empty');
});
});

View file

@ -14,6 +14,7 @@ import {
} from '../../src/core/ingestion/workers/worker-pool.js';
import { pathToFileURL } from 'node:url';
import { spawn } from 'node:child_process';
import { createRequire } from 'node:module';
import path from 'node:path';
import fs from 'node:fs';
import os from 'node:os';
@ -245,10 +246,8 @@ describe('worker pool integration', () => {
const results = await pool.dispatch<any, any>(files);
// All 7 files fit one default sub-batch (size 200 / budget 8MB),
// so the dispatch returns exactly one chunk result regardless of
// pool size.
expect(results).toHaveLength(1);
// Small inputs must still split across the available workers.
expect(results.length).toBeGreaterThan(1);
// Total files parsed should match input
const totalParsed = results.reduce((sum: number, r: any) => sum + r.fileCount, 0);
@ -793,6 +792,204 @@ describe('worker pool integration', () => {
}
});
it.each([2, 4, 7, 9])(
'uses available workers for a small %i-file cache pack',
async (fileCount) => {
const { tempDir, workerPath } = writeTempWorker(
'gitnexus-worker-small-pack-',
`
const { parentPort, threadId } = require('node:worker_threads');
let paths = [];
parentPort.on('message', (msg) => {
if (msg && msg.type === 'sub-batch') {
paths = msg.files.map((file) => file.path);
parentPort.postMessage({ type: 'progress', filesProcessed: paths.length });
parentPort.postMessage({ type: 'sub-batch-done' });
} else if (msg && msg.type === 'flush') {
parentPort.postMessage({ type: 'result', data: { paths, threadId } });
}
});
`,
);
pool = createWorkerPool(pathToFileURL(workerPath), 4, {
subBatchMaxBytes: 256 * 1024,
});
try {
// Stable cache packs are often smaller than the byte budget. They must
// still use the pool, including when a later dispatch has fewer files.
for (const count of [fileCount, 1]) {
const files = Array.from({ length: count }, (_, i) => ({
path: `file-${i}.ts`,
content: 'export const value = 1;',
}));
const progress: number[] = [];
const results = await pool.dispatch<
(typeof files)[number],
{ paths: string[]; threadId: number }
>(files, (completed) => progress.push(completed), `pack-${count}`);
expect(new Set(results.map((result) => result.threadId)).size).toBe(Math.min(4, count));
expect(results.flatMap((result) => result.paths)).toEqual(files.map((file) => file.path));
expect(progress).toEqual([...progress].sort((a, b) => a - b));
expect(progress.at(-1)).toBe(count);
}
} finally {
await pool.terminate();
fs.rmSync(tempDir, { recursive: true, force: true });
}
},
);
it('splits packs into one round, keeping results and chunk hashes per group', async () => {
const { tempDir, workerPath } = writeTempWorker(
'gitnexus-worker-dispatch-groups-',
`
const { parentPort, threadId } = require('node:worker_threads');
let paths = [];
parentPort.on('message', (msg) => {
if (msg && msg.type === 'sub-batch') {
for (const file of msg.files) paths.push(file.path);
parentPort.postMessage({ type: 'progress', filesProcessed: msg.files.length });
parentPort.postMessage({ type: 'sub-batch-done' });
} else if (msg && msg.type === 'flush') {
parentPort.postMessage({
type: 'result',
data: { paths, threadId, chunkHash: msg.chunkHash },
});
paths = [];
}
});
`,
);
pool = createWorkerPool(pathToFileURL(workerPath), 4);
try {
// Shaped like real cache packs: mostly tiny, one larger. A single round
// must still keep every result attributable to the pack that owns it.
const groups = [
{ chunkHash: 'hash-a', items: [{ path: 'a0.ts', content: 'export const a0 = 1;' }] },
{ chunkHash: 'hash-b', items: [{ path: 'b0.ts', content: 'export const b0 = 1;' }] },
{
chunkHash: 'hash-c',
items: Array.from({ length: 12 }, (_, i) => ({
path: `c${i}.ts`,
content: 'export const c = 1;',
})),
},
{ chunkHash: 'hash-d', items: [{ path: 'd0.ts', content: 'export const d0 = 1;' }] },
];
const progress: number[] = [];
const perGroup = await pool.dispatchGroups<
(typeof groups)[number]['items'][number],
{ paths: string[]; threadId: number; chunkHash?: string }
>(groups, (completed) => progress.push(completed));
expect(perGroup).toHaveLength(groups.length);
// No job straddles a pack: every path comes back under its own group,
// and every result carries that group's chunk hash (the cache key).
for (const [index, group] of groups.entries()) {
expect(perGroup[index].flatMap((result) => result.paths).sort()).toEqual(
group.items.map((file) => file.path).sort(),
);
for (const result of perGroup[index]) expect(result.chunkHash).toBe(group.chunkHash);
}
// The whole round shares the pool rather than one pack per barrier.
const threads = new Set(perGroup.flat().map((result) => result.threadId));
expect(threads.size).toBeGreaterThan(1);
expect(progress).toEqual([...progress].sort((a, b) => a - b));
expect(progress.at(-1)).toBe(15);
} finally {
await pool.terminate();
fs.rmSync(tempDir, { recursive: true, force: true });
}
});
it('returns an empty result array for a group whose items were all quarantined', async () => {
const { tempDir, workerPath } = writeTempWorker(
'gitnexus-worker-groups-quarantine-',
`
const { parentPort } = require('node:worker_threads');
let paths = [];
parentPort.on('message', (msg) => {
if (msg && msg.type === 'sub-batch') {
for (const file of msg.files) {
if (file.path === 'poison.ts') process.exit(134);
paths.push(file.path);
}
parentPort.postMessage({ type: 'progress', filesProcessed: msg.files.length });
parentPort.postMessage({ type: 'sub-batch-done' });
} else if (msg && msg.type === 'flush') {
parentPort.postMessage({ type: 'result', data: { paths } });
paths = [];
}
});
`,
);
pool = createWorkerPool(pathToFileURL(workerPath), 2);
try {
await pool.dispatch([{ path: 'poison.ts', content: '' }]);
expect(pool.getQuarantinedPaths?.()).toContain('poison.ts');
// Group alignment must survive quarantine filtering — an all-quarantined
// group still occupies its slot so results line up with the input packs.
const perGroup = await pool.dispatchGroups<
{ path: string; content: string },
{ paths: string[] }
>([
{ chunkHash: 'poisoned', items: [{ path: 'poison.ts', content: '' }] },
{ chunkHash: 'healthy', items: [{ path: 'ok.ts', content: 'export const ok = 1;' }] },
]);
expect(perGroup).toHaveLength(2);
expect(perGroup[0]).toEqual([]);
expect(perGroup[1].flatMap((result) => result.paths)).toEqual(['ok.ts']);
} finally {
await pool.terminate();
fs.rmSync(tempDir, { recursive: true, force: true });
}
});
it('rejects a second dispatch while one is still in flight', async () => {
const { tempDir, workerPath } = writeTempWorker(
'gitnexus-worker-reentrant-dispatch-',
`
const { parentPort } = require('node:worker_threads');
let paths = [];
parentPort.on('message', (msg) => {
if (msg && msg.type === 'sub-batch') {
paths = msg.files.map((file) => file.path);
parentPort.postMessage({ type: 'progress', filesProcessed: paths.length });
parentPort.postMessage({ type: 'sub-batch-done' });
} else if (msg && msg.type === 'flush') {
setTimeout(() => parentPort.postMessage({ type: 'result', data: { paths } }), 150);
}
});
`,
);
pool = createWorkerPool(pathToFileURL(workerPath), 2);
try {
// Concurrent dispatches hand the same slots out twice; both then stall
// until every worker idle-times out. Fail at the call, not 10s later.
const first = pool.dispatch<{ path: string; content: string }, { paths: string[] }>([
{ path: 'first.ts', content: 'export const first = 1;' },
]);
await expect(
pool.dispatch([{ path: 'second.ts', content: 'export const second = 1;' }]),
).rejects.toThrow(/not reentrant/);
await expect(first).resolves.toHaveLength(1);
// The guard clears once the in-flight dispatch settles.
await expect(
pool.dispatch([{ path: 'third.ts', content: 'export const third = 1;' }]),
).resolves.toHaveLength(1);
} finally {
await pool.terminate();
fs.rmSync(tempDir, { recursive: true, force: true });
}
});
it('bounds worker jobs by byte budget as well as file count', async () => {
const { tempDir, workerPath } = writeTempWorker(
'gitnexus-worker-byte-budget-',
@ -831,6 +1028,69 @@ describe('worker pool integration', () => {
}
});
it.skipIf(!hasDistWorker)(
'reuses compiled queries across jobs while keeping TS and TSX grammars separate',
async () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-query-cache-'));
const workerPath = path.join(tempDir, 'worker.cjs');
const parserPath = createRequire(import.meta.url).resolve('tree-sitter');
fs.writeFileSync(
workerPath,
`
const { parentPort } = require('node:worker_threads');
const Parser = require(${JSON.stringify(parserPath)});
let queryCompilations = 0;
Parser.Query = new Proxy(Parser.Query, {
construct(target, args, newTarget) {
queryCompilations++;
return Reflect.construct(target, args, newTarget);
},
});
const send = parentPort.postMessage.bind(parentPort);
parentPort.postMessage = (message, ...args) => {
if (message.type === 'result') message.data.queryCompilations = queryCompilations;
return send(message, ...args);
};
import(${JSON.stringify(pathToFileURL(DIST_WORKER).href)});
`,
);
pool = createWorkerPool(pathToFileURL(workerPath), 1, { workerReadyTimeoutMs: 30_000 });
type QueryResult = {
queryCompilations: number;
fileCount: number;
nodes: Array<{ properties: { name: string } }>;
};
try {
const counts: number[] = [];
for (const [extension, name] of [
['ts', 'first'],
['ts', 'second'],
['tsx', 'view'],
['tsx', 'otherView'],
['ts', 'last'],
]) {
const file = {
path: `${name}.${extension}`,
content: `export function ${name}() { return ${extension === 'tsx' ? '<div />' : '1'}; }`,
};
const [result] = await pool.dispatch<typeof file, QueryResult>([file]);
expect(result.fileCount).toBe(1);
expect(result.nodes.map((node) => node.properties.name)).toContain(name);
counts.push(result.queryCompilations);
}
expect(counts[0]).toBeGreaterThan(0);
expect(counts[1]).toBe(counts[0]);
expect(counts[2]).toBeGreaterThan(counts[1]);
expect(counts[3]).toBe(counts[2]);
expect(counts[4]).toBe(counts[2]);
} finally {
await pool.terminate();
fs.rmSync(tempDir, { recursive: true, force: true });
}
},
60_000,
);
it.skipIf(!hasDistWorker)('createWorkerPool with size 0 creates pool with zero workers', () => {
const workerUrl = pathToFileURL(DIST_WORKER) as URL;
const zeroPool = createWorkerPool(workerUrl, 0);

View file

@ -299,6 +299,33 @@ describe('analyze-config (.gitnexusrc support, #243)', () => {
expect(() => validateBranchName('foo..bar', 'src')).toThrow(/must not contain ".."/);
});
it('validateBranchName rejects the ref shapes git check-ref-format rejects', () => {
// These previously passed validation and failed later in the git subprocess,
// which over HTTP meant a 202 and a background failure instead of a 400.
expect(() => validateBranchName('feature.lock', 'src')).toThrow(/must not end with "\.lock"/);
expect(() => validateBranchName('refs/heads.lock/x', 'src')).toThrow(
/must not end with "\.lock"/,
);
expect(() => validateBranchName('/feature', 'src')).toThrow(/must not start or end with "\/"/);
expect(() => validateBranchName('feature/', 'src')).toThrow(/must not start or end with "\/"/);
expect(() => validateBranchName('feature//next', 'src')).toThrow(/consecutive slashes/);
expect(() => validateBranchName('@', 'src')).toThrow(/single character "@"/);
expect(() => validateBranchName('feature@{1}', 'src')).toThrow(/must not contain "@\{"/);
expect(() => validateBranchName('.hidden', 'src')).toThrow(/starting with "\."/);
expect(() => validateBranchName('feature/.hidden', 'src')).toThrow(/starting with "\."/);
expect(() => validateBranchName('feature.', 'src')).toThrow(/end with "\."/);
});
it('validateBranchName still accepts the real branch shapes those rules must not catch', () => {
// A dot, a slash and an @ are all legal in the middle of a ref — the new
// rules must reject only what git itself would.
expect(validateBranchName('release/1.2.3', 'src')).toBe('release/1.2.3');
expect(validateBranchName('feature/lockfile-bump', 'src')).toBe('feature/lockfile-bump');
expect(validateBranchName('user@host', 'src')).toBe('user@host');
expect(validateBranchName('v1.0', 'src')).toBe('v1.0');
expect(validateBranchName('a/b/c', 'src')).toBe('a/b/c');
});
it('validateBranchName rejects a newline / control character', () => {
expect(() => validateBranchName('main\nrm -rf', 'src')).toThrow(/control or hidden|whitespace/);
});
@ -394,6 +421,20 @@ describe('analyze-config (.gitnexusrc support, #243)', () => {
expect(() => validateBranchName('a'.repeat(256), 'src')).toThrow(/too long/);
});
it('validateBranchName rejects HEAD (case-sensitive) and accepts head (#3199)', () => {
expect(() => validateBranchName('HEAD', 'src')).toThrow(GitNexusRcError);
expect(() => validateBranchName('HEAD', 'src')).toThrow(/must not be "HEAD"/);
expect(() => validateBranchName(' HEAD ', 'src')).toThrow(GitNexusRcError);
expect(validateBranchName('head', 'src')).toBe('head');
});
it('validateBranchName rejects a force-refspec "+" prefix (#3199)', () => {
expect(() => validateBranchName('+main', 'src')).toThrow(GitNexusRcError);
expect(() => validateBranchName('+main', 'src')).toThrow(/must not start with "\+"/);
expect(() => validateBranchName('+develop', 'src')).toThrow(GitNexusRcError);
expect(() => validateBranchName('+develop', 'src')).toThrow(/must not start with "\+"/);
});
it('rejects Markdown-significant characters in a config name, allows real names (#1996)', async () => {
await writeRc(JSON.stringify({ name: '**evil**' }));
expect(() => loadAnalyzeConfig(dir)).toThrow(/Markdown-significant/);

View file

@ -56,6 +56,65 @@ describe('JobManager', () => {
expect(job2.id).toBe(job1.id);
});
it('returns existing job for the same repoUrl AND the same branch', () => {
const job1 = manager.createJob({
repoUrl: 'https://github.com/user/repo',
branch: 'development',
});
manager.updateJob(job1.id, { status: 'analyzing' });
const job2 = manager.createJob({
repoUrl: 'https://github.com/user/repo',
branch: 'development',
});
expect(job2.id).toBe(job1.id);
});
it('does not return the active job to a caller asking for a different branch', () => {
const job1 = manager.createJob({
repoUrl: 'https://github.com/user/repo',
branch: 'development',
});
manager.updateJob(job1.id, { status: 'analyzing' });
// Handing job1 back would report branch "development" as the work being done
// for a caller that asked for "main". Falling through to the single-slot
// guard is the truthful answer.
expect(() =>
manager.createJob({ repoUrl: 'https://github.com/user/repo', branch: 'main' }),
).toThrow(/already in progress/);
});
it('treats an unpinned request as distinct from a branch-pinned one', () => {
const job1 = manager.createJob({
repoUrl: 'https://github.com/user/repo',
branch: 'development',
});
manager.updateJob(job1.id, { status: 'analyzing' });
expect(() => manager.createJob({ repoUrl: 'https://github.com/user/repo' })).toThrow(
/already in progress/,
);
});
it('keeps callers that omit branch deduping exactly as before', () => {
// The parameter is optional, so every pre-existing call site (upload route,
// embed manager, tests) compares undefined === undefined and is unaffected.
const job1 = manager.createJob({ repoPath: '/tmp/repo' });
manager.updateJob(job1.id, { status: 'analyzing' });
expect(manager.createJob({ repoPath: '/tmp/repo' }).id).toBe(job1.id);
});
it('carries the branch unchanged through the whole job lifecycle', () => {
// `branch` is part of dedup identity, so it must not drift mid-flight.
// `updateJob`'s Pick<> allowlist omits it, so no well-typed caller can
// change it; this pins that none of the updates the server actually
// performs (clone -> analyze -> terminal) disturbs it either.
const job = manager.createJob({ repoUrl: 'https://github.com/user/repo', branch: 'develop' });
manager.updateJob(job.id, { status: 'cloning' });
manager.updateJob(job.id, { repoPath: '/tmp/repo', status: 'analyzing' });
manager.updateJob(job.id, { progress: { phase: 'parsing', percent: 30, message: 'Parsing' } });
manager.updateJob(job.id, { status: 'complete', repoName: 'repo' });
expect(manager.getJob(job.id)?.branch).toBe('develop');
});
it('updates job progress', () => {
const job = manager.createJob({ repoUrl: 'https://github.com/user/repo' });
manager.updateJob(job.id, {

View file

@ -0,0 +1,238 @@
/**
* The finalization gate must watch the index the run actually WROTE.
*
* `registerRepo` always records the flat `.gitnexus` as `entry.storagePath`, but
* a pinned `--branch` run whose label differs from the flat slot's owner writes
* `lbug`/`gitnexus.json` under `branches/<slug>/`. The gate used to probe the
* flat path unconditionally, so for such a run it watched files this job never
* rewrote:
*
* - the gate never settles, so the job stays non-terminal for the full 60s;
* - meanwhile the worker, having already sent `complete`, calls
* `process.exit(0)` ~500ms later;
* - the exit handler saw a non-terminal job and classified that clean exit as
* a crash — retrying a SUCCESSFUL analysis three times before failing it
* with `Worker crashed 3 times (code 0)`.
*
* Reported twice on #3199 (maintainer review + @azizur100389's repro). These
* tests pin both halves: the gate follows the placement, and a terminal IPC
* makes a subsequent exit 0 settlement rather than a crash.
*
* The filesystem mock is deliberately PATH-SENSITIVE — only the branch sub-slot
* looks freshly written. A gate that probes the flat path therefore cannot pass
* these tests by accident.
*/
import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest';
import { EventEmitter } from 'node:events';
import path from 'node:path';
// `vi.hoisted` is lifted above the imports, so nothing in here may reference
// them — these stay plain literals and `path` is only used below.
const H = vi.hoisted(() => ({
forkMock: vi.fn(),
STORAGE_PATH: '/tmp/gitnexus-settle-storage',
REPO_PATH: '/tmp/gitnexus-settle-repo',
METADATA_FILE: 'gitnexus.json',
// Set per-test: the only directory the fake filesystem reports as freshly
// written. Anything else looks stale, exactly like a slot this job skipped.
settledDir: '',
}));
const { forkMock, REPO_PATH } = H;
vi.mock('child_process', async () => {
const actual = await vi.importActual<typeof import('child_process')>('child_process');
return { ...actual, fork: H.forkMock };
});
vi.mock('../../src/storage/repo-manager.js', () => ({
canonicalizePath: (p: string) => p,
getStoragePath: () => H.STORAGE_PATH,
INDEX_METADATA_FILE: H.METADATA_FILE,
listRegisteredRepos: async () => [{ path: H.REPO_PATH, storagePath: H.STORAGE_PATH }],
registryPathEquals: (a: string, b: string) => a === b,
}));
vi.mock('node:fs', async () => {
const actual = await vi.importActual<typeof import('node:fs')>('node:fs');
return {
...actual,
statSync: (p: string) => {
// Fresh only inside the directory this run is pretending to have written.
// Plain string work rather than `path.dirname`: this factory is hoisted
// above the imports too.
const file = String(p);
const dir = file.slice(0, Math.max(file.lastIndexOf('/'), file.lastIndexOf('\\')));
if (H.settledDir && dir === H.settledDir) {
return { mtimeMs: Number.MAX_SAFE_INTEGER };
}
return { mtimeMs: 0 };
},
existsSync: () => false, // no WAL/shadow/checkpoint sidecars anywhere
};
});
import { createLaunchAnalysisWorker } from '../../src/server/analyze-launch.js';
import { JobManager } from '../../src/server/analyze-job.js';
import { projectAnalyzeResultForIpc } from '../../src/server/analyze-worker-ipc.js';
import { BRANCHES_DIR, branchSlug } from '../../src/storage/branch-index.js';
import type { AnalyzeResult } from '../../src/core/run-analyze.js';
import type { CompleteMessage } from '../../src/server/analyze-worker.js';
const BRANCH = 'feature/settle';
/** The `complete` message the worker really sends, via the production projection. */
const completeMessage = (
isPrimaryBranch: boolean,
extras?: { alreadyUpToDate?: boolean },
): CompleteMessage => {
const result = {
repoName: 'settle-fixture',
repoPath: REPO_PATH,
stats: { files: 3, nodes: 9, edges: 12 },
isPrimaryBranch,
...(extras?.alreadyUpToDate ? { alreadyUpToDate: true } : {}),
} satisfies Partial<AnalyzeResult> as AnalyzeResult;
return { type: 'complete', result: projectAnalyzeResultForIpc(result) };
};
interface FakeChild extends EventEmitter {
stderr: EventEmitter;
send: Mock<(msg: unknown) => boolean>;
kill: Mock<(signal?: NodeJS.Signals) => boolean>;
}
const makeChild = (): FakeChild => {
const child = new EventEmitter() as FakeChild;
child.stderr = new EventEmitter();
child.send = vi.fn();
child.kill = vi.fn();
return child;
};
describe('finalization gate follows the placement the run chose', () => {
let jobManager: JobManager;
let child: FakeChild;
let backendInit: Mock<() => Promise<unknown>>;
let closeDbHandle: Mock<() => Promise<void>>;
const launcher = () =>
createLaunchAnalysisWorker({
jobManager,
backend: { init: backendInit },
acquireRepoLock: () => null,
releaseRepoLock: () => {},
closeDbHandle,
});
beforeEach(() => {
jobManager = new JobManager();
child = makeChild();
forkMock.mockImplementation(() => child);
backendInit = vi.fn(async () => true);
closeDbHandle = vi.fn(async () => {});
H.settledDir = '';
});
afterEach(() => {
jobManager.dispose();
vi.restoreAllMocks();
forkMock.mockReset();
});
it('completes a branch sub-slot run, whose files are NOT in the flat slot', async () => {
// Only `branches/<slug>/` looks written. The flat slot is stale, so a gate
// probing it would spin to the 60s timeout instead of settling here.
H.settledDir = path.join(H.STORAGE_PATH, BRANCHES_DIR, branchSlug(BRANCH));
const job = jobManager.createJob({ repoPath: REPO_PATH, branch: BRANCH });
launcher()(job, REPO_PATH, { branch: BRANCH });
child.emit('message', completeMessage(false));
await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete'));
expect(jobManager.getJob(job.id)?.error).toBeUndefined();
expect(backendInit).toHaveBeenCalledTimes(1);
});
it('still settles a flat-slot run against the flat slot', async () => {
// Control: the primary-branch case must keep watching `entry.storagePath`.
H.settledDir = H.STORAGE_PATH;
const job = jobManager.createJob({ repoPath: REPO_PATH });
launcher()(job, REPO_PATH, {});
child.emit('message', completeMessage(true));
await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete'));
expect(backendInit).toHaveBeenCalledTimes(1);
});
it('settles a first-pin (branch SET, isPrimaryBranch true) against the flat slot', async () => {
// Fresh clone: first pin adopts the flat slot. The slug dir is stale, so a
// gate that does `branch ? slugDir : flat` would spin the 60s timeout here.
H.settledDir = H.STORAGE_PATH;
const job = jobManager.createJob({ repoPath: REPO_PATH, branch: BRANCH });
launcher()(job, REPO_PATH, { branch: BRANCH });
child.emit('message', completeMessage(true));
await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete'));
expect(jobManager.getJob(job.id)?.error).toBeUndefined();
expect(backendInit).toHaveBeenCalledTimes(1);
});
it('completes alreadyUpToDate quickly even when the slot is stale, without retrying', async () => {
// No directory looks freshly written. Without the alreadyUpToDate skip the
// mtime gate would hold the analyze slot for the full 60s settle timeout.
H.settledDir = '';
const job = jobManager.createJob({ repoPath: REPO_PATH });
launcher()(job, REPO_PATH, {});
child.emit('message', completeMessage(true, { alreadyUpToDate: true }));
child.emit('exit', 0);
await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete'), {
timeout: 2_000,
});
expect(jobManager.getJob(job.id)?.error).toBeUndefined();
expect(backendInit).toHaveBeenCalledTimes(1);
expect(forkMock).toHaveBeenCalledTimes(1);
expect(jobManager.getJob(job.id)?.retryCount).toBe(0);
});
it('does not fork a retry when the worker exits 0 after reporting complete', async () => {
// The worker exits ~500ms after the `complete` IPC, while the gate is still
// running and the job is deliberately non-terminal. That exit is the worker
// winding down, not dying.
H.settledDir = path.join(H.STORAGE_PATH, BRANCHES_DIR, branchSlug(BRANCH));
const job = jobManager.createJob({ repoPath: REPO_PATH, branch: BRANCH });
launcher()(job, REPO_PATH, { branch: BRANCH });
child.emit('message', completeMessage(false));
child.emit('exit', 0);
await vi.waitFor(() => expect(jobManager.getJob(job.id)?.status).toBe('complete'));
expect(jobManager.getJob(job.id)?.error).toBeUndefined();
// One fork for the run itself; a retry would be a second.
expect(forkMock).toHaveBeenCalledTimes(1);
expect(jobManager.getJob(job.id)?.retryCount).toBe(0);
});
it('still treats an exit with no terminal IPC as a crash worth retrying', async () => {
// The guard must not swallow real crashes: no `complete`/`error` was sent.
H.settledDir = H.STORAGE_PATH;
const job = jobManager.createJob({ repoPath: REPO_PATH });
launcher()(job, REPO_PATH, {});
child.emit('exit', 1);
// The first retry is scheduled on a 1s backoff, so this needs more than
// vi.waitFor's default budget.
await vi.waitFor(() => expect(forkMock).toHaveBeenCalledTimes(2), { timeout: 4_000 });
expect(jobManager.getJob(job.id)?.retryCount).toBe(1);
});
});

View file

@ -170,6 +170,50 @@ describe('createLaunchAnalysisWorker — collapsed index is never published', ()
);
});
it('forwards the index-branch selector to the worker', () => {
const launch = createLaunchAnalysisWorker({
jobManager,
backend: { init: backendInit },
acquireRepoLock: () => null,
releaseRepoLock: () => {},
closeDbHandle,
});
const job = jobManager.createJob({ repoPath: REPO_PATH });
launch(job, REPO_PATH, { branch: 'development' });
// `StartMessage.options` is typed as `AnalyzeOptions`, so this key IS
// `AnalyzeOptions.branch` — the field `resolveWriteTarget` reads to choose
// the run's storage slot. (It does not always mean a `branches/<slug>/`
// sub-slot: `resolveBranchPlacement` keeps the flat slot when that slot has
// no owner, or when its owner is already this label.) A rename breaks this
// test.
expect(child.send).toHaveBeenCalledWith(
expect.objectContaining({
options: expect.objectContaining({ branch: 'development' }),
}),
);
});
it('omits branch entirely when the caller did not select one', () => {
const launch = createLaunchAnalysisWorker({
jobManager,
backend: { init: backendInit },
acquireRepoLock: () => null,
releaseRepoLock: () => {},
closeDbHandle,
});
const job = jobManager.createJob({ repoPath: REPO_PATH });
launch(job, REPO_PATH, {});
// Not merely undefined: absent. `AnalyzeOptions.branch === undefined` is the
// documented signal for "target the flat workspace slot", so sending the key
// with an undefined value must not become the way that default is expressed.
const sent = child.send.mock.calls.at(0)?.[0] as { options: Record<string, unknown> };
expect(Object.hasOwn(sent.options, 'branch')).toBe(false);
});
afterEach(() => {
jobManager.dispose();
vi.restoreAllMocks();

View file

@ -15,6 +15,7 @@ vi.mock('../../src/core/logger.js', () => ({
}));
import {
analyzeCloneOptions,
extractRepoName,
extractWebRepoName,
getCloneDir,
@ -671,6 +672,359 @@ describe('git-clone', () => {
}
});
// `assertRemoteMatchesRequestedUrl` runs REAL git (it is not injectable), so
// these fixtures are real repositories with a matching origin; only the
// branch logic under test is driven through `runGitForTest`.
const REMOTE = 'https://github.com/owner/repo.git';
const makeExistingClone = async (root: string, branch: string) => {
const target = path.join(root, 'repo');
await fs.mkdir(target, { recursive: true });
await runGit(['init', `--initial-branch=${branch}`], target);
await runGit(['remote', 'add', 'origin', REMOTE], target);
return target;
};
it('re-indexing the SAME pinned branch fetches via a safe refspec instead of refusing a dirty tree', async () => {
// Analyze writes AGENTS.md / CLAUDE.md / .claude/ into the clone, so the
// tree is dirty from its own first run. Routing a same-branch request
// through the checkout path made every pinned RE-index fail asking for
// `overwrite_local_changes` — a flag the route will not pass because it
// would `git clean --force -d` the directory (#3199 review).
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'develop');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse') return 'develop\n';
if (args[0] === 'status') return ' M AGENTS.md\n?? .claude/\n'; // dirty, as analyze leaves it
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
const verbs = calls.map((c) => c[0]);
expect(calls).toContainEqual([
'fetch',
'--depth',
'1',
'origin',
'+refs/heads/develop:refs/remotes/origin/develop',
]);
expect(calls).toContainEqual(['checkout', '-B', 'develop', 'origin/develop']);
// Never a raw `origin <branch>` pull — `+develop` would be a force-fetch.
expect(
calls.some((c) => c[0] === 'pull' && c.includes('origin') && c.includes('develop')),
).toBe(false);
expect(verbs).not.toContain('status'); // so the dirty check never ran
expect(calls.some((c) => c[0] === 'merge')).toBe(false);
// Must not fall back to a bare `git pull --ff-only` — that follows
// `branch.<name>.merge`, which is not verified (only origin.url is).
expect(calls.some((c) => c[0] === 'pull' && c.length === 2)).toBe(false);
});
it('same-branch argv uses +refs/heads/develop:refs/remotes/origin/develop, never raw develop as a pull dest', async () => {
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'develop');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'develop\n';
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
const fetchCall = calls.find((c) => c[0] === 'fetch');
expect(fetchCall).toEqual([
'fetch',
'--depth',
'1',
'origin',
'+refs/heads/develop:refs/remotes/origin/develop',
]);
expect(fetchCall?.[4]).not.toBe('develop');
expect(calls.some((c) => c[0] === 'pull' && c[3] === 'develop')).toBe(false);
});
it('does not treat a leading-plus branch as a force-fetch pull refspec', async () => {
// If `+develop` somehow reached cloneOrPull, the heads/ mapping keeps
// the `+` inside the ref name. It is not git's force-fetch prefix.
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'main');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return '+develop\n';
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: '+develop',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls).toContainEqual([
'fetch',
'--depth',
'1',
'origin',
'+refs/heads/+develop:refs/remotes/origin/+develop',
]);
expect(
calls.some((c) => c[0] === 'pull' && c.some((a) => a === '+develop' || a.startsWith('+'))),
).toBe(false);
// Force prefix is on the mapping, not a force-update of `develop`.
expect(
calls.some(
(c) => c[0] === 'fetch' && c.includes('+refs/heads/develop:refs/remotes/origin/develop'),
),
).toBe(false);
});
it('restores a dirty AGENTS.md on the same branch without taking the switch refuse path', async () => {
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'develop');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'develop\n';
if (args[0] === 'ls-files' && args.includes('./AGENTS.md')) return 'AGENTS.md\n';
if (args[0] === 'status') return ' M AGENTS.md\n?? .claude/\n';
return '';
});
await expect(
cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
}),
).resolves.toBe(target);
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls).toContainEqual(['ls-files', '--', './AGENTS.md']);
expect(calls).toContainEqual(['checkout', 'HEAD', '--', './AGENTS.md']);
expect(calls).toContainEqual([
'clean',
'-fdx',
'--',
'./AGENTS.md',
'./CLAUDE.md',
'./.claude',
]);
expect(calls.some((c) => c[0] === 'status')).toBe(false);
expect(calls).toContainEqual(['checkout', '-B', 'develop', 'origin/develop']);
expect(calls.some((c) => c[0] === 'clean' && c.includes('/.gitnexus'))).toBe(false);
});
it('removes untracked AGENTS.md on the same branch so merge is not blocked', async () => {
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'develop');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'develop\n';
if (args[0] === 'ls-files') return ''; // untracked analyze output
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls).toContainEqual([
'clean',
'-fdx',
'--',
'./AGENTS.md',
'./CLAUDE.md',
'./.claude',
]);
expect(calls.some((c) => c[0] === 'checkout' && c.includes('HEAD'))).toBe(false);
expect(calls).toContainEqual(['checkout', '-B', 'develop', 'origin/develop']);
});
it('still switches — and still refuses a dirty tree — for a DIFFERENT branch', async () => {
// The refusal must survive where it matters: a real switch can discard
// local work, so the fast path above must not weaken it.
const root = await mkControlledRoot('gitnexus-controlled-root-');
try {
const target = await makeExistingClone(root, 'main');
const runGitForTest = vi.fn(async (args: string[]) => {
if (args[0] === 'rev-parse') return 'main\n'; // on a different branch
if (args[0] === 'status') return ' M src/index.ts\n';
return '';
});
await expect(
cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
}),
).rejects.toThrow(/local changes detected/);
} finally {
await fs.rm(root, { recursive: true, force: true });
}
});
it('treats a detached HEAD as needing the checkout path', async () => {
// `rev-parse --abbrev-ref HEAD` reports `HEAD` when detached. A SHA that
// does not match the requested ref is a real switch, not "already there".
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'main');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n';
if (args[0] === 'rev-parse' && args[1] === 'HEAD')
return 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1\n';
if (args[0] === 'rev-parse') return 'bbb222bbb222bbb222bbb222bbb222bbb222bbb2\n';
if (args[0] === 'status') return ''; // clean, so the checkout proceeds
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls.map((c) => c[0])).toContain('checkout');
expect(calls.some((c) => c[0] === 'checkout' && c.includes('-B'))).toBe(true);
});
it('does not switch or fetch when a detached HEAD SHA matches the requested ref', async () => {
// Tag / SHA pin: already at the commit. Fetching refs/heads/<name> would
// follow a same-named branch past the pin (#3199 review).
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
const sha = 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1';
try {
const target = await makeExistingClone(root, 'main');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n';
if (args[0] === 'rev-parse') return `${sha}\n`;
if (args[0] === 'status') return ' M AGENTS.md\n';
if (args[0] === 'ls-files' && args.includes('./AGENTS.md')) return 'AGENTS.md\n';
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls.some((c) => c[0] === 'status')).toBe(false);
expect(calls.some((c) => c[0] === 'checkout' && c.includes('-B'))).toBe(false);
expect(calls).toContainEqual(['checkout', 'HEAD', '--', './AGENTS.md']);
expect(calls.map((c) => c[0])).not.toContain('fetch');
expect(calls.map((c) => c[0])).not.toContain('merge');
});
it('peels an annotated tag so a tag pin is not treated as a switch', async () => {
// `rev-parse v1.0` is the tag object; HEAD is the peeled commit. Without
// `^{commit}` the SHA match misses and re-index porcelain-refuses.
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
const commit = 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1';
const tagObject = 'cccccccccccccccccccccccccccccccccccccccc';
try {
const target = await makeExistingClone(root, 'main');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n';
if (args[0] === 'rev-parse' && args[1] === 'HEAD') return `${commit}\n`;
if (args[0] === 'rev-parse' && args[1] === 'v1.0^{commit}') return `${commit}\n`;
if (args[0] === 'rev-parse' && args[1] === 'v1.0') return `${tagObject}\n`;
if (args[0] === 'rev-parse') throw new Error('unknown ref');
if (args[0] === 'status') return ' M AGENTS.md\n';
return '';
});
await cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'v1.0',
runGitForTest,
});
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls).toContainEqual(['rev-parse', 'v1.0^{commit}']);
expect(calls.some((c) => c[0] === 'status')).toBe(false);
expect(calls.some((c) => c[0] === 'checkout' && c.includes('-B'))).toBe(false);
expect(calls.map((c) => c[0])).not.toContain('fetch');
});
it('still refuses a dirty tree when a detached HEAD SHA does not match the requested ref', async () => {
const root = await mkControlledRoot('gitnexus-controlled-root-');
const calls: string[][] = [];
try {
const target = await makeExistingClone(root, 'main');
const runGitForTest = vi.fn(async (args: string[]) => {
calls.push(args);
if (args[0] === 'rev-parse' && args[1] === '--abbrev-ref') return 'HEAD\n';
if (args[0] === 'rev-parse' && args[1] === 'HEAD')
return 'aaa111aaa111aaa111aaa111aaa111aaa111aaa1\n';
if (args[0] === 'rev-parse') return 'bbb222bbb222bbb222bbb222bbb222bbb222bbb2\n';
if (args[0] === 'status') return ' M src/index.ts\n';
return '';
});
await expect(
cloneOrPull(REMOTE, target, undefined, {
allowedCloneRoot: root,
expectedRepoName: 'repo',
branch: 'develop',
runGitForTest,
}),
).rejects.toThrow(/local changes detected/);
} finally {
await fs.rm(root, { recursive: true, force: true });
}
expect(calls.some((c) => c[0] === 'status')).toBe(true);
expect(calls.some((c) => c[0] === 'checkout')).toBe(false);
});
it('allows auto-sync SSH SCP clone URLs with a per-repo timeout', async () => {
const root = await mkControlledRoot('gitnexus-controlled-root-');
const target = path.join(root, 'repo');
@ -1367,3 +1721,115 @@ describe('git-clone', () => {
});
});
});
describe('getCloneDir — a branch-pinned analyze gets its own working tree', () => {
it('keeps the historic directory for an unpinned request', () => {
// Backward compatibility: existing installs must keep using the dir they have.
expect(getCloneDir('Hello-World')).toBe(getCloneDir('Hello-World', undefined));
});
it('gives a pinned request a different directory from the unpinned one', () => {
// This separation is what stops (a) a pinned request failing on the dirty
// tree an earlier analyze left, and (b) a later unpinned request pulling on
// the branch a pin left checked out and indexing it as the default.
expect(getCloneDir('Hello-World', 'development')).not.toBe(getCloneDir('Hello-World'));
});
it('gives two different branches two different directories', () => {
expect(getCloneDir('Hello-World', 'development')).not.toBe(
getCloneDir('Hello-World', 'release/1.2'),
);
});
it('is stable for the same branch', () => {
expect(getCloneDir('Hello-World', 'release/1.2')).toBe(
getCloneDir('Hello-World', 'release/1.2'),
);
});
it('keeps a slash-bearing ref inside a single path segment under the clone root', () => {
// `release/1.2` must not become a nested directory, or the containment
// guarantees around CLONE_ROOT would be reasoning about a different path.
const dir = getCloneDir('Hello-World', 'release/1.2');
const base = getCloneDir('Hello-World');
expect(path.dirname(dir)).toBe(path.dirname(base));
expect(path.basename(dir)).not.toContain('/');
});
it('round-trips its own directory name, which is how DELETE re-derives it', () => {
// DELETE /api/repo calls getCloneDir(entry.name); the pinned repo is
// registered under this basename, so the two must agree.
const dir = getCloneDir('Hello-World', 'development');
expect(getCloneDir(path.basename(dir))).toBe(dir);
});
it('pinned basename differs from the GitHub stem (mid-job repoName / registerRepo)', () => {
// POST /api/analyze must set job.repoName and registryName to
// path.basename(targetPath), not extractWebRepoName(url). The stem is
// only getCloneDir's first argument (#3199 review).
const stem = 'Hello-World';
const basename = path.basename(getCloneDir(stem, 'development'));
expect(basename).not.toBe(stem);
expect(basename.startsWith(`${stem}__`)).toBe(true);
});
it('keeps the directory name inside the 255-byte filesystem limit', () => {
// validateBranchName allows a 255-char ref and branchSlug appends 9 more,
// so the naive `<repo>__<slug>` reached 267 and the clone could not create
// its target directory.
const longBranch = 'b'.repeat(255);
const base = path.basename(getCloneDir('Hello-World', longBranch));
expect(base.length).toBeLessThanOrEqual(255);
});
it('still separates two long branches that share a prefix', () => {
// Trimming keeps the hash, which is a digest of the FULL ref — otherwise
// two long branches would collapse onto one directory and silently share
// an index.
const a = 'b'.repeat(250) + 'one';
const b = 'b'.repeat(250) + 'two';
expect(getCloneDir('Hello-World', a)).not.toBe(getCloneDir('Hello-World', b));
expect(path.basename(getCloneDir('Hello-World', a)).length).toBeLessThanOrEqual(255);
});
it('round-trips a trimmed directory name too', () => {
const dir = getCloneDir('Hello-World', 'b'.repeat(255));
expect(getCloneDir(path.basename(dir))).toBe(dir);
});
it('still rejects a traversal attempt in the repo name', () => {
expect(() => getCloneDir('..', 'development')).toThrow(/Invalid repository name/);
expect(() => getCloneDir('a/b', 'development')).toThrow(/Invalid repository name/);
});
});
describe('analyzeCloneOptions — the /api/analyze glue for #3198', () => {
// The route passes the result straight to `cloneOrPull`. Inline, a regression
// that dropped `branch` for token-less URLs left every other test green while
// silently reindexing the default branch — so each combination is pinned.
it('returns undefined when neither a token nor a branch is supplied', () => {
expect(analyzeCloneOptions(undefined, undefined)).toBeUndefined();
});
it('carries a token on its own', () => {
expect(analyzeCloneOptions('ghp_token', undefined)).toEqual({ token: 'ghp_token' });
});
it('carries a branch on its own — the public-repo case', () => {
// The regression that would reopen #3198: a branch requested for a public
// URL must still reach cloneOrPull, with no token in play.
expect(analyzeCloneOptions(undefined, 'development')).toEqual({ branch: 'development' });
});
it('carries both together', () => {
expect(analyzeCloneOptions('ghp_token', 'development')).toEqual({
token: 'ghp_token',
branch: 'development',
});
});
it('treats an empty branch as absent rather than sending an empty ref', () => {
expect(analyzeCloneOptions('ghp_token', '')).toEqual({ token: 'ghp_token' });
expect(analyzeCloneOptions('', '')).toBeUndefined();
});
});

View file

@ -0,0 +1,15 @@
import { describe, expect, it } from 'vitest';
import { InvalidBranchError, validateBranchName } from '../../src/core/git-ref.js';
describe('core/git-ref', () => {
it('throws InvalidBranchError with name "InvalidBranchError"', () => {
expect(() => validateBranchName('HEAD', 'src')).toThrow(InvalidBranchError);
try {
validateBranchName('HEAD', 'src');
throw new Error('expected InvalidBranchError');
} catch (err) {
expect(err).toBeInstanceOf(InvalidBranchError);
expect((err as Error).name).toBe('InvalidBranchError');
}
});
});

View file

@ -383,6 +383,24 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => {
}
});
it('does not cache a chunk whose durable generation could not be reset', async () => {
// The reset failed, so the previous generation's shards are still on disk.
// Caching this chunk would let a future warm hit union those stale shards
// with the new ones. Same posture as a quarantined chunk: leave it uncached
// so the next run re-dispatches into a directory it can actually clear.
const f = writeFile('src/stale-generation.ts', 'export function stale() { return 1; }\n');
const cache = newCache();
prepareOverride.impl = () => Promise.reject(new Error('EACCES: simulated cache failure'));
try {
await expect(run(cache, [f])).resolves.toBeDefined();
} finally {
prepareOverride.impl = undefined;
}
// Nothing was written under any key -- neither on disk nor in memory.
expect(cache.onDiskKeys.size + cache.entries.size).toBe(0);
});
it('retains worker ParsedFiles when the main-thread run-store write fails', async () => {
const f = writeFile(
'src/persist-fallback.ts',

View file

@ -103,6 +103,50 @@ parentPort.on('message', (msg) => {
);
};
/**
* Like `writeResultWorker`, but each spawned instance writes its OWN marker
* keyed by `threadId`. The shared single-marker workers above can only prove
* "at least one worker started"; counting files in `markerDir` gives the actual
* pool size the parse phase asked `createWorkerPool` for, which is the only
* thing that distinguishes a clamped pool from an honored override.
*/
const writeSpawnCountingWorker = (workerPath: string, markerDir: string): void => {
fs.writeFileSync(
workerPath,
`
const fs = require('node:fs');
const path = require('node:path');
const { parentPort, threadId } = require('node:worker_threads');
fs.mkdirSync(${JSON.stringify(markerDir)}, { recursive: true });
fs.writeFileSync(path.join(${JSON.stringify(markerDir)}, 'worker-' + threadId), 'spawned');
parentPort.postMessage({ type: 'ready' });
const accumulated = {
nodes: [], relationships: [], symbols: [], imports: [], calls: [], assignments: [], heritage: [],
routes: [], fetchCalls: [], fetchWrapperDefs: [], decoratorRoutes: [], routerIncludes: [], routerImports: [], toolDefs: [], ormQueries: [], constructorBindings: [],
fileScopeBindings: [], parsedFiles: [], skippedLanguages: {}, fileCount: 0,
};
parentPort.on('message', (msg) => {
if (msg && msg.type === 'sub-batch') {
for (const file of msg.files) {
const filePath = file.path;
const name = filePath.split('/').pop().replace(/\.ts$/, '');
accumulated.nodes.push({
id: 'Function:' + filePath + ':' + name,
label: 'Function',
properties: { name, filePath, startLine: 1, endLine: 1, language: 'typescript' },
});
accumulated.fileCount++;
}
parentPort.postMessage({ type: 'progress', filesProcessed: accumulated.fileCount });
parentPort.postMessage({ type: 'sub-batch-done' });
return;
}
if (msg && msg.type === 'flush') parentPort.postMessage({ type: 'result', data: accumulated });
});
`,
);
};
const writeExitBeforeReadyWorker = (workerPath: string): void => {
fs.writeFileSync(workerPath, `process.exit(1);\n`);
};
@ -240,6 +284,56 @@ describe('parse-impl worker pool lazy startup', () => {
expect(Array.from(graph.nodes.values()).some((n) => n.properties.name === 'fatal')).toBe(false);
});
it('honors GITNEXUS_WORKER_POOL_SIZE above the work-proportional cap', async () => {
// The auto pool size is bounded by source bytes so a tiny repo does not
// spawn a full idle pool. That bound must apply to the AUTO default only —
// it used to clamp the operator's env override too, so an operator asking
// for more workers silently got the byte-derived number while `--workers`
// was honored.
//
// This asserts the pool the PARSE PHASE actually builds, not the resolver in
// isolation: `resolveAutoPoolSize()` already honored the env var before the
// fix, so a test at that level stays green through a revert.
const saved = process.env.GITNEXUS_WORKER_POOL_SIZE;
process.env.GITNEXUS_WORKER_POOL_SIZE = '3';
try {
// Four tiny files: total bytes are far under one CHUNK_BYTES_PER_WORKER so
// the work-proportional cap is 1, while the parseable count stays above the
// requested 3 (the pool never exceeds the number of files to parse).
const rels = ['src/a.ts', 'src/b.ts', 'src/c.ts', 'src/d.ts'];
const scanned = rels.map((rel) => {
const full = path.join(repoDir, rel);
fs.mkdirSync(path.dirname(full), { recursive: true });
fs.writeFileSync(full, `export function ${path.basename(rel, '.ts')}() { return 1; }\n`);
return { path: rel, size: fs.statSync(full).size };
});
const markerDir = path.join(tempDir, 'pool-size-markers');
const workerPath = path.join(tempDir, 'pool-size-worker.js');
writeSpawnCountingWorker(workerPath, markerDir);
const result = await runChunkedParseAndResolve(
createKnowledgeGraph(),
scanned,
rels,
rels.length,
repoDir,
Date.now(),
() => {},
// No `workerPoolSize`: the env var is the only override in play.
{ workerUrlForTest: pathToFileURL(workerPath) },
);
expect(result.usedWorkerPool).toBe(true);
// 3, not the byte-derived 1. Exactly this assertion fails on the clamped
// parent commit, which is what makes it a regression test for the fix.
expect(fs.readdirSync(markerDir)).toHaveLength(3);
} finally {
if (saved === undefined) delete process.env.GITNEXUS_WORKER_POOL_SIZE;
else process.env.GITNEXUS_WORKER_POOL_SIZE = saved;
}
});
it('throws when GITNEXUS_WORKER_POOL_SIZE=0 and no --workers flag (sequential parsing removed)', async () => {
const saved = process.env.GITNEXUS_WORKER_POOL_SIZE;
process.env.GITNEXUS_WORKER_POOL_SIZE = '0';

View file

@ -86,6 +86,64 @@ describe('parsedfile-store', () => {
}
});
it('repeated loads with different wantPaths each see their own files (shard-listing memo)', async () => {
// Scope resolution calls loadParsedFilesForPaths once per LANGUAGE over the
// same store, so the second and later passes reuse the shard path listings
// the first pass authenticated instead of re-reading every shard. The
// failure mode that memo introduces is a FALSE SKIP: pass 2 concludes a
// shard holds nothing it wants, and those files silently never reach the
// graph — an exit-0 wrong answer, not a crash. Each pass below wants files
// the previous pass did not, so a listing carried over from the wrong shard
// shows up as a missing file here.
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
try {
await persistParsedFileChunk(dir, 'chunk-0', [makeParsedFile('a.c'), makeParsedFile('b.c')]);
await persistParsedFileChunk(dir, 'chunk-1', [makeParsedFile('c.c')]);
await persistParsedFileChunk(dir, 'chunk-2', [makeParsedFile('d.c')]);
const first = await loadParsedFilesForPaths(dir, new Set(['a.c']));
expect([...first.keys()]).toEqual(['a.c']);
// Key-set asserts alone still pass if the memo never skipped: the
// envelope listing would open the shard and skip deserialize. Spy
// open+deserialize on a later miss so a no-memo path fails.
const deserialize = vi.spyOn(v8, 'deserialize');
const open = vi.spyOn(nodeFsPromises, 'open');
try {
const second = await loadParsedFilesForPaths(dir, new Set(['c.c', 'd.c']));
expect([...second.keys()].sort()).toEqual(['c.c', 'd.c']);
expect(open.mock.calls.map(([file]) => path.basename(String(file))).sort()).toEqual([
'chunk-1.v8',
'chunk-2.v8',
]);
expect(deserialize).toHaveBeenCalledTimes(2);
open.mockClear();
deserialize.mockClear();
const third = await loadParsedFilesForPaths(dir, new Set(['b.c']));
expect([...third.keys()]).toEqual(['b.c']);
expect(open.mock.calls.map(([file]) => path.basename(String(file)))).toEqual([
'chunk-0.v8',
]);
expect(deserialize).toHaveBeenCalledTimes(1);
const fourth = await loadParsedFilesForPaths(dir, new Set(['a.c', 'b.c', 'c.c', 'd.c']));
expect([...fourth.keys()].sort()).toEqual(['a.c', 'b.c', 'c.c', 'd.c']);
} finally {
deserialize.mockRestore();
open.mockRestore();
}
// A shard written AFTER the memo was populated is still found: the memo
// holds listings, not the shard roster, and the roster is re-read per call.
await persistParsedFileChunk(dir, 'chunk-3', [makeParsedFile('e.c')]);
const fifth = await loadParsedFilesForPaths(dir, new Set(['e.c', 'a.c']));
expect([...fifth.keys()].sort()).toEqual(['a.c', 'e.c']);
} finally {
await rm(dir, { recursive: true, force: true });
}
});
it('writes no shard for an empty chunk', async () => {
const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-'));
try {

View file

@ -37,7 +37,8 @@ describe('processParsing — worker-pool error propagation (U20)', () => {
const graph = createKnowledgeGraph();
const workerPool: WorkerPool = {
size: 1,
dispatch: vi.fn(async () => {
dispatch: vi.fn(async () => []),
dispatchGroups: vi.fn(async () => {
throw new Error('replacement worker failed');
}),
terminate: vi.fn(async () => undefined),
@ -63,7 +64,8 @@ describe('processParsing — worker-pool error propagation (U20)', () => {
const graph = createKnowledgeGraph();
const workerPool: WorkerPool = {
size: 1,
dispatch: vi.fn(async () => {
dispatch: vi.fn(async () => []),
dispatchGroups: vi.fn(async () => {
throw new WorkerPoolDispatchError(
'Worker pool circuit breaker tripped: 2 consecutive failures on slot 0',
['src/poison.ts'],
@ -114,6 +116,7 @@ describe('processParsing — worker-pool error propagation (U20)', () => {
const workerPool: WorkerPool = {
size: 1,
dispatch: vi.fn(async () => []),
dispatchGroups: vi.fn(async () => []),
terminate: vi.fn(async () => undefined),
getQuarantinedPaths: () => ['src/poison.ts'],
};

View file

@ -4,8 +4,24 @@ import { isProcessAlive, readProcessStartTime } from '../../src/utils/process-id
afterEach(() => {
vi.restoreAllMocks();
vi.doUnmock('node:child_process');
vi.resetModules();
});
/**
* Loads a fresh copy of the module (fresh memo) over a counted `execFileSync`,
* so "how many times did we actually shell out" is observable. `doMock` is not
* hoisted, so the statically imported functions used by the other tests keep
* the real implementation.
*/
async function withCountedProbe(probe: () => string) {
const execFileSync = vi.fn(probe);
vi.doMock('node:child_process', () => ({ execFileSync }));
vi.resetModules();
const identity = await import('../../src/utils/process-identity.js');
return { execFileSync, readProcessStartTimeCached: identity.readProcessStartTimeCached };
}
describe('process identity', () => {
it('treats only ESRCH as a dead process', () => {
const kill = vi.spyOn(process, 'kill');
@ -39,4 +55,33 @@ describe('process identity', () => {
}
},
);
it('probes this process once and re-probes a foreign pid every time', async () => {
const { execFileSync, readProcessStartTimeCached } = await withCountedProbe(() => 'STAMP\n');
expect(readProcessStartTimeCached(process.pid)).toBe('STAMP');
expect(readProcessStartTimeCached(process.pid)).toBe('STAMP');
// On Windows each probe is a powershell.exe spawn plus a WMI query.
expect(execFileSync).toHaveBeenCalledTimes(1);
// A foreign process can exit and its pid be reused — caching that stamp
// would blind the reuse check the stamp exists for.
expect(readProcessStartTimeCached(process.pid + 1)).toBe('STAMP');
expect(readProcessStartTimeCached(process.pid + 1)).toBe('STAMP');
expect(execFileSync).toHaveBeenCalledTimes(3);
});
it('retries after a failed self probe instead of caching the failure', async () => {
const { execFileSync, readProcessStartTimeCached } = await withCountedProbe(() => 'STAMP\n');
execFileSync.mockImplementationOnce(() => {
throw new Error('probe unavailable');
});
// A cached failure would leave acquireFileLock throwing "Unable to
// determine process start time" for the rest of the process's life.
expect(readProcessStartTimeCached(process.pid)).toBeUndefined();
expect(readProcessStartTimeCached(process.pid)).toBe('STAMP');
expect(readProcessStartTimeCached(process.pid)).toBe('STAMP');
expect(execFileSync).toHaveBeenCalledTimes(2);
});
});

View file

@ -270,12 +270,13 @@ describe('gitnexus skill-evolution workflow contract', () => {
});
it('passes the cell concurrency through to the benchmark', () => {
// The lane is serial unless told otherwise: concurrency only pays off when
// the runner has the vCPUs for it, and a cell starved of CPU drifts toward
// its session timeout, which the gate counts as an excluded run.
// Dispatch defaults to 3. Scheduled runs still fall back to serial unless
// GITNEXUS_EVOLUTION_WORKERS is set — a cell starved of CPU that hits the
// session ceiling is an excluded run the gate refuses.
expect(evolveJob?.env?.WORKERS).toBe(
"${{ inputs.workers || vars.GITNEXUS_EVOLUTION_WORKERS || '1' }}",
);
expect(workflow).toMatch(/workers:\n(?:[^\n]*\n){0,4} default: '3'/);
});
it('seeds from the newest usable completed main run, including failed runs', () => {
@ -572,6 +573,27 @@ exit 1`);
// uploads. The job must finish inside that window even when the schedule
// fires late (the 2026-08-01 run was queued 65 minutes after the cron).
expect(jobBudget as number).toBeLessThanOrEqual(21 * 60);
// A Friday dispatch inherits leftover uptime. The shared entrypoint — not
// the workflow YAML — must cap the sweep so it fails in-process and the
// always() upload still runs (run 33962002890).
const script = readFileSync(
path.join(REPO_ROOT, 'eval/workflow_bench/run-evolution.sh'),
'utf8',
);
// The flag, not a precomputed number: the CLI reads /proc/uptime in the
// same breath as it starts the clock the cap is measured against, so
// nothing between the two can be charged to the sweep.
expect(script).toContain('--max-runtime-from-instance-window');
expect(script).not.toContain('instance_window_budget_from_proc');
expect(script).toContain('export RUNTIME_DIGEST');
// The workflow's half of that contract is calling the entrypoint, not
// naming the flag. The YAML never mentions --max-runtime-from-instance-window
// at all — only the prose at gitnexus-skill-evolution.yml:179-184 describing
// the cap — so an assertion on flag text there would test a comment, and a
// loop step that had stopped invoking the script would still pass it.
expect(stepRun('Run the propose → benchmark → gate loop')).toContain(
'./workflow_bench/run-evolution.sh --apply',
);
});
it('uploads benchmark evidence unconditionally, on a path it addresses itself', () => {

View file

@ -58,7 +58,7 @@ let workerInstances: FakeWorker[] = [];
class FakeWorker extends EventEmitter {
readonly seenMessages: unknown[] = [];
constructor() {
constructor(startupExitCode?: number) {
super();
workerInstances.push(this);
// Real Worker fires 'online' asynchronously after the runtime is ready;
@ -67,6 +67,10 @@ class FakeWorker extends EventEmitter {
// message instead — emit that too so replacement-worker tests don't
// hit the WORKER_READY_TIMEOUT_MS budget (5s).
queueMicrotask(() => {
if (startupExitCode !== undefined) {
this.emit('exit', startupExitCode);
return;
}
this.emit('online');
this.emit('message', { type: 'ready' });
});
@ -262,8 +266,8 @@ describe('worker pool resilience', () => {
consecutiveFailureThreshold: 10,
maxRespawnsPerSlot: 1,
});
// Slot 0 dies twice, exceeding budget=1; slot 1 succeeds with the
// requeued remainder.
// Separate dispatches target the first live slot twice, independently
// of how multi-file packs are split across workers.
nextActions.push({ kind: 'crash-exit', code: 134, afterStartingFiles: 1 });
nextActions.push({ kind: 'crash-exit', code: 134, afterStartingFiles: 1 });
nextActions.push({
@ -272,9 +276,10 @@ describe('worker pool resilience', () => {
result: { fileCount: 2 },
});
await pool.dispatch([{ path: 'src/a.ts', content: '' }]);
await pool.dispatch([{ path: 'src/b.ts', content: '' }]);
expect(pool.getStats().activeSlots).toBe(1);
const results = await pool.dispatch<{ path: string; content: string }, unknown>([
{ path: 'src/a.ts', content: '' },
{ path: 'src/b.ts', content: '' },
{ path: 'src/c.ts', content: '' },
{ path: 'src/d.ts', content: '' },
]);
@ -495,18 +500,13 @@ describe('worker pool resilience', () => {
await pool.terminate();
});
it('drops slot when waitForWorkerOnline rejects (replaceWorker failure path)', async () => {
it('drops slot when replacement readiness rejects', async () => {
let factoryCallCount = 0;
const pool = createWorkerPool(workerUrl, 2, {
workerFactory: () => {
factoryCallCount++;
const worker = new FakeWorker();
// Slot 0's initial worker is healthy; the replacement (3rd factory
// call after slot 0 dies once) exits before emitting 'online'.
if (factoryCallCount === 3) {
// Override the queued 'online' microtask with an immediate 'exit'.
queueMicrotask(() => worker.emit('exit', 1));
}
// The replacement exits INSTEAD OF reporting ready.
const worker = new FakeWorker(factoryCallCount === 3 ? 1 : undefined);
return worker as unknown as import('node:worker_threads').Worker;
},
consecutiveFailureThreshold: 10,
@ -519,8 +519,9 @@ describe('worker pool resilience', () => {
result: { fileCount: 2 },
});
await pool.dispatch([{ path: 'src/a.ts', content: '' }]);
expect(pool.getStats().activeSlots).toBe(1);
const results = await pool.dispatch<{ path: string; content: string }, unknown>([
{ path: 'src/a.ts', content: '' },
{ path: 'src/b.ts', content: '' },
{ path: 'src/c.ts', content: '' },
]);
@ -532,6 +533,33 @@ describe('worker pool resilience', () => {
await pool.terminate();
});
it('finishes recovery when a failed worker never acknowledges termination', async () => {
const pool = createWorkerPool(workerUrl, 2, {
workerFactory: () => {
const worker = new FakeWorker();
if (workerInstances.length === 1) {
worker.terminate = () => new Promise<number>(() => {});
}
return worker as unknown as import('node:worker_threads').Worker;
},
shutdownDrainMs: 10,
});
nextActions.push({ kind: 'crash-exit', code: 134, afterStartingFiles: 1 });
nextActions.push({ kind: 'parse-ok', files: [{ path: 'src/good.ts' }] });
try {
const results = await pool.dispatch([
{ path: 'src/bad.ts', content: '' },
{ path: 'src/good.ts', content: '' },
]);
expect(results).toEqual([{ fileCount: 1 }]);
expect(pool.getQuarantinedPaths()).toEqual(['src/bad.ts']);
expect(pool.getStats().slotGenerations).toEqual([1, 0]);
expect(pool.getStats().activeSlots).toBe(2);
} finally {
await pool.terminate();
}
}, 1000);
it('trips the breaker when all slots exhaust their respawn budget', async () => {
const pool = createWorkerPool(workerUrl, 2, {
workerFactory: () => new FakeWorker() as unknown as import('node:worker_threads').Worker,