fix(ci): require every test to execute across the CI matrix (#3479)

This commit is contained in:
Gergő Magyar 2026-10-06 07:34:17 +03:00 • committed by GitHub
parent e8a09d067b
commit df49f90889
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
32 changed files with 1478 additions and 244 deletions

View file

@ -287,6 +287,7 @@ jobs:
# ── Locate test results ──
RESULTS_FILE=$(find_first "$DIR/test-reports" "test-results.json")
WEB_RESULTS_FILE=$(find_first "$DIR/test-reports" "web-test-results.json")
PYTEST_RESULTS_FILE=$(find_first "$DIR/test-reports" "pytest-results.json")
sum_results() {
local file=$1
@ -303,12 +304,21 @@ jobs:
# framework, not as a top-line metric).
read -r CLI_T CLI_P CLI_F CLI_S _ CLI_D <<< "$(sum_results "$RESULTS_FILE")"
read -r WEB_T WEB_P WEB_F WEB_S _ WEB_D <<< "$(sum_results "$WEB_RESULTS_FILE")"
read -r PY_T PY_P PY_F PY_S _ PY_D <<< "$(sum_results "$PYTEST_RESULTS_FILE")"
TOTAL=$((CLI_T + WEB_T))
PASSED=$((CLI_P + WEB_P))
FAILED=$((CLI_F + WEB_F))
SKIPPED=$((CLI_S + WEB_S))
TOTAL=$((CLI_T + WEB_T + PY_T))
PASSED=$((CLI_P + WEB_P + PY_P))
FAILED=$((CLI_F + WEB_F + PY_F))
SKIPPED=$((CLI_S + WEB_S + PY_S))
DURATION=$((CLI_D > WEB_D ? CLI_D : WEB_D))
DURATION=$((DURATION > PY_D ? DURATION : PY_D))
EXECUTION_ERRORS=0
for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE" "$PYTEST_RESULTS_FILE"; do
if [ -n "$rf" ] && [ -f "$rf" ]; then
errors=$(jq -r '(.executionFailures // []) | length' "$rf")
EXECUTION_ERRORS=$((EXECUTION_ERRORS + errors))
fi
done
# ── Status helpers ──
status_icon() {
@ -385,22 +395,22 @@ jobs:
if [ "$TOTAL" -gt 0 ] 2>/dev/null; then
echo "### Test Results"
echo ""
echo "| Tests | Passed | Failed | Skipped | Duration |"
echo "| Tests | Passed | Failed | Unverified | Duration |"
echo "|-------|--------|--------|---------|----------|"
echo "| ${TOTAL} | ${PASSED} | ${FAILED} | ${SKIPPED} | ${DURATION}s |"
echo ""
if [ "$FAILED" = "0" ]; then
if [[ "$FAILED" == "0" && "$SKIPPED" == "0" && "$EXECUTION_ERRORS" == "0" && "$TESTS" == "success" ]]; then
echo "✅ All **${PASSED}** tests passed"
else
echo "❌ **${FAILED}** failed / **${PASSED}** passed"
echo "❌ Test execution is incomplete or failed: **${FAILED}** failed, **${SKIPPED}** unverified, **${EXECUTION_ERRORS}** runner/suite errors."
fi
if [ "$SKIPPED" != "0" ]; then
echo ""
echo "<details>"
echo "<summary>${SKIPPED} test(s) skipped — expand for details</summary>"
echo "<summary>${SKIPPED} test(s) have no recorded pass — expand for details</summary>"
echo ""
for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE"; do
for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE" "$PYTEST_RESULTS_FILE"; do
if [ -n "$rf" ] && [ -f "$rf" ]; then
jq -r '
.testResults[]

View file

@ -32,6 +32,7 @@ jobs:
env:
GITNEXUS_REQUIRE_FTS: '1'
GITNEXUS_REQUIRE_ZIG: '1'
GITNEXUS_REQUIRE_VECTOR: '1'
steps:
# persist-credentials: false — runs tests + uploads a blob artifact; the
# default-persisted token must not be capturable through it (zizmor
@ -115,7 +116,7 @@ jobs:
run: >-
npx vitest --mergeReports
--reporter=default
--reporter=json
--reporter=./scripts/execution-reporter.ts
--outputFile=test-results.json
--coverage
--coverage.reporter=json-summary
@ -131,7 +132,8 @@ jobs:
run: >-
npx vitest run
--reporter=default
--reporter=json
--reporter=../gitnexus/scripts/execution-reporter.ts
--includeTaskLocation
--outputFile=web-test-results.json
working-directory: gitnexus-web
- name: Run docker-server integration tests
@ -140,7 +142,7 @@ jobs:
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: test-reports
name: coverage-reports
path: |
gitnexus/coverage/coverage-summary.json
gitnexus/coverage/coverage-final.json
@ -162,7 +164,7 @@ jobs:
steps:
- id: gen
run: |
TOTAL=3 # cross-platform (windows/macOS) shards per OS
TOTAL=4 # cross-platform (windows/macOS) shards per OS
COV_TOTAL=3 # ubuntu coverage shards (merged before thresholds)
if [ "$TOTAL" -lt 1 ] || [ "$COV_TOTAL" -lt 1 ]; then
echo "shard totals must be >= 1" >&2; exit 1
@ -254,8 +256,48 @@ jobs:
shell: bash
env:
SHARD: ${{ matrix.shard }}/${{ needs.shard-plan.outputs.total }}
GITNEXUS_TEST_REPORT: ${{ matrix.os }}-${{ matrix.shard }}.json
run: npx tsx scripts/run-cross-platform.ts --shard="$SHARD"
working-directory: gitnexus
- name: Upload platform execution report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: execution-${{ matrix.os }}-${{ matrix.shard }}
path: gitnexus/${{ matrix.os }}-${{ matrix.shard }}.json
if-no-files-found: error
retention-days: 5
# Branch protection still requires these six names from the three-shard
# matrix. Keep them as aggregate gates as the real matrix grows: every native
# shard and the complete execution audit must pass before any gate succeeds.
# These jobs only verify dependency results; tests run in cross-platform above.
protected-platform-checks:
name: ${{ matrix.os }} (platform-sensitive) ${{ matrix.check }}/3
needs: [cross-platform, test-completeness]
if: always()
permissions: {}
strategy:
fail-fast: false
matrix:
os: [windows-latest, macos-latest]
check: [1, 2, 3]
runs-on: ubuntu-latest
timeout-minutes: 2
steps:
- name: Require all native shards and the execution audit
shell: bash
env:
NATIVE_RESULT: ${{ needs.cross-platform.result }}
EXECUTION_RESULT: ${{ needs.test-completeness.result }}
run: |
echo "Compatibility gate for the former three-shard check name."
echo "All native shards: $NATIVE_RESULT"
echo "Complete execution audit: $EXECUTION_RESULT"
if [[ "$NATIVE_RESULT" != "success" || "$EXECUTION_RESULT" != "success" ]]; then
echo "::error::Every native shard and the execution audit must succeed."
exit 1
fi
# Tree-sitter ABI gate (#1922). Two halves, both blocking:
# 1. Static, offline: assert every grammar's compiled ABI loads on the
@ -482,11 +524,8 @@ jobs:
# heap, so parallel forks both skew the timings and OOM the worker pool — they
# must run one file at a time.
#
# go-pipeline-benchmark.test.ts is deliberately NOT included: its
# worker-pool (#1848) suite spins a real worker pool that exits unexpectedly
# under vitest's fork pool (reproduced in validation), which would make this
# gate flaky. Go is already guarded by its non-gated O(n^2) tripwire (runs in
# the main coverage job) plus its golden capture-parity test.
# Every opt-in suite is discovered by run-benchmarks.ts. A failing worker
# benchmark is a failure to investigate, never a reason to omit that suite.
benchmarks:
name: benchmarks (GITNEXUS_BENCH)
runs-on: ubuntu-latest
@ -876,28 +915,18 @@ jobs:
- name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial)
if: ${{ !cancelled() }}
# cpp-adl-benchmark.test.ts and csharp-razor-view-components-benchmark.test.ts
# are not `*-pipeline-benchmark.test.ts` files but belong here for the
# same reason: they are skipIf-gated on GITNEXUS_BENCH, so the scaling
# guards they hold never run in the main coverage job.
env:
GITNEXUS_BENCH: '1'
GITNEXUS_WORKER_READY_TIMEOUT_MS: '60000'
run: >-
npx vitest run --no-file-parallelism
test/integration/cobol-pipeline-benchmark.test.ts
test/integration/csharp-pipeline-benchmark.test.ts
test/integration/csharp-razor-view-components-benchmark.test.ts
test/integration/objective-c-pipeline-benchmark.test.ts
test/integration/cpp-adl-benchmark.test.ts
test/integration/data-route-table-benchmark.test.ts
test/integration/instance-ownership-pipeline-benchmark.test.ts
test/integration/spring-bean-resource-benchmark.test.ts
test/integration/spring-dynamic-lookup-benchmark.test.ts
test/integration/rust-pipeline-benchmark.test.ts
test/integration/php-pipeline-benchmark.test.ts
test/integration/ruby-pipeline-benchmark.test.ts
run: npm run test:benchmarks
working-directory: gitnexus
- name: Upload benchmark execution report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: execution-benchmarks
path: gitnexus/benchmarks.json
if-no-files-found: error
retention-days: 5
# Locked eval suite. setup-uv and uv itself are immutable so CI exercises
# exactly the dependency graph developers run from eval/uv.lock.
@ -916,8 +945,88 @@ jobs:
python-version: '3.13'
enable-cache: true
cache-dependency-glob: eval/uv.lock
- run: uv run --locked --extra dev python -m pytest tests -q
- run: uv run --locked --extra dev python -m pytest tests -q --junitxml=pytest-locked.xml
working-directory: eval
- name: Upload pytest execution report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: execution-pytest-locked
path: eval/pytest-locked.xml
if-no-files-found: error
retention-days: 5
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
- name: Exercise real workflow evidence preflight
if: ${{ !cancelled() }}
run: >-
npx vitest run test/unit/skill-evolution-workflow.test.ts
--reporter=default --reporter=./scripts/execution-reporter.ts --outputFile=preflight.json
working-directory: gitnexus
- name: Upload preflight execution report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: execution-preflight
path: gitnexus/preflight.json
if-no-files-found: error
retention-days: 5
test-completeness:
name: every test executed
needs:
[
shard-plan,
coverage-merge,
cross-platform,
benchmarks,
eval-tests,
eval-containment-linux,
eval-containment-windows,
]
if: ${{ !cancelled() }}
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
persist-credentials: false
- uses: ./.github/actions/setup-gitnexus
- name: Download coverage and web reports
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
name: coverage-reports
- name: Download all execution receipts
if: ${{ !cancelled() }}
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8
with:
pattern: execution-*
path: execution-reports
merge-multiple: true
- name: Require a real pass for every collected test
env:
PLATFORM_SHARDS: ${{ needs.shard-plan.outputs.total }}
run: |
cp gitnexus/test-results.json execution-reports/coverage.json
cd gitnexus
node --import tsx scripts/test-completeness.ts ../execution-reports test-results.json "$PLATFORM_SHARDS"
- name: Require a real pass for every pytest case
if: ${{ !cancelled() }}
run: python3 eval/check_test_execution.py execution-reports eval/pytest-results.json
- name: Upload complete test reports
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: test-reports
path: |
gitnexus/coverage/coverage-summary.json
gitnexus/coverage/coverage-final.json
gitnexus/test-results.json
gitnexus-web/web-test-results.json
eval/pytest-results.json
if-no-files-found: error
retention-days: 5
# Native Linux ownership and Bubblewrap boundary. The environment flag makes
# the real namespace test mandatory; a missing/blocked bwrap is a failure.
@ -992,8 +1101,19 @@ jobs:
tests/test_workflow_bench_sessions.py
tests/test_ce_plugin_runtime.py
tests/test_offline_sweep_integration.py
tests/test_mock_provider.py -q
tests/test_mock_provider.py
tests/test_evolve.py::test_outer_runner_pid_namespace_kills_setsid_descendant
tests/test_oracle_assets.py::test_hidden_vitest_config_executes_sibling_oracle_against_candidate_checkout
-q --junitxml=pytest-ubuntu.xml
working-directory: eval
- name: Upload pytest execution report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: execution-pytest-ubuntu
path: eval/pytest-ubuntu.xml
if-no-files-found: error
retention-days: 5
# Native Windows Job Object canary. POSIX-only tests skip by platform, while
# the grandchild delayed-write test must execute and pass on this runner.
@ -1015,5 +1135,13 @@ jobs:
run: >-
uv run --locked --extra dev python -m pytest
tests/test_process_control.py tests/test_model_gateway.py
-k "not locked_litellm" -q
-k "not locked_litellm" -q --junitxml=pytest-windows.xml
working-directory: eval
- name: Upload pytest execution report
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: execution-pytest-windows
path: eval/pytest-windows.xml
if-no-files-found: error
retention-days: 5

8
.gitignore vendored
View file

@ -131,5 +131,13 @@ local_docs/
.context/
gitnexus/web/
# Local copies of CI execution receipts.
gitnexus/benchmarks.json
gitnexus/preflight.json
gitnexus/test-results.json
gitnexus/windows-latest-*.json
gitnexus/macos-latest-*.json
gitnexus-web/web-test-results.json
# Machine-local skill-evolution evidence (consumed by eval/workflow_bench/evolve.py)
eval/workflow_bench/learnings.jsonl

View file

@ -611,6 +611,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
| Variable | Default | Effect | Tune when… |
| ----------------------------------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `GITNEXUS_TEST_REPORT` | unset | Write a JSON execution receipt from `scripts/run-cross-platform.ts`; accepts a basename ending in `.json`. | CI platform shards use distinct filenames so the completeness check can require every report. |
| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers <n>`. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. |
| `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. |
| `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. |

View file

@ -125,11 +125,46 @@ GitHub Actions (`.github/workflows/ci.yml`) orchestrate:
| `ci-scope-parity.yml` | discover, parity | Scope-resolution parity for all migrated languages |
| `ci-e2e.yml` | e2e (chromium) | Playwright E2E, gated on `gitnexus-web/**` changes |
The `CI Gate` job in `ci.yml` is the single required check for branch protection. It requires quality, tests, e2e, and scope-parity to all pass.
The `CI Gate` job in `ci.yml` requires the quality and test workflows to pass.
The browser E2E workflow must pass or be skipped because no web files changed.
Branch protection also requires six platform check names from the former
three-shard matrix. These names remain as aggregate gates: all native shards
and the `every test executed` audit must succeed before any of them passes.
Failed, cancelled, skipped, or missing dependency results fail these gates.
The actual native tests run in the current Windows/macOS shard matrix.
The `typecheck` job runs both the production compiler check and
`npm run typecheck:tests`. A type error in either check fails the job and the CI gate.
### Complete execution, including platform and benchmark tests
The required `every test executed` job reconciles execution receipts from Ubuntu
coverage, every Windows/macOS shard, the serial benchmark run, and the real Python
workflow preflight. It requires a recorded pass for every collected test. A skip
on Linux is satisfied only by a pass of that exact test in another required job.
Test identities include the file, suite/title and source location; ambiguous
parameterized cases must have unique titles. Missing receipts, missing test files,
unhandled runner errors, failed hooks, failed assertions, and tests with no pass
all fail the gate. Web tests are checked separately with the same rules.
The locked pytest suite and Linux/Windows containment jobs also upload JUnit
receipts. The same gate checks every Python test file was collected and every
case passed in at least one job, while preserving failures from any job. The
Linux containment job supplies Bubblewrap, the pinned CLI, and built Vitest
dependencies for tests that cannot run in the basic Python job. Python results
are included in the combined PR report.
The PR report shows the reconciled result as **Unverified**. Zero means every test
has execution evidence; individual OS logs still show tests that require another
OS as skipped. Failed executions remain failures even if another job passes.
`npm run test:benchmarks` discovers all tests gated by `GITNEXUS_BENCH` and runs
them serially. Keep timing measurements out of parallel coverage workers. The
`eval-tests` job installs the locked Python dependencies and runs the workflow
preflight Vitest tests as well as pytest; those tests must exercise real Python
validation, never a stubbed success.
## Regression testing
Re-run the full relevant suite when:

View file

@ -0,0 +1,121 @@
"""Require a real passing execution of every pytest case across the CI jobs."""
from __future__ import annotations
import json
from pathlib import Path
import sys
import xml.etree.ElementTree as ET
RECEIPTS = ("pytest-locked.xml", "pytest-ubuntu.xml", "pytest-windows.xml")
def reconcile(reports: list[Path], expected_files: list[str]) -> dict:
if not reports:
raise ValueError("No pytest execution reports supplied")
tests: dict[tuple[str, str], set[str]] = {}
durations: list[float] = []
for report in reports:
root = ET.parse(report).getroot()
suites = [root] if root.tag == "testsuite" else list(root.findall("testsuite"))
if root.tag not in {"testsuite", "testsuites"} or not suites:
raise ValueError(f"Invalid pytest report: {report}")
seen: set[tuple[str, str]] = set()
duration = 0.0
for suite in suites:
cases = suite.findall("testcase")
if not cases:
raise ValueError(f"Empty pytest suite in {report}")
counts = {"tests": len(cases), "failures": 0, "errors": 0, "skipped": 0}
for case in cases:
key = (case.get("classname", ""), case.get("name", ""))
if not all(key) or key in seen:
raise ValueError(f"Missing or ambiguous pytest identity in {report}: {key}")
seen.add(key)
for status in ("failures", "errors", "skipped"):
tag = {"failures": "failure", "errors": "error", "skipped": "skipped"}[status]
counts[status] += len(case.findall(tag))
status = (
"failed"
if case.find("failure") is not None or case.find("error") is not None
else "pending"
if case.find("skipped") is not None
else "passed"
)
tests.setdefault(key, set()).add(status)
for name, count in counts.items():
if int(suite.get(name, "-1")) != count:
raise ValueError(f"Inconsistent pytest {name} count in {report}")
duration += float(suite.get("time", "0"))
durations.append(duration)
modules = {module for module, _ in tests}
for file in expected_files:
module = file.removesuffix(".py").replace("/", ".")
if not any(name == module or name.startswith(module + ".") for name in modules):
raise ValueError(f"Pytest file was never collected: {file}")
suites_by_module: dict[str, dict] = {}
failed: list[str] = []
pending = 0
for (module, name), statuses in sorted(tests.items()):
status = "failed" if "failed" in statuses else "passed" if "passed" in statuses else "pending"
full_name = f"{module}::{name}"
if status == "failed":
failed.append(full_name)
pending += status == "pending"
suite = suites_by_module.setdefault(
module,
{
"name": module,
"status": "passed",
"assertionResults": [],
"endTime": round(max(durations) * 1000),
},
)
suite["assertionResults"].append(
{
"fullName": full_name,
"ancestorTitles": [module],
"title": name,
"status": status,
}
)
if status == "failed":
suite["status"] = "failed"
return {
"success": not failed and pending == 0,
"numTotalTests": len(tests),
"numPassedTests": len(tests) - len(failed) - pending,
"numFailedTests": len(failed),
"numPendingTests": pending,
"numTotalTestSuites": len(suites_by_module),
"numFailedTestSuites": sum(s["status"] == "failed" for s in suites_by_module.values()),
"executionFailures": failed,
"startTime": 0,
"testResults": list(suites_by_module.values()),
}
def main() -> int:
if len(sys.argv) != 3:
raise ValueError("Usage: check_test_execution.py <reports-dir> <output.json>")
directory, output = map(Path, sys.argv[1:])
root = Path(__file__).resolve().parent
expected = [p.relative_to(root).as_posix() for p in (root / "tests").rglob("test_*.py")]
report = reconcile([directory / name for name in RECEIPTS], expected)
output.write_text(json.dumps(report), encoding="utf-8")
print(
f"Eval execution: {report['numPassedTests']}/{report['numTotalTests']} passed; "
f"{report['numPendingTests']} unverified; {report['numFailedTests']} failures"
)
for suite in report["testResults"]:
for case in suite["assertionResults"]:
if case["status"] != "passed":
print(f"{case['status']}: {case['fullName']}", file=sys.stderr)
return 0 if report["success"] else 1
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,142 @@
"""The execution gate must reject absent, skipped, failed, or ambiguous evidence."""
from pathlib import Path
import json
import shutil
import subprocess
import sys
import xml.etree.ElementTree as ET
import pytest
from check_test_execution import reconcile
def receipt(tmp_path: Path, filename: str, cases: list[tuple[str, str]], module="tests.test_native") -> Path:
suite = ET.Element("testsuite", tests=str(len(cases)), failures="0", errors="0", skipped="0", time="1.5")
for name, status in cases:
case = ET.SubElement(suite, "testcase", classname=module, name=name)
if status != "passed":
ET.SubElement(case, status)
count = {"failure": "failures", "error": "errors", "skipped": "skipped"}[status]
suite.set(count, str(int(suite.get(count)) + 1))
path = tmp_path / filename
ET.ElementTree(suite).write(path, encoding="utf-8")
return path
def test_platform_skip_requires_a_matching_real_pass(tmp_path):
linux = receipt(tmp_path, "linux.xml", [("test_windows", "skipped"), ("test_posix", "passed")])
windows = receipt(tmp_path, "windows.xml", [("test_windows", "passed"), ("test_posix", "skipped")])
report = reconcile([linux, windows], ["tests/test_native.py"])
assert report["success"] is True
assert report["numPassedTests"] == report["numTotalTests"] == 2
assert report["numPendingTests"] == report["numFailedTests"] == 0
def test_unresolved_skip_is_not_a_pass(tmp_path):
path = receipt(tmp_path, "report.xml", [("test_missing_runtime", "skipped")])
report = reconcile([path], [])
assert report["success"] is False
assert report["numPendingTests"] == 1
assert report["numPassedTests"] == 0
assert report["testResults"][0]["assertionResults"][0]["ancestorTitles"] == ["tests.test_native"]
@pytest.mark.parametrize("status", ["failure", "error"])
def test_failure_is_not_hidden_by_another_job_passing(tmp_path, status):
failed = receipt(tmp_path, "failed.xml", [("test_shared", status)])
passed = receipt(tmp_path, "passed.xml", [("test_shared", "passed")])
report = reconcile([failed, passed], [])
assert report["success"] is False
assert report["numFailedTests"] == 1
assert report["numPassedTests"] == 0
assert report["executionFailures"] == ["tests.test_native::test_shared"]
def test_same_title_in_another_file_does_not_resolve_a_skip(tmp_path):
pending = receipt(tmp_path, "pending.xml", [("test_same", "skipped")])
unrelated = receipt(tmp_path, "unrelated.xml", [("test_same", "passed")], module="tests.test_other")
report = reconcile([pending, unrelated], [])
assert report["numTotalTests"] == 2
assert report["numPendingTests"] == 1
def test_duplicate_identity_is_rejected(tmp_path):
path = receipt(tmp_path, "duplicate.xml", [("test_same", "passed"), ("test_same", "skipped")])
with pytest.raises(ValueError, match="ambiguous pytest identity"):
reconcile([path], [])
def test_missing_receipt_is_rejected(tmp_path):
with pytest.raises(FileNotFoundError):
reconcile([tmp_path / "missing.xml"], [])
def test_empty_receipt_is_rejected(tmp_path):
path = receipt(tmp_path, "empty.xml", [])
with pytest.raises(ValueError, match="Empty pytest suite"):
reconcile([path], [])
def test_missing_file_is_rejected(tmp_path):
path = receipt(tmp_path, "report.xml", [("test_present", "passed")])
with pytest.raises(ValueError, match="never collected: tests/test_missing.py"):
reconcile([path], ["tests/test_missing.py"])
def test_inconsistent_counts_are_rejected(tmp_path):
path = receipt(tmp_path, "report.xml", [("test_present", "passed")])
path.write_text(path.read_text().replace('tests="1"', 'tests="2"'))
with pytest.raises(ValueError, match="Inconsistent pytest tests count"):
reconcile([path], [])
def test_pytest_testsuites_wrapper_and_test_classes(tmp_path):
path = receipt(tmp_path, "report.xml", [("test_method", "passed")], module="tests.test_native.TestNative")
suite = ET.parse(path).getroot()
root = ET.Element("testsuites")
root.append(suite)
ET.ElementTree(root).write(path, encoding="utf-8")
assert reconcile([path], ["tests/test_native.py"])["success"] is True
def test_missing_report_list_is_rejected():
with pytest.raises(ValueError, match="No pytest execution reports"):
reconcile([], [])
@pytest.mark.parametrize("scenario", ["passed", "pending", "failed", "missing-receipt", "missing-file"])
def test_cli_requires_complete_execution_evidence(tmp_path, scenario):
script = tmp_path / "check_test_execution.py"
shutil.copyfile(Path(__file__).parents[1] / script.name, script)
tests = tmp_path / "tests"
tests.mkdir()
(tests / "test_native.py").write_text("# inventory fixture\n")
reports = tmp_path / "reports"
reports.mkdir()
receipt(reports, "pytest-locked.xml", [("test_native", "skipped" if scenario == "pending" else "passed")])
receipt(reports, "pytest-ubuntu.xml", [("test_native", "error" if scenario == "failed" else "skipped")])
if scenario != "missing-receipt":
receipt(reports, "pytest-windows.xml", [("test_native", "skipped")])
if scenario == "missing-file":
(tests / "test_missing.py").write_text("# must be collected\n")
output = tmp_path / "pytest-results.json"
result = subprocess.run(
[sys.executable, str(script), str(reports), str(output)],
capture_output=True,
text=True,
timeout=10,
)
assert result.returncode == (0 if scenario == "passed" else 1)
if scenario.startswith("missing-"):
assert not output.exists()
assert ("pytest-windows.xml" if scenario == "missing-receipt" else "never collected") in result.stderr
else:
report = json.loads(output.read_text())
assert report["success"] is (scenario == "passed")
assert report["numTotalTests"] == 1
assert report["numPassedTests"] == (scenario == "passed")
assert report["numPendingTests"] == (scenario == "pending")
assert report["numFailedTests"] == (scenario == "failed")
assert "Eval execution:" in result.stdout

View file

@ -5,6 +5,7 @@ import json
import os
import subprocess
import sys
import threading
import time
from contextlib import contextmanager
from datetime import UTC, datetime, timedelta
@ -1428,21 +1429,32 @@ def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path):
pytest.skip(str(exc))
raise AssertionError("pytest.skip() returned unexpectedly")
sentinel = tmp_path / "escaped"
ready = tmp_path / "descendant-ready"
cancel = threading.Event()
descendant = (
"import time,pathlib; "
f"pathlib.Path({str(ready)!r}).touch(); print('ready', flush=True); "
f"time.sleep(1); pathlib.Path({str(sentinel)!r}).touch()"
)
child = (
"import os,subprocess,sys,time; "
f"subprocess.Popen([sys.executable,'-c',\"import time,pathlib;time.sleep(1);pathlib.Path({str(sentinel)!r}).touch()\"],preexec_fn=os.setsid); "
f"subprocess.Popen([sys.executable,'-c',{descendant!r}],preexec_fn=os.setsid); "
"time.sleep(10)"
)
result = run_managed(
pid_namespace_command([sys.executable, "-c", child], bwrap_bin=bwrap),
timeout=0.15,
# Cancel only after the escaped-session descendant has actually started.
# A timeout during bwrap/Python startup must not count as containment.
timeout=10,
terminate_grace=0.1,
require_pid_namespace=True,
cancel_event=cancel,
stdout_observer=lambda _chunk: cancel.set(),
)
time.sleep(1.1)
assert ready.exists(), "the setsid descendant must start before it can be contained"
assert not result.ok
assert result.state in {"timeout", "forced-kill"}
assert result.state == "cancelled"
time.sleep(1.1)
assert not sentinel.exists()

View file

@ -250,7 +250,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
# Carries the real-CLI identity probe, which needs CLAUDE_CANARY_BIN -
# set only on this job. Omitted from this list it skipped everywhere.
"tests/test_mock_provider.py",
# These also need runtime dependencies absent from the locked pytest job.
"tests/test_evolve.py::test_outer_runner_pid_namespace_kills_setsid_descendant",
"tests/test_oracle_assets.py::test_hidden_vitest_config_executes_sibling_oracle_against_candidate_checkout",
"-q",
"--junitxml=pytest-ubuntu.xml",
]
bwrap_canary_marker = re.compile(
r'@pytest\.mark\.skipif\(\s*os\.environ\.get\("GITNEXUS_REQUIRE_BWRAP_CANARY"\)',
@ -266,6 +270,72 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
assert "eval-containment-windows:" in workflow
@pytest.fixture(scope="module")
def ci_test_jobs():
repo_root = Path(__file__).resolve().parents[2]
workflow = yaml.safe_load((repo_root / ".github" / "workflows" / "ci-tests.yml").read_text())
return workflow["jobs"]
def test_platform_ci_preserves_required_check_names_with_complete_execution_gates(ci_test_jobs):
jobs = ci_test_jobs
gate = jobs["protected-platform-checks"]
matrix = gate["strategy"]["matrix"]
names = {
"tests / " + gate["name"].replace("${{ matrix.os }}", platform).replace("${{ matrix.check }}", str(check))
for platform in matrix["os"]
for check in matrix["check"]
}
assert names == {
f"tests / {platform} (platform-sensitive) {check}/3"
for platform in ("windows-latest", "macos-latest")
for check in (1, 2, 3)
}
plan = jobs["shard-plan"]["steps"][0]["run"]
total = int(re.search(r"^TOTAL=(\d+)\b", plan, re.MULTILINE)[1])
native_names = {
"tests / "
+ jobs["cross-platform"]["name"]
.replace("${{ matrix.os }}", platform)
.replace("${{ matrix.shard }}", str(shard))
.replace("${{ needs.shard-plan.outputs.total }}", str(total))
for platform in jobs["cross-platform"]["strategy"]["matrix"]["os"]
for shard in range(1, total + 1)
}
assert names.isdisjoint(native_names)
assert set(gate["needs"]) == {"cross-platform", "test-completeness"}
assert gate["if"] == "always()"
assert not gate.get("continue-on-error", False)
assert not jobs["cross-platform"].get("continue-on-error", False)
assert not jobs["test-completeness"].get("continue-on-error", False)
assert gate["permissions"] == {}
assert len(gate["steps"]) == 1
step = gate["steps"][0]
assert not step.get("continue-on-error", False)
assert "if" not in step
assert step["shell"] == "bash"
assert step["env"] == {
"NATIVE_RESULT": "${{ needs.cross-platform.result }}",
"EXECUTION_RESULT": "${{ needs.test-completeness.result }}",
}
@pytest.mark.parametrize("native_result", ["success", "failure", "cancelled", "skipped", ""])
@pytest.mark.parametrize("execution_result", ["success", "failure", "cancelled", "skipped", ""])
def test_required_platform_check_fails_unless_every_dependency_passed(ci_test_jobs, native_result, execution_result):
step = ci_test_jobs["protected-platform-checks"]["steps"][0]
result = subprocess.run(
["bash", "--noprofile", "--norc", "-e", "-o", "pipefail", "-c", step["run"]],
env={**os.environ, "NATIVE_RESULT": native_result, "EXECUTION_RESULT": execution_result},
capture_output=True,
text=True,
timeout=10,
check=False,
)
expected = 0 if native_result == execution_result == "success" else 1
assert result.returncode == expected, result.stdout + result.stderr
def test_shipped_scenarios_opt_out_the_cross_module_cell_and_rebuild_graph_assets():
task_file = Path(__file__).resolve().parents[1] / "workflow_bench" / "tasks.scenarios.yaml"
tasks = yaml.safe_load(task_file.read_text())["tasks"]
@ -1092,4 +1162,3 @@ def test_an_uninvoked_skill_still_counts_toward_the_arm_median():
# The invariant that makes the half-fix unsafe: the median and the run count
# the gate reads must cover the same rows.
assert agg["valid_runs"] == 2

View file

@ -59,6 +59,7 @@
"test:watch": "vitest",
"test:coverage": "vitest run --coverage",
"test:cross-platform": "tsx scripts/run-cross-platform.ts",
"test:benchmarks": "node --import tsx scripts/run-benchmarks.ts",
"postinstall": "node scripts/build-tree-sitter-grammars.cjs",
"assert-publish-coverage": "node scripts/assert-publish-grammar-coverage.cjs",
"prepare": "node scripts/build.js",

View file

@ -206,9 +206,8 @@ const LBUG_NATIVE = [
// (the same reason fts-extension-e2e.test.ts is registered below), so the
// FTS-unavailable branch has to run on a real Windows/macOS runner rather
// than only on Ubuntu where FTS always loads. And its both-blocked case is
// gated on GITNEXUS_REQUIRE_VECTOR=1, which ci-tests.yml sets ONLY on this
// job — everywhere else an unavailable VECTOR extension skips instead of
// failing. Budget: four real analyze runs, so expect it to sit alongside the
// gated on GITNEXUS_REQUIRE_VECTOR=1, which ci-tests.yml requires in both
// coverage and platform jobs so an unavailable VECTOR fails loudly. Budget: four real analyze runs, so expect it to sit alongside the
// VECTOR sibling's ~87s Windows measurement.
'test/unit/incremental-index-extension-dml-gate.test.ts',
];
@ -308,6 +307,10 @@ const NATIVE_ADDON_SMOKE = [
// Filesystem behavior tests — exercise operations that vary across
// platforms (CRLF, symlinks, permissions, temp dirs)
const FILESYSTEM = [
// The deletion-guard cases in this file require real Windows path semantics.
'test/unit/canonicalize-path-long-path-prefix.test.ts',
'test/unit/storage-resolver.test.ts',
'test/unit/dart-package-imports.test.ts',
// Cargo membership uses path normalization, descriptor validation, symlinks,
// and Rust native parsing (including long Windows source strings).
'test/unit/scope-resolution/rust-cargo-targets.test.ts',

View file

@ -0,0 +1,33 @@
import { JsonReporter, type TestRunEndReason } from 'vitest/node';
/** Vitest's stock JSON omits unhandled errors, even when success is true. */
export default class ExecutionReporter extends JsonReporter {
private executionErrors = 0;
private executionReason: TestRunEndReason = 'interrupted';
constructor() {
super({});
}
override async onTestRunEnd(
modules: Parameters<JsonReporter['onTestRunEnd']>[0],
errors: readonly unknown[] = [],
reason: TestRunEndReason = 'interrupted',
): Promise<void> {
this.executionErrors = errors.length;
this.executionReason = reason;
await super.onTestRunEnd(modules);
}
override async writeReport(json: string): Promise<void> {
const report = JSON.parse(json);
await super.writeReport(
JSON.stringify({
...report,
success: report.success && this.executionErrors === 0 && this.executionReason === 'passed',
executionErrors: this.executionErrors,
executionReason: this.executionReason,
}),
);
}
}

View file

@ -0,0 +1,32 @@
import { execFileSync } from 'node:child_process';
import { globSync, readFileSync } from 'node:fs';
import { fileURLToPath } from 'node:url';
const root = fileURLToPath(new URL('../', import.meta.url));
// Discover the opt-in suites from source so a new benchmark cannot silently
// miss a hand-maintained workflow list. Keep wall-clock measurements serial.
const files = globSync('test/**/*.test.ts', { cwd: root })
.filter((file) =>
/process\.env(?:\.GITNEXUS_BENCH|\[['"]GITNEXUS_BENCH['"]\])/.test(
readFileSync(new URL(`../${file.replaceAll('\\', '/')}`, import.meta.url), 'utf8'),
),
)
.sort();
if (!files.length) throw new Error('No benchmark test suites discovered');
execFileSync(
process.execPath,
[
fileURLToPath(new URL('../node_modules/vitest/vitest.mjs', import.meta.url)),
'run',
'--no-file-parallelism',
...files,
'--reporter=default',
'--reporter=./scripts/execution-reporter.ts',
'--outputFile=benchmarks.json',
],
{
cwd: root,
stdio: 'inherit',
env: { ...process.env, GITNEXUS_BENCH: '1' },
},
);

View file

@ -80,8 +80,19 @@ console.log(
);
const startedAt = Date.now();
const reportName = process.env.GITNEXUS_TEST_REPORT;
if (reportName && !/^[a-zA-Z0-9_-]+\.json$/.test(reportName)) {
throw new Error('GITNEXUS_TEST_REPORT must be a JSON basename');
}
const reportArgs = reportName
? [
'--reporter=default',
'--reporter=./scripts/execution-reporter.ts',
`--outputFile=${reportName}`,
]
: [];
try {
execFileSync('npx', ['vitest', 'run', ...files], {
execFileSync('npx', ['vitest', 'run', ...files, ...reportArgs], {
cwd: ROOT,
stdio: 'inherit',
timeout: timeoutMs,

View file

@ -0,0 +1,203 @@
import { globSync, readFileSync, writeFileSync } from 'node:fs';
import path from 'node:path';
import { fileURLToPath, pathToFileURL } from 'node:url';
import type { TestRunEndReason } from 'vitest/node';
type Assertion = {
ancestorTitles: string[];
title: string;
fullName: string;
status: string;
location?: { line: number; column: number };
};
type Suite = {
name: string;
status: string;
assertionResults: Assertion[];
endTime?: number;
message?: string;
};
type Report = {
success?: boolean;
numTotalTests?: number;
startTime?: number;
executionErrors: number;
executionReason: TestRunEndReason;
testResults: Suite[];
};
/** Stable across GitHub's Linux, macOS and Windows checkout roots. */
function testPath(name: string, packageName: string): string {
const normalized = name.replaceAll('\\', '/');
const relative =
normalized.match(new RegExp(`(?:^|/)${packageName}/((?:test|src)/.+)$`))?.[1] ?? normalized;
if (!/^(?:test|src)\/.+\.test\.[jt]sx?$/.test(relative)) {
throw new Error(`Invalid test report path: ${name}`);
}
return relative;
}
/** A skip is resolved only by a recorded pass of that same test in another job. */
export function reconcileTestReports(
inputs: unknown[],
expectedFiles: string[] = [],
packageName = 'gitnexus',
) {
if (inputs.length === 0) throw new Error('No execution reports supplied');
const tests = new Map<
string,
{ file: string; assertion: Assertion; passed: boolean; failed: boolean }
>();
const failures = new Set<string>();
const failedSuites = new Set<string>();
const startTimes: number[] = [];
const endTimes: number[] = [];
for (const [index, input] of inputs.entries()) {
const report = input as Report;
if (!report || !Array.isArray(report.testResults) || report.testResults.length === 0) {
throw new Error(`Missing or empty execution report ${index}`);
}
if (report.success === false) failures.add(`Report ${index}: unsuccessful runner result`);
if (
!Number.isInteger(report.executionErrors) ||
report.executionErrors < 0 ||
!['passed', 'failed', 'interrupted'].includes(report.executionReason)
) {
throw new Error(`Report ${index}: missing execution reporter evidence`);
}
if (report.executionErrors || report.executionReason !== 'passed') {
failures.add(
`Report ${index}: ${report.executionErrors} unhandled errors; ${report.executionReason}`,
);
}
if (typeof report.startTime === 'number') startTimes.push(report.startTime);
let count = 0;
const seen = new Set<string>();
for (const suite of report.testResults) {
const file = testPath(suite.name, packageName);
if (!Array.isArray(suite.assertionResults) || suite.assertionResults.length === 0) {
throw new Error(`No collected tests in ${file}`);
}
if (suite.status === 'failed') {
failures.add(`${file}: failed suite`);
failedSuites.add(file);
}
if (typeof suite.endTime === 'number') endTimes.push(suite.endTime);
for (const assertion of suite.assertionResults) {
if (
!['passed', 'failed', 'pending', 'skipped', 'todo', 'disabled'].includes(
assertion.status,
) ||
!Array.isArray(assertion.ancestorTitles) ||
typeof assertion.title !== 'string' ||
(assertion.location !== undefined &&
(!Number.isInteger(assertion.location.line) ||
!Number.isInteger(assertion.location.column)))
) {
throw new Error(`Malformed assertion in ${file}`);
}
// Collection order differs across platforms. Source location, not an
// ordinal, distinguishes equal titles declared at different locations.
// Helper-generated tests may have no location; their complete title
// must then be unique within the file (the duplicate check still applies).
const title = JSON.stringify([...assertion.ancestorTitles, assertion.title]);
const key = JSON.stringify([
file,
title,
assertion.location?.line ?? null,
assertion.location?.column ?? null,
]);
if (seen.has(key))
throw new Error(
`Ambiguous test identity in ${file}: ${title}; give parameterized cases unique titles`,
);
seen.add(key);
const result = tests.get(key) ?? { file, assertion, passed: false, failed: false };
result.passed ||= assertion.status === 'passed';
result.failed ||= assertion.status === 'failed';
tests.set(key, result);
if (result.failed) failures.add(`${file}: ${assertion.fullName ?? title}`);
count++;
}
}
if (typeof report.numTotalTests === 'number' && report.numTotalTests !== count) {
throw new Error(`Report ${index}: expected ${report.numTotalTests} tests, found ${count}`);
}
}
const unverified = [...tests.values()]
.filter((test) => !test.passed && !test.failed)
.map((test) => `${test.file}: ${test.assertion.fullName}`);
const suites = new Map<string, Suite>();
for (const test of tests.values()) {
const status = test.failed ? 'failed' : test.passed ? 'passed' : 'pending';
const suite = suites.get(test.file) ?? {
name: test.file,
status: 'passed',
assertionResults: [],
};
suite.assertionResults.push({ ...test.assertion, status });
if (status === 'failed' || failedSuites.has(test.file)) suite.status = 'failed';
suites.set(test.file, suite);
}
for (const file of expectedFiles) {
if (!suites.has(file.replaceAll('\\', '/')))
throw new Error(`Test file was never collected: ${file}`);
}
const passed = [...tests.values()].filter((test) => test.passed && !test.failed).length;
const endTime = endTimes.length ? Math.max(...endTimes) : 0;
return {
total: tests.size,
passed,
unverified,
failures: [...failures],
report: {
success: failures.size === 0 && unverified.length === 0,
numTotalTests: tests.size,
numPassedTests: passed,
numFailedTests: [...tests.values()].filter((test) => test.failed).length,
numPendingTests: unverified.length,
numTotalTestSuites: suites.size,
numFailedTestSuites: failedSuites.size,
executionFailures: [...failures],
startTime: startTimes.length ? Math.min(...startTimes) : 0,
testResults: [...suites.values()].map((suite) => ({ ...suite, endTime })),
},
};
}
if (process.argv[1] && import.meta.url === pathToFileURL(path.resolve(process.argv[1])).href) {
const [directory, output, shardsText] = process.argv.slice(2);
const shards = Number(shardsText);
if (!directory || !output || !Number.isInteger(shards) || shards < 1) {
throw new Error('Usage: test-completeness.ts <reports-dir> <output.json> <platform-shards>');
}
const names = ['coverage.json', 'benchmarks.json', 'preflight.json'];
for (const os of ['windows-latest', 'macos-latest']) {
for (let shard = 1; shard <= shards; shard++) names.push(`${os}-${shard}.json`);
}
// Exact expected receipts: an absent/cancelled job is never an empty success.
const root = fileURLToPath(new URL('../', import.meta.url));
const expected = globSync('test/**/*.test.ts', { cwd: root });
const result = reconcileTestReports(
names.map((name) => JSON.parse(readFileSync(path.join(directory, name), 'utf8'))),
expected,
);
writeFileSync(output, JSON.stringify(result.report));
console.log(
`Test execution: ${result.passed}/${result.total} passed; ${result.unverified.length} unverified; ${result.failures.length} failures`,
);
for (const message of [...result.unverified, ...result.failures]) console.error(message);
if (!result.report.success) process.exitCode = 1;
const webPath = path.join(root, '../gitnexus-web/web-test-results.json');
const web = reconcileTestReports(
[JSON.parse(readFileSync(webPath, 'utf8'))],
globSync('{test,src}/**/*.test.{ts,tsx}', { cwd: path.join(root, '../gitnexus-web') }),
'gitnexus-web',
);
writeFileSync(webPath, JSON.stringify(web.report));
console.log(
`Web execution: ${web.passed}/${web.total} passed; ${web.unverified.length} unverified; ${web.failures.length} failures`,
);
for (const message of [...web.unverified, ...web.failures]) console.error(message);
if (!web.report.success) process.exitCode = 1;
}

View file

@ -39,6 +39,11 @@ export interface DartPackageConfigOptions {
* is listed and before its entries are opened.
*/
readonly beforeEntryOpen?: (relativePath: string) => void | Promise<void>;
/**
* Test seam. Production calls omit it. Invoked after the directory is
* opened and before it is listed.
*/
readonly beforeDirectoryList?: (relativePath: string) => void | Promise<void>;
}
type ManifestRead =
@ -123,12 +128,13 @@ function directoryIdentity(stat: BigIntStats): string {
/**
* Path that lists the directory inode already open on `fd`.
* Linux uses `/proc/self/fd/N` and macOS uses `/dev/fd/N`.
* Other platforms have no such path; callers refuse instead of listing by name.
* Only Linux has one: `/proc/self/fd/N`. macOS refuses `opendir` on
* `/dev/fd/N` for a directory (ENOTDIR) and Node has no `fdopendir`, so the
* walker lists the lexical path and verifies the pinned chain around it.
* Every other platform returns null and is not walked.
*/
export function descriptorDirectoryPath(fd: number): string | null {
if (process.platform === 'linux') return `/proc/self/fd/${fd}`;
if (process.platform === 'darwin') return `/dev/fd/${fd}`;
return null;
}
@ -264,11 +270,33 @@ async function openChildFile(
throw Object.assign(new Error('no descriptor anchor'), { code: 'ENOTSUP' });
}
async function listOpenedDirectory(handle: FileHandle, entryLimit: number): Promise<Dirent[]> {
const listing = descriptorDirectoryPath(handle.fd);
if (listing === null) {
/**
* List the directory open on the last frame of `frames`. Linux lists the
* pinned inode through its descriptor. macOS re-checks the pinned chain, lists
* the lexical path, then re-checks the chain, so a path replaced around the
* listing fails the walk instead of being listed.
*/
async function listOpenedDirectory(
frames: readonly WalkFrame[],
entryLimit: number,
beforeList?: () => void | Promise<void>,
): Promise<Dirent[]> {
const frame = frames[frames.length - 1];
if (frame === undefined) {
throw Object.assign(new Error('no directory to list'), { code: 'EINVAL' });
}
const anchored = descriptorDirectoryPath(frame.handle.fd);
if (anchored === null && process.platform !== 'darwin') {
throw Object.assign(new Error('no descriptor listing'), { code: 'ENOTSUP' });
}
if (anchored === null) await assertPinnedChain(frames);
if (beforeList) await beforeList();
const entries = await readDirectoryBounded(anchored ?? frame.absolute, entryLimit);
if (anchored === null) await assertPinnedChain(frames);
return entries;
}
async function readDirectoryBounded(listing: string, entryLimit: number): Promise<Dirent[]> {
const dir = await opendir(listing);
const entries: Dirent[] = [];
try {
@ -295,8 +323,16 @@ async function listOpenedDirectory(handle: FileHandle, entryLimit: number): Prom
*/
export async function readDirectoryNoFollow(directory: string): Promise<Dirent[]> {
const opened = await openVerifiedDirectory(directory);
const frame: WalkFrame = {
relative: '',
absolute: directory,
handle: opened.handle,
identity: opened.identity,
entries: [],
next: 0,
};
try {
return await listOpenedDirectory(opened.handle, DART_PUBSPEC_DIRECTORY_ENTRY_LIMIT);
return await listOpenedDirectory([frame], DART_PUBSPEC_DIRECTORY_ENTRY_LIMIT);
} finally {
await opened.handle.close();
}
@ -345,7 +381,12 @@ export async function loadDartPackageConfig(
const fillFrame = async (frame: WalkFrame): Promise<void> => {
if (++visited > directoryLimit) return incomplete('directory-limit', frame.relative || '.');
try {
frame.entries = await listOpenedDirectory(frame.handle, directoryEntryLimit);
const beforeList = options?.beforeDirectoryList;
frame.entries = await listOpenedDirectory(
stack,
directoryEntryLimit,
beforeList ? () => beforeList(frame.relative) : undefined,
);
} catch (error) {
const reason =
(error as NodeJS.ErrnoException).code === 'E2BIG' ? 'directory-entries' : 'read-directory';

View file

@ -473,13 +473,10 @@ describe('Go scope-capture O(n^2) regression tripwire', () => {
*/
function generateSyntheticInterfaceData(interfaceCount: number, structCount: number): ParsedFile[] {
const defs: SymbolDefinition[] = [];
const ifaceIds: string[] = [];
const structIds: string[] = [];
// Interfaces
for (let i = 0; i < interfaceCount; i++) {
const ifaceId = `iface:Repo${i}`;
ifaceIds.push(ifaceId);
defs.push({
nodeId: ifaceId,
filePath: 'repo.go',
@ -515,37 +512,34 @@ function generateSyntheticInterfaceData(interfaceCount: number, structCount: num
// Structs — each implements all interfaces
for (let s = 0; s < structCount; s++) {
const structId = `struct:Impl${s}`;
structIds.push(structId);
defs.push({
nodeId: structId,
filePath: 'repo.go',
type: 'Struct',
qualifiedName: `Impl${s}`,
});
for (let i = 0; i < interfaceCount; i++) {
defs.push({
nodeId: `struct:Impl${s}.Repo${i}.Find`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Impl${s}.Find`,
ownerId: structId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['string'],
returnType: 'User',
});
defs.push({
nodeId: `struct:Impl${s}.Repo${i}.Save`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Impl${s}.Save`,
ownerId: structId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['User'],
returnType: 'error',
});
}
defs.push({
nodeId: `struct:Impl${s}.Find`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Impl${s}.Find`,
ownerId: structId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['string'],
returnType: 'User',
});
defs.push({
nodeId: `struct:Impl${s}.Save`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Impl${s}.Save`,
ownerId: structId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['User'],
returnType: 'error',
});
}
// BadStructs — wrong Save signature (string instead of User), should NOT match
@ -557,31 +551,29 @@ function generateSyntheticInterfaceData(interfaceCount: number, structCount: num
type: 'Struct',
qualifiedName: `Bad${b}`,
});
for (let i = 0; i < interfaceCount; i++) {
defs.push({
nodeId: `struct:Bad${b}.Repo${i}.Find`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Bad${b}.Find`,
ownerId: badId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['string'],
returnType: 'User',
});
// Mismatched Save: string param instead of User
defs.push({
nodeId: `struct:Bad${b}.Repo${i}.Save`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Bad${b}.Save`,
ownerId: badId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['string'],
returnType: 'error',
});
}
defs.push({
nodeId: `struct:Bad${b}.Find`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Bad${b}.Find`,
ownerId: badId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['string'],
returnType: 'User',
});
// Mismatched Save: string param instead of User
defs.push({
nodeId: `struct:Bad${b}.Save`,
filePath: 'repo.go',
type: 'Method',
qualifiedName: `Bad${b}.Save`,
ownerId: badId,
parameterCount: 1,
requiredParameterCount: 1,
parameterTypes: ['string'],
returnType: 'error',
});
}
return [
@ -606,6 +598,15 @@ function generateSyntheticInterfaceData(interfaceCount: number, structCount: num
* path (structIdsByMethodName intersection) keeps this well under budget.
*/
describe('Go structural interface detection O(n²) regression tripwire', () => {
it('uses legal Go method sets with one declaration per receiver and method name', () => {
const [parsed] = generateSyntheticInterfaceData(3, 3);
const methods = parsed.localDefs.filter((def) => def.type === 'Method');
const identities = methods.map((def) => `${def.ownerId}:${def.qualifiedName}`);
expect(new Set(identities).size).toBe(methods.length);
// Three interfaces, three matching structs and three negative controls.
expect(methods).toHaveLength(18);
});
it('detects implementations for 100 interfaces × 100 structs within budget', () => {
const IFACE_COUNT = 100;
const STRUCT_COUNT = 100;
@ -779,7 +780,7 @@ describe.skipIf(!BENCH_ENABLED)('Go structural interface detection split-phase b
console.log('\nScaling analysis (total time ratio / pair-count ratio):');
// We can't split phases without exporting internals, but we can report
// how total time scales relative to pair count.
// The detection loop is O(I×C×M×O_a) where O_a grows with I in this
// synthetic case, so pair-count ratio alone underestimates expected growth.
// Each receiver declares Find and Save once, as real Go requires.
// Required signatures stay constant while interface/struct pairs grow.
}, 300_000);
});

View file

@ -575,14 +575,17 @@ describe('LocalBackend.callTool', () => {
['impact', { name: 'validate', symbol: 'login', direction: 'upstream' }],
['impact', { target: 'validate', direction: 'upstream', maxDepth: 3, depth: 1 }],
['context', { name: 'validate', file_path: 'src/auth.ts', file: 'src/login.ts' }],
])('rejects conflicting %s aliases before repository resolution', async (method, params) => {
const resolveSpy = vi.spyOn(backend, 'selectToolRepository');
])(
'rejects conflicting %s aliases before repository resolution (case %#)',
async (method, params) => {
const resolveSpy = vi.spyOn(backend, 'selectToolRepository');
const result = await backend.callTool(method, params);
const result = await backend.callTool(method, params);
expect(result.error).toMatch(/conflicting mcp parameters/i);
expect(resolveSpy).not.toHaveBeenCalled();
});
expect(result.error).toMatch(/conflicting mcp parameters/i);
expect(resolveSpy).not.toHaveBeenCalled();
},
);
it.each([
['impact', { name: 42, direction: 'upstream' }],

View file

@ -3,7 +3,7 @@
*
* This is the Linux-runnable guard for the `canonicalizePath` wiring. The
* companion assertions in `repo-manager.test.ts` exercise the real
* `realpathSync.native` and are therefore `it.skipIf(win32)`, so they only run on
* `realpathSync.native` and are therefore Windows-only, so they only run on
* the windows-latest matrix leg — leaving the Ubuntu gate with no coverage of the
* behaviour at all. This file closes that hole by injecting the two platform
* primitives and nothing else:
@ -26,8 +26,8 @@
* and the realpath case passes, which is exactly the asymmetry the fix targets:
* the realpath branch never leaked, because libuv strips the prefix itself.
*
* Deliberately NOT registered in `scripts/cross-platform-tests.ts`: it simulates
* Windows rather than needing it, so its home is the Ubuntu suite.
* Registered in `scripts/cross-platform-tests.ts` as well: the injected Windows
* path rules must agree with the native Windows run.
*/
import { describe, it, expect, vi } from 'vitest';
@ -139,13 +139,10 @@ describe('canonicalizePath vs the `\\\\?\\` long-path prefix (#2667)', () => {
});
});
// The guard in front of `fs.rm(recursive)` in remove.ts / clean.ts. It compares
// `path.resolve` forms on both sides and deliberately does NOT canonicalize, so a
// prefixed entry stays self-consistent while a mixed-form entry fails closed.
// Pinned here because "complete the fix by stripping here too" is the tempting
// follow-up refactor, and it would widen what the recursive delete accepts.
// The storage resolver now compares canonical paths for repository-local slots.
// Equivalent prefix spellings name the same .gitnexus directory; normalization
// must still reject the repository itself, its parents, and unowned external paths.
describe('assertSafeStoragePath vs the `\\\\?\\` prefix (#2667)', () => {
const itOnWindows = process.platform === 'win32' ? it : it.skip;
const base: Omit<RegistryEntry, 'storagePath'> = {
name: 'repo',
path: '\\\\?\\D:\\Projects\\repo',
@ -153,16 +150,29 @@ describe('assertSafeStoragePath vs the `\\\\?\\` prefix (#2667)', () => {
lastCommit: 'deadbee',
};
itOnWindows('accepts an entry whose path and storagePath share the prefix', async () => {
it('accepts an entry whose path and storagePath share the prefix', async () => {
await expect(
assertSafeStoragePath({ ...base, storagePath: '\\\\?\\D:\\Projects\\repo\\.gitnexus' }),
).resolves.toBeUndefined();
});
itOnWindows('rejects a mixed-form entry instead of deleting through it', async () => {
it('accepts mixed prefix spellings of the same repository-local slot', async () => {
await expect(
assertSafeStoragePath({ ...base, storagePath: 'D:\\Projects\\repo\\.gitnexus' }),
).rejects.toThrow();
).resolves.toBeUndefined();
});
it.each([
'D:\\Projects\\repo',
'\\\\?\\D:\\Projects\\repo',
'D:\\Projects',
'D:\\',
'D:\\Projects\\other\\.gitnexus',
'\\\\?\\D:\\Projects\\other\\.gitnexus',
])('rejects an unsafe deletion target despite prefix normalization: %s', async (storagePath) => {
await expect(assertSafeStoragePath({ ...base, storagePath })).rejects.toThrow(
/Refusing to remove storage path for safety/,
);
});
});

View file

@ -13,6 +13,7 @@
* reproduce the outage exactly.
*/
import { readFileSync } from 'node:fs';
import { describe, it, expect } from 'vitest';
import {
shardFiles,
@ -22,7 +23,12 @@ import {
} from '../../scripts/cross-platform-shard.js';
import { ALL_CROSS_PLATFORM } from '../../scripts/cross-platform-tests.js';
const SHARD_TOTAL = 3;
// Replay the actual CI partition, including shard-count changes as the suite grows.
const workflow = readFileSync(
new URL('../../../.github/workflows/ci-tests.yml', import.meta.url),
'utf8',
);
const SHARD_TOTAL = Number(workflow.match(/^\s+TOTAL=(\d+)\b/m)?.[1]);
/** Every shard of a split, as file lists. */
const allShards = (files: readonly string[], total: number): readonly (readonly string[])[] =>
@ -54,7 +60,7 @@ describe('cross-platform shard partition', () => {
'test/unit/incremental-index-extension-dml-gate.test.ts',
].map((file) => shards.findIndex((files) => files.includes(file)));
expect(heavyweightLocations).not.toContain(-1);
expect(new Set(heavyweightLocations).size).toBe(SHARD_TOTAL);
expect(new Set(heavyweightLocations).size).toBe(heavyweightLocations.length);
expect(
shards.filter(
(s) =>

View file

@ -1,5 +1,5 @@
import { execFileSync } from 'node:child_process';
import { afterEach, describe, expect, it } from 'vitest';
import { afterEach, describe, expect, it, vi } from 'vitest';
import { constants } from 'node:fs';
import {
mkdtemp,
@ -463,47 +463,52 @@ describe.skipIf(!pubspecWalkAnchored())('Dart pubspec package discovery', () =>
},
);
it('does not follow a listed directory replaced by a symlink on macOS', async () => {
if (process.platform !== 'darwin') return;
const outside = await fixture({
'pubspec.yaml': 'name: foreign',
'nested/pubspec.yaml': 'name: foreign',
});
const root = await fixture({
'pkg/pubspec.yaml': 'name: data',
'pkg/nested/pubspec.yaml': 'name: nested_data',
});
const pkg = path.join(root, 'pkg');
await expect(
loadDartPackageConfig(root, {
beforeEntryOpen: async (relative) => {
if (relative !== 'pkg') return;
await rename(pkg, path.join(root, 'pkg-moved'));
await symlink(outside, pkg, 'dir');
},
}),
).rejects.toThrow(/Dart pubspec discovery failed \((read-directory|read-pubspec)\)/);
});
it.skipIf(process.platform !== 'darwin')(
'does not follow a listed directory replaced by a symlink on macOS',
async () => {
const outside = await fixture({
'pubspec.yaml': 'name: foreign',
'nested/pubspec.yaml': 'name: foreign',
});
const root = await fixture({
'pkg/pubspec.yaml': 'name: data',
'pkg/nested/pubspec.yaml': 'name: nested_data',
});
const pkg = path.join(root, 'pkg');
await expect(
loadDartPackageConfig(root, {
beforeEntryOpen: async (relative) => {
if (relative !== 'pkg') return;
await rename(pkg, path.join(root, 'pkg-moved'));
await symlink(outside, pkg, 'dir');
},
}),
).rejects.toThrow(/Dart pubspec discovery failed \((read-directory|read-pubspec)\)/);
},
);
it('reads listed manifests from the opened directory inode after that path is replaced', async () => {
if (descriptorEntryPath(0, 'pubspec.yaml') === null) return;
const outside = await fixture({
'pubspec.yaml': 'name: foreign',
'nested/pubspec.yaml': 'name: foreign',
});
// Linux lists the opened inode through /proc/self/fd/N. macOS cannot list a
// descriptor (opendir on /dev/fd/N is ENOTDIR), so it lists the path and
// re-checks the pinned chain afterwards; the swap must fail that check.
it('lists the opened directory, or refuses, when its path is replaced before listing', async () => {
const outside = await fixture({ 'pubspec.yaml': 'name: foreign' });
const root = await fixture({
'pkg/pubspec.yaml': 'name: data',
'pkg/nested/pubspec.yaml': 'name: nested_data',
});
const pkg = path.join(root, 'pkg');
const loaded = await loadDartPackageConfig(root, {
beforeEntryOpen: async (relative) => {
const load = loadDartPackageConfig(root, {
beforeDirectoryList: async (relative) => {
if (relative !== 'pkg') return;
await rename(pkg, path.join(root, 'pkg-moved'));
await symlink(outside, pkg, 'dir');
},
});
expect(loaded.packages).toEqual(
if (process.platform === 'darwin') {
await expect(load).rejects.toThrow('Dart pubspec discovery failed (read-directory): pkg');
return;
}
expect((await load).packages).toEqual(
new Map([
['data', 'pkg/lib'],
['nested_data', 'pkg/nested/lib'],
@ -511,26 +516,55 @@ describe.skipIf(!pubspecWalkAnchored())('Dart pubspec package discovery', () =>
);
});
it('lists the opened directory inode after its path becomes a symlink', async () => {
const listingRoot = descriptorDirectoryPath(0);
if (listingRoot === null) return;
const outside = await fixture({ 'outside.txt': 'out' });
const root = await fixture({});
const real = path.join(root, 'real');
await mkdir(real);
await writeFile(path.join(real, 'inside.txt'), 'in');
const handle = await open(real, directoryOpenFlags());
try {
const listed = descriptorDirectoryPath(handle.fd);
if (listed === null) throw new Error('descriptor listing path missing');
await rename(real, path.join(root, 'real-moved'));
await symlink(outside, real, process.platform === 'win32' ? 'junction' : 'dir');
await expect(readDirectoryNoFollow(real)).rejects.toThrow();
expect(await readdir(listed)).toEqual(['inside.txt']);
} finally {
await handle.close();
}
});
it.skipIf(descriptorEntryPath(0, 'pubspec.yaml') === null)(
'reads listed manifests from the opened directory inode after that path is replaced',
async () => {
const outside = await fixture({
'pubspec.yaml': 'name: foreign',
'nested/pubspec.yaml': 'name: foreign',
});
const root = await fixture({
'pkg/pubspec.yaml': 'name: data',
'pkg/nested/pubspec.yaml': 'name: nested_data',
});
const pkg = path.join(root, 'pkg');
const loaded = await loadDartPackageConfig(root, {
beforeEntryOpen: async (relative) => {
if (relative !== 'pkg') return;
await rename(pkg, path.join(root, 'pkg-moved'));
await symlink(outside, pkg, 'dir');
},
});
expect(loaded.packages).toEqual(
new Map([
['data', 'pkg/lib'],
['nested_data', 'pkg/nested/lib'],
]),
);
},
);
it.skipIf(descriptorDirectoryPath(0) === null)(
'lists the opened directory inode after its path becomes a symlink',
async () => {
const outside = await fixture({ 'outside.txt': 'out' });
const root = await fixture({});
const real = path.join(root, 'real');
await mkdir(real);
await writeFile(path.join(real, 'inside.txt'), 'in');
const handle = await open(real, directoryOpenFlags());
try {
const listed = descriptorDirectoryPath(handle.fd);
if (listed === null) throw new Error('descriptor listing path missing');
await rename(real, path.join(root, 'real-moved'));
await symlink(outside, real, process.platform === 'win32' ? 'junction' : 'dir');
await expect(readDirectoryNoFollow(real)).rejects.toThrow();
expect(await readdir(listed)).toEqual(['inside.txt']);
} finally {
await handle.close();
}
},
);
});
describe('directory symlink refusal', () => {
@ -549,13 +583,18 @@ describe('directory symlink refusal', () => {
await expect(readDirectoryNoFollow(link)).rejects.toThrow();
});
it.skipIf(pubspecWalkAnchored())(
'does not discover packages when the walk cannot honor no-follow',
async () => {
const root = await mkdtemp(path.join(os.tmpdir(), 'gitnexus-dart-pubspec-'));
roots.push(root);
await writeFile(path.join(root, 'pubspec.yaml'), 'name: app\n');
it('does not discover packages when the walk cannot honor no-follow', async () => {
const root = await mkdtemp(path.join(os.tmpdir(), 'gitnexus-dart-pubspec-'));
roots.push(root);
await writeFile(path.join(root, 'pubspec.yaml'), 'name: app\n');
// Exercise the unsupported-platform branch on every host. The real
// capability check must refuse before attempting any directory walk.
const platform = vi.spyOn(process, 'platform', 'get').mockReturnValue('win32');
try {
expect(pubspecWalkAnchored()).toBe(false);
expect((await loadDartPackageConfig(root)).packages.size).toBe(0);
},
);
} finally {
platform.mockRestore();
}
});
});

View file

@ -22,7 +22,7 @@ describe('isEvalServerBindRestriction', () => {
['Error: listen EADDRNOTAVAIL: address not available 192.168.1.99:0'],
['something then bind EACCES: permission denied'],
['LISTEN EACCES: permission denied'],
])('matches %p', (stderr) => {
])('matches %s', (stderr) => {
expect(isEvalServerBindRestriction(stderr)).toBe(true);
});
});
@ -35,7 +35,7 @@ describe('isEvalServerBindRestriction', () => {
['GITNEXUS_EVAL_SERVER_READY:127.0.0.1:5173'],
[''],
['unknown option --host'],
])('does not match %p', (stderr) => {
])('does not match %s', (stderr) => {
expect(isEvalServerBindRestriction(stderr)).toBe(false);
});
});

View file

@ -953,20 +953,23 @@ SAFE_WRITE_FIXTURES('generated-plan safe writer', () => {
['--expected-plan-path', SAFE_PLAN_PATH],
/--expected-plan-path and --expected-plan-digest require --replace/,
],
] as const)('rejects command-inapplicable CLI options for %s', (command, extra, pattern) => {
const repo = createBaseRepo('gitnexus-plan-cli-options-');
try {
const result = spawnSync(
process.execPath,
[PLAN_HELPER, command, '--repo', repo, '--generated-plan', SAFE_PLAN_PATH, ...extra],
{ encoding: 'utf8', input: '# plan\n' },
);
expect(result.status).toBe(1);
expect(result.stderr).toMatch(pattern);
} finally {
fs.rmSync(repo, { recursive: true, force: true });
}
});
] as const)(
'rejects command-inapplicable CLI options for %s (case %#)',
(command, extra, pattern) => {
const repo = createBaseRepo('gitnexus-plan-cli-options-');
try {
const result = spawnSync(
process.execPath,
[PLAN_HELPER, command, '--repo', repo, '--generated-plan', SAFE_PLAN_PATH, ...extra],
{ encoding: 'utf8', input: '# plan\n' },
);
expect(result.status).toBe(1);
expect(result.stderr).toMatch(pattern);
} finally {
fs.rmSync(repo, { recursive: true, force: true });
}
},
);
it('never overwrites a destination created immediately before initial publication', async () => {
const repo = createBaseRepo('gitnexus-plan-writer-');

View file

@ -283,7 +283,7 @@ describe('MCP repository policy', () => {
);
it.each([{ GITNEXUS_MCP_ALLOWED_REPOS: ' ' }, { GITNEXUS_MCP_DEFAULT_REPO: ' ' }])(
'fails closed for explicitly blank repository configuration',
'fails closed for explicitly blank repository configuration (case %#)',
async (env) => {
await expect(createMcpRepositoryPolicy(createBackend(), env)).rejects.toThrow(
/must not be blank/i,

View file

@ -1959,7 +1959,7 @@ describe('assertSafeStoragePath (#1003)', () => {
await expect(assertSafeStoragePath(entry)).rejects.toBeInstanceOf(UnsafeStoragePathError);
});
it.each([null, 42])('rejects malformed storagePath %p with the safety error', async (value) => {
it.each([null, 42])('rejects malformed storagePath %s with the safety error', async (value) => {
const entry = {
...base,
storagePath: value,

View file

@ -1,6 +1,7 @@
import { execSync } from 'child_process';
import fs from 'fs/promises';
import { basename } from 'node:path';
import { pathToFileURL } from 'node:url';
import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest';
import {
getStoragePaths,
@ -2747,6 +2748,7 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () =>
vi.doUnmock('../../src/core/embeddings/embedding-identity.js');
vi.doUnmock('../../src/core/embeddings/embedding-pipeline.js');
vi.doUnmock('../../src/core/embeddings/staged-embedding-recovery.js');
vi.doUnmock('../../src/core/analyzer-identity.js');
vi.restoreAllMocks();
vi.resetModules();
vi.clearAllMocks();
@ -3404,30 +3406,58 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () =>
await fs.writeFile(`${storagePath}/${source.filename}`, source.contents);
}
await fs.writeFile(lbugPath, 'previous published index');
// This unit fixture switches process.platform to exercise Windows
// recovery. Hash an immutable, test-owned analyzer runtime instead of
// the shared checkout under concurrent CI activity. Keep the real
// identity validation and finalization.
const identity = await import('../../src/core/analyzer-identity.js');
const runtimeRoot = `${tmpRepo.dbPath}/analyzer-runtime`;
await fs.mkdir(`${runtimeRoot}/src`, { recursive: true });
await fs.writeFile(
`${runtimeRoot}/package.json`,
'{"name":"recovery-fixture","version":"1.0.0"}',
);
await fs.writeFile(`${runtimeRoot}/package-lock.json`, '{"lockfileVersion":3}');
await fs.writeFile(`${runtimeRoot}/src/analyzer.ts`, 'export const analyzer = 1;');
const runtimeUrl = pathToFileURL(`${runtimeRoot}/src/analyzer.ts`).href;
const identityOptions = { cacheDirectory: `${tmpRepo.dbPath}/identity-cache` };
const resolveRunnerIdentity = vi.fn(() =>
identity.resolveAnalyzerRunnerIdentity(runtimeUrl, identityOptions),
);
vi.doMock('../../src/core/analyzer-identity.js', () => ({
...identity,
resolveAnalyzerRunnerIdentity: resolveRunnerIdentity,
finalizeAnalyzerRunnerIdentity: (
_url: string,
startedWith: NonNullable<RepoMeta['runnerIdentity']>,
) => identity.finalizeAnalyzerRunnerIdentity(runtimeUrl, startedWith, identityOptions),
}));
const { normalizeCachedEmbeddings } =
await import('../../src/core/embeddings/embedding-restore-spill.js');
vi.doMock(
'../../src/core/embeddings/staged-embedding-recovery.js',
async (importActual) => ({
...(await importActual<
typeof import('../../src/core/embeddings/staged-embedding-recovery.js')
>()),
recoverStagedEmbeddings: vi.fn(async () =>
normalizeCachedEmbeddings({
embeddings: [
{
nodeId: RESILIENCE_NODE_ID,
chunkIndex: 0,
startLine: 1,
endLine: 2,
contentHash: 'current-hash',
embedding: new Array(EMBEDDING_DIMS).fill(0),
},
],
}),
),
// Resolve the real merge implementation on the host, before selecting
// the Windows branch. A lazy importActual factory otherwise loads and
// transforms real modules while process.platform is temporarily win32.
const recovery = await vi.importActual<
typeof import('../../src/core/embeddings/staged-embedding-recovery.js')
>('../../src/core/embeddings/staged-embedding-recovery.js');
const recoverStagedEmbeddings = vi.fn(async () =>
normalizeCachedEmbeddings({
embeddings: [
{
nodeId: RESILIENCE_NODE_ID,
chunkIndex: 0,
startLine: 1,
endLine: 2,
contentHash: 'current-hash',
embedding: new Array(EMBEDDING_DIMS).fill(0),
},
],
}),
);
vi.doMock('../../src/core/embeddings/staged-embedding-recovery.js', () => ({
...recovery,
recoverStagedEmbeddings,
}));
const snapshots: Array<RepoMeta | null> = [];
const rename = fs.rename.bind(fs);
const { batchInsertEmbeddings } = mockResilienceHarness({
@ -3463,16 +3493,25 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () =>
// branch. The native adapter is mocked; source files and lock cleanup are real.
await import('../../src/core/run-analyze.js');
vi.stubEnv('GITNEXUS_ATOMIC_WINDOWS_SWAP', '0');
const logs: string[] = [];
Object.defineProperty(process, 'platform', { value: 'win32', configurable: true });
const error = await runAnalyze(
tmpRepo.dbPath,
{ force: true, embeddings: true, skipAgentsMd: true, skipSkills: true },
[],
logs,
);
if (actualPlatformDescriptor) {
Object.defineProperty(process, 'platform', actualPlatformDescriptor);
}
expect(batchInsertEmbeddings).toHaveBeenCalled();
expect(error, logs.join('\n')).toEqual(
outcome === 'success' ? null : expect.objectContaining({ message: outcome }),
);
expect(resolveRunnerIdentity, logs.join('\n')).toHaveBeenCalledOnce();
expect(recoverStagedEmbeddings, logs.join('\n')).toHaveBeenCalledOnce();
expect(logs).toContain(
'Recovered 1 complete staged embedding chunk(s) for 1 node(s); unchanged content can reuse them.',
);
expect(batchInsertEmbeddings, logs.join('\n')).toHaveBeenCalled();
expect(vi.mocked(adapter.initLbug).mock.calls.at(-1)?.[0]).toBe(lbugPath);
const finalMeta = await loadMeta(storagePath);
// Reacquiring the real lock performs the next retry's orphan sweep.
@ -3481,7 +3520,6 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () =>
const lock = await acquireIndexLock(storagePath, { timeoutMs: 1000 });
try {
if (outcome === 'success') {
expect(error).toBeNull();
expect(finalMeta?.embeddingCheckpoint).toBeUndefined();
expect(finalMeta?.stats?.embeddings).toBe(9);
expect(await fs.readFile(lbugPath, 'utf8')).toBe('in-place replacement');
@ -3491,7 +3529,6 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () =>
});
}
} else {
expect(error).toMatchObject({ message: outcome });
for (const source of sourceFiles) {
expect(await fs.readFile(`${storagePath}/${source.filename}`, 'utf8')).toBe(
source.contents,

View file

@ -389,7 +389,7 @@ describe('Rust: isGlobalNameFallbackPlausible', () => {
['src/a/b.rs', 'super::unique_helper_xyz', 'src/a.rs'],
['src/a/b/c.rs', 'super::super::unique_helper_xyz', 'src/a/mod.rs'],
['src/a/b.rs', 'self::child::unique_helper_xyz', 'src/a/b/child.rs'],
])('resolves relative imports from %s', (caller, target, candidate) => {
])('resolves relative imports from %s via %s', (caller, target, candidate) => {
expect(
rustIsGlobalNameFallbackPlausible({
site: BARE_SITE,

View file

@ -150,7 +150,7 @@ describe('Cargo manifest target metadata', () => {
`${PACKAGE}build=1\n`,
`${PACKAGE}[[bin]]\npath="custom/entry.rs"\n`,
`${PACKAGE}[[test]]\npath="custom/entry.rs"\n`,
])('does not manufacture evidence from malformed metadata', (manifest) => {
])('does not manufacture evidence from malformed metadata (case %#)', (manifest) => {
expect(cargoTargetRoots('Cargo.toml', manifest, files)).toBeUndefined();
});
});

View file

@ -326,7 +326,7 @@ describe('Spring AOP persisted reason contract (#2416)', () => {
},
];
it.each(reasons)('round-trips the $kind reason', (reason) => {
it.each(reasons)('round-trips the $kind reason for $annotation', (reason) => {
expect(decodeSpringAopReason(encodeSpringAopReason(reason))).toEqual(reason);
});

View file

@ -0,0 +1,284 @@
import { describe, expect, it } from 'vitest';
import { spawnSync } from 'node:child_process';
import { copyFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
import { tmpdir } from 'node:os';
import path from 'node:path';
import { fileURLToPath, pathToFileURL } from 'node:url';
import { reconcileTestReports } from '../../scripts/test-completeness.js';
const report = (name: string, statuses: string[]) => ({
success: !statuses.includes('failed'),
executionErrors: 0,
executionReason: 'passed',
testResults: [
{
name,
status: statuses.includes('failed') ? 'failed' : 'passed',
assertionResults: statuses.map((status, i) => ({
ancestorTitles: ['contract'],
title: `case ${i}`,
fullName: `contract case ${i}`,
status,
location: { line: i + 1, column: 1 },
})),
},
],
});
describe('required test execution across CI jobs', () => {
it('fails when a collected test never executes, including todo', () => {
const result = reconcileTestReports([
report('/home/runner/work/GitNexus/GitNexus/gitnexus/test/unit/a.test.ts', [
'passed',
'pending',
'todo',
]),
]);
expect(result.total).toBe(3);
expect(result.passed).toBe(1);
expect(result.unverified).toHaveLength(2);
});
it('accepts a platform skip only with an actual pass from another job', () => {
const result = reconcileTestReports([
report('/home/runner/work/GitNexus/GitNexus/gitnexus/test/unit/a.test.ts', [
'passed',
'pending',
]),
report('D:\\a\\GitNexus\\GitNexus\\gitnexus\\test\\unit\\a.test.ts', ['pending', 'passed']),
]);
expect(result.total).toBe(2);
expect(result.passed).toBe(2);
expect(result.unverified).toEqual([]);
expect(result.failures).toEqual([]);
});
it('never lets a passing job erase a failure on another platform', () => {
const result = reconcileTestReports([
report('/repo/gitnexus/test/unit/a.test.ts', ['failed']),
report('/repo/gitnexus/test/unit/a.test.ts', ['passed']),
]);
expect(result.failures.some((message) => message.includes('contract case 0'))).toBe(true);
});
it('counts failed tests separately from tests that never executed', () => {
const result = reconcileTestReports([
report('/repo/gitnexus/test/unit/a.test.ts', ['passed', 'failed', 'pending']),
]);
expect(result.report.numPassedTests).toBe(1);
expect(result.report.numFailedTests).toBe(1);
expect(result.report.numPendingTests).toBe(1);
expect(result.unverified).toHaveLength(1);
expect(result.report.success).toBe(false);
});
it('does not collapse identical titles from different files or repeated cases', () => {
const repeated = report('/repo/gitnexus/test/unit/a.test.ts', ['passed', 'pending']);
repeated.testResults[0].assertionResults[1].title = 'case 0';
repeated.testResults[0].assertionResults[1].fullName = 'contract case 0';
const result = reconcileTestReports([
repeated,
report('/repo/gitnexus/test/unit/b.test.ts', ['passed']),
]);
expect(result.total).toBe(3);
expect(result.unverified).toHaveLength(1);
});
it('matches source locations when repeated titles are collected in a different order', () => {
const first = report('/repo/gitnexus/test/unit/a.test.ts', ['pending', 'passed']);
first.testResults[0].assertionResults[1].title = 'case 0';
first.testResults[0].assertionResults[1].fullName = 'contract case 0';
const second = structuredClone(first);
second.testResults[0].assertionResults.reverse();
const result = reconcileTestReports([first, second]);
expect(result.total).toBe(2);
expect(result.unverified).toHaveLength(1);
});
it('rejects ambiguous parameterized cases with the same name and location', () => {
const input = report('/repo/gitnexus/test/unit/a.test.ts', ['passed']);
input.testResults[0].assertionResults.push({ ...input.testResults[0].assertionResults[0] });
expect(() => reconcileTestReports([input])).toThrow(/ambiguous/i);
});
it('requires an unambiguous title and a real pass for helper-generated tests without locations', () => {
const input = report('/repo/gitnexus/test/unit/a.test.ts', ['pending']);
Reflect.deleteProperty(input.testResults[0].assertionResults[0], 'location');
expect(reconcileTestReports([input]).unverified).toHaveLength(1);
const passing = structuredClone(input);
passing.testResults[0].assertionResults[0].status = 'passed';
expect(reconcileTestReports([input, passing]).report.success).toBe(true);
passing.testResults[0].assertionResults.push({ ...passing.testResults[0].assertionResults[0] });
expect(() => reconcileTestReports([passing])).toThrow(/ambiguous/i);
});
it('reconciles source-adjacent web tests on Linux and Windows', () => {
const result = reconcileTestReports(
[
report('/repo/gitnexus-web/src/lib/upload-filter.test.ts', ['passed']),
report('C:/repo/gitnexus-web/src/lib/upload-filter.test.ts', ['passed']),
],
['src/lib/upload-filter.test.ts'],
'gitnexus-web',
);
expect(result.report.success).toBe(true);
expect(result.total).toBe(1);
expect(() =>
reconcileTestReports(
[report('/repo/gitnexus-web/test/unit/a.test.ts', ['passed'])],
['test/unit/a.test.ts', 'src/lib/upload-filter.test.ts'],
'gitnexus-web',
),
).toThrow(/never collected/);
});
it('requires every shard receipt and the full web inventory through the CLI', () => {
const temp = mkdtempSync(path.join(tmpdir(), 'gitnexus-completeness-cli-'));
const packageRoot = fileURLToPath(new URL('../../', import.meta.url));
const root = path.join(temp, 'gitnexus');
const webRoot = path.join(temp, 'gitnexus-web');
const receipts = path.join(temp, 'receipts');
const output = path.join(temp, 'combined.json');
for (const directory of [
path.join(root, 'scripts'),
path.join(root, 'test/unit'),
path.join(webRoot, 'src/lib'),
receipts,
]) {
mkdirSync(directory, { recursive: true });
}
const script = path.join(root, 'scripts/test-completeness.ts');
copyFileSync(path.join(packageRoot, 'scripts/test-completeness.ts'), script);
writeFileSync(path.join(temp, 'package.json'), '{"type":"module"}');
writeFileSync(path.join(root, 'test/unit/a.test.ts'), '// inventory fixture');
writeFileSync(path.join(webRoot, 'src/lib/a.test.ts'), '// inventory fixture');
const core = report(path.join(root, 'test/unit/a.test.ts'), ['passed']);
const web = report(path.join(webRoot, 'src/lib/a.test.ts'), ['passed']);
const webReport = path.join(webRoot, 'web-test-results.json');
for (const name of [
'coverage',
'benchmarks',
'preflight',
'windows-latest-1',
'windows-latest-2',
'macos-latest-1',
'macos-latest-2',
]) {
writeFileSync(path.join(receipts, `${name}.json`), JSON.stringify(core));
}
const run = () => {
writeFileSync(webReport, JSON.stringify(web));
return spawnSync(process.execPath, ['--import', 'tsx', script, receipts, output, '2'], {
cwd: packageRoot,
encoding: 'utf8',
timeout: 30_000,
});
};
try {
const complete = run();
expect(complete.error, complete.stderr).toBeUndefined();
expect(complete.status, complete.stderr).toBe(0);
expect(JSON.parse(readFileSync(output, 'utf8')).numPendingTests).toBe(0);
writeFileSync(path.join(webRoot, 'src/lib/missing.test.ts'), '// missing receipt');
const missingSource = run();
expect(missingSource.status).not.toBe(0);
expect(missingSource.stderr).toContain(
'Test file was never collected: src/lib/missing.test.ts',
);
rmSync(path.join(webRoot, 'src/lib/missing.test.ts'));
rmSync(path.join(receipts, 'windows-latest-2.json'));
const missingShard = run();
expect(missingShard.status).not.toBe(0);
expect(missingShard.stderr).toContain('windows-latest-2.json');
} finally {
rmSync(temp, { recursive: true, force: true });
}
});
it('rejects test files missing from the collected inventory', () => {
expect(() =>
reconcileTestReports(
[report('/repo/gitnexus/test/unit/a.test.ts', ['passed'])],
['test/unit/a.test.ts', 'test/unit/b.test.ts'],
),
).toThrow(/b.test.ts/);
});
it('rejects missing, empty or malformed reports instead of reporting zero skips', () => {
expect(() => reconcileTestReports([])).toThrow();
expect(() => reconcileTestReports([{ testResults: [] }])).toThrow();
expect(() =>
reconcileTestReports([{ testResults: [{ name: 'a', assertionResults: [] }] }]),
).toThrow();
expect(() =>
reconcileTestReports([report('/repo/gitnexus/test/unit/a.test.ts', ['unknown'])]),
).toThrow();
});
it('keeps failed collection and unhandled runner errors fatal', () => {
const broken = report('/repo/gitnexus/test/unit/a.test.ts', ['passed']);
broken.success = false;
expect(reconcileTestReports([broken]).failures).not.toHaveLength(0);
broken.success = true;
broken.executionErrors = 1;
expect(reconcileTestReports([broken]).failures).not.toHaveLength(0);
broken.executionErrors = 0;
broken.testResults[0].status = 'failed';
expect(reconcileTestReports([broken]).failures).not.toHaveLength(0);
expect(reconcileTestReports([broken]).report.numFailedTestSuites).toBe(1);
expect(reconcileTestReports([broken]).report.testResults[0].status).toBe('failed');
});
it('captures a real unhandled rejection even when Vitest exits successfully', () => {
const temp = mkdtempSync(path.join(tmpdir(), 'gitnexus-receipt-'));
const packageRoot = fileURLToPath(new URL('../../', import.meta.url));
const root = path.join(temp, 'gitnexus');
const output = path.join(temp, 'receipt.json');
mkdirSync(path.join(root, 'test/unit'), { recursive: true });
const vitestImport = pathToFileURL(
path.join(packageRoot, 'node_modules/vitest/dist/index.js'),
).href;
writeFileSync(
path.join(root, 'test/unit/rejection.test.ts'),
`
import { it, expect } from ${JSON.stringify(vitestImport)};
it('executes its assertion but leaks a rejection', async () => {
expect(2 + 2).toBe(4);
setTimeout(() => { void Promise.reject(new Error('receipt rejection probe')); }, 0);
await new Promise(resolve => setTimeout(resolve, 50));
});
`,
);
const config = path.join(root, 'vitest.config.mjs');
writeFileSync(
config,
'export default { test: { include: ["test/**/*.test.ts"], includeTaskLocation: true, dangerouslyIgnoreUnhandledErrors: true } };',
);
try {
const child = spawnSync(
process.execPath,
[
path.join(packageRoot, 'node_modules/vitest/vitest.mjs'),
'run',
'--root',
root,
'--config',
config,
'--reporter',
path.join(packageRoot, 'scripts/execution-reporter.ts'),
'--outputFile',
output,
],
{ cwd: packageRoot, encoding: 'utf8', timeout: 30_000 },
);
expect(child.error, child.stderr).toBeUndefined();
expect(child.status, child.stderr).toBe(0);
const receipt = JSON.parse(readFileSync(output, 'utf8'));
expect(receipt.numPassedTests).toBe(1);
expect(receipt.executionErrors).toBeGreaterThan(0);
expect(reconcileTestReports([receipt]).report.success).toBe(false);
} finally {
rmSync(temp, { recursive: true, force: true });
}
});
});

View file

@ -4519,11 +4519,10 @@ void processRepoMap(std::map<std::string, Repo> repoMap) {
});
});
describe('known limitations (documented skip tests)', () => {
it.skip('Ruby block parameter: users.each { |user| } — closure param inference, different feature', () => {
// Not a for-loop; .each { |user| } is a method call with a block.
// Requires closure parameter inference — a different feature category
// applicable to Ruby, Swift closures, Kotlin lambdas, and Java lambdas.
describe('unknown Ruby block parameter types', () => {
it('does not invent a type for a block parameter from an untyped receiver', () => {
// Neither the parameter name nor calling `save` proves users contains
// User instances. Inferring User here would introduce false call edges.
const tree = parse(
`
def process(users)
@ -4533,7 +4532,7 @@ end
Ruby,
);
const typeEnv = buildTypeEnv(tree, SupportedLanguages.Ruby);
expect(flatGet(typeEnv, 'user')).toBe('User');
expect(flatGet(typeEnv, 'user')).toBeUndefined();
});
});

View file

@ -8,6 +8,8 @@ export default defineConfig({
hookTimeout: 120000,
pool: 'forks',
globals: true,
// Stable test identity across the Linux/macOS/Windows execution receipts.
includeTaskLocation: true,
teardownTimeout: 3000,
// E2E harnesses pin a small NODE_OPTIONS heap so spawned CLI children
// stay light; without this opt-out the #2649 auto-heap override would