From df49f908893ab42cafab1b1ed09633ede93c2260 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 6 Oct 2026 07:34:17 +0300 Subject: [PATCH] fix(ci): require every test to execute across the CI matrix (#3479) --- .github/workflows/ci-report.yml | 28 +- .github/workflows/ci-tests.yml | 190 ++++++++++-- .gitignore | 8 + README.md | 1 + TESTING.md | 37 ++- eval/check_test_execution.py | 121 ++++++++ eval/tests/test_check_test_execution.py | 142 +++++++++ eval/tests/test_evolve.py | 22 +- eval/tests/test_workflow_bench.py | 71 ++++- gitnexus/package.json | 1 + gitnexus/scripts/cross-platform-tests.ts | 9 +- gitnexus/scripts/execution-reporter.ts | 33 ++ gitnexus/scripts/run-benchmarks.ts | 32 ++ gitnexus/scripts/run-cross-platform.ts | 13 +- gitnexus/scripts/test-completeness.ts | 203 +++++++++++++ .../languages/dart/package-config.ts | 57 +++- .../integration/go-pipeline-benchmark.test.ts | 111 +++---- gitnexus/test/unit/calltool-dispatch.test.ts | 15 +- ...canonicalize-path-long-path-prefix.test.ts | 34 ++- .../test/unit/cross-platform-shard.test.ts | 10 +- .../test/unit/dart-package-imports.test.ts | 157 ++++++---- .../unit/eval-server-bind-restriction.test.ts | 4 +- .../unit/evidence-provenance-helper.test.ts | 31 +- .../test/unit/mcp-repository-policy.test.ts | 2 +- gitnexus/test/unit/repo-manager.test.ts | 2 +- .../test/unit/run-analyze-fts-repair.test.ts | 85 ++++-- .../name-fallback-visibility.test.ts | 2 +- .../rust-cargo-targets.test.ts | 2 +- gitnexus/test/unit/spring-aop.test.ts | 2 +- gitnexus/test/unit/test-completeness.test.ts | 284 ++++++++++++++++++ gitnexus/test/unit/type-env.test.ts | 11 +- gitnexus/vitest.config.ts | 2 + 32 files changed, 1478 insertions(+), 244 deletions(-) create mode 100644 eval/check_test_execution.py create mode 100644 eval/tests/test_check_test_execution.py create mode 100644 gitnexus/scripts/execution-reporter.ts create mode 100644 gitnexus/scripts/run-benchmarks.ts create mode 100644 gitnexus/scripts/test-completeness.ts create mode 100644 gitnexus/test/unit/test-completeness.test.ts diff --git a/.github/workflows/ci-report.yml b/.github/workflows/ci-report.yml index a16cb1103..83a05c35e 100644 --- a/.github/workflows/ci-report.yml +++ b/.github/workflows/ci-report.yml @@ -287,6 +287,7 @@ jobs: # ── Locate test results ── RESULTS_FILE=$(find_first "$DIR/test-reports" "test-results.json") WEB_RESULTS_FILE=$(find_first "$DIR/test-reports" "web-test-results.json") + PYTEST_RESULTS_FILE=$(find_first "$DIR/test-reports" "pytest-results.json") sum_results() { local file=$1 @@ -303,12 +304,21 @@ jobs: # framework, not as a top-line metric). read -r CLI_T CLI_P CLI_F CLI_S _ CLI_D <<< "$(sum_results "$RESULTS_FILE")" read -r WEB_T WEB_P WEB_F WEB_S _ WEB_D <<< "$(sum_results "$WEB_RESULTS_FILE")" + read -r PY_T PY_P PY_F PY_S _ PY_D <<< "$(sum_results "$PYTEST_RESULTS_FILE")" - TOTAL=$((CLI_T + WEB_T)) - PASSED=$((CLI_P + WEB_P)) - FAILED=$((CLI_F + WEB_F)) - SKIPPED=$((CLI_S + WEB_S)) + TOTAL=$((CLI_T + WEB_T + PY_T)) + PASSED=$((CLI_P + WEB_P + PY_P)) + FAILED=$((CLI_F + WEB_F + PY_F)) + SKIPPED=$((CLI_S + WEB_S + PY_S)) DURATION=$((CLI_D > WEB_D ? CLI_D : WEB_D)) + DURATION=$((DURATION > PY_D ? DURATION : PY_D)) + EXECUTION_ERRORS=0 + for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE" "$PYTEST_RESULTS_FILE"; do + if [ -n "$rf" ] && [ -f "$rf" ]; then + errors=$(jq -r '(.executionFailures // []) | length' "$rf") + EXECUTION_ERRORS=$((EXECUTION_ERRORS + errors)) + fi + done # ── Status helpers ── status_icon() { @@ -385,22 +395,22 @@ jobs: if [ "$TOTAL" -gt 0 ] 2>/dev/null; then echo "### Test Results" echo "" - echo "| Tests | Passed | Failed | Skipped | Duration |" + echo "| Tests | Passed | Failed | Unverified | Duration |" echo "|-------|--------|--------|---------|----------|" echo "| ${TOTAL} | ${PASSED} | ${FAILED} | ${SKIPPED} | ${DURATION}s |" echo "" - if [ "$FAILED" = "0" ]; then + if [[ "$FAILED" == "0" && "$SKIPPED" == "0" && "$EXECUTION_ERRORS" == "0" && "$TESTS" == "success" ]]; then echo "✅ All **${PASSED}** tests passed" else - echo "❌ **${FAILED}** failed / **${PASSED}** passed" + echo "❌ Test execution is incomplete or failed: **${FAILED}** failed, **${SKIPPED}** unverified, **${EXECUTION_ERRORS}** runner/suite errors." fi if [ "$SKIPPED" != "0" ]; then echo "" echo "
" - echo "${SKIPPED} test(s) skipped — expand for details" + echo "${SKIPPED} test(s) have no recorded pass — expand for details" echo "" - for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE"; do + for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE" "$PYTEST_RESULTS_FILE"; do if [ -n "$rf" ] && [ -f "$rf" ]; then jq -r ' .testResults[] diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index e6d93f74b..4862fc6ec 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -32,6 +32,7 @@ jobs: env: GITNEXUS_REQUIRE_FTS: '1' GITNEXUS_REQUIRE_ZIG: '1' + GITNEXUS_REQUIRE_VECTOR: '1' steps: # persist-credentials: false — runs tests + uploads a blob artifact; the # default-persisted token must not be capturable through it (zizmor @@ -115,7 +116,7 @@ jobs: run: >- npx vitest --mergeReports --reporter=default - --reporter=json + --reporter=./scripts/execution-reporter.ts --outputFile=test-results.json --coverage --coverage.reporter=json-summary @@ -131,7 +132,8 @@ jobs: run: >- npx vitest run --reporter=default - --reporter=json + --reporter=../gitnexus/scripts/execution-reporter.ts + --includeTaskLocation --outputFile=web-test-results.json working-directory: gitnexus-web - name: Run docker-server integration tests @@ -140,7 +142,7 @@ jobs: if: always() uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: - name: test-reports + name: coverage-reports path: | gitnexus/coverage/coverage-summary.json gitnexus/coverage/coverage-final.json @@ -162,7 +164,7 @@ jobs: steps: - id: gen run: | - TOTAL=3 # cross-platform (windows/macOS) shards per OS + TOTAL=4 # cross-platform (windows/macOS) shards per OS COV_TOTAL=3 # ubuntu coverage shards (merged before thresholds) if [ "$TOTAL" -lt 1 ] || [ "$COV_TOTAL" -lt 1 ]; then echo "shard totals must be >= 1" >&2; exit 1 @@ -254,8 +256,48 @@ jobs: shell: bash env: SHARD: ${{ matrix.shard }}/${{ needs.shard-plan.outputs.total }} + GITNEXUS_TEST_REPORT: ${{ matrix.os }}-${{ matrix.shard }}.json run: npx tsx scripts/run-cross-platform.ts --shard="$SHARD" working-directory: gitnexus + - name: Upload platform execution report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: execution-${{ matrix.os }}-${{ matrix.shard }} + path: gitnexus/${{ matrix.os }}-${{ matrix.shard }}.json + if-no-files-found: error + retention-days: 5 + + # Branch protection still requires these six names from the three-shard + # matrix. Keep them as aggregate gates as the real matrix grows: every native + # shard and the complete execution audit must pass before any gate succeeds. + # These jobs only verify dependency results; tests run in cross-platform above. + protected-platform-checks: + name: ${{ matrix.os }} (platform-sensitive) ${{ matrix.check }}/3 + needs: [cross-platform, test-completeness] + if: always() + permissions: {} + strategy: + fail-fast: false + matrix: + os: [windows-latest, macos-latest] + check: [1, 2, 3] + runs-on: ubuntu-latest + timeout-minutes: 2 + steps: + - name: Require all native shards and the execution audit + shell: bash + env: + NATIVE_RESULT: ${{ needs.cross-platform.result }} + EXECUTION_RESULT: ${{ needs.test-completeness.result }} + run: | + echo "Compatibility gate for the former three-shard check name." + echo "All native shards: $NATIVE_RESULT" + echo "Complete execution audit: $EXECUTION_RESULT" + if [[ "$NATIVE_RESULT" != "success" || "$EXECUTION_RESULT" != "success" ]]; then + echo "::error::Every native shard and the execution audit must succeed." + exit 1 + fi # Tree-sitter ABI gate (#1922). Two halves, both blocking: # 1. Static, offline: assert every grammar's compiled ABI loads on the @@ -482,11 +524,8 @@ jobs: # heap, so parallel forks both skew the timings and OOM the worker pool — they # must run one file at a time. # - # go-pipeline-benchmark.test.ts is deliberately NOT included: its - # worker-pool (#1848) suite spins a real worker pool that exits unexpectedly - # under vitest's fork pool (reproduced in validation), which would make this - # gate flaky. Go is already guarded by its non-gated O(n^2) tripwire (runs in - # the main coverage job) plus its golden capture-parity test. + # Every opt-in suite is discovered by run-benchmarks.ts. A failing worker + # benchmark is a failure to investigate, never a reason to omit that suite. benchmarks: name: benchmarks (GITNEXUS_BENCH) runs-on: ubuntu-latest @@ -876,28 +915,18 @@ jobs: - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) if: ${{ !cancelled() }} - # cpp-adl-benchmark.test.ts and csharp-razor-view-components-benchmark.test.ts - # are not `*-pipeline-benchmark.test.ts` files but belong here for the - # same reason: they are skipIf-gated on GITNEXUS_BENCH, so the scaling - # guards they hold never run in the main coverage job. env: - GITNEXUS_BENCH: '1' GITNEXUS_WORKER_READY_TIMEOUT_MS: '60000' - run: >- - npx vitest run --no-file-parallelism - test/integration/cobol-pipeline-benchmark.test.ts - test/integration/csharp-pipeline-benchmark.test.ts - test/integration/csharp-razor-view-components-benchmark.test.ts - test/integration/objective-c-pipeline-benchmark.test.ts - test/integration/cpp-adl-benchmark.test.ts - test/integration/data-route-table-benchmark.test.ts - test/integration/instance-ownership-pipeline-benchmark.test.ts - test/integration/spring-bean-resource-benchmark.test.ts - test/integration/spring-dynamic-lookup-benchmark.test.ts - test/integration/rust-pipeline-benchmark.test.ts - test/integration/php-pipeline-benchmark.test.ts - test/integration/ruby-pipeline-benchmark.test.ts + run: npm run test:benchmarks working-directory: gitnexus + - name: Upload benchmark execution report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: execution-benchmarks + path: gitnexus/benchmarks.json + if-no-files-found: error + retention-days: 5 # Locked eval suite. setup-uv and uv itself are immutable so CI exercises # exactly the dependency graph developers run from eval/uv.lock. @@ -916,8 +945,88 @@ jobs: python-version: '3.13' enable-cache: true cache-dependency-glob: eval/uv.lock - - run: uv run --locked --extra dev python -m pytest tests -q + - run: uv run --locked --extra dev python -m pytest tests -q --junitxml=pytest-locked.xml working-directory: eval + - name: Upload pytest execution report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: execution-pytest-locked + path: eval/pytest-locked.xml + if-no-files-found: error + retention-days: 5 + - uses: ./.github/actions/setup-gitnexus + with: + build: 'true' + - name: Exercise real workflow evidence preflight + if: ${{ !cancelled() }} + run: >- + npx vitest run test/unit/skill-evolution-workflow.test.ts + --reporter=default --reporter=./scripts/execution-reporter.ts --outputFile=preflight.json + working-directory: gitnexus + - name: Upload preflight execution report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: execution-preflight + path: gitnexus/preflight.json + if-no-files-found: error + retention-days: 5 + + test-completeness: + name: every test executed + needs: + [ + shard-plan, + coverage-merge, + cross-platform, + benchmarks, + eval-tests, + eval-containment-linux, + eval-containment-windows, + ] + if: ${{ !cancelled() }} + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: ./.github/actions/setup-gitnexus + - name: Download coverage and web reports + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + name: coverage-reports + - name: Download all execution receipts + if: ${{ !cancelled() }} + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 + with: + pattern: execution-* + path: execution-reports + merge-multiple: true + - name: Require a real pass for every collected test + env: + PLATFORM_SHARDS: ${{ needs.shard-plan.outputs.total }} + run: | + cp gitnexus/test-results.json execution-reports/coverage.json + cd gitnexus + node --import tsx scripts/test-completeness.ts ../execution-reports test-results.json "$PLATFORM_SHARDS" + - name: Require a real pass for every pytest case + if: ${{ !cancelled() }} + run: python3 eval/check_test_execution.py execution-reports eval/pytest-results.json + - name: Upload complete test reports + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: test-reports + path: | + gitnexus/coverage/coverage-summary.json + gitnexus/coverage/coverage-final.json + gitnexus/test-results.json + gitnexus-web/web-test-results.json + eval/pytest-results.json + if-no-files-found: error + retention-days: 5 # Native Linux ownership and Bubblewrap boundary. The environment flag makes # the real namespace test mandatory; a missing/blocked bwrap is a failure. @@ -992,8 +1101,19 @@ jobs: tests/test_workflow_bench_sessions.py tests/test_ce_plugin_runtime.py tests/test_offline_sweep_integration.py - tests/test_mock_provider.py -q + tests/test_mock_provider.py + tests/test_evolve.py::test_outer_runner_pid_namespace_kills_setsid_descendant + tests/test_oracle_assets.py::test_hidden_vitest_config_executes_sibling_oracle_against_candidate_checkout + -q --junitxml=pytest-ubuntu.xml working-directory: eval + - name: Upload pytest execution report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: execution-pytest-ubuntu + path: eval/pytest-ubuntu.xml + if-no-files-found: error + retention-days: 5 # Native Windows Job Object canary. POSIX-only tests skip by platform, while # the grandchild delayed-write test must execute and pass on this runner. @@ -1015,5 +1135,13 @@ jobs: run: >- uv run --locked --extra dev python -m pytest tests/test_process_control.py tests/test_model_gateway.py - -k "not locked_litellm" -q + -k "not locked_litellm" -q --junitxml=pytest-windows.xml working-directory: eval + - name: Upload pytest execution report + if: always() + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: execution-pytest-windows + path: eval/pytest-windows.xml + if-no-files-found: error + retention-days: 5 diff --git a/.gitignore b/.gitignore index 873f4082f..ffa32d659 100644 --- a/.gitignore +++ b/.gitignore @@ -131,5 +131,13 @@ local_docs/ .context/ gitnexus/web/ +# Local copies of CI execution receipts. +gitnexus/benchmarks.json +gitnexus/preflight.json +gitnexus/test-results.json +gitnexus/windows-latest-*.json +gitnexus/macos-latest-*.json +gitnexus-web/web-test-results.json + # Machine-local skill-evolution evidence (consumed by eval/workflow_bench/evolve.py) eval/workflow_bench/learnings.jsonl diff --git a/README.md b/README.md index a761716e7..d89a7ef05 100644 --- a/README.md +++ b/README.md @@ -611,6 +611,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | Variable | Default | Effect | Tune when… | | ----------------------------------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `GITNEXUS_TEST_REPORT` | unset | Write a JSON execution receipt from `scripts/run-cross-platform.ts`; accepts a basename ending in `.json`. | CI platform shards use distinct filenames so the completeness check can require every report. | | `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers `. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. | | `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | diff --git a/TESTING.md b/TESTING.md index 4be0f73c5..450a0c792 100644 --- a/TESTING.md +++ b/TESTING.md @@ -125,11 +125,46 @@ GitHub Actions (`.github/workflows/ci.yml`) orchestrate: | `ci-scope-parity.yml` | discover, parity | Scope-resolution parity for all migrated languages | | `ci-e2e.yml` | e2e (chromium) | Playwright E2E, gated on `gitnexus-web/**` changes | -The `CI Gate` job in `ci.yml` is the single required check for branch protection. It requires quality, tests, e2e, and scope-parity to all pass. +The `CI Gate` job in `ci.yml` requires the quality and test workflows to pass. +The browser E2E workflow must pass or be skipped because no web files changed. + +Branch protection also requires six platform check names from the former +three-shard matrix. These names remain as aggregate gates: all native shards +and the `every test executed` audit must succeed before any of them passes. +Failed, cancelled, skipped, or missing dependency results fail these gates. +The actual native tests run in the current Windows/macOS shard matrix. The `typecheck` job runs both the production compiler check and `npm run typecheck:tests`. A type error in either check fails the job and the CI gate. +### Complete execution, including platform and benchmark tests + +The required `every test executed` job reconciles execution receipts from Ubuntu +coverage, every Windows/macOS shard, the serial benchmark run, and the real Python +workflow preflight. It requires a recorded pass for every collected test. A skip +on Linux is satisfied only by a pass of that exact test in another required job. +Test identities include the file, suite/title and source location; ambiguous +parameterized cases must have unique titles. Missing receipts, missing test files, +unhandled runner errors, failed hooks, failed assertions, and tests with no pass +all fail the gate. Web tests are checked separately with the same rules. + +The locked pytest suite and Linux/Windows containment jobs also upload JUnit +receipts. The same gate checks every Python test file was collected and every +case passed in at least one job, while preserving failures from any job. The +Linux containment job supplies Bubblewrap, the pinned CLI, and built Vitest +dependencies for tests that cannot run in the basic Python job. Python results +are included in the combined PR report. + +The PR report shows the reconciled result as **Unverified**. Zero means every test +has execution evidence; individual OS logs still show tests that require another +OS as skipped. Failed executions remain failures even if another job passes. + +`npm run test:benchmarks` discovers all tests gated by `GITNEXUS_BENCH` and runs +them serially. Keep timing measurements out of parallel coverage workers. The +`eval-tests` job installs the locked Python dependencies and runs the workflow +preflight Vitest tests as well as pytest; those tests must exercise real Python +validation, never a stubbed success. + ## Regression testing Re-run the full relevant suite when: diff --git a/eval/check_test_execution.py b/eval/check_test_execution.py new file mode 100644 index 000000000..1d5e0f71c --- /dev/null +++ b/eval/check_test_execution.py @@ -0,0 +1,121 @@ +"""Require a real passing execution of every pytest case across the CI jobs.""" + +from __future__ import annotations + +import json +from pathlib import Path +import sys +import xml.etree.ElementTree as ET + + +RECEIPTS = ("pytest-locked.xml", "pytest-ubuntu.xml", "pytest-windows.xml") + + +def reconcile(reports: list[Path], expected_files: list[str]) -> dict: + if not reports: + raise ValueError("No pytest execution reports supplied") + tests: dict[tuple[str, str], set[str]] = {} + durations: list[float] = [] + for report in reports: + root = ET.parse(report).getroot() + suites = [root] if root.tag == "testsuite" else list(root.findall("testsuite")) + if root.tag not in {"testsuite", "testsuites"} or not suites: + raise ValueError(f"Invalid pytest report: {report}") + seen: set[tuple[str, str]] = set() + duration = 0.0 + for suite in suites: + cases = suite.findall("testcase") + if not cases: + raise ValueError(f"Empty pytest suite in {report}") + counts = {"tests": len(cases), "failures": 0, "errors": 0, "skipped": 0} + for case in cases: + key = (case.get("classname", ""), case.get("name", "")) + if not all(key) or key in seen: + raise ValueError(f"Missing or ambiguous pytest identity in {report}: {key}") + seen.add(key) + for status in ("failures", "errors", "skipped"): + tag = {"failures": "failure", "errors": "error", "skipped": "skipped"}[status] + counts[status] += len(case.findall(tag)) + status = ( + "failed" + if case.find("failure") is not None or case.find("error") is not None + else "pending" + if case.find("skipped") is not None + else "passed" + ) + tests.setdefault(key, set()).add(status) + for name, count in counts.items(): + if int(suite.get(name, "-1")) != count: + raise ValueError(f"Inconsistent pytest {name} count in {report}") + duration += float(suite.get("time", "0")) + durations.append(duration) + + modules = {module for module, _ in tests} + for file in expected_files: + module = file.removesuffix(".py").replace("/", ".") + if not any(name == module or name.startswith(module + ".") for name in modules): + raise ValueError(f"Pytest file was never collected: {file}") + + suites_by_module: dict[str, dict] = {} + failed: list[str] = [] + pending = 0 + for (module, name), statuses in sorted(tests.items()): + status = "failed" if "failed" in statuses else "passed" if "passed" in statuses else "pending" + full_name = f"{module}::{name}" + if status == "failed": + failed.append(full_name) + pending += status == "pending" + suite = suites_by_module.setdefault( + module, + { + "name": module, + "status": "passed", + "assertionResults": [], + "endTime": round(max(durations) * 1000), + }, + ) + suite["assertionResults"].append( + { + "fullName": full_name, + "ancestorTitles": [module], + "title": name, + "status": status, + } + ) + if status == "failed": + suite["status"] = "failed" + return { + "success": not failed and pending == 0, + "numTotalTests": len(tests), + "numPassedTests": len(tests) - len(failed) - pending, + "numFailedTests": len(failed), + "numPendingTests": pending, + "numTotalTestSuites": len(suites_by_module), + "numFailedTestSuites": sum(s["status"] == "failed" for s in suites_by_module.values()), + "executionFailures": failed, + "startTime": 0, + "testResults": list(suites_by_module.values()), + } + + +def main() -> int: + if len(sys.argv) != 3: + raise ValueError("Usage: check_test_execution.py ") + directory, output = map(Path, sys.argv[1:]) + root = Path(__file__).resolve().parent + expected = [p.relative_to(root).as_posix() for p in (root / "tests").rglob("test_*.py")] + report = reconcile([directory / name for name in RECEIPTS], expected) + output.write_text(json.dumps(report), encoding="utf-8") + print( + f"Eval execution: {report['numPassedTests']}/{report['numTotalTests']} passed; " + f"{report['numPendingTests']} unverified; {report['numFailedTests']} failures" + ) + for suite in report["testResults"]: + for case in suite["assertionResults"]: + if case["status"] != "passed": + print(f"{case['status']}: {case['fullName']}", file=sys.stderr) + return 0 if report["success"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/eval/tests/test_check_test_execution.py b/eval/tests/test_check_test_execution.py new file mode 100644 index 000000000..22950081f --- /dev/null +++ b/eval/tests/test_check_test_execution.py @@ -0,0 +1,142 @@ +"""The execution gate must reject absent, skipped, failed, or ambiguous evidence.""" + +from pathlib import Path +import json +import shutil +import subprocess +import sys +import xml.etree.ElementTree as ET + +import pytest + +from check_test_execution import reconcile + + +def receipt(tmp_path: Path, filename: str, cases: list[tuple[str, str]], module="tests.test_native") -> Path: + suite = ET.Element("testsuite", tests=str(len(cases)), failures="0", errors="0", skipped="0", time="1.5") + for name, status in cases: + case = ET.SubElement(suite, "testcase", classname=module, name=name) + if status != "passed": + ET.SubElement(case, status) + count = {"failure": "failures", "error": "errors", "skipped": "skipped"}[status] + suite.set(count, str(int(suite.get(count)) + 1)) + path = tmp_path / filename + ET.ElementTree(suite).write(path, encoding="utf-8") + return path + + +def test_platform_skip_requires_a_matching_real_pass(tmp_path): + linux = receipt(tmp_path, "linux.xml", [("test_windows", "skipped"), ("test_posix", "passed")]) + windows = receipt(tmp_path, "windows.xml", [("test_windows", "passed"), ("test_posix", "skipped")]) + report = reconcile([linux, windows], ["tests/test_native.py"]) + assert report["success"] is True + assert report["numPassedTests"] == report["numTotalTests"] == 2 + assert report["numPendingTests"] == report["numFailedTests"] == 0 + + +def test_unresolved_skip_is_not_a_pass(tmp_path): + path = receipt(tmp_path, "report.xml", [("test_missing_runtime", "skipped")]) + report = reconcile([path], []) + assert report["success"] is False + assert report["numPendingTests"] == 1 + assert report["numPassedTests"] == 0 + assert report["testResults"][0]["assertionResults"][0]["ancestorTitles"] == ["tests.test_native"] + + +@pytest.mark.parametrize("status", ["failure", "error"]) +def test_failure_is_not_hidden_by_another_job_passing(tmp_path, status): + failed = receipt(tmp_path, "failed.xml", [("test_shared", status)]) + passed = receipt(tmp_path, "passed.xml", [("test_shared", "passed")]) + report = reconcile([failed, passed], []) + assert report["success"] is False + assert report["numFailedTests"] == 1 + assert report["numPassedTests"] == 0 + assert report["executionFailures"] == ["tests.test_native::test_shared"] + + +def test_same_title_in_another_file_does_not_resolve_a_skip(tmp_path): + pending = receipt(tmp_path, "pending.xml", [("test_same", "skipped")]) + unrelated = receipt(tmp_path, "unrelated.xml", [("test_same", "passed")], module="tests.test_other") + report = reconcile([pending, unrelated], []) + assert report["numTotalTests"] == 2 + assert report["numPendingTests"] == 1 + + +def test_duplicate_identity_is_rejected(tmp_path): + path = receipt(tmp_path, "duplicate.xml", [("test_same", "passed"), ("test_same", "skipped")]) + with pytest.raises(ValueError, match="ambiguous pytest identity"): + reconcile([path], []) + + +def test_missing_receipt_is_rejected(tmp_path): + with pytest.raises(FileNotFoundError): + reconcile([tmp_path / "missing.xml"], []) + + +def test_empty_receipt_is_rejected(tmp_path): + path = receipt(tmp_path, "empty.xml", []) + with pytest.raises(ValueError, match="Empty pytest suite"): + reconcile([path], []) + + +def test_missing_file_is_rejected(tmp_path): + path = receipt(tmp_path, "report.xml", [("test_present", "passed")]) + with pytest.raises(ValueError, match="never collected: tests/test_missing.py"): + reconcile([path], ["tests/test_missing.py"]) + + +def test_inconsistent_counts_are_rejected(tmp_path): + path = receipt(tmp_path, "report.xml", [("test_present", "passed")]) + path.write_text(path.read_text().replace('tests="1"', 'tests="2"')) + with pytest.raises(ValueError, match="Inconsistent pytest tests count"): + reconcile([path], []) + + +def test_pytest_testsuites_wrapper_and_test_classes(tmp_path): + path = receipt(tmp_path, "report.xml", [("test_method", "passed")], module="tests.test_native.TestNative") + suite = ET.parse(path).getroot() + root = ET.Element("testsuites") + root.append(suite) + ET.ElementTree(root).write(path, encoding="utf-8") + assert reconcile([path], ["tests/test_native.py"])["success"] is True + + +def test_missing_report_list_is_rejected(): + with pytest.raises(ValueError, match="No pytest execution reports"): + reconcile([], []) + + +@pytest.mark.parametrize("scenario", ["passed", "pending", "failed", "missing-receipt", "missing-file"]) +def test_cli_requires_complete_execution_evidence(tmp_path, scenario): + script = tmp_path / "check_test_execution.py" + shutil.copyfile(Path(__file__).parents[1] / script.name, script) + tests = tmp_path / "tests" + tests.mkdir() + (tests / "test_native.py").write_text("# inventory fixture\n") + reports = tmp_path / "reports" + reports.mkdir() + receipt(reports, "pytest-locked.xml", [("test_native", "skipped" if scenario == "pending" else "passed")]) + receipt(reports, "pytest-ubuntu.xml", [("test_native", "error" if scenario == "failed" else "skipped")]) + if scenario != "missing-receipt": + receipt(reports, "pytest-windows.xml", [("test_native", "skipped")]) + if scenario == "missing-file": + (tests / "test_missing.py").write_text("# must be collected\n") + output = tmp_path / "pytest-results.json" + result = subprocess.run( + [sys.executable, str(script), str(reports), str(output)], + capture_output=True, + text=True, + timeout=10, + ) + assert result.returncode == (0 if scenario == "passed" else 1) + if scenario.startswith("missing-"): + assert not output.exists() + assert ("pytest-windows.xml" if scenario == "missing-receipt" else "never collected") in result.stderr + else: + report = json.loads(output.read_text()) + assert report["success"] is (scenario == "passed") + assert report["numTotalTests"] == 1 + assert report["numPassedTests"] == (scenario == "passed") + assert report["numPendingTests"] == (scenario == "pending") + assert report["numFailedTests"] == (scenario == "failed") + assert "Eval execution:" in result.stdout diff --git a/eval/tests/test_evolve.py b/eval/tests/test_evolve.py index 32d50fd85..9267a6c6b 100644 --- a/eval/tests/test_evolve.py +++ b/eval/tests/test_evolve.py @@ -5,6 +5,7 @@ import json import os import subprocess import sys +import threading import time from contextlib import contextmanager from datetime import UTC, datetime, timedelta @@ -1428,21 +1429,32 @@ def test_outer_runner_pid_namespace_kills_setsid_descendant(tmp_path): pytest.skip(str(exc)) raise AssertionError("pytest.skip() returned unexpectedly") sentinel = tmp_path / "escaped" + ready = tmp_path / "descendant-ready" + cancel = threading.Event() + descendant = ( + "import time,pathlib; " + f"pathlib.Path({str(ready)!r}).touch(); print('ready', flush=True); " + f"time.sleep(1); pathlib.Path({str(sentinel)!r}).touch()" + ) child = ( "import os,subprocess,sys,time; " - f"subprocess.Popen([sys.executable,'-c',\"import time,pathlib;time.sleep(1);pathlib.Path({str(sentinel)!r}).touch()\"],preexec_fn=os.setsid); " + f"subprocess.Popen([sys.executable,'-c',{descendant!r}],preexec_fn=os.setsid); " "time.sleep(10)" ) result = run_managed( pid_namespace_command([sys.executable, "-c", child], bwrap_bin=bwrap), - timeout=0.15, + # Cancel only after the escaped-session descendant has actually started. + # A timeout during bwrap/Python startup must not count as containment. + timeout=10, terminate_grace=0.1, require_pid_namespace=True, + cancel_event=cancel, + stdout_observer=lambda _chunk: cancel.set(), ) - time.sleep(1.1) - + assert ready.exists(), "the setsid descendant must start before it can be contained" assert not result.ok - assert result.state in {"timeout", "forced-kill"} + assert result.state == "cancelled" + time.sleep(1.1) assert not sentinel.exists() diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index 4b197d9dc..4a03ed243 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -250,7 +250,11 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): # Carries the real-CLI identity probe, which needs CLAUDE_CANARY_BIN - # set only on this job. Omitted from this list it skipped everywhere. "tests/test_mock_provider.py", + # These also need runtime dependencies absent from the locked pytest job. + "tests/test_evolve.py::test_outer_runner_pid_namespace_kills_setsid_descendant", + "tests/test_oracle_assets.py::test_hidden_vitest_config_executes_sibling_oracle_against_candidate_checkout", "-q", + "--junitxml=pytest-ubuntu.xml", ] bwrap_canary_marker = re.compile( r'@pytest\.mark\.skipif\(\s*os\.environ\.get\("GITNEXUS_REQUIRE_BWRAP_CANARY"\)', @@ -266,6 +270,72 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): assert "eval-containment-windows:" in workflow +@pytest.fixture(scope="module") +def ci_test_jobs(): + repo_root = Path(__file__).resolve().parents[2] + workflow = yaml.safe_load((repo_root / ".github" / "workflows" / "ci-tests.yml").read_text()) + return workflow["jobs"] + + +def test_platform_ci_preserves_required_check_names_with_complete_execution_gates(ci_test_jobs): + jobs = ci_test_jobs + gate = jobs["protected-platform-checks"] + matrix = gate["strategy"]["matrix"] + names = { + "tests / " + gate["name"].replace("${{ matrix.os }}", platform).replace("${{ matrix.check }}", str(check)) + for platform in matrix["os"] + for check in matrix["check"] + } + assert names == { + f"tests / {platform} (platform-sensitive) {check}/3" + for platform in ("windows-latest", "macos-latest") + for check in (1, 2, 3) + } + plan = jobs["shard-plan"]["steps"][0]["run"] + total = int(re.search(r"^TOTAL=(\d+)\b", plan, re.MULTILINE)[1]) + native_names = { + "tests / " + + jobs["cross-platform"]["name"] + .replace("${{ matrix.os }}", platform) + .replace("${{ matrix.shard }}", str(shard)) + .replace("${{ needs.shard-plan.outputs.total }}", str(total)) + for platform in jobs["cross-platform"]["strategy"]["matrix"]["os"] + for shard in range(1, total + 1) + } + assert names.isdisjoint(native_names) + assert set(gate["needs"]) == {"cross-platform", "test-completeness"} + assert gate["if"] == "always()" + assert not gate.get("continue-on-error", False) + assert not jobs["cross-platform"].get("continue-on-error", False) + assert not jobs["test-completeness"].get("continue-on-error", False) + assert gate["permissions"] == {} + assert len(gate["steps"]) == 1 + step = gate["steps"][0] + assert not step.get("continue-on-error", False) + assert "if" not in step + assert step["shell"] == "bash" + assert step["env"] == { + "NATIVE_RESULT": "${{ needs.cross-platform.result }}", + "EXECUTION_RESULT": "${{ needs.test-completeness.result }}", + } + + +@pytest.mark.parametrize("native_result", ["success", "failure", "cancelled", "skipped", ""]) +@pytest.mark.parametrize("execution_result", ["success", "failure", "cancelled", "skipped", ""]) +def test_required_platform_check_fails_unless_every_dependency_passed(ci_test_jobs, native_result, execution_result): + step = ci_test_jobs["protected-platform-checks"]["steps"][0] + result = subprocess.run( + ["bash", "--noprofile", "--norc", "-e", "-o", "pipefail", "-c", step["run"]], + env={**os.environ, "NATIVE_RESULT": native_result, "EXECUTION_RESULT": execution_result}, + capture_output=True, + text=True, + timeout=10, + check=False, + ) + expected = 0 if native_result == execution_result == "success" else 1 + assert result.returncode == expected, result.stdout + result.stderr + + def test_shipped_scenarios_opt_out_the_cross_module_cell_and_rebuild_graph_assets(): task_file = Path(__file__).resolve().parents[1] / "workflow_bench" / "tasks.scenarios.yaml" tasks = yaml.safe_load(task_file.read_text())["tasks"] @@ -1092,4 +1162,3 @@ def test_an_uninvoked_skill_still_counts_toward_the_arm_median(): # The invariant that makes the half-fix unsafe: the median and the run count # the gate reads must cover the same rows. assert agg["valid_runs"] == 2 - diff --git a/gitnexus/package.json b/gitnexus/package.json index 99422e958..afb4b5f37 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -59,6 +59,7 @@ "test:watch": "vitest", "test:coverage": "vitest run --coverage", "test:cross-platform": "tsx scripts/run-cross-platform.ts", + "test:benchmarks": "node --import tsx scripts/run-benchmarks.ts", "postinstall": "node scripts/build-tree-sitter-grammars.cjs", "assert-publish-coverage": "node scripts/assert-publish-grammar-coverage.cjs", "prepare": "node scripts/build.js", diff --git a/gitnexus/scripts/cross-platform-tests.ts b/gitnexus/scripts/cross-platform-tests.ts index 3fdb26ee0..55bade8f0 100644 --- a/gitnexus/scripts/cross-platform-tests.ts +++ b/gitnexus/scripts/cross-platform-tests.ts @@ -206,9 +206,8 @@ const LBUG_NATIVE = [ // (the same reason fts-extension-e2e.test.ts is registered below), so the // FTS-unavailable branch has to run on a real Windows/macOS runner rather // than only on Ubuntu where FTS always loads. And its both-blocked case is - // gated on GITNEXUS_REQUIRE_VECTOR=1, which ci-tests.yml sets ONLY on this - // job — everywhere else an unavailable VECTOR extension skips instead of - // failing. Budget: four real analyze runs, so expect it to sit alongside the + // gated on GITNEXUS_REQUIRE_VECTOR=1, which ci-tests.yml requires in both + // coverage and platform jobs so an unavailable VECTOR fails loudly. Budget: four real analyze runs, so expect it to sit alongside the // VECTOR sibling's ~87s Windows measurement. 'test/unit/incremental-index-extension-dml-gate.test.ts', ]; @@ -308,6 +307,10 @@ const NATIVE_ADDON_SMOKE = [ // Filesystem behavior tests — exercise operations that vary across // platforms (CRLF, symlinks, permissions, temp dirs) const FILESYSTEM = [ + // The deletion-guard cases in this file require real Windows path semantics. + 'test/unit/canonicalize-path-long-path-prefix.test.ts', + 'test/unit/storage-resolver.test.ts', + 'test/unit/dart-package-imports.test.ts', // Cargo membership uses path normalization, descriptor validation, symlinks, // and Rust native parsing (including long Windows source strings). 'test/unit/scope-resolution/rust-cargo-targets.test.ts', diff --git a/gitnexus/scripts/execution-reporter.ts b/gitnexus/scripts/execution-reporter.ts new file mode 100644 index 000000000..83a93a035 --- /dev/null +++ b/gitnexus/scripts/execution-reporter.ts @@ -0,0 +1,33 @@ +import { JsonReporter, type TestRunEndReason } from 'vitest/node'; + +/** Vitest's stock JSON omits unhandled errors, even when success is true. */ +export default class ExecutionReporter extends JsonReporter { + private executionErrors = 0; + private executionReason: TestRunEndReason = 'interrupted'; + + constructor() { + super({}); + } + + override async onTestRunEnd( + modules: Parameters[0], + errors: readonly unknown[] = [], + reason: TestRunEndReason = 'interrupted', + ): Promise { + this.executionErrors = errors.length; + this.executionReason = reason; + await super.onTestRunEnd(modules); + } + + override async writeReport(json: string): Promise { + const report = JSON.parse(json); + await super.writeReport( + JSON.stringify({ + ...report, + success: report.success && this.executionErrors === 0 && this.executionReason === 'passed', + executionErrors: this.executionErrors, + executionReason: this.executionReason, + }), + ); + } +} diff --git a/gitnexus/scripts/run-benchmarks.ts b/gitnexus/scripts/run-benchmarks.ts new file mode 100644 index 000000000..e2a58d5a3 --- /dev/null +++ b/gitnexus/scripts/run-benchmarks.ts @@ -0,0 +1,32 @@ +import { execFileSync } from 'node:child_process'; +import { globSync, readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; + +const root = fileURLToPath(new URL('../', import.meta.url)); +// Discover the opt-in suites from source so a new benchmark cannot silently +// miss a hand-maintained workflow list. Keep wall-clock measurements serial. +const files = globSync('test/**/*.test.ts', { cwd: root }) + .filter((file) => + /process\.env(?:\.GITNEXUS_BENCH|\[['"]GITNEXUS_BENCH['"]\])/.test( + readFileSync(new URL(`../${file.replaceAll('\\', '/')}`, import.meta.url), 'utf8'), + ), + ) + .sort(); +if (!files.length) throw new Error('No benchmark test suites discovered'); +execFileSync( + process.execPath, + [ + fileURLToPath(new URL('../node_modules/vitest/vitest.mjs', import.meta.url)), + 'run', + '--no-file-parallelism', + ...files, + '--reporter=default', + '--reporter=./scripts/execution-reporter.ts', + '--outputFile=benchmarks.json', + ], + { + cwd: root, + stdio: 'inherit', + env: { ...process.env, GITNEXUS_BENCH: '1' }, + }, +); diff --git a/gitnexus/scripts/run-cross-platform.ts b/gitnexus/scripts/run-cross-platform.ts index 5404a87c3..630fd0b93 100644 --- a/gitnexus/scripts/run-cross-platform.ts +++ b/gitnexus/scripts/run-cross-platform.ts @@ -80,8 +80,19 @@ console.log( ); const startedAt = Date.now(); +const reportName = process.env.GITNEXUS_TEST_REPORT; +if (reportName && !/^[a-zA-Z0-9_-]+\.json$/.test(reportName)) { + throw new Error('GITNEXUS_TEST_REPORT must be a JSON basename'); +} +const reportArgs = reportName + ? [ + '--reporter=default', + '--reporter=./scripts/execution-reporter.ts', + `--outputFile=${reportName}`, + ] + : []; try { - execFileSync('npx', ['vitest', 'run', ...files], { + execFileSync('npx', ['vitest', 'run', ...files, ...reportArgs], { cwd: ROOT, stdio: 'inherit', timeout: timeoutMs, diff --git a/gitnexus/scripts/test-completeness.ts b/gitnexus/scripts/test-completeness.ts new file mode 100644 index 000000000..89ca86892 --- /dev/null +++ b/gitnexus/scripts/test-completeness.ts @@ -0,0 +1,203 @@ +import { globSync, readFileSync, writeFileSync } from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath, pathToFileURL } from 'node:url'; +import type { TestRunEndReason } from 'vitest/node'; + +type Assertion = { + ancestorTitles: string[]; + title: string; + fullName: string; + status: string; + location?: { line: number; column: number }; +}; +type Suite = { + name: string; + status: string; + assertionResults: Assertion[]; + endTime?: number; + message?: string; +}; +type Report = { + success?: boolean; + numTotalTests?: number; + startTime?: number; + executionErrors: number; + executionReason: TestRunEndReason; + testResults: Suite[]; +}; + +/** Stable across GitHub's Linux, macOS and Windows checkout roots. */ +function testPath(name: string, packageName: string): string { + const normalized = name.replaceAll('\\', '/'); + const relative = + normalized.match(new RegExp(`(?:^|/)${packageName}/((?:test|src)/.+)$`))?.[1] ?? normalized; + if (!/^(?:test|src)\/.+\.test\.[jt]sx?$/.test(relative)) { + throw new Error(`Invalid test report path: ${name}`); + } + return relative; +} + +/** A skip is resolved only by a recorded pass of that same test in another job. */ +export function reconcileTestReports( + inputs: unknown[], + expectedFiles: string[] = [], + packageName = 'gitnexus', +) { + if (inputs.length === 0) throw new Error('No execution reports supplied'); + const tests = new Map< + string, + { file: string; assertion: Assertion; passed: boolean; failed: boolean } + >(); + const failures = new Set(); + const failedSuites = new Set(); + const startTimes: number[] = []; + const endTimes: number[] = []; + for (const [index, input] of inputs.entries()) { + const report = input as Report; + if (!report || !Array.isArray(report.testResults) || report.testResults.length === 0) { + throw new Error(`Missing or empty execution report ${index}`); + } + if (report.success === false) failures.add(`Report ${index}: unsuccessful runner result`); + if ( + !Number.isInteger(report.executionErrors) || + report.executionErrors < 0 || + !['passed', 'failed', 'interrupted'].includes(report.executionReason) + ) { + throw new Error(`Report ${index}: missing execution reporter evidence`); + } + if (report.executionErrors || report.executionReason !== 'passed') { + failures.add( + `Report ${index}: ${report.executionErrors} unhandled errors; ${report.executionReason}`, + ); + } + if (typeof report.startTime === 'number') startTimes.push(report.startTime); + let count = 0; + const seen = new Set(); + for (const suite of report.testResults) { + const file = testPath(suite.name, packageName); + if (!Array.isArray(suite.assertionResults) || suite.assertionResults.length === 0) { + throw new Error(`No collected tests in ${file}`); + } + if (suite.status === 'failed') { + failures.add(`${file}: failed suite`); + failedSuites.add(file); + } + if (typeof suite.endTime === 'number') endTimes.push(suite.endTime); + for (const assertion of suite.assertionResults) { + if ( + !['passed', 'failed', 'pending', 'skipped', 'todo', 'disabled'].includes( + assertion.status, + ) || + !Array.isArray(assertion.ancestorTitles) || + typeof assertion.title !== 'string' || + (assertion.location !== undefined && + (!Number.isInteger(assertion.location.line) || + !Number.isInteger(assertion.location.column))) + ) { + throw new Error(`Malformed assertion in ${file}`); + } + // Collection order differs across platforms. Source location, not an + // ordinal, distinguishes equal titles declared at different locations. + // Helper-generated tests may have no location; their complete title + // must then be unique within the file (the duplicate check still applies). + const title = JSON.stringify([...assertion.ancestorTitles, assertion.title]); + const key = JSON.stringify([ + file, + title, + assertion.location?.line ?? null, + assertion.location?.column ?? null, + ]); + if (seen.has(key)) + throw new Error( + `Ambiguous test identity in ${file}: ${title}; give parameterized cases unique titles`, + ); + seen.add(key); + const result = tests.get(key) ?? { file, assertion, passed: false, failed: false }; + result.passed ||= assertion.status === 'passed'; + result.failed ||= assertion.status === 'failed'; + tests.set(key, result); + if (result.failed) failures.add(`${file}: ${assertion.fullName ?? title}`); + count++; + } + } + if (typeof report.numTotalTests === 'number' && report.numTotalTests !== count) { + throw new Error(`Report ${index}: expected ${report.numTotalTests} tests, found ${count}`); + } + } + const unverified = [...tests.values()] + .filter((test) => !test.passed && !test.failed) + .map((test) => `${test.file}: ${test.assertion.fullName}`); + const suites = new Map(); + for (const test of tests.values()) { + const status = test.failed ? 'failed' : test.passed ? 'passed' : 'pending'; + const suite = suites.get(test.file) ?? { + name: test.file, + status: 'passed', + assertionResults: [], + }; + suite.assertionResults.push({ ...test.assertion, status }); + if (status === 'failed' || failedSuites.has(test.file)) suite.status = 'failed'; + suites.set(test.file, suite); + } + for (const file of expectedFiles) { + if (!suites.has(file.replaceAll('\\', '/'))) + throw new Error(`Test file was never collected: ${file}`); + } + const passed = [...tests.values()].filter((test) => test.passed && !test.failed).length; + const endTime = endTimes.length ? Math.max(...endTimes) : 0; + return { + total: tests.size, + passed, + unverified, + failures: [...failures], + report: { + success: failures.size === 0 && unverified.length === 0, + numTotalTests: tests.size, + numPassedTests: passed, + numFailedTests: [...tests.values()].filter((test) => test.failed).length, + numPendingTests: unverified.length, + numTotalTestSuites: suites.size, + numFailedTestSuites: failedSuites.size, + executionFailures: [...failures], + startTime: startTimes.length ? Math.min(...startTimes) : 0, + testResults: [...suites.values()].map((suite) => ({ ...suite, endTime })), + }, + }; +} + +if (process.argv[1] && import.meta.url === pathToFileURL(path.resolve(process.argv[1])).href) { + const [directory, output, shardsText] = process.argv.slice(2); + const shards = Number(shardsText); + if (!directory || !output || !Number.isInteger(shards) || shards < 1) { + throw new Error('Usage: test-completeness.ts '); + } + const names = ['coverage.json', 'benchmarks.json', 'preflight.json']; + for (const os of ['windows-latest', 'macos-latest']) { + for (let shard = 1; shard <= shards; shard++) names.push(`${os}-${shard}.json`); + } + // Exact expected receipts: an absent/cancelled job is never an empty success. + const root = fileURLToPath(new URL('../', import.meta.url)); + const expected = globSync('test/**/*.test.ts', { cwd: root }); + const result = reconcileTestReports( + names.map((name) => JSON.parse(readFileSync(path.join(directory, name), 'utf8'))), + expected, + ); + writeFileSync(output, JSON.stringify(result.report)); + console.log( + `Test execution: ${result.passed}/${result.total} passed; ${result.unverified.length} unverified; ${result.failures.length} failures`, + ); + for (const message of [...result.unverified, ...result.failures]) console.error(message); + if (!result.report.success) process.exitCode = 1; + const webPath = path.join(root, '../gitnexus-web/web-test-results.json'); + const web = reconcileTestReports( + [JSON.parse(readFileSync(webPath, 'utf8'))], + globSync('{test,src}/**/*.test.{ts,tsx}', { cwd: path.join(root, '../gitnexus-web') }), + 'gitnexus-web', + ); + writeFileSync(webPath, JSON.stringify(web.report)); + console.log( + `Web execution: ${web.passed}/${web.total} passed; ${web.unverified.length} unverified; ${web.failures.length} failures`, + ); + for (const message of [...web.unverified, ...web.failures]) console.error(message); + if (!web.report.success) process.exitCode = 1; +} diff --git a/gitnexus/src/core/ingestion/languages/dart/package-config.ts b/gitnexus/src/core/ingestion/languages/dart/package-config.ts index 4503a57bd..2040d6d94 100644 --- a/gitnexus/src/core/ingestion/languages/dart/package-config.ts +++ b/gitnexus/src/core/ingestion/languages/dart/package-config.ts @@ -39,6 +39,11 @@ export interface DartPackageConfigOptions { * is listed and before its entries are opened. */ readonly beforeEntryOpen?: (relativePath: string) => void | Promise; + /** + * Test seam. Production calls omit it. Invoked after the directory is + * opened and before it is listed. + */ + readonly beforeDirectoryList?: (relativePath: string) => void | Promise; } type ManifestRead = @@ -123,12 +128,13 @@ function directoryIdentity(stat: BigIntStats): string { /** * Path that lists the directory inode already open on `fd`. - * Linux uses `/proc/self/fd/N` and macOS uses `/dev/fd/N`. - * Other platforms have no such path; callers refuse instead of listing by name. + * Only Linux has one: `/proc/self/fd/N`. macOS refuses `opendir` on + * `/dev/fd/N` for a directory (ENOTDIR) and Node has no `fdopendir`, so the + * walker lists the lexical path and verifies the pinned chain around it. + * Every other platform returns null and is not walked. */ export function descriptorDirectoryPath(fd: number): string | null { if (process.platform === 'linux') return `/proc/self/fd/${fd}`; - if (process.platform === 'darwin') return `/dev/fd/${fd}`; return null; } @@ -264,11 +270,33 @@ async function openChildFile( throw Object.assign(new Error('no descriptor anchor'), { code: 'ENOTSUP' }); } -async function listOpenedDirectory(handle: FileHandle, entryLimit: number): Promise { - const listing = descriptorDirectoryPath(handle.fd); - if (listing === null) { +/** + * List the directory open on the last frame of `frames`. Linux lists the + * pinned inode through its descriptor. macOS re-checks the pinned chain, lists + * the lexical path, then re-checks the chain, so a path replaced around the + * listing fails the walk instead of being listed. + */ +async function listOpenedDirectory( + frames: readonly WalkFrame[], + entryLimit: number, + beforeList?: () => void | Promise, +): Promise { + const frame = frames[frames.length - 1]; + if (frame === undefined) { + throw Object.assign(new Error('no directory to list'), { code: 'EINVAL' }); + } + const anchored = descriptorDirectoryPath(frame.handle.fd); + if (anchored === null && process.platform !== 'darwin') { throw Object.assign(new Error('no descriptor listing'), { code: 'ENOTSUP' }); } + if (anchored === null) await assertPinnedChain(frames); + if (beforeList) await beforeList(); + const entries = await readDirectoryBounded(anchored ?? frame.absolute, entryLimit); + if (anchored === null) await assertPinnedChain(frames); + return entries; +} + +async function readDirectoryBounded(listing: string, entryLimit: number): Promise { const dir = await opendir(listing); const entries: Dirent[] = []; try { @@ -295,8 +323,16 @@ async function listOpenedDirectory(handle: FileHandle, entryLimit: number): Prom */ export async function readDirectoryNoFollow(directory: string): Promise { const opened = await openVerifiedDirectory(directory); + const frame: WalkFrame = { + relative: '', + absolute: directory, + handle: opened.handle, + identity: opened.identity, + entries: [], + next: 0, + }; try { - return await listOpenedDirectory(opened.handle, DART_PUBSPEC_DIRECTORY_ENTRY_LIMIT); + return await listOpenedDirectory([frame], DART_PUBSPEC_DIRECTORY_ENTRY_LIMIT); } finally { await opened.handle.close(); } @@ -345,7 +381,12 @@ export async function loadDartPackageConfig( const fillFrame = async (frame: WalkFrame): Promise => { if (++visited > directoryLimit) return incomplete('directory-limit', frame.relative || '.'); try { - frame.entries = await listOpenedDirectory(frame.handle, directoryEntryLimit); + const beforeList = options?.beforeDirectoryList; + frame.entries = await listOpenedDirectory( + stack, + directoryEntryLimit, + beforeList ? () => beforeList(frame.relative) : undefined, + ); } catch (error) { const reason = (error as NodeJS.ErrnoException).code === 'E2BIG' ? 'directory-entries' : 'read-directory'; diff --git a/gitnexus/test/integration/go-pipeline-benchmark.test.ts b/gitnexus/test/integration/go-pipeline-benchmark.test.ts index 095c1f330..04af7af9b 100644 --- a/gitnexus/test/integration/go-pipeline-benchmark.test.ts +++ b/gitnexus/test/integration/go-pipeline-benchmark.test.ts @@ -473,13 +473,10 @@ describe('Go scope-capture O(n^2) regression tripwire', () => { */ function generateSyntheticInterfaceData(interfaceCount: number, structCount: number): ParsedFile[] { const defs: SymbolDefinition[] = []; - const ifaceIds: string[] = []; - const structIds: string[] = []; // Interfaces for (let i = 0; i < interfaceCount; i++) { const ifaceId = `iface:Repo${i}`; - ifaceIds.push(ifaceId); defs.push({ nodeId: ifaceId, filePath: 'repo.go', @@ -515,37 +512,34 @@ function generateSyntheticInterfaceData(interfaceCount: number, structCount: num // Structs — each implements all interfaces for (let s = 0; s < structCount; s++) { const structId = `struct:Impl${s}`; - structIds.push(structId); defs.push({ nodeId: structId, filePath: 'repo.go', type: 'Struct', qualifiedName: `Impl${s}`, }); - for (let i = 0; i < interfaceCount; i++) { - defs.push({ - nodeId: `struct:Impl${s}.Repo${i}.Find`, - filePath: 'repo.go', - type: 'Method', - qualifiedName: `Impl${s}.Find`, - ownerId: structId, - parameterCount: 1, - requiredParameterCount: 1, - parameterTypes: ['string'], - returnType: 'User', - }); - defs.push({ - nodeId: `struct:Impl${s}.Repo${i}.Save`, - filePath: 'repo.go', - type: 'Method', - qualifiedName: `Impl${s}.Save`, - ownerId: structId, - parameterCount: 1, - requiredParameterCount: 1, - parameterTypes: ['User'], - returnType: 'error', - }); - } + defs.push({ + nodeId: `struct:Impl${s}.Find`, + filePath: 'repo.go', + type: 'Method', + qualifiedName: `Impl${s}.Find`, + ownerId: structId, + parameterCount: 1, + requiredParameterCount: 1, + parameterTypes: ['string'], + returnType: 'User', + }); + defs.push({ + nodeId: `struct:Impl${s}.Save`, + filePath: 'repo.go', + type: 'Method', + qualifiedName: `Impl${s}.Save`, + ownerId: structId, + parameterCount: 1, + requiredParameterCount: 1, + parameterTypes: ['User'], + returnType: 'error', + }); } // BadStructs — wrong Save signature (string instead of User), should NOT match @@ -557,31 +551,29 @@ function generateSyntheticInterfaceData(interfaceCount: number, structCount: num type: 'Struct', qualifiedName: `Bad${b}`, }); - for (let i = 0; i < interfaceCount; i++) { - defs.push({ - nodeId: `struct:Bad${b}.Repo${i}.Find`, - filePath: 'repo.go', - type: 'Method', - qualifiedName: `Bad${b}.Find`, - ownerId: badId, - parameterCount: 1, - requiredParameterCount: 1, - parameterTypes: ['string'], - returnType: 'User', - }); - // Mismatched Save: string param instead of User - defs.push({ - nodeId: `struct:Bad${b}.Repo${i}.Save`, - filePath: 'repo.go', - type: 'Method', - qualifiedName: `Bad${b}.Save`, - ownerId: badId, - parameterCount: 1, - requiredParameterCount: 1, - parameterTypes: ['string'], - returnType: 'error', - }); - } + defs.push({ + nodeId: `struct:Bad${b}.Find`, + filePath: 'repo.go', + type: 'Method', + qualifiedName: `Bad${b}.Find`, + ownerId: badId, + parameterCount: 1, + requiredParameterCount: 1, + parameterTypes: ['string'], + returnType: 'User', + }); + // Mismatched Save: string param instead of User + defs.push({ + nodeId: `struct:Bad${b}.Save`, + filePath: 'repo.go', + type: 'Method', + qualifiedName: `Bad${b}.Save`, + ownerId: badId, + parameterCount: 1, + requiredParameterCount: 1, + parameterTypes: ['string'], + returnType: 'error', + }); } return [ @@ -606,6 +598,15 @@ function generateSyntheticInterfaceData(interfaceCount: number, structCount: num * path (structIdsByMethodName intersection) keeps this well under budget. */ describe('Go structural interface detection O(n²) regression tripwire', () => { + it('uses legal Go method sets with one declaration per receiver and method name', () => { + const [parsed] = generateSyntheticInterfaceData(3, 3); + const methods = parsed.localDefs.filter((def) => def.type === 'Method'); + const identities = methods.map((def) => `${def.ownerId}:${def.qualifiedName}`); + expect(new Set(identities).size).toBe(methods.length); + // Three interfaces, three matching structs and three negative controls. + expect(methods).toHaveLength(18); + }); + it('detects implementations for 100 interfaces × 100 structs within budget', () => { const IFACE_COUNT = 100; const STRUCT_COUNT = 100; @@ -779,7 +780,7 @@ describe.skipIf(!BENCH_ENABLED)('Go structural interface detection split-phase b console.log('\nScaling analysis (total time ratio / pair-count ratio):'); // We can't split phases without exporting internals, but we can report // how total time scales relative to pair count. - // The detection loop is O(I×C×M×O_a) where O_a grows with I in this - // synthetic case, so pair-count ratio alone underestimates expected growth. + // Each receiver declares Find and Save once, as real Go requires. + // Required signatures stay constant while interface/struct pairs grow. }, 300_000); }); diff --git a/gitnexus/test/unit/calltool-dispatch.test.ts b/gitnexus/test/unit/calltool-dispatch.test.ts index df5895120..aa4bc9d82 100644 --- a/gitnexus/test/unit/calltool-dispatch.test.ts +++ b/gitnexus/test/unit/calltool-dispatch.test.ts @@ -575,14 +575,17 @@ describe('LocalBackend.callTool', () => { ['impact', { name: 'validate', symbol: 'login', direction: 'upstream' }], ['impact', { target: 'validate', direction: 'upstream', maxDepth: 3, depth: 1 }], ['context', { name: 'validate', file_path: 'src/auth.ts', file: 'src/login.ts' }], - ])('rejects conflicting %s aliases before repository resolution', async (method, params) => { - const resolveSpy = vi.spyOn(backend, 'selectToolRepository'); + ])( + 'rejects conflicting %s aliases before repository resolution (case %#)', + async (method, params) => { + const resolveSpy = vi.spyOn(backend, 'selectToolRepository'); - const result = await backend.callTool(method, params); + const result = await backend.callTool(method, params); - expect(result.error).toMatch(/conflicting mcp parameters/i); - expect(resolveSpy).not.toHaveBeenCalled(); - }); + expect(result.error).toMatch(/conflicting mcp parameters/i); + expect(resolveSpy).not.toHaveBeenCalled(); + }, + ); it.each([ ['impact', { name: 42, direction: 'upstream' }], diff --git a/gitnexus/test/unit/canonicalize-path-long-path-prefix.test.ts b/gitnexus/test/unit/canonicalize-path-long-path-prefix.test.ts index 0e4730a00..9783d9553 100644 --- a/gitnexus/test/unit/canonicalize-path-long-path-prefix.test.ts +++ b/gitnexus/test/unit/canonicalize-path-long-path-prefix.test.ts @@ -3,7 +3,7 @@ * * This is the Linux-runnable guard for the `canonicalizePath` wiring. The * companion assertions in `repo-manager.test.ts` exercise the real - * `realpathSync.native` and are therefore `it.skipIf(win32)`, so they only run on + * `realpathSync.native` and are therefore Windows-only, so they only run on * the windows-latest matrix leg — leaving the Ubuntu gate with no coverage of the * behaviour at all. This file closes that hole by injecting the two platform * primitives and nothing else: @@ -26,8 +26,8 @@ * and the realpath case passes, which is exactly the asymmetry the fix targets: * the realpath branch never leaked, because libuv strips the prefix itself. * - * Deliberately NOT registered in `scripts/cross-platform-tests.ts`: it simulates - * Windows rather than needing it, so its home is the Ubuntu suite. + * Registered in `scripts/cross-platform-tests.ts` as well: the injected Windows + * path rules must agree with the native Windows run. */ import { describe, it, expect, vi } from 'vitest'; @@ -139,13 +139,10 @@ describe('canonicalizePath vs the `\\\\?\\` long-path prefix (#2667)', () => { }); }); -// The guard in front of `fs.rm(recursive)` in remove.ts / clean.ts. It compares -// `path.resolve` forms on both sides and deliberately does NOT canonicalize, so a -// prefixed entry stays self-consistent while a mixed-form entry fails closed. -// Pinned here because "complete the fix by stripping here too" is the tempting -// follow-up refactor, and it would widen what the recursive delete accepts. +// The storage resolver now compares canonical paths for repository-local slots. +// Equivalent prefix spellings name the same .gitnexus directory; normalization +// must still reject the repository itself, its parents, and unowned external paths. describe('assertSafeStoragePath vs the `\\\\?\\` prefix (#2667)', () => { - const itOnWindows = process.platform === 'win32' ? it : it.skip; const base: Omit = { name: 'repo', path: '\\\\?\\D:\\Projects\\repo', @@ -153,16 +150,29 @@ describe('assertSafeStoragePath vs the `\\\\?\\` prefix (#2667)', () => { lastCommit: 'deadbee', }; - itOnWindows('accepts an entry whose path and storagePath share the prefix', async () => { + it('accepts an entry whose path and storagePath share the prefix', async () => { await expect( assertSafeStoragePath({ ...base, storagePath: '\\\\?\\D:\\Projects\\repo\\.gitnexus' }), ).resolves.toBeUndefined(); }); - itOnWindows('rejects a mixed-form entry instead of deleting through it', async () => { + it('accepts mixed prefix spellings of the same repository-local slot', async () => { await expect( assertSafeStoragePath({ ...base, storagePath: 'D:\\Projects\\repo\\.gitnexus' }), - ).rejects.toThrow(); + ).resolves.toBeUndefined(); + }); + + it.each([ + 'D:\\Projects\\repo', + '\\\\?\\D:\\Projects\\repo', + 'D:\\Projects', + 'D:\\', + 'D:\\Projects\\other\\.gitnexus', + '\\\\?\\D:\\Projects\\other\\.gitnexus', + ])('rejects an unsafe deletion target despite prefix normalization: %s', async (storagePath) => { + await expect(assertSafeStoragePath({ ...base, storagePath })).rejects.toThrow( + /Refusing to remove storage path for safety/, + ); }); }); diff --git a/gitnexus/test/unit/cross-platform-shard.test.ts b/gitnexus/test/unit/cross-platform-shard.test.ts index 66ea6a430..81722c941 100644 --- a/gitnexus/test/unit/cross-platform-shard.test.ts +++ b/gitnexus/test/unit/cross-platform-shard.test.ts @@ -13,6 +13,7 @@ * reproduce the outage exactly. */ +import { readFileSync } from 'node:fs'; import { describe, it, expect } from 'vitest'; import { shardFiles, @@ -22,7 +23,12 @@ import { } from '../../scripts/cross-platform-shard.js'; import { ALL_CROSS_PLATFORM } from '../../scripts/cross-platform-tests.js'; -const SHARD_TOTAL = 3; +// Replay the actual CI partition, including shard-count changes as the suite grows. +const workflow = readFileSync( + new URL('../../../.github/workflows/ci-tests.yml', import.meta.url), + 'utf8', +); +const SHARD_TOTAL = Number(workflow.match(/^\s+TOTAL=(\d+)\b/m)?.[1]); /** Every shard of a split, as file lists. */ const allShards = (files: readonly string[], total: number): readonly (readonly string[])[] => @@ -54,7 +60,7 @@ describe('cross-platform shard partition', () => { 'test/unit/incremental-index-extension-dml-gate.test.ts', ].map((file) => shards.findIndex((files) => files.includes(file))); expect(heavyweightLocations).not.toContain(-1); - expect(new Set(heavyweightLocations).size).toBe(SHARD_TOTAL); + expect(new Set(heavyweightLocations).size).toBe(heavyweightLocations.length); expect( shards.filter( (s) => diff --git a/gitnexus/test/unit/dart-package-imports.test.ts b/gitnexus/test/unit/dart-package-imports.test.ts index 0e33bc463..c127838d1 100644 --- a/gitnexus/test/unit/dart-package-imports.test.ts +++ b/gitnexus/test/unit/dart-package-imports.test.ts @@ -1,5 +1,5 @@ import { execFileSync } from 'node:child_process'; -import { afterEach, describe, expect, it } from 'vitest'; +import { afterEach, describe, expect, it, vi } from 'vitest'; import { constants } from 'node:fs'; import { mkdtemp, @@ -463,47 +463,52 @@ describe.skipIf(!pubspecWalkAnchored())('Dart pubspec package discovery', () => }, ); - it('does not follow a listed directory replaced by a symlink on macOS', async () => { - if (process.platform !== 'darwin') return; - const outside = await fixture({ - 'pubspec.yaml': 'name: foreign', - 'nested/pubspec.yaml': 'name: foreign', - }); - const root = await fixture({ - 'pkg/pubspec.yaml': 'name: data', - 'pkg/nested/pubspec.yaml': 'name: nested_data', - }); - const pkg = path.join(root, 'pkg'); - await expect( - loadDartPackageConfig(root, { - beforeEntryOpen: async (relative) => { - if (relative !== 'pkg') return; - await rename(pkg, path.join(root, 'pkg-moved')); - await symlink(outside, pkg, 'dir'); - }, - }), - ).rejects.toThrow(/Dart pubspec discovery failed \((read-directory|read-pubspec)\)/); - }); + it.skipIf(process.platform !== 'darwin')( + 'does not follow a listed directory replaced by a symlink on macOS', + async () => { + const outside = await fixture({ + 'pubspec.yaml': 'name: foreign', + 'nested/pubspec.yaml': 'name: foreign', + }); + const root = await fixture({ + 'pkg/pubspec.yaml': 'name: data', + 'pkg/nested/pubspec.yaml': 'name: nested_data', + }); + const pkg = path.join(root, 'pkg'); + await expect( + loadDartPackageConfig(root, { + beforeEntryOpen: async (relative) => { + if (relative !== 'pkg') return; + await rename(pkg, path.join(root, 'pkg-moved')); + await symlink(outside, pkg, 'dir'); + }, + }), + ).rejects.toThrow(/Dart pubspec discovery failed \((read-directory|read-pubspec)\)/); + }, + ); - it('reads listed manifests from the opened directory inode after that path is replaced', async () => { - if (descriptorEntryPath(0, 'pubspec.yaml') === null) return; - const outside = await fixture({ - 'pubspec.yaml': 'name: foreign', - 'nested/pubspec.yaml': 'name: foreign', - }); + // Linux lists the opened inode through /proc/self/fd/N. macOS cannot list a + // descriptor (opendir on /dev/fd/N is ENOTDIR), so it lists the path and + // re-checks the pinned chain afterwards; the swap must fail that check. + it('lists the opened directory, or refuses, when its path is replaced before listing', async () => { + const outside = await fixture({ 'pubspec.yaml': 'name: foreign' }); const root = await fixture({ 'pkg/pubspec.yaml': 'name: data', 'pkg/nested/pubspec.yaml': 'name: nested_data', }); const pkg = path.join(root, 'pkg'); - const loaded = await loadDartPackageConfig(root, { - beforeEntryOpen: async (relative) => { + const load = loadDartPackageConfig(root, { + beforeDirectoryList: async (relative) => { if (relative !== 'pkg') return; await rename(pkg, path.join(root, 'pkg-moved')); await symlink(outside, pkg, 'dir'); }, }); - expect(loaded.packages).toEqual( + if (process.platform === 'darwin') { + await expect(load).rejects.toThrow('Dart pubspec discovery failed (read-directory): pkg'); + return; + } + expect((await load).packages).toEqual( new Map([ ['data', 'pkg/lib'], ['nested_data', 'pkg/nested/lib'], @@ -511,26 +516,55 @@ describe.skipIf(!pubspecWalkAnchored())('Dart pubspec package discovery', () => ); }); - it('lists the opened directory inode after its path becomes a symlink', async () => { - const listingRoot = descriptorDirectoryPath(0); - if (listingRoot === null) return; - const outside = await fixture({ 'outside.txt': 'out' }); - const root = await fixture({}); - const real = path.join(root, 'real'); - await mkdir(real); - await writeFile(path.join(real, 'inside.txt'), 'in'); - const handle = await open(real, directoryOpenFlags()); - try { - const listed = descriptorDirectoryPath(handle.fd); - if (listed === null) throw new Error('descriptor listing path missing'); - await rename(real, path.join(root, 'real-moved')); - await symlink(outside, real, process.platform === 'win32' ? 'junction' : 'dir'); - await expect(readDirectoryNoFollow(real)).rejects.toThrow(); - expect(await readdir(listed)).toEqual(['inside.txt']); - } finally { - await handle.close(); - } - }); + it.skipIf(descriptorEntryPath(0, 'pubspec.yaml') === null)( + 'reads listed manifests from the opened directory inode after that path is replaced', + async () => { + const outside = await fixture({ + 'pubspec.yaml': 'name: foreign', + 'nested/pubspec.yaml': 'name: foreign', + }); + const root = await fixture({ + 'pkg/pubspec.yaml': 'name: data', + 'pkg/nested/pubspec.yaml': 'name: nested_data', + }); + const pkg = path.join(root, 'pkg'); + const loaded = await loadDartPackageConfig(root, { + beforeEntryOpen: async (relative) => { + if (relative !== 'pkg') return; + await rename(pkg, path.join(root, 'pkg-moved')); + await symlink(outside, pkg, 'dir'); + }, + }); + expect(loaded.packages).toEqual( + new Map([ + ['data', 'pkg/lib'], + ['nested_data', 'pkg/nested/lib'], + ]), + ); + }, + ); + + it.skipIf(descriptorDirectoryPath(0) === null)( + 'lists the opened directory inode after its path becomes a symlink', + async () => { + const outside = await fixture({ 'outside.txt': 'out' }); + const root = await fixture({}); + const real = path.join(root, 'real'); + await mkdir(real); + await writeFile(path.join(real, 'inside.txt'), 'in'); + const handle = await open(real, directoryOpenFlags()); + try { + const listed = descriptorDirectoryPath(handle.fd); + if (listed === null) throw new Error('descriptor listing path missing'); + await rename(real, path.join(root, 'real-moved')); + await symlink(outside, real, process.platform === 'win32' ? 'junction' : 'dir'); + await expect(readDirectoryNoFollow(real)).rejects.toThrow(); + expect(await readdir(listed)).toEqual(['inside.txt']); + } finally { + await handle.close(); + } + }, + ); }); describe('directory symlink refusal', () => { @@ -549,13 +583,18 @@ describe('directory symlink refusal', () => { await expect(readDirectoryNoFollow(link)).rejects.toThrow(); }); - it.skipIf(pubspecWalkAnchored())( - 'does not discover packages when the walk cannot honor no-follow', - async () => { - const root = await mkdtemp(path.join(os.tmpdir(), 'gitnexus-dart-pubspec-')); - roots.push(root); - await writeFile(path.join(root, 'pubspec.yaml'), 'name: app\n'); + it('does not discover packages when the walk cannot honor no-follow', async () => { + const root = await mkdtemp(path.join(os.tmpdir(), 'gitnexus-dart-pubspec-')); + roots.push(root); + await writeFile(path.join(root, 'pubspec.yaml'), 'name: app\n'); + // Exercise the unsupported-platform branch on every host. The real + // capability check must refuse before attempting any directory walk. + const platform = vi.spyOn(process, 'platform', 'get').mockReturnValue('win32'); + try { + expect(pubspecWalkAnchored()).toBe(false); expect((await loadDartPackageConfig(root)).packages.size).toBe(0); - }, - ); + } finally { + platform.mockRestore(); + } + }); }); diff --git a/gitnexus/test/unit/eval-server-bind-restriction.test.ts b/gitnexus/test/unit/eval-server-bind-restriction.test.ts index 0f96e301e..0e46a05cb 100644 --- a/gitnexus/test/unit/eval-server-bind-restriction.test.ts +++ b/gitnexus/test/unit/eval-server-bind-restriction.test.ts @@ -22,7 +22,7 @@ describe('isEvalServerBindRestriction', () => { ['Error: listen EADDRNOTAVAIL: address not available 192.168.1.99:0'], ['something then bind EACCES: permission denied'], ['LISTEN EACCES: permission denied'], - ])('matches %p', (stderr) => { + ])('matches %s', (stderr) => { expect(isEvalServerBindRestriction(stderr)).toBe(true); }); }); @@ -35,7 +35,7 @@ describe('isEvalServerBindRestriction', () => { ['GITNEXUS_EVAL_SERVER_READY:127.0.0.1:5173'], [''], ['unknown option --host'], - ])('does not match %p', (stderr) => { + ])('does not match %s', (stderr) => { expect(isEvalServerBindRestriction(stderr)).toBe(false); }); }); diff --git a/gitnexus/test/unit/evidence-provenance-helper.test.ts b/gitnexus/test/unit/evidence-provenance-helper.test.ts index 7159a38b8..84597e23b 100644 --- a/gitnexus/test/unit/evidence-provenance-helper.test.ts +++ b/gitnexus/test/unit/evidence-provenance-helper.test.ts @@ -953,20 +953,23 @@ SAFE_WRITE_FIXTURES('generated-plan safe writer', () => { ['--expected-plan-path', SAFE_PLAN_PATH], /--expected-plan-path and --expected-plan-digest require --replace/, ], - ] as const)('rejects command-inapplicable CLI options for %s', (command, extra, pattern) => { - const repo = createBaseRepo('gitnexus-plan-cli-options-'); - try { - const result = spawnSync( - process.execPath, - [PLAN_HELPER, command, '--repo', repo, '--generated-plan', SAFE_PLAN_PATH, ...extra], - { encoding: 'utf8', input: '# plan\n' }, - ); - expect(result.status).toBe(1); - expect(result.stderr).toMatch(pattern); - } finally { - fs.rmSync(repo, { recursive: true, force: true }); - } - }); + ] as const)( + 'rejects command-inapplicable CLI options for %s (case %#)', + (command, extra, pattern) => { + const repo = createBaseRepo('gitnexus-plan-cli-options-'); + try { + const result = spawnSync( + process.execPath, + [PLAN_HELPER, command, '--repo', repo, '--generated-plan', SAFE_PLAN_PATH, ...extra], + { encoding: 'utf8', input: '# plan\n' }, + ); + expect(result.status).toBe(1); + expect(result.stderr).toMatch(pattern); + } finally { + fs.rmSync(repo, { recursive: true, force: true }); + } + }, + ); it('never overwrites a destination created immediately before initial publication', async () => { const repo = createBaseRepo('gitnexus-plan-writer-'); diff --git a/gitnexus/test/unit/mcp-repository-policy.test.ts b/gitnexus/test/unit/mcp-repository-policy.test.ts index 3c55fcc19..1e4322837 100644 --- a/gitnexus/test/unit/mcp-repository-policy.test.ts +++ b/gitnexus/test/unit/mcp-repository-policy.test.ts @@ -283,7 +283,7 @@ describe('MCP repository policy', () => { ); it.each([{ GITNEXUS_MCP_ALLOWED_REPOS: ' ' }, { GITNEXUS_MCP_DEFAULT_REPO: ' ' }])( - 'fails closed for explicitly blank repository configuration', + 'fails closed for explicitly blank repository configuration (case %#)', async (env) => { await expect(createMcpRepositoryPolicy(createBackend(), env)).rejects.toThrow( /must not be blank/i, diff --git a/gitnexus/test/unit/repo-manager.test.ts b/gitnexus/test/unit/repo-manager.test.ts index fddafbe39..d168aef57 100644 --- a/gitnexus/test/unit/repo-manager.test.ts +++ b/gitnexus/test/unit/repo-manager.test.ts @@ -1959,7 +1959,7 @@ describe('assertSafeStoragePath (#1003)', () => { await expect(assertSafeStoragePath(entry)).rejects.toBeInstanceOf(UnsafeStoragePathError); }); - it.each([null, 42])('rejects malformed storagePath %p with the safety error', async (value) => { + it.each([null, 42])('rejects malformed storagePath %s with the safety error', async (value) => { const entry = { ...base, storagePath: value, diff --git a/gitnexus/test/unit/run-analyze-fts-repair.test.ts b/gitnexus/test/unit/run-analyze-fts-repair.test.ts index b6dda595c..007cbc391 100644 --- a/gitnexus/test/unit/run-analyze-fts-repair.test.ts +++ b/gitnexus/test/unit/run-analyze-fts-repair.test.ts @@ -1,6 +1,7 @@ import { execSync } from 'child_process'; import fs from 'fs/promises'; import { basename } from 'node:path'; +import { pathToFileURL } from 'node:url'; import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest'; import { getStoragePaths, @@ -2747,6 +2748,7 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () => vi.doUnmock('../../src/core/embeddings/embedding-identity.js'); vi.doUnmock('../../src/core/embeddings/embedding-pipeline.js'); vi.doUnmock('../../src/core/embeddings/staged-embedding-recovery.js'); + vi.doUnmock('../../src/core/analyzer-identity.js'); vi.restoreAllMocks(); vi.resetModules(); vi.clearAllMocks(); @@ -3404,30 +3406,58 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () => await fs.writeFile(`${storagePath}/${source.filename}`, source.contents); } await fs.writeFile(lbugPath, 'previous published index'); + // This unit fixture switches process.platform to exercise Windows + // recovery. Hash an immutable, test-owned analyzer runtime instead of + // the shared checkout under concurrent CI activity. Keep the real + // identity validation and finalization. + const identity = await import('../../src/core/analyzer-identity.js'); + const runtimeRoot = `${tmpRepo.dbPath}/analyzer-runtime`; + await fs.mkdir(`${runtimeRoot}/src`, { recursive: true }); + await fs.writeFile( + `${runtimeRoot}/package.json`, + '{"name":"recovery-fixture","version":"1.0.0"}', + ); + await fs.writeFile(`${runtimeRoot}/package-lock.json`, '{"lockfileVersion":3}'); + await fs.writeFile(`${runtimeRoot}/src/analyzer.ts`, 'export const analyzer = 1;'); + const runtimeUrl = pathToFileURL(`${runtimeRoot}/src/analyzer.ts`).href; + const identityOptions = { cacheDirectory: `${tmpRepo.dbPath}/identity-cache` }; + const resolveRunnerIdentity = vi.fn(() => + identity.resolveAnalyzerRunnerIdentity(runtimeUrl, identityOptions), + ); + vi.doMock('../../src/core/analyzer-identity.js', () => ({ + ...identity, + resolveAnalyzerRunnerIdentity: resolveRunnerIdentity, + finalizeAnalyzerRunnerIdentity: ( + _url: string, + startedWith: NonNullable, + ) => identity.finalizeAnalyzerRunnerIdentity(runtimeUrl, startedWith, identityOptions), + })); const { normalizeCachedEmbeddings } = await import('../../src/core/embeddings/embedding-restore-spill.js'); - vi.doMock( - '../../src/core/embeddings/staged-embedding-recovery.js', - async (importActual) => ({ - ...(await importActual< - typeof import('../../src/core/embeddings/staged-embedding-recovery.js') - >()), - recoverStagedEmbeddings: vi.fn(async () => - normalizeCachedEmbeddings({ - embeddings: [ - { - nodeId: RESILIENCE_NODE_ID, - chunkIndex: 0, - startLine: 1, - endLine: 2, - contentHash: 'current-hash', - embedding: new Array(EMBEDDING_DIMS).fill(0), - }, - ], - }), - ), + // Resolve the real merge implementation on the host, before selecting + // the Windows branch. A lazy importActual factory otherwise loads and + // transforms real modules while process.platform is temporarily win32. + const recovery = await vi.importActual< + typeof import('../../src/core/embeddings/staged-embedding-recovery.js') + >('../../src/core/embeddings/staged-embedding-recovery.js'); + const recoverStagedEmbeddings = vi.fn(async () => + normalizeCachedEmbeddings({ + embeddings: [ + { + nodeId: RESILIENCE_NODE_ID, + chunkIndex: 0, + startLine: 1, + endLine: 2, + contentHash: 'current-hash', + embedding: new Array(EMBEDDING_DIMS).fill(0), + }, + ], }), ); + vi.doMock('../../src/core/embeddings/staged-embedding-recovery.js', () => ({ + ...recovery, + recoverStagedEmbeddings, + })); const snapshots: Array = []; const rename = fs.rename.bind(fs); const { batchInsertEmbeddings } = mockResilienceHarness({ @@ -3463,16 +3493,25 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () => // branch. The native adapter is mocked; source files and lock cleanup are real. await import('../../src/core/run-analyze.js'); vi.stubEnv('GITNEXUS_ATOMIC_WINDOWS_SWAP', '0'); + const logs: string[] = []; Object.defineProperty(process, 'platform', { value: 'win32', configurable: true }); const error = await runAnalyze( tmpRepo.dbPath, { force: true, embeddings: true, skipAgentsMd: true, skipSkills: true }, - [], + logs, ); if (actualPlatformDescriptor) { Object.defineProperty(process, 'platform', actualPlatformDescriptor); } - expect(batchInsertEmbeddings).toHaveBeenCalled(); + expect(error, logs.join('\n')).toEqual( + outcome === 'success' ? null : expect.objectContaining({ message: outcome }), + ); + expect(resolveRunnerIdentity, logs.join('\n')).toHaveBeenCalledOnce(); + expect(recoverStagedEmbeddings, logs.join('\n')).toHaveBeenCalledOnce(); + expect(logs).toContain( + 'Recovered 1 complete staged embedding chunk(s) for 1 node(s); unchanged content can reuse them.', + ); + expect(batchInsertEmbeddings, logs.join('\n')).toHaveBeenCalled(); expect(vi.mocked(adapter.initLbug).mock.calls.at(-1)?.[0]).toBe(lbugPath); const finalMeta = await loadMeta(storagePath); // Reacquiring the real lock performs the next retry's orphan sweep. @@ -3481,7 +3520,6 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () => const lock = await acquireIndexLock(storagePath, { timeoutMs: 1000 }); try { if (outcome === 'success') { - expect(error).toBeNull(); expect(finalMeta?.embeddingCheckpoint).toBeUndefined(); expect(finalMeta?.stats?.embeddings).toBe(9); expect(await fs.readFile(lbugPath, 'utf8')).toBe('in-place replacement'); @@ -3491,7 +3529,6 @@ describe('runFullAnalysis embedding-checkpoint resilience (#2790 review)', () => }); } } else { - expect(error).toMatchObject({ message: outcome }); for (const source of sourceFiles) { expect(await fs.readFile(`${storagePath}/${source.filename}`, 'utf8')).toBe( source.contents, diff --git a/gitnexus/test/unit/scope-resolution/name-fallback-visibility.test.ts b/gitnexus/test/unit/scope-resolution/name-fallback-visibility.test.ts index 228fcead8..a02d40f2d 100644 --- a/gitnexus/test/unit/scope-resolution/name-fallback-visibility.test.ts +++ b/gitnexus/test/unit/scope-resolution/name-fallback-visibility.test.ts @@ -389,7 +389,7 @@ describe('Rust: isGlobalNameFallbackPlausible', () => { ['src/a/b.rs', 'super::unique_helper_xyz', 'src/a.rs'], ['src/a/b/c.rs', 'super::super::unique_helper_xyz', 'src/a/mod.rs'], ['src/a/b.rs', 'self::child::unique_helper_xyz', 'src/a/b/child.rs'], - ])('resolves relative imports from %s', (caller, target, candidate) => { + ])('resolves relative imports from %s via %s', (caller, target, candidate) => { expect( rustIsGlobalNameFallbackPlausible({ site: BARE_SITE, diff --git a/gitnexus/test/unit/scope-resolution/rust-cargo-targets.test.ts b/gitnexus/test/unit/scope-resolution/rust-cargo-targets.test.ts index 7458c5fcc..97ece0be7 100644 --- a/gitnexus/test/unit/scope-resolution/rust-cargo-targets.test.ts +++ b/gitnexus/test/unit/scope-resolution/rust-cargo-targets.test.ts @@ -150,7 +150,7 @@ describe('Cargo manifest target metadata', () => { `${PACKAGE}build=1\n`, `${PACKAGE}[[bin]]\npath="custom/entry.rs"\n`, `${PACKAGE}[[test]]\npath="custom/entry.rs"\n`, - ])('does not manufacture evidence from malformed metadata', (manifest) => { + ])('does not manufacture evidence from malformed metadata (case %#)', (manifest) => { expect(cargoTargetRoots('Cargo.toml', manifest, files)).toBeUndefined(); }); }); diff --git a/gitnexus/test/unit/spring-aop.test.ts b/gitnexus/test/unit/spring-aop.test.ts index 93e42436c..407ef86b6 100644 --- a/gitnexus/test/unit/spring-aop.test.ts +++ b/gitnexus/test/unit/spring-aop.test.ts @@ -326,7 +326,7 @@ describe('Spring AOP persisted reason contract (#2416)', () => { }, ]; - it.each(reasons)('round-trips the $kind reason', (reason) => { + it.each(reasons)('round-trips the $kind reason for $annotation', (reason) => { expect(decodeSpringAopReason(encodeSpringAopReason(reason))).toEqual(reason); }); diff --git a/gitnexus/test/unit/test-completeness.test.ts b/gitnexus/test/unit/test-completeness.test.ts new file mode 100644 index 000000000..93cf5466f --- /dev/null +++ b/gitnexus/test/unit/test-completeness.test.ts @@ -0,0 +1,284 @@ +import { describe, expect, it } from 'vitest'; +import { spawnSync } from 'node:child_process'; +import { copyFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import { fileURLToPath, pathToFileURL } from 'node:url'; +import { reconcileTestReports } from '../../scripts/test-completeness.js'; + +const report = (name: string, statuses: string[]) => ({ + success: !statuses.includes('failed'), + executionErrors: 0, + executionReason: 'passed', + testResults: [ + { + name, + status: statuses.includes('failed') ? 'failed' : 'passed', + assertionResults: statuses.map((status, i) => ({ + ancestorTitles: ['contract'], + title: `case ${i}`, + fullName: `contract case ${i}`, + status, + location: { line: i + 1, column: 1 }, + })), + }, + ], +}); + +describe('required test execution across CI jobs', () => { + it('fails when a collected test never executes, including todo', () => { + const result = reconcileTestReports([ + report('/home/runner/work/GitNexus/GitNexus/gitnexus/test/unit/a.test.ts', [ + 'passed', + 'pending', + 'todo', + ]), + ]); + expect(result.total).toBe(3); + expect(result.passed).toBe(1); + expect(result.unverified).toHaveLength(2); + }); + + it('accepts a platform skip only with an actual pass from another job', () => { + const result = reconcileTestReports([ + report('/home/runner/work/GitNexus/GitNexus/gitnexus/test/unit/a.test.ts', [ + 'passed', + 'pending', + ]), + report('D:\\a\\GitNexus\\GitNexus\\gitnexus\\test\\unit\\a.test.ts', ['pending', 'passed']), + ]); + expect(result.total).toBe(2); + expect(result.passed).toBe(2); + expect(result.unverified).toEqual([]); + expect(result.failures).toEqual([]); + }); + + it('never lets a passing job erase a failure on another platform', () => { + const result = reconcileTestReports([ + report('/repo/gitnexus/test/unit/a.test.ts', ['failed']), + report('/repo/gitnexus/test/unit/a.test.ts', ['passed']), + ]); + expect(result.failures.some((message) => message.includes('contract case 0'))).toBe(true); + }); + + it('counts failed tests separately from tests that never executed', () => { + const result = reconcileTestReports([ + report('/repo/gitnexus/test/unit/a.test.ts', ['passed', 'failed', 'pending']), + ]); + expect(result.report.numPassedTests).toBe(1); + expect(result.report.numFailedTests).toBe(1); + expect(result.report.numPendingTests).toBe(1); + expect(result.unverified).toHaveLength(1); + expect(result.report.success).toBe(false); + }); + + it('does not collapse identical titles from different files or repeated cases', () => { + const repeated = report('/repo/gitnexus/test/unit/a.test.ts', ['passed', 'pending']); + repeated.testResults[0].assertionResults[1].title = 'case 0'; + repeated.testResults[0].assertionResults[1].fullName = 'contract case 0'; + const result = reconcileTestReports([ + repeated, + report('/repo/gitnexus/test/unit/b.test.ts', ['passed']), + ]); + expect(result.total).toBe(3); + expect(result.unverified).toHaveLength(1); + }); + + it('matches source locations when repeated titles are collected in a different order', () => { + const first = report('/repo/gitnexus/test/unit/a.test.ts', ['pending', 'passed']); + first.testResults[0].assertionResults[1].title = 'case 0'; + first.testResults[0].assertionResults[1].fullName = 'contract case 0'; + const second = structuredClone(first); + second.testResults[0].assertionResults.reverse(); + const result = reconcileTestReports([first, second]); + expect(result.total).toBe(2); + expect(result.unverified).toHaveLength(1); + }); + + it('rejects ambiguous parameterized cases with the same name and location', () => { + const input = report('/repo/gitnexus/test/unit/a.test.ts', ['passed']); + input.testResults[0].assertionResults.push({ ...input.testResults[0].assertionResults[0] }); + expect(() => reconcileTestReports([input])).toThrow(/ambiguous/i); + }); + + it('requires an unambiguous title and a real pass for helper-generated tests without locations', () => { + const input = report('/repo/gitnexus/test/unit/a.test.ts', ['pending']); + Reflect.deleteProperty(input.testResults[0].assertionResults[0], 'location'); + expect(reconcileTestReports([input]).unverified).toHaveLength(1); + const passing = structuredClone(input); + passing.testResults[0].assertionResults[0].status = 'passed'; + expect(reconcileTestReports([input, passing]).report.success).toBe(true); + passing.testResults[0].assertionResults.push({ ...passing.testResults[0].assertionResults[0] }); + expect(() => reconcileTestReports([passing])).toThrow(/ambiguous/i); + }); + + it('reconciles source-adjacent web tests on Linux and Windows', () => { + const result = reconcileTestReports( + [ + report('/repo/gitnexus-web/src/lib/upload-filter.test.ts', ['passed']), + report('C:/repo/gitnexus-web/src/lib/upload-filter.test.ts', ['passed']), + ], + ['src/lib/upload-filter.test.ts'], + 'gitnexus-web', + ); + expect(result.report.success).toBe(true); + expect(result.total).toBe(1); + expect(() => + reconcileTestReports( + [report('/repo/gitnexus-web/test/unit/a.test.ts', ['passed'])], + ['test/unit/a.test.ts', 'src/lib/upload-filter.test.ts'], + 'gitnexus-web', + ), + ).toThrow(/never collected/); + }); + + it('requires every shard receipt and the full web inventory through the CLI', () => { + const temp = mkdtempSync(path.join(tmpdir(), 'gitnexus-completeness-cli-')); + const packageRoot = fileURLToPath(new URL('../../', import.meta.url)); + const root = path.join(temp, 'gitnexus'); + const webRoot = path.join(temp, 'gitnexus-web'); + const receipts = path.join(temp, 'receipts'); + const output = path.join(temp, 'combined.json'); + for (const directory of [ + path.join(root, 'scripts'), + path.join(root, 'test/unit'), + path.join(webRoot, 'src/lib'), + receipts, + ]) { + mkdirSync(directory, { recursive: true }); + } + const script = path.join(root, 'scripts/test-completeness.ts'); + copyFileSync(path.join(packageRoot, 'scripts/test-completeness.ts'), script); + writeFileSync(path.join(temp, 'package.json'), '{"type":"module"}'); + writeFileSync(path.join(root, 'test/unit/a.test.ts'), '// inventory fixture'); + writeFileSync(path.join(webRoot, 'src/lib/a.test.ts'), '// inventory fixture'); + const core = report(path.join(root, 'test/unit/a.test.ts'), ['passed']); + const web = report(path.join(webRoot, 'src/lib/a.test.ts'), ['passed']); + const webReport = path.join(webRoot, 'web-test-results.json'); + for (const name of [ + 'coverage', + 'benchmarks', + 'preflight', + 'windows-latest-1', + 'windows-latest-2', + 'macos-latest-1', + 'macos-latest-2', + ]) { + writeFileSync(path.join(receipts, `${name}.json`), JSON.stringify(core)); + } + const run = () => { + writeFileSync(webReport, JSON.stringify(web)); + return spawnSync(process.execPath, ['--import', 'tsx', script, receipts, output, '2'], { + cwd: packageRoot, + encoding: 'utf8', + timeout: 30_000, + }); + }; + try { + const complete = run(); + expect(complete.error, complete.stderr).toBeUndefined(); + expect(complete.status, complete.stderr).toBe(0); + expect(JSON.parse(readFileSync(output, 'utf8')).numPendingTests).toBe(0); + writeFileSync(path.join(webRoot, 'src/lib/missing.test.ts'), '// missing receipt'); + const missingSource = run(); + expect(missingSource.status).not.toBe(0); + expect(missingSource.stderr).toContain( + 'Test file was never collected: src/lib/missing.test.ts', + ); + rmSync(path.join(webRoot, 'src/lib/missing.test.ts')); + rmSync(path.join(receipts, 'windows-latest-2.json')); + const missingShard = run(); + expect(missingShard.status).not.toBe(0); + expect(missingShard.stderr).toContain('windows-latest-2.json'); + } finally { + rmSync(temp, { recursive: true, force: true }); + } + }); + + it('rejects test files missing from the collected inventory', () => { + expect(() => + reconcileTestReports( + [report('/repo/gitnexus/test/unit/a.test.ts', ['passed'])], + ['test/unit/a.test.ts', 'test/unit/b.test.ts'], + ), + ).toThrow(/b.test.ts/); + }); + + it('rejects missing, empty or malformed reports instead of reporting zero skips', () => { + expect(() => reconcileTestReports([])).toThrow(); + expect(() => reconcileTestReports([{ testResults: [] }])).toThrow(); + expect(() => + reconcileTestReports([{ testResults: [{ name: 'a', assertionResults: [] }] }]), + ).toThrow(); + expect(() => + reconcileTestReports([report('/repo/gitnexus/test/unit/a.test.ts', ['unknown'])]), + ).toThrow(); + }); + + it('keeps failed collection and unhandled runner errors fatal', () => { + const broken = report('/repo/gitnexus/test/unit/a.test.ts', ['passed']); + broken.success = false; + expect(reconcileTestReports([broken]).failures).not.toHaveLength(0); + broken.success = true; + broken.executionErrors = 1; + expect(reconcileTestReports([broken]).failures).not.toHaveLength(0); + broken.executionErrors = 0; + broken.testResults[0].status = 'failed'; + expect(reconcileTestReports([broken]).failures).not.toHaveLength(0); + expect(reconcileTestReports([broken]).report.numFailedTestSuites).toBe(1); + expect(reconcileTestReports([broken]).report.testResults[0].status).toBe('failed'); + }); + + it('captures a real unhandled rejection even when Vitest exits successfully', () => { + const temp = mkdtempSync(path.join(tmpdir(), 'gitnexus-receipt-')); + const packageRoot = fileURLToPath(new URL('../../', import.meta.url)); + const root = path.join(temp, 'gitnexus'); + const output = path.join(temp, 'receipt.json'); + mkdirSync(path.join(root, 'test/unit'), { recursive: true }); + const vitestImport = pathToFileURL( + path.join(packageRoot, 'node_modules/vitest/dist/index.js'), + ).href; + writeFileSync( + path.join(root, 'test/unit/rejection.test.ts'), + ` +import { it, expect } from ${JSON.stringify(vitestImport)}; +it('executes its assertion but leaks a rejection', async () => { + expect(2 + 2).toBe(4); + setTimeout(() => { void Promise.reject(new Error('receipt rejection probe')); }, 0); + await new Promise(resolve => setTimeout(resolve, 50)); +}); +`, + ); + const config = path.join(root, 'vitest.config.mjs'); + writeFileSync( + config, + 'export default { test: { include: ["test/**/*.test.ts"], includeTaskLocation: true, dangerouslyIgnoreUnhandledErrors: true } };', + ); + try { + const child = spawnSync( + process.execPath, + [ + path.join(packageRoot, 'node_modules/vitest/vitest.mjs'), + 'run', + '--root', + root, + '--config', + config, + '--reporter', + path.join(packageRoot, 'scripts/execution-reporter.ts'), + '--outputFile', + output, + ], + { cwd: packageRoot, encoding: 'utf8', timeout: 30_000 }, + ); + expect(child.error, child.stderr).toBeUndefined(); + expect(child.status, child.stderr).toBe(0); + const receipt = JSON.parse(readFileSync(output, 'utf8')); + expect(receipt.numPassedTests).toBe(1); + expect(receipt.executionErrors).toBeGreaterThan(0); + expect(reconcileTestReports([receipt]).report.success).toBe(false); + } finally { + rmSync(temp, { recursive: true, force: true }); + } + }); +}); diff --git a/gitnexus/test/unit/type-env.test.ts b/gitnexus/test/unit/type-env.test.ts index 62830ccac..a472ce12e 100644 --- a/gitnexus/test/unit/type-env.test.ts +++ b/gitnexus/test/unit/type-env.test.ts @@ -4519,11 +4519,10 @@ void processRepoMap(std::map repoMap) { }); }); - describe('known limitations (documented skip tests)', () => { - it.skip('Ruby block parameter: users.each { |user| } — closure param inference, different feature', () => { - // Not a for-loop; .each { |user| } is a method call with a block. - // Requires closure parameter inference — a different feature category - // applicable to Ruby, Swift closures, Kotlin lambdas, and Java lambdas. + describe('unknown Ruby block parameter types', () => { + it('does not invent a type for a block parameter from an untyped receiver', () => { + // Neither the parameter name nor calling `save` proves users contains + // User instances. Inferring User here would introduce false call edges. const tree = parse( ` def process(users) @@ -4533,7 +4532,7 @@ end Ruby, ); const typeEnv = buildTypeEnv(tree, SupportedLanguages.Ruby); - expect(flatGet(typeEnv, 'user')).toBe('User'); + expect(flatGet(typeEnv, 'user')).toBeUndefined(); }); }); diff --git a/gitnexus/vitest.config.ts b/gitnexus/vitest.config.ts index 2b1d43c38..73da0755f 100644 --- a/gitnexus/vitest.config.ts +++ b/gitnexus/vitest.config.ts @@ -8,6 +8,8 @@ export default defineConfig({ hookTimeout: 120000, pool: 'forks', globals: true, + // Stable test identity across the Linux/macOS/Windows execution receipts. + includeTaskLocation: true, teardownTimeout: 3000, // E2E harnesses pin a small NODE_OPTIONS heap so spawned CLI children // stay light; without this opt-out the #2649 auto-heap override would