diff --git a/.github/workflows/ci-report.yml b/.github/workflows/ci-report.yml index 03908bb2b..03c933ad0 100644 --- a/.github/workflows/ci-report.yml +++ b/.github/workflows/ci-report.yml @@ -95,31 +95,33 @@ jobs: # Validate PR number is a positive integer (artifact comes from # untrusted fork code, so treat contents defensively). - PR_NUM=$(cat "$DIR/pr_number" | tr -d '[:space:]') + PR_NUM=$(tr -d '[:space:]' < "$DIR/pr_number") if ! [[ "$PR_NUM" =~ ^[0-9]+$ ]]; then echo "skip=true" >> "$GITHUB_OUTPUT" echo "::error::Invalid PR number in artifact: '$PR_NUM'" exit 0 fi - echo "skip=false" >> "$GITHUB_OUTPUT" - echo "pr_number=$PR_NUM" >> "$GITHUB_OUTPUT" # Validate job-result strings against known GitHub Actions values. # Artifact contents come from the PR workflow (potentially untrusted # fork code), so we whitelist to prevent newline injection into # GITHUB_OUTPUT. validate_result() { local val - val=$(cat "$1" | tr -d '[:space:]') + val=$(tr -d '[:space:]' < "$1") case "$val" in success|failure|cancelled|skipped) echo "$val" ;; *) echo "unknown" ;; esac } - echo "quality=$(validate_result "$DIR/quality_result")" >> "$GITHUB_OUTPUT" - echo "tests=$(validate_result "$DIR/tests_result")" >> "$GITHUB_OUTPUT" - echo "e2e=$(validate_result "$DIR/e2e_result")" >> "$GITHUB_OUTPUT" + { + echo "skip=false" + echo "pr_number=$PR_NUM" + echo "quality=$(validate_result "$DIR/quality_result")" + echo "tests=$(validate_result "$DIR/tests_result")" + echo "e2e=$(validate_result "$DIR/e2e_result")" + } >> "$GITHUB_OUTPUT" - name: Checkout (for vitest config) if: steps.meta.outputs.skip != 'true' @@ -279,14 +281,17 @@ jobs: fi } - read CLI_T CLI_P CLI_F CLI_S CLI_SU CLI_D <<< "$(sum_results "$RESULTS_FILE")" - read WEB_T WEB_P WEB_F WEB_S WEB_SU WEB_D <<< "$(sum_results "$WEB_RESULTS_FILE")" + # `_` placeholder for the suite-count column — positional + # readability for sum_results' 6-field output, but the value + # isn't surfaced in the report (suites are tracked per-test + # framework, not as a top-line metric). + read -r CLI_T CLI_P CLI_F CLI_S _ CLI_D <<< "$(sum_results "$RESULTS_FILE")" + read -r WEB_T WEB_P WEB_F WEB_S _ WEB_D <<< "$(sum_results "$WEB_RESULTS_FILE")" TOTAL=$((CLI_T + WEB_T)) PASSED=$((CLI_P + WEB_P)) FAILED=$((CLI_F + WEB_F)) SKIPPED=$((CLI_S + WEB_S)) - SUITES=$((CLI_SU + WEB_SU)) DURATION=$((CLI_D > WEB_D ? CLI_D : WEB_D)) # ── Status helpers ── diff --git a/.github/workflows/pr-autofix-publish.yml b/.github/workflows/pr-autofix-publish.yml new file mode 100644 index 000000000..a22f2cc7a --- /dev/null +++ b/.github/workflows/pr-autofix-publish.yml @@ -0,0 +1,314 @@ +name: PR Autofix (publish) + +# TRUSTED HALF of the autofix pipeline. +# +# Triggered by `pr-autofix.yml` completing on a PR (including fork PRs). +# Downloads the diff artifact produced by the untrusted job and posts +# inline review-comment suggestions to the PR using `reviewdog`. This +# job NEVER checks out fork code — it only consumes the diff (data) and +# calls the GitHub API. That isolation is what makes it safe to run +# under `pull-requests: write` on fork-triggered events. +# +# Also posts (or edits) a single sticky summary comment so contributors +# and AI agents have one stable, machine-readable signal that says +# whether autofix had anything to suggest. Look for the heading +# "## :sparkles: PR Autofix" in the PR's top-level comments. +# +# Reviewdog reporter: `github-pr-review` reads $REVIEWDOG_GITHUB_API_TOKEN +# and posts via the GraphQL/REST PR-review API. It does not need a +# checkout because the diff itself encodes file paths + line numbers. + +on: + workflow_run: + workflows: ['PR Autofix'] + types: [completed] + +concurrency: + # Key on PR identity, NOT workflow_run.id — workflow_run.id is per-run + # unique, which would defeat serialization and let two parallel + # publishes both POST a sticky summary comment. CONTRIBUTING.md + # § GitHub Actions — Concurrency Convention names this anti-pattern + # explicitly. For fork PRs, `pull_requests[]` is empty in the + # workflow_run payload, so we fall back to head-repo + head-branch. + group: ${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }} + cancel-in-progress: false + +permissions: {} + +jobs: + publish: + name: publish-autofix + if: >- + github.event.workflow_run.event == 'pull_request' + && github.event.workflow_run.conclusion == 'success' + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + pull-requests: write + # Required by actions/download-artifact to fetch artifacts produced + # by a different workflow run. + actions: read + # Required to create the `gitnexus/autofix` Check Run that reports + # the outcome (clean / suggestions-posted / skipped-too-large) to + # the PR's Checks tab. Branch protection or agents can grep the + # conclusion + output title without parsing the sticky comment. + checks: write + steps: + # Pinned to v8.0.1. Verify SHA via: + # gh api repos/actions/download-artifact/git/refs/tags/v8.0.1 + - name: Download autofix artifact + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: autofix + run-id: ${{ github.event.workflow_run.id }} + github-token: ${{ secrets.GITHUB_TOKEN }} + path: autofix-in + + - name: Read and validate metadata + id: meta + shell: bash + run: | + set -euo pipefail + test -f autofix-in/metadata.json + jq . autofix-in/metadata.json + + # The artifact comes from the untrusted half running fork code. + # Every field is allowlist-validated before it can flow into + # $GITHUB_OUTPUT. A newline in head_ref would otherwise let a + # malicious branch name inject a second `pr_number=N` line and + # redirect this job's reviewdog suggestions / sticky summary + # comment onto a victim PR under github-actions[bot] with + # pull-requests: write. + assert_field() { + local key="$1" pattern="$2" value + value=$(jq -r ".${key} // empty" autofix-in/metadata.json) + if [ -z "$value" ] || ! [[ "$value" =~ $pattern ]]; then + echo "::error::metadata.${key} failed allowlist (got: $(printf '%q' "$value"))" + exit 1 + fi + printf '%s' "$value" + } + + SCHEMA=$(assert_field schema '^gitnexus\.pr-autofix/v[0-9]+$') + PR_NUMBER=$(assert_field pr_number '^[0-9]+$') + HEAD_SHA=$(assert_field head_sha '^[0-9a-f]{40}$') + HEAD_REF=$(assert_field head_ref '^[A-Za-z0-9._/-]+$') + HEAD_REPO=$(assert_field head_repo '^[A-Za-z0-9._-]+/[A-Za-z0-9._-]+$') + BASE_REPO=$(assert_field base_repo '^[A-Za-z0-9._-]+/[A-Za-z0-9._-]+$') + CHANGED=$(assert_field changed_lines '^[0-9]+$') + + # Defence-in-depth: refuse to act if the artifact claims to + # belong to a different repo than the one that triggered us. + if [ "$BASE_REPO" != "${GITHUB_REPOSITORY}" ]; then + echo "::error::Artifact base_repo does not match \$GITHUB_REPOSITORY — refusing to publish." + exit 1 + fi + + { + echo "schema=${SCHEMA}" + echo "pr_number=${PR_NUMBER}" + echo "head_sha=${HEAD_SHA}" + echo "head_ref=${HEAD_REF}" + echo "head_repo=${HEAD_REPO}" + echo "base_repo=${BASE_REPO}" + echo "changed_lines=${CHANGED}" + } >> "$GITHUB_OUTPUT" + + # Pinned to v1.5.0. Verify SHA via: + # gh api repos/reviewdog/action-setup/git/refs/tags/v1.5.0 + # (annotated tag — resolve via .../git/tags/ --jq .object) + - name: Install reviewdog + if: steps.meta.outputs.changed_lines != '0' + uses: reviewdog/action-setup@d8a7baabd7f3e8544ee4dbde3ee41d0011c3a93f # v1.5.0 + with: + # Pin the binary, not just the action SHA — a bad reviewdog + # release otherwise breaks every PR with no rollback. Bump + # this knob deliberately when validating a new release. + reviewdog_version: v0.21.0 + + - name: Post inline suggestions + id: suggest + if: steps.meta.outputs.changed_lines != '0' + env: + REVIEWDOG_GITHUB_API_TOKEN: ${{ secrets.GITHUB_TOKEN }} + CI_REPO_OWNER: ${{ github.repository_owner }} + CI_REPO_NAME: ${{ github.event.repository.name }} + CI_PULL_REQUEST: ${{ steps.meta.outputs.pr_number }} + CI_COMMIT: ${{ steps.meta.outputs.head_sha }} + # Pull `changed_lines` through env so bash gets a real + # variable (and shellcheck SC2170 doesn't fire on `-gt` against + # a `${{ }}`-interpolated literal). + CHANGED_LINES: ${{ steps.meta.outputs.changed_lines }} + shell: bash + run: | + set -euo pipefail + patch=autofix-in/autofix.patch + if [ ! -s "$patch" ]; then + echo "Empty patch — nothing to suggest." + echo "posted=false" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # GitHub's review-comment API returns 406 on diffs above ~3k + # changed lines. Bail out gracefully and let the summary + # comment carry the signal instead. + if [ "$CHANGED_LINES" -gt 3000 ]; then + echo "Diff too large ($CHANGED_LINES lines) — skipping inline suggestions." + echo "posted=skipped-too-large" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # `-f.diff.strip=1` matches `git diff` output (a/foo b/foo). + # `-filter-mode=added` only suggests on lines the PR added, + # which avoids re-suggesting on already-resolved threads when + # the contributor re-adds the autoformat label. + reviewdog \ + -f=diff -f.diff.strip=1 \ + -name="prettier+eslint" \ + -reporter=github-pr-review \ + -filter-mode=added \ + -level=warning \ + -fail-on-error=false < "$patch" + + echo "posted=true" >> "$GITHUB_OUTPUT" + + - name: Upsert sticky summary comment + # Only post when ci-quality found something fixable (= the + # autofix patch is non-empty). When prettier/eslint are clean + # the patch is zero bytes and the sticky comment is pure noise, + # so we skip it. When the diff was too large for inline + # suggestions, the sticky is the only signal the contributor + # gets, so we still post in that case. + if: >- + always() + && steps.meta.outputs.pr_number != '' + && steps.meta.outputs.changed_lines != '0' + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GH_REPO: ${{ github.repository }} + PR: ${{ steps.meta.outputs.pr_number }} + CHANGED: ${{ steps.meta.outputs.changed_lines }} + HEAD_SHA: ${{ steps.meta.outputs.head_sha }} + SCHEMA: ${{ steps.meta.outputs.schema }} + POSTED: ${{ steps.suggest.outputs.posted }} + RUN_ID: ${{ github.run_id }} + shell: bash + run: | + set -euo pipefail + + # Stable heading + marker — agents grep for these exact strings. + marker="" + heading="## :sparkles: PR Autofix" + + if [ "${POSTED}" = "skipped-too-large" ]; then + ui_state="skipped-too-large" + prose="Diff is **${CHANGED}** lines — too large for inline suggestions (GitHub caps the review-comment API at ~3000). Run locally: \`npm run lint:fix && npm run format\`." + else + ui_state="suggestions-posted" + prose="Posted formatting / unused-import suggestions inline. Click **Apply suggestion** on each, or run locally: \`npm run lint:fix && npm run format\`." + fi + + # Machine-readable JSON block — agents parse this instead of + # regexing English. Fenced code-block info string is + # `gitnexus-autofix` so agents can locate it without ambiguity. + json=$(jq -n -c \ + --arg schema "${SCHEMA}" \ + --arg state "${ui_state}" \ + --argjson pr_number "${PR}" \ + --argjson changed_lines "${CHANGED}" \ + --arg head_sha "${HEAD_SHA}" \ + --arg run_id "${RUN_ID}" \ + '{schema:$schema, state:$state, pr_number:$pr_number, changed_lines:$changed_lines, head_sha:$head_sha, run_id:$run_id}') + + # Multi-line quoted string instead of a column-0 heredoc — YAML's + # `run: |` block ends as soon as a content line dedents below the + # block's first-line indent, which would mis-parse the workflow. + body="${marker} + ${heading} + + ${prose} + + \`\`\`gitnexus-autofix + ${json} + \`\`\`" + # Strip the leading 10-space indent that the YAML block requires + # so the rendered comment body starts at column 0. + body="$(printf '%s\n' "$body" | sed 's/^ //')" + + # Small retry wrapper for transient 5xx / rate-limit responses + # on the GitHub REST API. Three tries with linear backoff. We + # only retry GET (idempotent) and PATCH on a known comment id + # (idempotent). POST is NOT wrapped — retrying a comment-create + # would create duplicates if the first attempt actually landed. + gh_retry() { + local n=0 max=3 + while true; do + if gh "$@"; then return 0; fi + n=$((n+1)) + if [ "$n" -ge "$max" ]; then return 1; fi + sleep $((n * 2)) + done + } + + # Find existing bot comment by the marker and edit-in-place; else create. + # CRITICAL: filter by `.user.login == "github-actions[bot]"`. A regular + # user posting a comment containing the marker would otherwise be the + # `head -n1` match; PATCH on someone else's comment 403s, `set -e` + # aborts, and the bot is permanently DoS'd for that PR. + existing=$(gh_retry api "repos/${GH_REPO}/issues/${PR}/comments" \ + --paginate --jq ".[] | select(.user.login == \"github-actions[bot]\" and (.body | contains(\"${marker}\"))) | .id" \ + | head -n1 || true) + + if [ -n "${existing}" ]; then + gh_retry api -X PATCH "repos/${GH_REPO}/issues/comments/${existing}" \ + -f body="${body}" >/dev/null + echo "Updated comment ${existing}." + else + gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \ + -f body="${body}" >/dev/null + echo "Created summary comment." + fi + + - name: Emit gitnexus/autofix Check Run + # Stable check name `gitnexus/autofix` so PR-watching agents can + # `gh pr checks ` and read the conclusion + title without + # parsing the sticky comment. Three outcomes: + # clean → conclusion: success + # suggestions-posted → conclusion: neutral (review suggestions) + # skipped-too-large → conclusion: neutral (diff > 3000 lines) + # `neutral` does not block branch-protection required-checks but + # is visually distinct from a green pass. + if: always() && steps.meta.outputs.head_sha != '' + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GH_REPO: ${{ github.repository }} + HEAD_SHA: ${{ steps.meta.outputs.head_sha }} + CHANGED: ${{ steps.meta.outputs.changed_lines }} + POSTED: ${{ steps.suggest.outputs.posted }} + shell: bash + run: | + set -euo pipefail + + if [ "${CHANGED}" = "0" ]; then + conclusion="success" + title="Formatting clean" + summary="Prettier and ESLint --fix produced no changes." + elif [ "${POSTED}" = "skipped-too-large" ]; then + conclusion="neutral" + title="Diff too large for inline suggestions (${CHANGED} lines)" + summary="GitHub caps the review-comment API at ~3000 lines. Run \`npm run lint:fix && npm run format\` locally." + else + conclusion="neutral" + title="Suggestions posted" + summary="Inline review-comment suggestions posted. Click **Apply suggestion** on each, or run \`npm run lint:fix && npm run format\` locally." + fi + + gh api -X POST "repos/${GH_REPO}/check-runs" \ + -f name="gitnexus/autofix" \ + -f head_sha="${HEAD_SHA}" \ + -f status="completed" \ + -f conclusion="${conclusion}" \ + -f "output[title]=${title}" \ + -f "output[summary]=${summary}" \ + >/dev/null + echo "Posted check-run gitnexus/autofix=${conclusion} (${title})" diff --git a/.github/workflows/pr-autofix.yml b/.github/workflows/pr-autofix.yml new file mode 100644 index 000000000..f15e8c0e6 --- /dev/null +++ b/.github/workflows/pr-autofix.yml @@ -0,0 +1,146 @@ +name: PR Autofix + +# UNTRUSTED HALF of the autofix pipeline. +# +# Runs `npm run lint:fix` + `npm run format` against the PR head +# (including fork heads) and uploads the resulting diff as an artifact. +# This job has NO privileged token and CANNOT post to the PR. The trusted +# `pr-autofix-publish.yml` workflow downloads the artifact via +# `workflow_run` and posts the inline review-comment suggestions. +# +# Why the split: +# ESLint loads plugins from fork-controlled `node_modules`, so running +# it in a job with `pull-requests: write` would let a malicious fork PR +# ship a poisoned eslint plugin and execute arbitrary code under that +# token. By keeping fork code execution in this job (token: read-only) +# and posting from a separate trusted job that never touches fork +# code, we get the inline-suggestion UX for fork PRs without the +# supply-chain hole. (See autofix.ci for the same pattern.) +# +# Removes unused imports via `eslint-plugin-unused-imports`, already in +# devDependencies and wired into the `lint` config. + +on: + pull_request: + types: [opened, synchronize, reopened] + # Skip lockfile / generated-file PRs entirely — `action-suggester` + # cannot post on diffs > ~3k lines (GitHub returns 406) and these + # paths produce massive diffs no human wants suggested back inline. + paths-ignore: + - '**/package-lock.json' + - '**/*.snap' + - '**/dist/**' + - '**/node_modules/**' + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number }} + # Don't cancel in-flight runs; the publish workflow may already be + # downloading the artifact and a cancelled untrusted run produces no + # signal at all (worse DX than waiting). + cancel-in-progress: false + +# This workflow runs untrusted fork code. Top-level deny-all and NO +# job-level grants — the job can only read its own checkout. +permissions: {} + +jobs: + autofix: + name: autofix + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + # PR head commit (not the synthetic merge ref) — we need the + # exact tree the contributor pushed so suggestions line up. + ref: ${{ github.event.pull_request.head.sha }} + repository: ${{ github.event.pull_request.head.repo.full_name }} + persist-credentials: false + + - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 + with: + node-version: 20 + cache: npm + cache-dependency-path: package-lock.json + + # `--ignore-scripts` blocks pre/postinstall lifecycle hooks. ESLint + # plugins still load from node_modules (that is the actual escape + # hatch on a typical fork), but this job has no token to abuse — + # which is the whole point of the split. + - run: npm ci --ignore-scripts + + - name: ESLint --fix (removes unused imports) + run: npm run lint:fix + # Lint errors that --fix can't auto-resolve must not block the + # diff artifact — partial fixes are still useful as suggestions. + continue-on-error: true + + - name: Prettier --write + run: npm run format + continue-on-error: true + + - name: Capture diff and metadata + id: capture + # Pass GitHub-context values via env: rather than `${{ }}` + # interpolated directly into the bash body. `head.ref` and + # `head.repo.full_name` are fork-controlled strings; expanding + # them into shell source is the canonical template-injection + # vector zizmor flags. Even though this job has `permissions: {}`, + # routing through env: makes it impossible for a future scope + # grant to turn into RCE. Inside bash, reference as `$HEAD_REF` + # etc. — the values are then plain strings, not code. + env: + PR_NUMBER: ${{ github.event.pull_request.number }} + HEAD_SHA: ${{ github.event.pull_request.head.sha }} + HEAD_REF: ${{ github.event.pull_request.head.ref }} + HEAD_REPO: ${{ github.event.pull_request.head.repo.full_name }} + BASE_REPO: ${{ github.repository }} + shell: bash + run: | + set -euo pipefail + mkdir -p autofix-out + + # Produce a unified diff of the working tree vs. the PR head. + # Empty diff => nothing to suggest; the publish job short-circuits. + git diff --no-color > autofix-out/autofix.patch + + # NOTE: `changed_lines` is the line-count of the patch file, + # which includes hunk headers and context lines — NOT the + # added/removed source-line count. The 3000-line cap in + # pr-autofix-publish.yml is therefore conservative (fires + # before reviewdog hits GitHub's ~3k review-comment API + # ceiling). That bias is intentional. + changed_lines=$(wc -l < autofix-out/autofix.patch | tr -d ' ') + echo "changed_lines=${changed_lines}" >> "$GITHUB_OUTPUT" + + # Carry PR identity over to the trusted job. workflow_run + # context is base-repo-only, so the publish job needs these + # to call the GitHub PR API on the right resource. + # CONTRACT: keep this schema in sync with pr-autofix-publish.yml's + # `assert_field` validators and the agent-facing JSON block in + # the sticky comment. Bump `schema` when changing field names. + jq -n \ + --arg schema 'gitnexus.pr-autofix/v1' \ + --argjson pr_number "${PR_NUMBER}" \ + --arg head_sha "${HEAD_SHA}" \ + --arg head_ref "${HEAD_REF}" \ + --arg head_repo "${HEAD_REPO}" \ + --arg base_repo "${BASE_REPO}" \ + --argjson changed_lines "${changed_lines}" \ + '{schema:$schema, pr_number:$pr_number, head_sha:$head_sha, head_ref:$head_ref, head_repo:$head_repo, base_repo:$base_repo, changed_lines:$changed_lines}' \ + > autofix-out/metadata.json + + echo "--- metadata ---" + cat autofix-out/metadata.json + echo "--- diff (head) ---" + head -c 2000 autofix-out/autofix.patch || true + + # Pinned to v7.0.1. Verify SHA via: + # gh api repos/actions/upload-artifact/git/refs/tags/v7.0.1 + - name: Upload autofix artifact + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: autofix + path: autofix-out/ + retention-days: 1 + if-no-files-found: error diff --git a/.github/workflows/release-candidate.yml b/.github/workflows/release-candidate.yml index 9ab00c0bd..36f37b5c6 100644 --- a/.github/workflows/release-candidate.yml +++ b/.github/workflows/release-candidate.yml @@ -294,9 +294,11 @@ jobs: fi fi - echo "base=$BASE" >> "$GITHUB_OUTPUT" - echo "rc_n=$NEXT_N" >> "$GITHUB_OUTPUT" - echo "rc_version=$RC_VERSION" >> "$GITHUB_OUTPUT" + { + echo "base=$BASE" + echo "rc_n=$NEXT_N" + echo "rc_version=$RC_VERSION" + } >> "$GITHUB_OUTPUT" - name: Apply rc version in-CI shell: bash @@ -354,9 +356,11 @@ jobs: # remote ref, the push fails and we stop before npm publish. git push --atomic origin "refs/tags/$VTAG" "refs/tags/$MARKER" - echo "vtag=$VTAG" >> "$GITHUB_OUTPUT" - echo "marker=$MARKER" >> "$GITHUB_OUTPUT" - echo "release_sha=$RELEASE_SHA" >> "$GITHUB_OUTPUT" + { + echo "vtag=$VTAG" + echo "marker=$MARKER" + echo "release_sha=$RELEASE_SHA" + } >> "$GITHUB_OUTPUT" - name: Publish to npm (rc dist-tag) run: npm publish --provenance --access public --tag rc diff --git a/.github/workflows/workflow-lint.yml b/.github/workflows/workflow-lint.yml index 97f388d28..7b38b8ddd 100644 --- a/.github/workflows/workflow-lint.yml +++ b/.github/workflows/workflow-lint.yml @@ -1,10 +1,13 @@ -name: Workflow Lint (zizmor) +name: Workflow Lint -# Lints .github/workflows/** for known GitHub Actions security misconfigurations: -# unpinned Actions, dangerous ${{ ... }} interpolation in run: blocks, -# missing per-job permissions:, etc. +# Lints .github/workflows/** for both: +# - actionlint: YAML syntax, expression typing, shellcheck inside `run:` +# blocks, unknown contexts, deprecated runner labels. +# - zizmor: security misconfigurations — unpinned actions, dangerous +# `${{ }}` interpolation, missing per-job permissions, etc. # -# Scoped to PRs that touch .github/** only — keeps off the typical PR critical path. +# Scoped to PRs that touch .github/** only — keeps off the typical PR +# critical path. on: pull_request: @@ -17,6 +20,27 @@ concurrency: cancel-in-progress: true jobs: + actionlint: + name: actionlint + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + contents: read + steps: + - name: Checkout + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + persist-credentials: false + + # Pinned to v2.1.2. Verify SHA via: + # gh api repos/raven-actions/actionlint/git/refs/tags/v2.1.2 + # The action wraps the upstream `rhysd/actionlint` binary and emits + # GitHub-annotation-formatted findings on PRs. + - name: Run actionlint + uses: raven-actions/actionlint@205b530c5d9fa8f44ae9ed59f341a0db994aa6f8 # v2.1.2 + with: + fail-on-error: true + zizmor: runs-on: ubuntu-latest timeout-minutes: 10 diff --git a/.github/zizmor.yml b/.github/zizmor.yml index c534679c1..b2f89e3ba 100644 --- a/.github/zizmor.yml +++ b/.github/zizmor.yml @@ -14,6 +14,15 @@ rules: # no checkout of fork code occurs. Header comment in the file documents. - ci-report.yml + # workflow_run is the trusted half of the autofix pipeline. The + # untrusted half (pr-autofix.yml) runs fork code with permissions:{} + # and produces only a diff artifact (data, not executable code). The + # publish job consumes the artifact, allowlist-validates every field + # of metadata.json before exporting to $GITHUB_OUTPUT, never checks + # out fork code, and never executes anything fork-controlled. Header + # comment in the file documents the split. + - pr-autofix-publish.yml + # pull_request_target needed by claude-code-action to access secrets # and post review comments on fork PRs. Mitigated by: PR checkouts pin # the fork's HEAD SHA (not the branch ref) to prevent TOCTOU races, diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 0ed2aeba5..99930cf0a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -103,6 +103,24 @@ Every workflow under `.github/workflows/` MUST declare a top-level `concurrency: - When adding a new workflow, copy the concurrency block from an existing workflow of the same event shape. +## CI automation contracts + +Two workflows produce machine-readable signals on every PR. Coding agents and humans alike can rely on the names and shapes below — change them with intent. + +### `gitnexus/autofix` + +`pr-autofix.yml` (untrusted) + `pr-autofix-publish.yml` (trusted) run `prettier --write` and `eslint --fix` against the PR head and surface the diff as inline review-comment suggestions. Three signals are emitted: + +| Surface | Where | Notes | +|---|---|---| +| Sticky PR comment | Top-level comment with the HTML marker `` and heading `## :sparkles: PR Autofix`. Only posted when there is something to fix; clean PRs stay silent. | Edit-in-place via marker; one comment per PR. | +| Fenced JSON block | Inside the sticky, fenced as `gitnexus-autofix`. Schema `gitnexus.pr-autofix/v1` with fields `state` (`suggestions-posted` \| `skipped-too-large`), `pr_number`, `head_sha`, `changed_lines`, `run_id`. | Parseable signal — preferred over regexing prose. | +| Check Run | Stable name `gitnexus/autofix` on the PR head SHA. Conclusion: `success` (clean) or `neutral` (suggestions-posted / skipped-too-large). The output title disambiguates the two `neutral` cases. | Surfaced under PR Checks; readable via `gh pr checks `. | + +To detect outcome from an agent: `gh pr checks --json name,conclusion,output | jq '.[] | select(.name == "gitnexus/autofix")'`. + +Forks are supported. The untrusted half runs fork code with `permissions: {}` and ships the diff as an artifact; the trusted publish job consumes only the diff (data, not code) and posts the comment + check run. + ## AI-assisted contributions If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes. diff --git a/README.md b/README.md index 5b7a442b0..0cdb017f4 100644 --- a/README.md +++ b/README.md @@ -214,6 +214,7 @@ gitnexus clean --all --force # Delete all indexes gitnexus wiki [path] # Generate repository wiki from knowledge graph gitnexus wiki --model # Wiki with custom LLM model (default: gpt-4o-mini) gitnexus wiki --base-url # Wiki with custom LLM API base URL +gitnexus publish # Notify the understand-quickly registry (opt-in, see below) # Repository groups (multi-repo / monorepo service tracking) gitnexus group create # Create a repository group @@ -228,6 +229,12 @@ gitnexus group status # Check staleness of repos in a group If `analyze` reports a worker parse timeout on a large or unusual repository, it keeps running and falls back safely. To give slow worker jobs more time, use `gitnexus analyze --worker-timeout 60` or set `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=60000`. For very large files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` controls the worker job byte budget. +#### Publishing to understand-quickly (opt-in) + +[`looptech-ai/understand-quickly`](https://github.com/looptech-ai/understand-quickly) is a public registry of code-knowledge graphs that lists `gitnexus@1` as a first-class format. After registering your repo once (`npx @understand-quickly/cli add` or the [wizard](https://looptech-ai.github.io/understand-quickly/add.html)), `gitnexus publish` fires a single `repository_dispatch` event so the registry resyncs your entry on demand instead of waiting for the nightly job. + +It is opt-in and a no-op without `UNDERSTAND_QUICKLY_TOKEN` — a fine-grained GitHub PAT with `Repository dispatches: write` on the registry repo. Nothing else happens; no graph file is uploaded. See the [protocol spec](https://github.com/looptech-ai/understand-quickly/blob/main/docs/integrations/protocol.md) for the full contract. + ### What Your AI Agent Gets **16 tools** exposed via MCP (11 per-repo + 5 group): diff --git a/gitnexus-shared/src/index.ts b/gitnexus-shared/src/index.ts index ea66c3855..faf136fe7 100644 --- a/gitnexus-shared/src/index.ts +++ b/gitnexus-shared/src/index.ts @@ -143,6 +143,18 @@ export type { ScopeTree } from './scope-resolution/scope-tree.js'; export { buildPositionIndex } from './scope-resolution/position-index.js'; export type { PositionIndex } from './scope-resolution/position-index.js'; +// Understand-Quickly registry integration (opt-in) +export { + UNDERSTAND_QUICKLY_DISPATCH_URL, + UNDERSTAND_QUICKLY_EVENT_TYPE, + UNDERSTAND_QUICKLY_TOKEN_ENV, + buildUqDispatchPayload, + isValidOwnerRepo, + parseOwnerRepoFromRemote, + stripGitSuffix, +} from './integrations/understand-quickly.js'; +export type { UqDispatchPayload } from './integrations/understand-quickly.js'; + // Shadow-mode diff + aggregation (RFC §6.3; Ring 2 SHARED #918) export { diffResolutions } from './scope-resolution/shadow/diff.js'; export type { diff --git a/gitnexus-shared/src/integrations/understand-quickly.ts b/gitnexus-shared/src/integrations/understand-quickly.ts new file mode 100644 index 000000000..f30e7461b --- /dev/null +++ b/gitnexus-shared/src/integrations/understand-quickly.ts @@ -0,0 +1,151 @@ +/** + * Understand-Quickly registry integration helpers. + * + * Pure, runtime-agnostic logic for opting in to publishing a GitNexus + * index to the [`looptech-ai/understand-quickly`](https://github.com/looptech-ai/understand-quickly) + * registry. Lives in `gitnexus-shared` so both the Node CLI and any + * future browser-side surface can construct identical dispatch payloads. + * + * Network I/O lives in the CLI command (`gitnexus/src/cli/publish.ts`) + * to keep this module free of Node-only imports — see the comment at + * the top of `gitnexus-shared/src/graph/types.ts`. + * + * The protocol contract (single dispatch event, no graph upload) is + * documented at: + * https://github.com/looptech-ai/understand-quickly/blob/main/docs/integrations/protocol.md + */ + +/** + * URL of the registry repo's repository_dispatch endpoint. Hardcoded + * because the registry is the canonical home for this integration — + * users who want a private registry can fork and patch. + */ +export const UNDERSTAND_QUICKLY_DISPATCH_URL = + 'https://api.github.com/repos/looptech-ai/understand-quickly/dispatches'; + +/** + * Event type the registry's sync workflow listens for. + * See `looptech-ai/understand-quickly/.github/workflows/sync.yml`. + */ +export const UNDERSTAND_QUICKLY_EVENT_TYPE = 'sync-entry'; + +/** Environment variable that gates the dispatch. */ +export const UNDERSTAND_QUICKLY_TOKEN_ENV = 'UNDERSTAND_QUICKLY_TOKEN'; + +export interface UqDispatchPayload { + event_type: typeof UNDERSTAND_QUICKLY_EVENT_TYPE; + client_payload: { + /** `/` shape — must match the registered entry. */ + id: string; + }; +} + +/** + * Build the JSON body for the `repository_dispatch` ping. Pure — no + * env reads, no network. Validates that `id` looks like `owner/repo` + * (one slash, no whitespace, both halves non-empty) so a misconfigured + * caller fails loudly before the round-trip. + */ +export function buildUqDispatchPayload(id: string): UqDispatchPayload { + if (!isValidOwnerRepo(id)) { + throw new Error( + `[understand-quickly] expected id of the form "owner/repo", got "${id}". ` + + `The registry uses this string to look up your entry in registry.json — ` + + `it must match the GitHub owner/repo of the source code, not a local path.`, + ); + } + return { + event_type: UNDERSTAND_QUICKLY_EVENT_TYPE, + client_payload: { id }, + }; +} + +/** + * `owner/repo` validation. Conservative on purpose: GitHub's actual + * naming rules are looser, but we want to catch local paths + * (`/Users/...`), bare slugs (`my-repo`), and accidental whitespace. + * + * Matches GitHub's published slug rules: + * owner: starts with alnum, then alnum/hyphen only, must end with + * alnum (no trailing hyphen — GitHub rejects this at account + * creation, so a `my-org-/repo` input would otherwise pass us + * and 422 from GitHub). No underscore, no dot. Length cap 39. + * repo: any of alnum/dot/hyphen/underscore. Length cap 100. + */ +export function isValidOwnerRepo(id: string): boolean { + return /^[A-Za-z0-9](?:[A-Za-z0-9-]{0,37}[A-Za-z0-9])?\/[A-Za-z0-9._-]{1,100}$/.test(id); +} + +/** + * Strip a single trailing `.git` (case-insensitive) and any trailing + * slashes from a URL-ish string. Bounded linear: each character is + * visited at most twice, no backtracking. + * + * Replaces `s.replace(/\.git\/*$/i, '').replace(/\/+$/, '')` which + * CodeQL's polynomial-regex check (codeql/js/polynomial-redos) flags as + * a worst-case O(n²) on adversarial input like "////.../x". + */ +export function stripGitSuffix(input: string): string { + let end = input.length; + // Trim trailing '/'. + while (end > 0 && input.charCodeAt(end - 1) === 0x2f) end--; + // Drop one trailing '.git' (case-insensitive). + if (end >= 4) { + const tail = input.slice(end - 4, end).toLowerCase(); + if (tail === '.git') end -= 4; + } + // Trim trailing '/' that may have sat between '.git' and the rest. + while (end > 0 && input.charCodeAt(end - 1) === 0x2f) end--; + return input.slice(0, end); +} + +/** + * Parse `owner/repo` out of a git remote URL. Mirrors the heuristic in + * `gitnexus/src/storage/git.ts:parseRepoNameFromUrl` but keeps both + * halves so we can build a registry id. Returns `null` on shapes we + * don't recognise. + * + * Examples: + * git@github.com:looptech-ai/understand-quickly.git + * https://github.com/looptech-ai/understand-quickly + * ssh://git@github.com/looptech-ai/understand-quickly.git + */ +export function parseOwnerRepoFromRemote(url: string | null | undefined): string | null { + if (!url) return null; + const trimmed = url.trim(); + if (!trimmed) return null; + // Strip a trailing `.git` (case-insensitive) and any trailing slashes + // so https://h/o/r and https://h/o/r.git collapse to the same id. + // Bounded-linear helper avoids the polynomial-regex CodeQL alert. + const stripped = stripGitSuffix(trimmed); + + // SCP-form SSH (`git@host:owner/repo`). Capture host so we can reject + // non-GitHub remotes — a GitLab origin like + // `https://gitlab.example.com/group/sub/project.git` would otherwise + // silently dispatch the wrong id (LOW 9). + const ssh = stripped.match(/^[^@]+@([^:]+):([^/]+)\/([^/]+)$/); + if (ssh) { + const host = ssh[1].toLowerCase(); + if (host !== 'github.com' && host !== 'www.github.com') return null; + return `${ssh[2]}/${ssh[3]}`; + } + + // URL forms (https://, ssh://, git://, file://) — last two path segments. + const url2 = stripped.match(/^[a-zA-Z][a-zA-Z0-9+.-]*:\/\/([^/]+)\/(.+)$/); + if (url2) { + // Strip optional `userinfo@` (e.g. `ssh://git@github.com/...`). + const authority = url2[1]; + const atIdx = authority.lastIndexOf('@'); + const hostAndPort = atIdx >= 0 ? authority.slice(atIdx + 1) : authority; + // Strip `:port` suffix if present. + const colonIdx = hostAndPort.indexOf(':'); + const host = (colonIdx >= 0 ? hostAndPort.slice(0, colonIdx) : hostAndPort).toLowerCase(); + if (host !== 'github.com' && host !== 'www.github.com') return null; + const segments = url2[2].split('/').filter(Boolean); + if (segments.length >= 2) { + const [owner, repo] = segments.slice(-2); + return `${owner}/${repo}`; + } + } + return null; +} diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index b07ccd7cb..04233b1a1 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -2065,9 +2065,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "25.6.1", - "resolved": "https://registry.npmjs.org/@types/node/-/node-25.6.1.tgz", - "integrity": "sha512-coJCN8O1q4AGyyqCAUSP06P+SrMTu18BkEj3NVAK07q6QUneD2wzj3CLv9+yP+BMeZQlMvneXqqvDe3w+xcq7g==", + "version": "25.6.2", + "resolved": "https://registry.npmjs.org/@types/node/-/node-25.6.2.tgz", + "integrity": "sha512-sokuT28dxf9JT5Kady1fsXOvI4HVpjZa95NKT5y9PNTIrs2AsobR4GFAA90ZG8M+nxVRLysCXsVj6eGC7Vbrlw==", "license": "MIT", "dependencies": { "undici-types": "~7.19.0" @@ -3110,9 +3110,9 @@ "license": "MIT" }, "node_modules/fast-uri": { - "version": "3.1.0", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.0.tgz", - "integrity": "sha512-iPeeDKJSWf4IEOasVVrknXpaBV0IApz/gp7S2bb7Z4Lljbl2MGJRqInZiUrQwV16cpzw/D3S5j5Julj/gT52AA==", + "version": "3.1.2", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz", + "integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==", "funding": [ { "type": "github", @@ -3476,9 +3476,9 @@ "license": "MIT" }, "node_modules/hono": { - "version": "4.12.16", - "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.16.tgz", - "integrity": "sha512-jN0ZewiNAWSe5khM3EyCmBb250+b40wWbwNILNfEvq84VREWwOIkuUsFONk/3i3nqkz7Oe1PcpM2mwQEK2L9Kg==", + "version": "4.12.18", + "resolved": "https://registry.npmjs.org/hono/-/hono-4.12.18.tgz", + "integrity": "sha512-RWzP96k/yv0PQfyXnWjs6zot20TqfpfsNXhOnev8d1InAxubW93L11/oNUc3tQqn2G0bSdAOBpX+2uDFHV7kdQ==", "license": "MIT", "engines": { "node": ">=16.9.0" @@ -4282,15 +4282,15 @@ } }, "node_modules/onnxruntime-common": { - "version": "1.25.1", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.25.1.tgz", - "integrity": "sha512-kKvYQFdos4LWJqhZ+nmKu3NT8NXzw8I5x9fNUKe1rNKcPfNKnYXUtW7JBpcKFsvLtrJashRgVYSbFap4cHxvNg==", + "version": "1.26.0", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.26.0.tgz", + "integrity": "sha512-qVyMR4lcWgbkc4getFV+GQijsTnbg/siteoqcDwa3sI/LxbrMSNw4ePyvCq/ymdQaRomCA7YuWmhzsswxvymdw==", "license": "MIT" }, "node_modules/onnxruntime-node": { - "version": "1.25.1", - "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.25.1.tgz", - "integrity": "sha512-N0M58CGTiTsLkPpx9bxmRFi24GT6r67Qei/GrBEIiDyntcYdXU5vQZp112ypydG9vEKRFgbgUYQJnEi+jll8dg==", + "version": "1.26.0", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.26.0.tgz", + "integrity": "sha512-OHl6PiOEOqxaLHL0N9eFrbzS7IGmu3BtJNH3RTEnRAheCIkfc3gjcjl4sGcjp9C22ZC9YTquDOxSdT/stBQ6BQ==", "hasInstallScript": true, "license": "MIT", "os": [ @@ -4301,7 +4301,7 @@ "dependencies": { "adm-zip": "^0.5.16", "global-agent": "^4.1.3", - "onnxruntime-common": "1.25.1" + "onnxruntime-common": "1.26.0" } }, "node_modules/onnxruntime-web": { diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 175db47e6..47745c575 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -27,6 +27,7 @@ import { warnMissingOptionalGrammars } from './optional-grammars.js'; import { glob } from 'glob'; import fs from 'fs/promises'; import { cliError } from './cli-message.js'; +import { isHfDownloadFailure } from '../core/embeddings/hf-env.js'; // Capture stderr.write at module load BEFORE anything (LadybugDB native // init, progress bar, console redirection) can monkey-patch it. The @@ -576,6 +577,26 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption return; } + // HF download failure — show clean guidance without the raw stack trace. + // Checked before writeFatalToStderr so the user sees one focused message + // rather than a stack-trace dump followed by a second remediation block. + if (isHfDownloadFailure(msg) || msg.includes('Failed to download embedding model')) { + cliError( + ` The embedding model could not be downloaded.\n` + + ` huggingface.co may be unreachable from your network\n` + + ` (e.g. behind a corporate proxy or a regional firewall).\n` + + ` Suggestions:\n` + + ` 1. Set HF_ENDPOINT to a mirror and retry:\n` + + ` HF_ENDPOINT=https://hf-mirror.com npx gitnexus analyze --embeddings\n` + + ` (Windows: set HF_ENDPOINT=https://hf-mirror.com && npx gitnexus analyze --embeddings)\n` + + ` 2. Check your proxy / VPN settings.\n` + + ` 3. Once downloaded the model is cached — future runs work offline.\n`, + { recoveryHint: 'hf-endpoint-unreachable' }, + ); + process.exitCode = 1; + return; + } + // Bypass the redirected console.error and write the full stack to // the real stderr captured at module load. The redirected // console.error wraps every line with `\\x1b[2K\\r` (ANSI clear-line) diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index bde0a25d6..5fb0e2384 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -163,6 +163,18 @@ program .description('Augment a search pattern with knowledge graph context (used by hooks)') .action(createLazyAction(() => import('./augment.js'), 'augmentCommand')); +program + .command('publish [path]') + .description( + 'Notify the understand-quickly registry that this repo has a fresh GitNexus index. ' + + 'Opt-in: requires UNDERSTAND_QUICKLY_TOKEN (fine-grained PAT with ' + + '`Repository dispatches: write` on looptech-ai/understand-quickly). ' + + 'No-op without the token. See https://github.com/looptech-ai/understand-quickly.', + ) + .option('--id ', 'Override the registry id (defaults to the origin remote)') + .option('--skip-git', 'Treat cwd as the repo root and skip parent git-root discovery') + .action(createLazyAction(() => import('./publish.js'), 'publishCommand')); + // ─── Direct Tool Commands (no MCP overhead) ──────────────────────── // These invoke LocalBackend directly for use in eval, scripts, and CI. diff --git a/gitnexus/src/cli/publish.ts b/gitnexus/src/cli/publish.ts new file mode 100644 index 000000000..8aedc9c35 --- /dev/null +++ b/gitnexus/src/cli/publish.ts @@ -0,0 +1,232 @@ +/** + * `gitnexus publish` — opt-in ping to the understand-quickly registry. + * + * Fires a single `repository_dispatch` event at + * `looptech-ai/understand-quickly` so the registry knows to refresh its + * entry for the current repo. Does NOT upload anything: per the + * understand-quickly protocol, the registry pulls the graph from a + * raw-GitHub URL the user controls. + * + * https://github.com/looptech-ai/understand-quickly/blob/main/docs/integrations/protocol.md + * + * Defaults: + * - Without `UNDERSTAND_QUICKLY_TOKEN` in the env, this is a no-op + * (prints one informational line, exit 0). Same shape as the + * `--publish` patterns in sibling tools. + * - With the token, fires the dispatch and reports the response code. + * + * The `id` is derived from the repo's `origin` remote unless the caller + * passes `--id ` explicitly. We deliberately do NOT auto-add + * the repo to the registry — registration is one-time and uses the + * `npx @understand-quickly/cli add` path documented in the protocol. + */ + +import path from 'path'; +import { + UNDERSTAND_QUICKLY_DISPATCH_URL, + UNDERSTAND_QUICKLY_TOKEN_ENV, + buildUqDispatchPayload, + isValidOwnerRepo, + parseOwnerRepoFromRemote, +} from 'gitnexus-shared'; +import { getGitRoot, getRemoteOriginUrl, getCurrentCommit } from '../storage/git.js'; +import { hasIndex } from '../storage/repo-manager.js'; +import { cliInfo, cliError } from './cli-message.js'; + +export interface PublishOptions { + /** Override the auto-derived `owner/repo` id. */ + id?: string; + /** Treat the cwd as the repo root (skip git-root walk). */ + skipGit?: boolean; +} + +const REGISTER_HINT = + 'Register your repo once with: npx @understand-quickly/cli add\n' + + 'Or use the wizard: https://looptech-ai.github.io/understand-quickly/add.html'; + +/** + * Hard cap on the dispatch fetch to keep CI publish steps from stalling + * for the OS TCP timeout (~2 min) when api.github.com is unreachable. + * Matches the pattern used in `src/core/embeddings/http-client.ts`. + */ +const DISPATCH_TIMEOUT_MS = 15_000; + +export const publishCommand = async ( + inputPath?: string, + options: PublishOptions = {}, +): Promise => { + // ── 0. Token gate FIRST — guarantees true no-op without the token. ── + // The README, CLI --help, and PR body all promise "exit 0 without + // UNDERSTAND_QUICKLY_TOKEN". Doing the index/repo-root checks before + // the token gate would make those promises false for users who haven't + // run `gitnexus analyze` yet but want to verify the command is wired. + const token = process.env[UNDERSTAND_QUICKLY_TOKEN_ENV]; + if (!token) { + cliInfo( + `[understand-quickly] ${UNDERSTAND_QUICKLY_TOKEN_ENV} is not set — skipping dispatch.\n` + + `Set it to a fine-grained PAT with "Repository dispatches: write" on ` + + `looptech-ai/understand-quickly to enable instant resync.\n` + + `(Without the token, the registry's nightly sync still picks up your entry.)`, + { skipped: 'no-token' }, + ); + return; + } + + // ── 1. Resolve the repo root (same precedence as `analyze`) ────────── + let repoPath: string; + if (inputPath) { + repoPath = path.resolve(inputPath); + } else if (options.skipGit) { + repoPath = path.resolve(process.cwd()); + } else { + const gitRoot = getGitRoot(process.cwd()); + if (!gitRoot) { + cliError( + '[understand-quickly] not inside a git repository.\n' + + 'Run from a repo, or pass --skip-git to publish from the current directory.', + ); + process.exitCode = 1; + return; + } + repoPath = gitRoot; + } + + // ── 2. Confirm a GitNexus index exists ─────────────────────────────── + // Publishing without an index is almost always a mistake — the + // registry's nightly sync would fetch a stale or missing graph file + // and mark the entry `missing`. Refuse loudly with a fix-it hint. + if (!(await hasIndex(repoPath))) { + cliError( + `[understand-quickly] no GitNexus index found at ${repoPath}/.gitnexus.\n` + + 'Run `gitnexus analyze` first, then re-run `gitnexus publish`.', + ); + process.exitCode = 1; + return; + } + + // ── 3. Derive the registry id ───────────────────────────────────────── + const id = + options.id ?? parseOwnerRepoFromRemote(getRemoteOriginUrl(repoPath) ?? undefined) ?? null; + if (!id || !isValidOwnerRepo(id)) { + cliError( + `[understand-quickly] could not derive a registry id from this repo.\n` + + `Pass --id explicitly (e.g. --id looptech-ai/${path.basename(repoPath)}).\n` + + REGISTER_HINT, + ); + process.exitCode = 1; + return; + } + + // ── 4. Fire the dispatch ───────────────────────────────────────────── + const payload = buildUqDispatchPayload(id); + let response: Response; + try { + response = await fetch(UNDERSTAND_QUICKLY_DISPATCH_URL, { + method: 'POST', + headers: { + Accept: 'application/vnd.github+json', + Authorization: `Bearer ${token}`, + 'X-GitHub-Api-Version': '2022-11-28', + 'Content-Type': 'application/json', + 'User-Agent': 'gitnexus-cli', + }, + body: JSON.stringify(payload), + signal: AbortSignal.timeout(DISPATCH_TIMEOUT_MS), + }); + } catch (err) { + // `AbortSignal.timeout()` throws a `DOMException` with `name === + // 'TimeoutError'` on Node 18.14+ (and on browsers/Bun). It is NOT + // a plain `AbortError`. Match the pattern used in + // gitnexus/src/core/embeddings/http-client.ts so the user sees the + // targeted "timed out" message instead of a generic "operation + // was aborted". + const isTimeout = err instanceof DOMException && err.name === 'TimeoutError'; + if (isTimeout) { + cliError( + `[understand-quickly] dispatch timed out after ${DISPATCH_TIMEOUT_MS}ms. ` + + `Check network access to api.github.com and retry.`, + { id }, + ); + } else { + const msg = err instanceof Error ? err.message : String(err); + cliError(`[understand-quickly] dispatch network error: ${msg}`, { id }); + } + process.exitCode = 1; + return; + } + + // GitHub returns 204 on success. Distinct branches for 401/403/404/422 + // so users debug without checking the docs. + if (response.status === 204) { + await response.body?.cancel().catch(() => {}); + // `getCurrentCommit` is only meaningful in the success path — moving + // it inside this branch removes a wasted child-process spawn on every + // error response (LOW 7). + const commit = getCurrentCommit(repoPath); + cliInfo( + `[understand-quickly] dispatched sync-entry for ${id}` + + (commit ? ` @ ${commit.slice(0, 7)}` : '') + + '.\n' + + `Note: a 204 only confirms GitHub accepted the dispatch. Whether the ` + + `registry workflow finds an entry for "${id}" is logged at ` + + `https://github.com/looptech-ai/understand-quickly/actions/workflows/sync.yml`, + { id, commit, status: response.status }, + ); + return; + } + + if (response.status === 401) { + cliError( + `[understand-quickly] dispatch returned 401 — the ${UNDERSTAND_QUICKLY_TOKEN_ENV} value is invalid or expired.\n` + + `Regenerate a fine-grained PAT at https://github.com/settings/personal-access-tokens ` + + `with Repository access scoped to looptech-ai/understand-quickly and the ` + + `"Repository dispatches: write" permission, then retry.`, + { id, status: response.status }, + ); + process.exitCode = 1; + return; + } + + if (response.status === 403) { + cliError( + `[understand-quickly] dispatch returned 403 — the token authenticated but ` + + `lacks the "Repository dispatches: write" permission on ` + + `looptech-ai/understand-quickly. Edit the PAT scopes and retry.`, + { id, status: response.status }, + ); + process.exitCode = 1; + return; + } + + if (response.status === 404) { + cliError( + `[understand-quickly] dispatch returned 404 — the token cannot reach ` + + `looptech-ai/understand-quickly. Verify the PAT has Repository access to ` + + `that exact repo (not just your own org).`, + { id, status: response.status }, + ); + process.exitCode = 1; + return; + } + + if (response.status === 422) { + // Malformed event_type / client_payload — a code bug in this CLI, + // not a user mistake. Surface so we get bug reports. + const body422 = await response.text().catch(() => ''); + cliError( + `[understand-quickly] dispatch returned 422 (this is a CLI bug; please report).\n` + + `Body: ${body422 || '(empty)'}`, + { id, status: response.status }, + ); + process.exitCode = 1; + return; + } + + // 5xx and anything else → bubble the body so the user has something to act on. + const body = await response.text().catch(() => ''); + cliError( + `[understand-quickly] dispatch failed with HTTP ${response.status}: ${body || '(empty body)'}`, + { id, status: response.status }, + ); + process.exitCode = 1; +}; diff --git a/gitnexus/src/config/ignore-service.ts b/gitnexus/src/config/ignore-service.ts index ce1fda913..2c3eebe3b 100644 --- a/gitnexus/src/config/ignore-service.ts +++ b/gitnexus/src/config/ignore-service.ts @@ -25,6 +25,8 @@ const DEFAULT_IGNORE_LIST = new Set([ 'bower_components', 'jspm_packages', 'vendor', // PHP/Go + 'third_party', // C/C++ (Google-style vendored dependencies) + '3rdparty', // C/C++ (alternate spelling, also Qt convention) // 'packages' removed - commonly used for monorepo source code (lerna, pnpm, yarn workspaces) 'venv', '.venv', diff --git a/gitnexus/src/core/augmentation/engine.ts b/gitnexus/src/core/augmentation/engine.ts index 896813c22..f97415cc9 100644 --- a/gitnexus/src/core/augmentation/engine.ts +++ b/gitnexus/src/core/augmentation/engine.ts @@ -104,7 +104,7 @@ export async function augment(pattern: string, cwd?: string): Promise { } // Step 1: BM25 search (fast, no embeddings) - const bm25Results = await searchFTSFromLbug(pattern, 10, repoId); + const { results: bm25Results } = await searchFTSFromLbug(pattern, 10, repoId); if (bm25Results.length === 0) return ''; diff --git a/gitnexus/src/core/embeddings/embedder.ts b/gitnexus/src/core/embeddings/embedder.ts index b37fb45f3..72ddcbd70 100644 --- a/gitnexus/src/core/embeddings/embedder.ts +++ b/gitnexus/src/core/embeddings/embedder.ts @@ -14,7 +14,12 @@ if (!process.env.ORT_LOG_LEVEL) { process.env.ORT_LOG_LEVEL = '3'; } -import { pipeline, env, type FeatureExtractionPipeline } from '@huggingface/transformers'; +import { + pipeline, + env, + type FeatureExtractionPipeline, + type ProgressInfo, +} from '@huggingface/transformers'; import { existsSync } from 'fs'; import { execFileSync } from 'child_process'; import { join, dirname } from 'path'; @@ -22,7 +27,7 @@ import { createRequire } from 'module'; import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types.js'; import { isHttpMode, getHttpDimensions, httpEmbed } from './http-client.js'; import { resolveEmbeddingConfig } from './config.js'; -import { applyHfEnvOverrides } from './hf-env.js'; +import { applyHfEnvOverrides, isHfDownloadFailure, withHfDownloadRetry } from './hf-env.js'; import { logger } from '../logger.js'; /** @@ -171,13 +176,18 @@ export const initEmbedder = async ( } const progressCallback = onProgress - ? (data: any) => { + ? (data: ProgressInfo) => { const progress: ModelProgress = { - status: data.status || 'progress', - file: data.file, - progress: data.progress, - loaded: data.loaded, - total: data.total, + // Map the `progress_total` aggregate event (not in ModelProgress.status) + // back to 'progress' so callers don't need to handle it separately. + status: + data.status === 'progress_total' + ? 'progress' + : ((data.status as ModelProgress['status']) ?? 'progress'), + file: 'file' in data ? data.file : undefined, + progress: 'progress' in data ? data.progress : undefined, + loaded: 'loaded' in data ? data.loaded : undefined, + total: 'total' in data ? data.total : undefined, }; onProgress(progress); } @@ -202,17 +212,29 @@ export const initEmbedder = async ( logger.info('🔧 Using WASM backend (slower)...'); } - embedderInstance = await (pipeline as any)('feature-extraction', finalConfig.modelId, { - device: device, - dtype: 'fp32', - progress_callback: progressCallback, - session_options: { - logSeverityLevel: 3, - intraOpNumThreads: finalConfig.threads, - interOpNumThreads: 1, - executionMode: 'sequential', + embedderInstance = await withHfDownloadRetry( + () => + pipeline('feature-extraction', finalConfig.modelId, { + device: device, + dtype: 'fp32', + progress_callback: progressCallback, + session_options: { + logSeverityLevel: 3, + intraOpNumThreads: finalConfig.threads, + interOpNumThreads: 1, + executionMode: 'sequential', + }, + }), + { + onRetry: isDev + ? (attempt, max, err) => + logger.warn( + { attempt, max, err: err.message }, + `⚠️ Model download network error (attempt ${attempt}/${max}), retrying…`, + ) + : undefined, }, - }); + ); currentDevice = device; if (isDev) { @@ -228,6 +250,20 @@ export const initEmbedder = async ( return embedderInstance!; } catch (deviceError) { + // Network errors and circuit-open errors are not device-specific — + // they will fail the same way on every device. Rethrow immediately + // with actionable HF_ENDPOINT guidance rather than silently falling + // back to the next device. + const errMsg = deviceError instanceof Error ? deviceError.message : String(deviceError); + if (isHfDownloadFailure(errMsg)) { + const endpointHint = process.env.HF_ENDPOINT + ? `The configured endpoint (${process.env.HF_ENDPOINT}) may be unreachable.` + : `huggingface.co may be unreachable from your network.\n` + + ` Set HF_ENDPOINT to a mirror and retry:\n` + + ` HF_ENDPOINT=https://hf-mirror.com npx gitnexus analyze --embeddings\n` + + ` (Windows: set HF_ENDPOINT=https://hf-mirror.com && npx gitnexus analyze --embeddings)`; + throw new Error(`Failed to download embedding model: ${errMsg}\n ${endpointHint}`); + } if (isDev && (device === 'cuda' || device === 'dml')) { const gpuType = device === 'dml' ? 'DirectML' : 'CUDA'; logger.info(`⚠️ ${gpuType} not available, falling back to CPU...`); diff --git a/gitnexus/src/core/embeddings/hf-env.ts b/gitnexus/src/core/embeddings/hf-env.ts index 6a977a76d..95548fff2 100644 --- a/gitnexus/src/core/embeddings/hf-env.ts +++ b/gitnexus/src/core/embeddings/hf-env.ts @@ -1,6 +1,25 @@ import os from 'node:os'; import { join } from 'node:path'; +// --------------------------------------------------------------------------- +// Download resilience defaults +// --------------------------------------------------------------------------- + +/** Per-attempt timeout for the full model download (5 minutes). */ +export const HF_DOWNLOAD_TIMEOUT_MS = 5 * 60 * 1_000; +/** Maximum total download attempts (1 initial + N-1 retries). */ +export const HF_MAX_ATTEMPTS = 3; +/** Initial delay between retry attempts; doubles on each subsequent retry. */ +export const HF_BASE_DELAY_MS = 2_000; +/** Number of consecutive failures required to open the circuit. */ +export const CB_FAILURE_THRESHOLD = 3; +/** How long the circuit stays open before transitioning to half-open. */ +export const CB_RESET_TIMEOUT_MS = 60_000; +/** Upper bound clamped on the env-override per-attempt timeout (30 minutes). */ +export const HF_MAX_TIMEOUT_MS = 30 * 60 * 1_000; +/** Upper bound clamped on the env-override attempt count. */ +export const HF_MAX_ATTEMPTS_CAP = 10; + /** * @internal Exported only for unit tests and the two embedder entry points * (`core/embeddings/embedder.ts` + `mcp/core/embedder.ts`). Not part of the @@ -60,3 +79,265 @@ export function applyHfEnvOverrides(env: HfEnvSubset): void { env.remoteHost = endpoint.endsWith('/') ? endpoint : endpoint + '/'; } } + +/** + * @internal Exported for unit tests and the two embedder entry points. + * + * Returns true when an error message indicates a network-level fetch failure + * during HuggingFace model download (e.g. `TypeError: fetch failed`, + * `ECONNREFUSED`, `ENOTFOUND`, `ETIMEDOUT`, `ECONNRESET`). + * + * These errors are not device-specific and cannot be fixed by falling back to + * a different ONNX device — the caller should rethrow immediately with + * guidance about `HF_ENDPOINT`. + */ +export function isNetworkFetchError(message: string): boolean { + return ( + message.includes('fetch failed') || + message.includes('ECONNREFUSED') || + message.includes('ENOTFOUND') || + message.includes('ETIMEDOUT') || + message.includes('ECONNRESET') + ); +} + +// --------------------------------------------------------------------------- +// Circuit breaker +// --------------------------------------------------------------------------- + +/** @internal Used by `withHfDownloadRetry` to mark a circuit-open rejection. */ +export const CIRCUIT_OPEN_TAG = 'hf-circuit-open'; + +/** Circuit-breaker states. */ +type CircuitState = 'closed' | 'open' | 'half-open'; + +/** + * Circuit breaker for HuggingFace model downloads. + * + * After `failureThreshold` consecutive network failures the circuit opens and + * all subsequent calls to `withHfDownloadRetry` fail immediately without + * issuing any network requests. After `resetTimeoutMs` the circuit enters the + * half-open state and the next call is attempted — if it succeeds the circuit + * closes again; if it fails the circuit re-opens. + * + * Exported for unit-testing; production code should use the module-level + * `hfDownloadCircuit` singleton. + */ +export class HfDownloadCircuitBreaker { + private _state: CircuitState = 'closed'; + private _failures = 0; + /** Timestamp of the last recorded failure (ms since epoch). */ + lastFailureAt = 0; + + constructor( + readonly failureThreshold: number = CB_FAILURE_THRESHOLD, + readonly resetTimeoutMs: number = CB_RESET_TIMEOUT_MS, + ) {} + + /** Effective state, factoring in the reset-timeout transition. */ + get state(): CircuitState { + if (this._state === 'open' && Date.now() - this.lastFailureAt > this.resetTimeoutMs) { + this._state = 'half-open'; + } + return this._state; + } + + /** Returns true when the circuit is open and calls should be rejected. */ + isOpen(): boolean { + return this.state === 'open'; + } + + /** Record a successful call — resets the failure counter and closes the circuit. */ + recordSuccess(): void { + this._failures = 0; + this._state = 'closed'; + } + + /** Record a failed call — increments the counter and opens the circuit when the threshold is reached. */ + recordFailure(): void { + this._failures++; + this.lastFailureAt = Date.now(); + if (this._failures >= this.failureThreshold) { + this._state = 'open'; + } + } + + /** @internal Reset to initial state (used in tests). */ + reset(): void { + this._failures = 0; + this._state = 'closed'; + this.lastFailureAt = 0; + } +} + +/** Module-level singleton shared by both embedder entry points. */ +export const hfDownloadCircuit = new HfDownloadCircuitBreaker(); + +// --------------------------------------------------------------------------- +// Retry + timeout wrapper +// --------------------------------------------------------------------------- + +/** @internal Returns true for errors that should abort without retry (circuit-open). */ +export function isHfCircuitOpenError(message: string): boolean { + return message.includes(CIRCUIT_OPEN_TAG); +} + +/** + * Returns true for any HuggingFace download failure that warrants showing the + * `HF_ENDPOINT` remediation hint: either a raw network error or a + * circuit-open rejection (which itself was caused by repeated network errors). + */ +export function isHfDownloadFailure(message: string): boolean { + return isNetworkFetchError(message) || isHfCircuitOpenError(message); +} + +/** @internal Wraps `fn` in a hard time-limit. The timeout error contains + * `ETIMEDOUT` so that `isNetworkFetchError` classifies it correctly. + */ +export function withDownloadTimeout(fn: () => Promise, timeoutMs: number): Promise { + return new Promise((resolve, reject) => { + const timer = setTimeout( + () => + reject( + new Error( + `ETIMEDOUT: model download timed out after ${Math.round(timeoutMs / 1000)}s — ` + + `check your network speed or set HF_ENDPOINT to a faster mirror`, + ), + ), + timeoutMs, + ); + fn().then( + (v) => { + clearTimeout(timer); + resolve(v); + }, + (e) => { + clearTimeout(timer); + reject(e); + }, + ); + }); +} + +/** @internal Async sleep (exposed for testing). */ +export function sleep(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + +export interface HfRetryOptions { + /** Maximum total attempts including the initial one (default: `HF_MAX_ATTEMPTS`). */ + maxAttempts?: number; + /** Delay before the first retry; doubles on each subsequent attempt (default: `HF_BASE_DELAY_MS`). */ + baseDelayMs?: number; + /** Per-attempt wall-clock timeout in ms (default: `HF_DOWNLOAD_TIMEOUT_MS`). */ + timeoutMs?: number; + /** + * Circuit-breaker instance to use. Defaults to the module-level + * `hfDownloadCircuit` singleton. Pass a fresh instance in tests. + */ + circuit?: HfDownloadCircuitBreaker; + /** + * Optional callback invoked before each retry (not the initial attempt). + * @param attempt - 1-based retry number + * @param max - total allowed attempts + * @param error - the error that triggered the retry + */ + onRetry?: (attempt: number, max: number, error: Error) => void; +} + +/** + * Retry wrapper for HuggingFace model downloads with per-attempt timeout and + * circuit-breaker protection. + * + * Behaviour: + * - If the circuit is **open**, fails immediately with a `CIRCUIT_OPEN_TAG` + * message (so `isHfDownloadFailure` still returns true and the caller can + * show `HF_ENDPOINT` guidance). + * - Each attempt is wrapped in `withDownloadTimeout`. + * - On a network-level error (`isNetworkFetchError`) the attempt is retried + * with exponential back-off; non-network errors (e.g. ONNX device failure) + * are rethrown immediately without retry. + * - Every network failure is recorded on the circuit breaker; a success resets + * it. + * - After all attempts are exhausted, the last network error is rethrown + * so the existing `isNetworkFetchError` / `isHfDownloadFailure` guards in + * the calling code still fire. + */ +export async function withHfDownloadRetry( + fn: () => Promise, + options: HfRetryOptions = {}, +): Promise { + // Resolve effective values — explicit options take precedence over env vars, + // which take precedence over built-in defaults. This lets users lower the + // per-attempt timeout without rebuilding (e.g. + // HF_DOWNLOAD_TIMEOUT_MS=60000 npx gitnexus analyze --embeddings + // reduces the worst-case wait from 15 minutes to ~3 minutes). + // + // Upper bounds are clamped to prevent accidental runaway configuration: + // - timeoutMs is capped at HF_MAX_TIMEOUT_MS (30 min) + // - maxAttempts is floored (fractional values → integer) and capped at + // HF_MAX_ATTEMPTS_CAP (10). Values ≤ 0, NaN, or Infinity fall back to + // the built-in defaults. + const envTimeout = Number(process.env.HF_DOWNLOAD_TIMEOUT_MS); + const envMaxAttempts = Number(process.env.HF_MAX_ATTEMPTS); + const resolvedTimeout = + Number.isFinite(envTimeout) && envTimeout > 0 + ? Math.min(envTimeout, HF_MAX_TIMEOUT_MS) + : HF_DOWNLOAD_TIMEOUT_MS; + const resolvedMaxAttempts = + Number.isFinite(envMaxAttempts) && envMaxAttempts > 0 + ? Math.min(Math.floor(envMaxAttempts), HF_MAX_ATTEMPTS_CAP) + : HF_MAX_ATTEMPTS; + const { + maxAttempts = resolvedMaxAttempts, + baseDelayMs = HF_BASE_DELAY_MS, + timeoutMs = resolvedTimeout, + circuit = hfDownloadCircuit, + onRetry, + } = options; + if (circuit.isOpen()) { + const secsUntilReset = Math.ceil( + (circuit.resetTimeoutMs - (Date.now() - circuit.lastFailureAt)) / 1000, + ); + throw new Error( + `${CIRCUIT_OPEN_TAG}: HuggingFace download circuit is open after repeated network failures` + + (secsUntilReset > 0 ? ` — will reset in ~${secsUntilReset}s` : ''), + ); + } + + let lastError: Error = new Error('unknown error'); + + for (let attempt = 0; attempt < maxAttempts; attempt++) { + try { + const result = await withDownloadTimeout(fn, timeoutMs); + circuit.recordSuccess(); + return result; + } catch (err) { + lastError = err instanceof Error ? err : new Error(String(err)); + + if (!isNetworkFetchError(lastError.message)) { + // Non-network error (e.g. CUDA unavailable) — propagate without retry + throw lastError; + } + + circuit.recordFailure(); + + if (circuit.isOpen()) { + // Circuit just tripped — fail fast, no more retries + throw new Error( + `${CIRCUIT_OPEN_TAG}: HuggingFace download circuit opened after ${circuit.failureThreshold} consecutive failures`, + ); + } + + if (attempt < maxAttempts - 1) { + const delay = baseDelayMs * Math.pow(2, attempt); + onRetry?.(attempt + 1, maxAttempts, lastError); + await sleep(delay); + } + } + } + + // All retries exhausted — throw the last network error so isNetworkFetchError + // patterns in the calling code still match and surface HF_ENDPOINT guidance. + throw lastError; +} diff --git a/gitnexus/src/core/git-staleness.ts b/gitnexus/src/core/git-staleness.ts index 96f70ddd6..c90cef85e 100644 --- a/gitnexus/src/core/git-staleness.ts +++ b/gitnexus/src/core/git-staleness.ts @@ -3,11 +3,14 @@ * Lives in core/ so application code does not depend on the MCP package layer. */ -import { execFileSync } from 'node:child_process'; +import { execFile, execFileSync } from 'node:child_process'; +import { promisify } from 'node:util'; import path from 'path'; import { readRegistry, type RegistryEntry, type CwdMatch } from '../storage/repo-manager.js'; import { findGitRootByDotGit, getCurrentCommit, getRemoteUrl } from '../storage/git.js'; +const execFileAsync = promisify(execFile); + export interface StalenessInfo { isStale: boolean; commitsBehind: number; @@ -41,6 +44,39 @@ export function checkStaleness(repoPath: string, lastCommit: string): StalenessI } } +/** + * Async variant of {@link checkStaleness} — spawns git as a child process + * instead of blocking the event loop. Used by `listRepos()` to check many + * repos in parallel (issue #1363: 200 repos × sync spawn ≈ 50 s). + */ +export async function checkStalenessAsync( + repoPath: string, + lastCommit: string, +): Promise { + try { + // Note: promisified execFile captures stdout/stderr by default (no stdio option needed, + // unlike the sync variant which requires explicit stdio: ['pipe','pipe','pipe']). + const { stdout } = await execFileAsync('git', ['rev-list', '--count', `${lastCommit}..HEAD`], { + cwd: repoPath, + encoding: 'utf-8', + }); + + const commitsBehind = parseInt(stdout.trim(), 10) || 0; + + if (commitsBehind > 0) { + return { + isStale: true, + commitsBehind, + hint: `⚠️ Index is ${commitsBehind} commit${commitsBehind > 1 ? 's' : ''} behind HEAD. Run analyze tool to update.`, + }; + } + + return { isStale: false, commitsBehind: 0 }; + } catch { + return { isStale: false, commitsBehind: 0 }; + } +} + /** * Compare a sibling-clone HEAD against an indexed `lastCommit`. Returns * `undefined` when the indexed commit is not reachable from the sibling diff --git a/gitnexus/src/core/group/config-parser.ts b/gitnexus/src/core/group/config-parser.ts index 73a9021b9..29c868171 100644 --- a/gitnexus/src/core/group/config-parser.ts +++ b/gitnexus/src/core/group/config-parser.ts @@ -4,9 +4,26 @@ import type { GroupConfig, GroupManifestLink, ContractType, ContractRole } from const _require = createRequire(import.meta.url); const yaml = _require('js-yaml') as typeof import('js-yaml'); -const VALID_CONTRACT_TYPES: ContractType[] = ['http', 'grpc', 'thrift', 'topic', 'lib', 'custom']; +const VALID_CONTRACT_TYPES: ContractType[] = [ + 'http', + 'grpc', + 'thrift', + 'topic', + 'lib', + 'custom', + 'include', +]; const VALID_ROLES: ContractRole[] = ['provider', 'consumer']; +// Defaults matter for backward compatibility: any group.yaml that omits a +// `detect.` key inherits its value from this constant. Adding a new +// extractor that defaults to `true` silently changes the behavior of every +// existing group on the next sync. New extractors must default to `false` +// (opt-in) so operators consciously enable them via group.yaml. +// +// `includes`: opt-in. The C/C++ IncludeExtractor (PR #1156) ships disabled by +// default; enable with `detect.includes: true` for groups containing C/C++ +// repos that need cross-repo header tracking. const DEFAULT_DETECT = { http: true, grpc: true, @@ -14,6 +31,7 @@ const DEFAULT_DETECT = { topics: true, shared_libs: true, embedding_fallback: true, + includes: false, workspace_deps: false, }; diff --git a/gitnexus/src/core/group/cross-impact.ts b/gitnexus/src/core/group/cross-impact.ts index f8625cdc5..eab942a62 100644 --- a/gitnexus/src/core/group/cross-impact.ts +++ b/gitnexus/src/core/group/cross-impact.ts @@ -91,6 +91,25 @@ function clampCrossDepth(raw: unknown): { depth: number; warning?: string } { return { depth: d }; } +/** + * Clamp the impact timeout to a sane bounded range. Callers can feed this + * via tool params, so an unclamped value lets a single request hold a + * timer slot for an arbitrarily long duration (CodeQL js/resource- + * exhaustion). 100ms lower bound preserves test-suite scenarios that + * exercise tight timeouts; 5min upper bound is well above any legitimate + * single-impact compute. Applied at the validate boundary so the + * downstream `deadline` (Date.now() + timeoutMs) and the local-leg + * `setTimeout` see the same clamped value — earlier shapes had a 1hr + * outer cap and a 5min inner clamp that disagreed. + */ +export const IMPACT_TIMEOUT_MIN_MS = 100; +export const IMPACT_TIMEOUT_MAX_MS = 5 * 60 * 1_000; + +export function clampTimeout(timeoutMs: number): number { + if (!Number.isFinite(timeoutMs) || timeoutMs <= 0) return IMPACT_TIMEOUT_MIN_MS; + return Math.min(IMPACT_TIMEOUT_MAX_MS, Math.max(IMPACT_TIMEOUT_MIN_MS, Math.trunc(timeoutMs))); +} + export function validateGroupImpactParams(params: Record): | { ok: true; @@ -143,13 +162,19 @@ export function validateGroupImpactParams(params: Record): const service = normalizeServicePrefix(params.service); const subgroup = typeof params.subgroup === 'string' ? params.subgroup : undefined; - let timeoutMs = + // Clamp at the validate boundary so the downstream `deadline` (line + // ~366) and `safeLocalImpact`'s `setTimeout` both see a single + // bounded value. Without this, the outer deadline budgeted Phase-2 + // cross-repo fanout up to 1hr while only the inner setTimeout was + // capped to 5min — the two halves of CodeQL #184's mitigation + // disagreed. + const rawTimeoutMs = typeof params.timeoutMs === 'number' && params.timeoutMs > 0 ? params.timeoutMs : typeof params.timeout === 'number' && params.timeout > 0 ? params.timeout : DEFAULT_LOCAL_IMPACT_TIMEOUT_MS; - if (timeoutMs > 3_600_000) timeoutMs = 3_600_000; + const timeoutMs = clampTimeout(rawTimeoutMs); return { ok: true, @@ -191,12 +216,13 @@ async function safeLocalImpact( impactParams: Parameters[1], timeoutMs: number, ): Promise<{ value: unknown; timedOut: boolean }> { + const safeTimeoutMs = clampTimeout(timeoutMs); let timer: ReturnType | undefined; const impactP = port.impact(repo, impactParams).catch((err) => ({ error: err instanceof Error ? err.message : String(err), })); const timeoutP = new Promise<'timeout'>((resolve) => { - timer = setTimeout(() => resolve('timeout'), timeoutMs); + timer = setTimeout(() => resolve('timeout'), safeTimeoutMs); }); const won = await Promise.race([ impactP.then((v) => ({ tag: 'impact' as const, v })), @@ -212,6 +238,65 @@ async function safeLocalImpact( return { value: won.v, timedOut: false }; } +/** + * Race a single Phase-2 `impactByUid` call against a remaining-budget + * timer. The Codex adversarial review on PR #1331 surfaced that the + * fanout loop only checked `Date.now() > deadline` *between* neighbor + * calls — once `await port.impactByUid(...)` was reached, a hung + * neighbor could pin the request indefinitely, and slow neighbors + * could compound past the 5-min `IMPACT_TIMEOUT_MAX_MS` cap. + * + * This helper wraps each call: a `setTimeout(remainingMs)` aborts an + * `AbortController` whose signal is forwarded to `impactByUid`, and a + * `Promise.race` resolves to `{ timedOut: true }` when the timer + * fires before the call completes. Implementors that ignore the + * signal (current local backend) still see their await resolved by + * the race; full cooperative cancellation inside the BFS is a future + * follow-up. On rejection, the value is `null` (matching the + * fanout's existing `if (fan == null)` truncation contract). + * + * Exported for direct unit testing — the helper IS the load-bearing + * mitigation surface, so the U3 regression test pins it directly + * rather than driving the full `runGroupImpact` path. + */ +export async function safeNeighborImpact( + port: GroupToolPort, + repoId: string, + uid: string, + direction: string, + opts: { + maxDepth: number; + relationTypes: string[]; + minConfidence: number; + includeTests: boolean; + }, + remainingMs: number, +): Promise<{ value: unknown; timedOut: boolean }> { + const controller = new AbortController(); + let timer: ReturnType | undefined; + const callP = port + .impactByUid(repoId, uid, direction, { ...opts, signal: controller.signal }) + .catch(() => null); + const timeoutP = new Promise<'timeout'>((resolve) => { + timer = setTimeout( + () => { + controller.abort(); + resolve('timeout'); + }, + Math.max(0, remainingMs), + ); + }); + const won = await Promise.race([ + callP.then((v) => ({ tag: 'impact' as const, v })), + timeoutP.then(() => ({ tag: 'timeout' as const })), + ]); + if (timer !== undefined) clearTimeout(timer); + if (won.tag === 'timeout') { + return { value: null, timedOut: true }; + } + return { value: won.v, timedOut: false }; +} + export function collectImpactSymbolUids( local: unknown, servicePrefix: string | undefined, @@ -476,7 +561,8 @@ export async function runGroupImpact( if (seen.has(key)) continue; seen.add(key); - if (Date.now() > deadline) { + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) { truncatedRepos.push(n.neighborRepo); continue; } @@ -492,13 +578,25 @@ export async function runGroupImpact( continue; } - const fan = await deps.port.impactByUid(neighborHandle.id, n.neighborUid, direction, { - maxDepth, - relationTypes: relationTypes ?? [], - minConfidence, - includeTests, - }); - if (fan == null) { + // Phase-2 hardening: race each impactByUid against a per-call + // timeout derived from the remaining budget. Without this wrap a + // single hung neighbor would pin the request past the clamped + // timeout, which Codex's adversarial review on PR #1331 flagged + // as the still-open half of CodeQL #184 / js/resource-exhaustion. + const { value: fan, timedOut: neighborTimedOut } = await safeNeighborImpact( + deps.port, + neighborHandle.id, + n.neighborUid, + direction, + { + maxDepth, + relationTypes: relationTypes ?? [], + minConfidence, + includeTests, + }, + remainingMs, + ); + if (neighborTimedOut || fan == null) { truncatedRepos.push(n.neighborRepo); continue; } diff --git a/gitnexus/src/core/group/extractors/include-extractor.ts b/gitnexus/src/core/group/extractors/include-extractor.ts new file mode 100644 index 000000000..7bbfd61ed --- /dev/null +++ b/gitnexus/src/core/group/extractors/include-extractor.ts @@ -0,0 +1,610 @@ +import * as path from 'node:path'; +import * as fs from 'node:fs/promises'; +import { glob } from 'glob'; +import Parser from 'tree-sitter'; +import C from 'tree-sitter-c'; +import Cpp from 'tree-sitter-cpp'; +import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js'; +import type { ExtractedContract, RepoHandle } from '../types.js'; +import { readSafe } from './fs-utils.js'; +import { buildSuffixIndex, type SuffixIndex } from '../../ingestion/import-resolvers/utils.js'; +import { createIgnoreFilter } from '../../../config/ignore-service.js'; +import { getMaxFileSizeBytes } from '../../ingestion/utils/max-file-size.js'; +import { logger } from '../../logger.js'; + +/** + * Cross-repo C/C++ `#include` dependency extractor. + * + * **Provider side:** registers every `.h/.hpp/.hxx/.hh` file in the repo + * as a provider contract with `include::`. + * + * **Consumer side:** parses all C/C++ source/header files for `#include "…"` + * directives, attempts suffix-based resolution against the repo's own file + * list (reusing the same algorithm as the single-repo ingestion pipeline), + * and emits unresolved include paths as consumer contracts. + * + * Matching: a consumer's `include::map/base/dice_map_view.h` in repo A + * matches a provider's `include::map/base/dice_map_view.h` in repo B via + * exact contract-id equality in `runExactMatch`. + */ + +// ---------- constants ---------- + +const HEADER_EXTENSIONS = new Set(['.h', '.hpp', '.hxx', '.hh']); + +// Source = headers (provider-eligible) ∪ implementation files (.c/.cpp/.cc/.cxx). +// Spread keeps the subset relationship explicit so a future contributor adding +// a new header extension to HEADER_EXTENSIONS does not have to remember to +// also add it here. +const SOURCE_EXTENSIONS = new Set([...HEADER_EXTENSIONS, '.c', '.cpp', '.cc', '.cxx']); + +const INCLUDE_QUERY_SRC = '(preproc_include path: (_) @import.source) @import'; + +/** + * Well-known C/C++ standard library headers that can appear in `#include "…"` + * form (some projects use quotes for system headers). + */ +const SYSTEM_HEADERS = new Set([ + // C standard + 'assert.h', + 'complex.h', + 'ctype.h', + 'errno.h', + 'fenv.h', + 'float.h', + 'inttypes.h', + 'iso646.h', + 'limits.h', + 'locale.h', + 'math.h', + 'setjmp.h', + 'signal.h', + 'stdalign.h', + 'stdarg.h', + 'stdatomic.h', + 'stdbool.h', + 'stddef.h', + 'stdint.h', + 'stdio.h', + 'stdlib.h', + 'stdnoreturn.h', + 'string.h', + 'tgmath.h', + 'threads.h', + 'time.h', + 'uchar.h', + 'wchar.h', + 'wctype.h', + // C++ standard (extensionless) + 'algorithm', + 'any', + 'array', + 'atomic', + 'barrier', + 'bit', + 'bitset', + 'cassert', + 'cctype', + 'cerrno', + 'cfenv', + 'cfloat', + 'charconv', + 'chrono', + 'cinttypes', + 'climits', + 'clocale', + 'cmath', + 'codecvt', + 'compare', + 'complex', + 'concepts', + 'condition_variable', + 'coroutine', + 'csetjmp', + 'csignal', + 'cstdarg', + 'cstddef', + 'cstdint', + 'cstdio', + 'cstdlib', + 'cstring', + 'ctime', + 'cuchar', + 'cwchar', + 'cwctype', + 'deque', + 'exception', + 'execution', + 'expected', + 'filesystem', + 'format', + 'forward_list', + 'fstream', + 'functional', + 'future', + 'generator', + 'initializer_list', + 'iomanip', + 'ios', + 'iosfwd', + 'iostream', + 'istream', + 'iterator', + 'latch', + 'limits', + 'list', + 'locale', + 'map', + 'mdspan', + 'memory', + 'memory_resource', + 'mutex', + 'new', + 'numbers', + 'numeric', + 'optional', + 'ostream', + 'print', + 'queue', + 'random', + 'ranges', + 'ratio', + 'regex', + 'scoped_allocator', + 'semaphore', + 'set', + 'shared_mutex', + 'source_location', + 'span', + 'spanstream', + 'sstream', + 'stack', + 'stacktrace', + 'stdexcept', + 'stdfloat', + 'stop_token', + 'streambuf', + 'string', + 'string_view', + 'strstream', + 'syncstream', + 'system_error', + 'thread', + 'tuple', + 'type_traits', + 'typeindex', + 'typeinfo', + 'unordered_map', + 'unordered_set', + 'utility', + 'valarray', + 'variant', + 'vector', + 'version', +]); + +/** Path prefixes that indicate system/kernel headers. */ +const SYSTEM_PATH_PREFIXES = [ + 'sys/', + 'net/', + 'netinet/', + 'arpa/', + 'linux/', + 'asm/', + 'bits/', + 'gnu/', + 'mach/', + 'machine/', + 'xlocale/', +]; + +/** Regex fallback for files that exceed tree-sitter's 32 KB parse limit. */ +const INCLUDE_REGEX = /^[ \t]*#\s*include\s*"([^"]+)"/gm; + +// ---------- helpers ---------- + +/** + * Normalize an include path to a canonical lowercase forward-slash form. + * + * IMPORTANT — case-folding caveat (PR #1156 review finding #3): + * Header paths are lowercased so consumer `#include "Foo/Bar.h"` and + * provider file `Foo/Bar.h` normalize to the same contract-id. This is + * the right trade-off on case-insensitive filesystems (macOS, Windows) + * but on case-sensitive Linux filesystems two distinct headers `Foo.h` + * and `foo.h` in the same repo will collide onto the same provider + * contract-id; only one survives `dedupe()`. The gain (reliable + * cross-platform matching) outweighs the cost (extremely rare header + * casing collisions inside a single repo). + */ +function normalizeIncludePath(raw: string): string { + return raw.replace(/\\/g, '/').replace(/^\.\//, '').replace(/\/+/g, '/').toLowerCase(); +} + +/** + * Strip C/C++ block comments from a source blob. Used only by the + * regex-fallback path to avoid emitting consumer contracts for + * commented-out #include directives. Line comments (`// …`) cannot hide + * #include directives because the regex anchors on start-of-line. + * See PR #1156 review finding #5. + */ +function stripBlockComments(src: string): string { + return src.replace(/\/\*[\s\S]*?\*\//g, ''); +} + +function isAngleBracketInclude(rawNodeText: string): boolean { + const trimmed = rawNodeText.trim(); + return trimmed.startsWith('<') && trimmed.endsWith('>'); +} + +function isSystemHeader(cleanedPath: string): boolean { + // Check well-known standard headers + if (SYSTEM_HEADERS.has(cleanedPath)) return true; + // Check system path prefixes + const lower = cleanedPath.toLowerCase(); + return SYSTEM_PATH_PREFIXES.some((prefix) => lower.startsWith(prefix)); +} + +function isHeaderFile(filePath: string): boolean { + return HEADER_EXTENSIONS.has(path.extname(filePath).toLowerCase()); +} + +function getLanguageForFile(filePath: string): unknown | null { + const ext = path.extname(filePath).toLowerCase(); + switch (ext) { + case '.c': + case '.h': + return C; + case '.cpp': + case '.cc': + case '.cxx': + case '.hpp': + case '.hxx': + case '.hh': + return Cpp; + default: + return null; + } +} + +/** + * Check whether an include path resolves to a file inside the local repo. + * + * Uses *exact full-path* matching on the suffix index — we never accept a + * truncated suffix match. For `#include "foo/bar.h"` this checks: + * (a) a file whose path ends with the full `foo/bar.h` + * (b) if the include omitted the extension, a file whose path ends with + * the include + one of the C/C++ header extensions + * + * Returns `true` when a local file matches — caller should suppress the + * cross-repo consumer contract. + * + * See PR #1156 review finding #4 (suffixResolve ambiguity). + */ +function isLocalInclude(cleaned: string, suffixIndex: SuffixIndex): boolean { + const candidates = [cleaned]; + if (!/\.[a-zA-Z0-9]+$/.test(cleaned)) { + for (const ext of ['.h', '.hpp', '.hxx', '.hh']) candidates.push(cleaned + ext); + } + for (const c of candidates) { + if (suffixIndex.get(c) || suffixIndex.getInsensitive(c)) return true; + } + return false; +} + +// ---------- main class ---------- + +export class IncludeExtractor implements ContractExtractor { + type = 'include' as const; + + /** + * Always returns `true`. NOT called by `sync.ts`, which gates extraction via + * `config.detect.includes` instead (see `sync.ts:174`). Kept solely to satisfy + * the `ContractExtractor` interface so the type stays uniform across extractors. + */ + async canExtract(_repo: RepoHandle): Promise { + return true; + } + + async extract( + dbExecutor: CypherExecutor | null, + repoPath: string, + _repo: RepoHandle, + ): Promise { + // 1. Build the local file list using the same discovery as ingestion + // (createIgnoreFilter + getMaxFileSizeBytes). This guarantees the + // universe of provider/consumer paths matches the universe of File + // nodes in the LadybugDB graph — so no cross-link points at a UID + // that group impact cannot fan out to. + // (PR #1156 Codex follow-up: discovery aligned with ingestion.) + const allFiles = await this.discoverIndexableFiles(repoPath); + const normalizedFiles = allFiles.map((f) => f.replace(/\\/g, '/')); + const suffixIndex = buildSuffixIndex(normalizedFiles, allFiles); + + // 2. Provider: register all header files + const providers = await this.extractProviders(dbExecutor, repoPath, allFiles); + + // 3. Consumer: filter the shared discovery list for source extensions + // and parse #include directives in those files. + const sourceFiles = allFiles.filter((f) => + SOURCE_EXTENSIONS.has(path.extname(f).toLowerCase()), + ); + const consumers = await this.extractConsumers(repoPath, sourceFiles, suffixIndex); + + return this.dedupe([...providers, ...consumers]); + } + + /** + * Discover repo-relative file paths using exactly the same rules the + * ingestion pipeline uses (`walkRepositoryPaths` in + * `gitnexus/src/core/ingestion/filesystem-walker.ts`): + * - `createIgnoreFilter` honors `.gitignore`, `.gitnexusignore`, the + * hardcoded ignore list, and `.gitnexusignore` last-match-wins + * negation. + * - `getMaxFileSizeBytes()` drops files larger than the cap so we + * never emit `File:` UIDs for files ingestion would skip. + * + * Uses sequential stat — there is no `READ_CONCURRENCY` batching here + * because group sync runs at startup-time, not the ingestion hot path, + * and parallelism gains are not worth the import-graph weight. + * + * MAINTENANCE: if `walkRepositoryPaths` changes its glob options, ignore + * filter shape, or size-cap logic, mirror those changes here. The two + * implementations exist because the consumers need different return + * shapes (string[] vs ScannedFile[]) and different concurrency, but + * they MUST agree on which files are reachable — that is what makes + * `File:` UIDs in cross-links correspond to graph File nodes. + */ + private async discoverIndexableFiles(repoPath: string): Promise { + const ignoreFilter = await createIgnoreFilter(repoPath); + const maxFileSizeBytes = getMaxFileSizeBytes(); + + const candidates = await glob('**/*', { + cwd: repoPath, + nodir: true, + dot: false, + ignore: ignoreFilter, + }); + + const survivors: string[] = []; + for (const rel of candidates) { + try { + const stat = await fs.stat(path.join(repoPath, rel)); + if (stat.size > maxFileSizeBytes) continue; + survivors.push(rel); + } catch (err) { + // ENOENT is the documented benign race (glob enumerated a file + // that was deleted before we stat'd it — same race + // walkRepositoryPaths absorbs via Promise.allSettled). Anything + // else (EACCES, EMFILE, EIO) deserves a warning so an operator + // can spot a permission/resource problem instead of silently + // shipping fewer contracts than expected. + const code = (err as NodeJS.ErrnoException | undefined)?.code; + if (code !== 'ENOENT') { + logger.warn( + { err: (err as Error).message, file: rel, repoPath }, + '⚠️ IncludeExtractor: stat failed during discovery; skipping file', + ); + } + } + } + return survivors; + } + + // ---------- provider extraction ---------- + + private async extractProviders( + dbExecutor: CypherExecutor | null, + repoPath: string, + allFiles: string[], + ): Promise { + // Strategy A: graph-assisted + if (dbExecutor) { + const graphProviders = await this.extractProvidersGraph(dbExecutor, repoPath); + if (graphProviders.length > 0) return graphProviders; + } + // Strategy B: filesystem fallback + return this.extractProvidersFallback(repoPath, allFiles); + } + + private async extractProvidersGraph( + db: CypherExecutor, + repoPath: string, + ): Promise { + try { + const rows = await db( + `MATCH (f:File) + WHERE f.filePath =~ '.*\\\\.(h|hpp|hxx|hh)$' + RETURN f.filePath AS filePath, f.id AS fileId`, + ); + // gitnexus analyze stores absolute paths in the File.filePath column. + // Provider contract IDs MUST be repo-relative — otherwise the consumer + // emits `include::map/base/view.h` and the provider emits + // `include::/abs/path/to/repo/map/base/view.h`, which never match + // through runExactMatch and the cross-link silently disappears. + // (PR #1156 follow-up review: graph provider absolute-path bug.) + const normalizedRepoPath = path.resolve(repoPath); + const out: ExtractedContract[] = []; + for (const r of rows) { + if (typeof r.filePath !== 'string' || !r.filePath) continue; + const absolute = r.filePath as string; + const rel = path.relative(normalizedRepoPath, absolute); + // Skip rows that resolve outside the repo (e.g., system headers + // somehow indexed, or stale absolute paths from a different machine). + // path.relative returns a `..`-prefixed path or an absolute path + // when the target is outside the base — both are wrong for our IDs. + if (!rel || rel.startsWith('..') || path.isAbsolute(rel)) continue; + const normalizedRel = rel.replace(/\\/g, '/'); + out.push({ + contractId: `include::${normalizeIncludePath(normalizedRel)}`, + type: 'include' as const, + role: 'provider' as const, + symbolUid: String(r.fileId ?? ''), + symbolRef: { filePath: normalizedRel, name: path.basename(normalizedRel) }, + symbolName: path.basename(normalizedRel), + confidence: 1.0, + meta: { source: 'graph' }, + }); + } + return out; + } catch { + return []; + } + } + + private extractProvidersFallback(_repoPath: string, allFiles: string[]): ExtractedContract[] { + return allFiles + .filter((f) => isHeaderFile(f)) + .map((f) => { + const filePath = f.replace(/\\/g, '/'); + return { + contractId: `include::${normalizeIncludePath(filePath)}`, + type: 'include' as const, + role: 'provider' as const, + symbolUid: `File:${filePath}`, + symbolRef: { filePath, name: path.basename(filePath) }, + symbolName: path.basename(filePath), + confidence: 0.95, + meta: { source: 'filesystem' }, + }; + }); + } + + // ---------- consumer extraction ---------- + + private async extractConsumers( + repoPath: string, + sourceFiles: string[], + suffixIndex: SuffixIndex, + ): Promise { + const parser = new Parser(); + const out: ExtractedContract[] = []; + // Compile the include query once per grammar to avoid re-compilation per file + const queryCache = new Map(); + + for (const rel of sourceFiles) { + const lang = getLanguageForFile(rel); + if (!lang) continue; + + const content = readSafe(repoPath, rel); + if (!content) continue; + + let query = queryCache.get(lang); + if (!query) { + try { + query = new Parser.Query(lang, INCLUDE_QUERY_SRC); + queryCache.set(lang, query); + } catch { + continue; + } + } + + // Collect raw include paths: tree-sitter first, regex fallback for large files. + // `extractionSource` is stamped on each emitted consumer contract so + // regex-fallback contracts stay auditable post-hoc (PR #1156 review finding #6). + let rawIncludes: string[]; + let extractionSource: 'tree_sitter' | 'regex_fallback'; + try { + parser.setLanguage(lang); + const tree = parser.parse(content); + let matches: Parser.QueryMatch[]; + try { + matches = query.matches(tree.rootNode); + } catch { + matches = []; + } + rawIncludes = []; + extractionSource = 'tree_sitter'; + for (const match of matches) { + const sourceNode = match.captures.find((c) => c.name === 'import.source'); + if (!sourceNode) continue; + const rawText = sourceNode.node.text; + if (isAngleBracketInclude(rawText)) continue; + const cleaned = rawText.replace(/['"<>]/g, ''); + if (cleaned && cleaned.length <= 2048) rawIncludes.push(cleaned); + } + } catch { + // tree-sitter failed (e.g. file > 32 KB) — fall back to regex. + // Strip block comments first so we don't emit a consumer contract + // for a commented-out #include (PR #1156 review finding #5). + rawIncludes = []; + extractionSource = 'regex_fallback'; + const scanTarget = stripBlockComments(content); + INCLUDE_REGEX.lastIndex = 0; + let m: RegExpExecArray | null; + while ((m = INCLUDE_REGEX.exec(scanTarget)) !== null) { + if (m[1] && m[1].length <= 2048) rawIncludes.push(m[1]); + } + } + + for (const cleaned of rawIncludes) { + // Filter: skip known system headers and system path prefixes + if (isSystemHeader(cleaned)) continue; + + // Skip relative-up includes: `#include "../include/foo.h"` is + // almost always an intra-repo reference. The suffix index is built + // from repo-relative paths, so isLocalInclude can never match + // `../foo.h`, and emitting it as a consumer contract just pollutes + // the registry with an entry no provider can ever satisfy. + // (PR #1156 follow-up review: `../` relative includes produce + // spurious consumer contracts.) + if (cleaned.startsWith('../') || cleaned.startsWith('..\\')) continue; + + // Skip macro-style includes: `#include PLATFORM_HEADER` parses as an + // identifier under tree-sitter's `(_) @import.source` wildcard. The + // identifier text passes the strip/clean step unchanged, so without + // this guard we would emit `include::platform_header` as a consumer + // contract — and no provider in any repo will ever expose a contract + // for a macro identifier (no file is named `PLATFORM_HEADER`). The + // contract would sit permanently orphaned in the registry. Real + // header references always contain a path separator (`/`, `\`) or an + // extension dot (`foo.h`), so an absent both is a reliable signal we + // are looking at a macro identifier. (PR #1156 follow-up review: + // macro includes emit orphaned consumer contracts.) + if (!/[./\\]/.test(cleaned)) continue; + + // Local resolution (PR #1156 review finding #4): only accept an + // exact-suffix match on the *full* include path. The generic + // suffixResolve() iterates all truncated suffixes, which would + // silently suppress a cross-repo `#include "map/base/view.h"` + // when the local repo has any `internal/view.h` — a realistic + // false-negative in large C++ codebases. Here we only resolve + // locally if a file path ends with the complete include string + // (optionally re-appending one of the C/C++ header extensions + // when the include already omits it). + if (isLocalInclude(cleaned, suffixIndex)) continue; + + // Unresolved: emit as consumer contract + const normalizedRel = rel.replace(/\\/g, '/'); + out.push({ + contractId: `include::${normalizeIncludePath(cleaned)}`, + type: 'include' as const, + role: 'consumer' as const, + symbolUid: `File:${normalizedRel}`, + symbolRef: { filePath: normalizedRel, name: cleaned }, + symbolName: cleaned, + confidence: 0.85, + meta: { + source: extractionSource, + includePath: cleaned, + }, + }); + } + } + + return out; + } + + // ---------- deduplication ---------- + + private dedupe(items: ExtractedContract[]): ExtractedContract[] { + const seen = new Set(); + const out: ExtractedContract[] = []; + for (const c of items) { + const k = `${c.contractId}|${c.role}|${c.symbolRef.filePath}`; + if (seen.has(k)) continue; + seen.add(k); + out.push(c); + } + return out; + } +} diff --git a/gitnexus/src/core/group/extractors/manifest-extractor.ts b/gitnexus/src/core/group/extractors/manifest-extractor.ts index 2af3db595..f4d0f77cf 100644 --- a/gitnexus/src/core/group/extractors/manifest-extractor.ts +++ b/gitnexus/src/core/group/extractors/manifest-extractor.ts @@ -274,6 +274,14 @@ export class ManifestExtractor { LIMIT 1`, { contract: link.contract }, ); + } else if (link.type === 'include') { + rows = await executor( + `MATCH (f:File) WHERE f.filePath = $contract + RETURN f.id AS uid, f.name AS name, f.filePath AS filePath + ORDER BY f.filePath ASC + LIMIT 1`, + { contract: link.contract }, + ); } else if (link.type === 'custom') { // Workspace extractors produce qualified contracts like "mathlex::Expression". // Graph nodes store the unqualified symbol name ("Expression"), so strip @@ -358,6 +366,8 @@ export class ManifestExtractor { return `lib::${contract}`; case 'custom': return `custom::${contract}`; + case 'include': + return `include::${contract}`; default: { const _exhaustive: never = type; throw new Error(`Unhandled ContractType: ${String(_exhaustive)}`); diff --git a/gitnexus/src/core/group/extractors/rust-workspace-extractor.ts b/gitnexus/src/core/group/extractors/rust-workspace-extractor.ts index 63fe7ea82..d58c3e08f 100644 --- a/gitnexus/src/core/group/extractors/rust-workspace-extractor.ts +++ b/gitnexus/src/core/group/extractors/rust-workspace-extractor.ts @@ -31,6 +31,32 @@ interface ImportedSymbol { filePath: string; } +/** + * Linear-time `[package].name = "..."` lookup. The previous regex + * `^\[package\]\s*\n(?:[^\[]*?\n)*?name\s*=\s*"([^"]+)"` had a nested + * lazy quantifier on `\n` that CodeQL js/redos flagged as exponential + * on inputs like `[package]\n` + many bare `\n`. We walk lines + * explicitly: scan from the first `[package]` header until we hit the + * next `[...]` section header, looking for the `name = "..."` line. + * O(n) with the line count. + * + * Exported so the U8 ReDoS regression test can drive the production + * line-walk directly with adversarial fixtures (multi-line strings, + * trailing sections, etc.) instead of duplicating it inline. + */ +export function parseCargoPackageName(content: string): string | null { + const lines = content.split('\n'); + const packageStart = lines.findIndex((l) => l.trim() === '[package]'); + if (packageStart < 0) return null; + for (let i = packageStart + 1; i < lines.length; i++) { + const line = lines[i].trimStart(); + if (line.startsWith('[')) break; // hit the next section header + const m = /^name\s*=\s*"([^"]+)"/.exec(line); + if (m) return m[1]; + } + return null; +} + /** * Parse a Cargo.toml to extract the crate name and workspace dependency * names. Uses simple line-based parsing — no TOML library needed for @@ -47,12 +73,9 @@ async function parseCrateManifest( return null; } - let name = ''; + const name = parseCargoPackageName(content) ?? ''; const workspaceDeps: string[] = []; - const nameMatch = content.match(/^\[package\]\s*\n(?:[^\[]*?\n)*?name\s*=\s*"([^"]+)"/m); - if (nameMatch) name = nameMatch[1]; - // Match dependencies that use workspace = true, which indicates they // are workspace-internal deps: // dep_name = { workspace = true } diff --git a/gitnexus/src/core/group/extractors/thrift-extractor.ts b/gitnexus/src/core/group/extractors/thrift-extractor.ts index cfd8fef02..709968790 100644 --- a/gitnexus/src/core/group/extractors/thrift-extractor.ts +++ b/gitnexus/src/core/group/extractors/thrift-extractor.ts @@ -217,6 +217,10 @@ export async function buildThriftContext(repoPath: string): Promise(); @@ -290,6 +294,10 @@ export class ThriftExtractor implements ContractExtractor { cwd: repoPath, absolute: false, nodir: true, + // TODO(#1156-followup): replace this hand-rolled list with createIgnoreFilter + // (the canonical ingestion ignore filter, like include-extractor.ts now uses). + // New entries to DEFAULT_IGNORE_LIST in src/config/ignore-service.ts (e.g. + // third_party, 3rdparty added in commit a9936a9b) silently do not apply here. ignore: ['**/node_modules/**', '**/.git/**', '**/vendor/**', '**/dist/**', '**/build/**'], }); diff --git a/gitnexus/src/core/group/matching.ts b/gitnexus/src/core/group/matching.ts index 3431f8ddf..0b27655c6 100644 --- a/gitnexus/src/core/group/matching.ts +++ b/gitnexus/src/core/group/matching.ts @@ -107,6 +107,8 @@ export function normalizeContractId(id: string): string { return `topic::${rest.trim().toLowerCase()}`; case 'lib': return `lib::${rest.toLowerCase()}`; + case 'include': + return `include::${rest.replace(/\\/g, '/').replace(/^\.\//, '').replace(/\/+/g, '/').toLowerCase()}`; default: return id; } diff --git a/gitnexus/src/core/group/service.ts b/gitnexus/src/core/group/service.ts index de324e70b..d0473048f 100644 --- a/gitnexus/src/core/group/service.ts +++ b/gitnexus/src/core/group/service.ts @@ -65,6 +65,15 @@ export interface GroupToolPort { relationTypes: string[]; minConfidence: number; includeTests: boolean; + // Optional cancellation signal. Callers (notably the cross-impact + // Phase-2 fanout) wrap this call in a Promise.race against a + // setTimeout-driven AbortController so a single hung neighbor + // cannot exceed the request's clamped timeout budget. Implementors + // may honor the signal cooperatively or simply let the caller's + // race resolve the await — the latter is sufficient for the + // resource-exhaustion mitigation. When the signal is absent or + // already aborted at call time, behavior is unchanged. + signal?: AbortSignal; }, ): Promise; context( diff --git a/gitnexus/src/core/group/storage.ts b/gitnexus/src/core/group/storage.ts index cc3dbfdc9..bc08fd7f9 100644 --- a/gitnexus/src/core/group/storage.ts +++ b/gitnexus/src/core/group/storage.ts @@ -4,6 +4,7 @@ import * as path from 'node:path'; import * as os from 'node:os'; import { randomBytes } from 'node:crypto'; import type { ContractRegistry } from './types.js'; +import { retryRename } from './bridge-db.js'; /** * Build an unpredictable suffix for atomic-write tmp files. Replaces the @@ -59,7 +60,13 @@ export async function writeContractRegistry( } finally { await handle.close(); } - await fsp.rename(tmpPath, targetPath); + // retryRename absorbs the documented Windows EPERM/EBUSY/EACCES race that + // fires when AV scanners or another concurrent rename briefly hold the + // destination handle between rename calls. Same helper bridge-db.ts uses + // (lines 304, 583, 587, 595, 605, 677) for the bridge.lbug atomic swap — + // single source of truth for the Windows-rename pattern across the group + // package. + await retryRename(tmpPath, targetPath); } export async function readContractRegistry(groupDir: string): Promise { diff --git a/gitnexus/src/core/group/sync.ts b/gitnexus/src/core/group/sync.ts index 7ed065131..cd64fdf8c 100644 --- a/gitnexus/src/core/group/sync.ts +++ b/gitnexus/src/core/group/sync.ts @@ -8,12 +8,14 @@ import { HttpRouteExtractor } from './extractors/http-route-extractor.js'; import { GrpcExtractor } from './extractors/grpc-extractor.js'; import { ThriftExtractor } from './extractors/thrift-extractor.js'; import { TopicExtractor } from './extractors/topic-extractor.js'; +import { IncludeExtractor } from './extractors/include-extractor.js'; import { ManifestExtractor } from './extractors/manifest-extractor.js'; import { discoverWorkspaceLinks } from './extractors/workspace-extractor.js'; import { buildProviderIndex, runExactMatch, runWildcardMatch } from './matching.js'; import { detectServiceBoundaries, assignService } from './service-boundary-detector.js'; import type { CypherExecutor } from './contract-extractor.js'; import { writeContractRegistry } from './storage.js'; +import { writeBridge } from './bridge-db.js'; import type { ContractRegistry } from './types.js'; import { logger } from '../logger.js'; @@ -100,6 +102,7 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis const grpcEx = new GrpcExtractor(); const thriftEx = new ThriftExtractor(); const topicEx = new TopicExtractor(); + const includeEx = new IncludeExtractor(); dbExecutors = new Map(); const openPoolIds: string[] = []; @@ -168,6 +171,17 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis } } + if (config.detect.includes) { + const extracted = await includeEx.extract(executor, handle.repoPath, handle); + for (const c of extracted) { + autoContracts.push({ + ...c, + repo: groupPath, + service: assignService(c.symbolRef.filePath, boundaries), + }); + } + } + const metaPath = path.join(handle.storagePath, 'meta.json'); try { const raw = await fs.readFile(metaPath, 'utf-8'); @@ -270,6 +284,28 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis if (opts?.groupDir && !opts.skipWrite) { await writeContractRegistry(opts.groupDir, registry); + // writeBridge failure (disk full, schema error, permission denied) must + // not mask the registry — contracts.json was just written successfully + // and is the canonical source of truth. A stale or absent bridge + // degrades impact queries to empty results, which is recoverable on + // the next sync. Surface the failure as a warning so operators can + // act, but do not propagate it. + // (PR #1156 follow-up review: writeBridge error in sync.ts propagates + // uncaught.) + try { + await writeBridge(opts.groupDir, { + contracts: allContracts, + crossLinks, + repoSnapshots, + missingRepos, + }); + } catch (err) { + const msg = err instanceof Error ? err.message : String(err); + logger.warn( + { err: msg, groupDir: opts.groupDir }, + '⚠️ writeBridge failed; contracts.json is intact but bridge.lbug is stale. Re-run `gitnexus group sync` to retry.', + ); + } } return { diff --git a/gitnexus/src/core/group/types.ts b/gitnexus/src/core/group/types.ts index 7d0a14251..8e43ff78f 100644 --- a/gitnexus/src/core/group/types.ts +++ b/gitnexus/src/core/group/types.ts @@ -1,4 +1,4 @@ -export type ContractType = 'http' | 'grpc' | 'thrift' | 'topic' | 'lib' | 'custom'; +export type ContractType = 'http' | 'grpc' | 'thrift' | 'topic' | 'lib' | 'custom' | 'include'; export type MatchType = 'exact' | 'manifest' | 'wildcard' | 'bm25' | 'embedding'; export type ContractRole = 'provider' | 'consumer'; @@ -28,6 +28,7 @@ export interface DetectConfig { topics: boolean; shared_libs: boolean; embedding_fallback: boolean; + includes: boolean; workspace_deps: boolean; } diff --git a/gitnexus/src/core/ingestion/call-processor.ts b/gitnexus/src/core/ingestion/call-processor.ts index 6c59578b0..bf0206057 100644 --- a/gitnexus/src/core/ingestion/call-processor.ts +++ b/gitnexus/src/core/ingestion/call-processor.ts @@ -769,9 +769,10 @@ export const processCalls = async ( let tree = astCache.get(file.path); if (!tree) { + const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; try { - tree = parser.parse(file.content, undefined, { - bufferSize: getTreeSitterBufferSize(file.content), + tree = parser.parse(parseContent, undefined, { + bufferSize: getTreeSitterBufferSize(parseContent), }); } catch (parseError) { continue; @@ -3280,9 +3281,10 @@ export const extractFetchCallsFromFiles = async ( let tree = astCache.get(file.path); if (!tree) { + const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; try { - tree = parser.parse(file.content, undefined, { - bufferSize: getTreeSitterBufferSize(file.content), + tree = parser.parse(parseContent, undefined, { + bufferSize: getTreeSitterBufferSize(parseContent), }); } catch { continue; diff --git a/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts b/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts index 23cedb99b..34be6bc03 100644 --- a/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts +++ b/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts @@ -369,9 +369,20 @@ const RE_USE_AFTER = /\bUSE\s+(?:AFTER\s+)?(?:STANDARD\s+)?(?:EXCEPTION|ERROR)\s+ON\s+([A-Z][A-Z0-9-]+|INPUT|OUTPUT|I-O|EXTEND)\b/i; // SET statement (condition, index) -const RE_SET_TO_TRUE = /\bSET\s+((?:[A-Z][A-Z0-9-]+(?:\s+OF\s+[A-Z][A-Z0-9-]+)?\s+)+)TO\s+TRUE\b/i; -const RE_SET_INDEX = - /\bSET\s+((?:[A-Z][A-Z0-9-]+\s+)+)(TO|UP\s+BY|DOWN\s+BY)\s+(\d+|[A-Z][A-Z0-9-]+)/i; +// +// Catastrophic-backtracking note (CodeQL js/redos): the previous shape +// `((?:[A-Z][A-Z0-9-]+(?:\s+OF\s+[A-Z][A-Z0-9-]+)?\s+)+)TO\s+TRUE` +// nested `\s+` quantifiers across alternations and was exponential on +// inputs like "SET a OF a OF a ... TO TRUE". Replaced with a lazy +// dot-match bounded by the explicit `\s+TO\s+TRUE` suffix — `.+?` is +// O(n) with the trailing anchor, and the captured group is parsed +// downstream the same way as before. +// Exported so the U8 ReDoS regression test can pin the exact production +// pattern. Direct import is the only way to ensure the test's +// pathological-input timing assertion exercises the production regex +// instead of an inline copy that drifts. +export const RE_SET_TO_TRUE = /\bSET\s+(.+?)\s+TO\s+TRUE\b/i; +export const RE_SET_INDEX = /\bSET\s+(.+?)\s+(TO|UP\s+BY|DOWN\s+BY)\s+(\d+|[A-Z][A-Z0-9-]+)/i; // INITIALIZE statement — data reset (captures targets before REPLACING/WITH clause) const RE_INITIALIZE = /\bINITIALIZE\s+([\s\S]*?)(?=\bREPLACING\b|\bWITH\b|\.\s*$|$)/i; diff --git a/gitnexus/src/core/ingestion/cpp-ue-preprocessor.ts b/gitnexus/src/core/ingestion/cpp-ue-preprocessor.ts new file mode 100644 index 000000000..baa5d17ff --- /dev/null +++ b/gitnexus/src/core/ingestion/cpp-ue-preprocessor.ts @@ -0,0 +1,265 @@ +/** + * Unreal Engine reflection-macro preprocessor for C++ source. + * + * Tree-sitter does not expand C preprocessor macros, so Unreal's reflection + * markers (`UCLASS(...)`, `UFUNCTION(...)`, `MODULENAME_API`, ...) are parsed + * verbatim. The result is mis-parsed declarations: in `class BRAWLUI_API + * UMyClass : public UObject`, tree-sitter-cpp captures `BRAWLUI_API` as the + * class name and the rest of the declaration becomes structurally wrong. + * + * This module elides those macros from the source text BEFORE tree-sitter + * parses it. Replacement is **length-preserving** (each elided byte becomes + * a space, newlines preserved) so byte offsets and line/column positions + * tree-sitter reports remain identical to the original file. Symbol + * locations in the graph stay accurate. + * + * A cheap detection guard short-circuits files that don't look like UE + * sources, so non-UE C++ codebases pay no cost. + * + * Pure function — no tree-sitter dependency, safe for worker threads. + */ +/** + * Strong UE markers — reflection macros that only Unreal Engine projects use. + * Presence of one of these is sufficient evidence that the file is a UE source + * and that `MODULENAME_API` tokens in it are intended as export macros. + * + * Importantly, `_API` tokens are NOT in this guard — `REST_API`, `HTTP_API`, + * `MY_LIB_API` and similar identifiers appear in plenty of non-UE C++ codebases + * as constants/enums/parameter names. We must not erase them just because the + * file mentions an `_API` token. + */ +const HAS_UE_HINT = + /\b(?:UCLASS|UFUNCTION|UPROPERTY|USTRUCT|UENUM|UINTERFACE|GENERATED_BODY|GENERATED_[A-Z_]+_BODY|UE_DEPRECATED|DECLARE_(?:DYNAMIC_)?(?:MULTICAST_)?DELEGATE)/; + +const SIMPLE_MACROS_NO_ARGS: readonly string[] = [ + 'GENERATED_BODY', + 'GENERATED_UCLASS_BODY', + 'GENERATED_USTRUCT_BODY', + 'GENERATED_UINTERFACE_BODY', + 'GENERATED_IINTERFACE_BODY', + 'DECLARE_CLASS', + 'GENERATED_BODY_LEGACY', +]; + +const PARENTHESIZED_MACROS: readonly string[] = [ + 'UCLASS', + 'UFUNCTION', + 'UPROPERTY', + 'USTRUCT', + 'UENUM', + 'UINTERFACE', + 'UMETA', + 'UE_DEPRECATED', +]; + +const DELEGATE_MACRO_RE = + /\bDECLARE_(?:DYNAMIC_)?(?:MULTICAST_)?DELEGATE(?:_(?:RetVal_OneParam|RetVal_TwoParams|RetVal_ThreeParams|RetVal_FourParams|RetVal_FiveParams|RetVal_SixParams|RetVal_SevenParams|RetVal_EightParams|RetVal_NineParams|RetVal|OneParam|TwoParams|ThreeParams|FourParams|FiveParams|SixParams|SevenParams|EightParams|NineParams|TenParams))?(?=\s*\()/g; + +/** + * Module export tokens like `BRAWLUI_API`, `ENGINE_API`, `COREUOBJECT_API`. + * Pattern: ALL_CAPS identifier ending in `_API`. The leading word boundary + * (`\b`) prevents matching mid-identifier. + */ +const API_MACRO_RE = /\b[A-Z][A-Z0-9_]*_API\b/g; + +/** Replace `[start, end)` of `chars` with spaces, preserving newlines. */ +function eraseRange(chars: string[], start: number, end: number): void { + for (let i = start; i < end; i++) { + if (chars[i] !== '\n' && chars[i] !== '\r') { + chars[i] = ' '; + } + } +} + +/** + * Find the matching close paren for an opening paren at index `openIdx`. + * Returns the index of `)` (inclusive end), or -1 if unbalanced. + * + * Handles nested parens and string/char literals so commas/parens inside + * strings don't throw off the match. Does not attempt to handle raw string + * literals (`R"(...)"`); UE reflection-macro arguments do not use them in + * practice. + */ +function findMatchingParen(source: string, openIdx: number): number { + if (source.charCodeAt(openIdx) !== 0x28) return -1; + let depth = 1; + let i = openIdx + 1; + const len = source.length; + while (i < len && depth > 0) { + const ch = source.charCodeAt(i); + // String literal + if (ch === 0x22) { + i++; + while (i < len) { + const c = source.charCodeAt(i); + if (c === 0x5c) { + i += 2; + continue; + } + if (c === 0x22) { + i++; + break; + } + i++; + } + continue; + } + // Char literal + if (ch === 0x27) { + i++; + while (i < len) { + const c = source.charCodeAt(i); + if (c === 0x5c) { + i += 2; + continue; + } + if (c === 0x27) { + i++; + break; + } + i++; + } + continue; + } + // Line comment + if (ch === 0x2f && source.charCodeAt(i + 1) === 0x2f) { + while (i < len && source.charCodeAt(i) !== 0x0a) i++; + continue; + } + // Block comment + if (ch === 0x2f && source.charCodeAt(i + 1) === 0x2a) { + i += 2; + while (i < len) { + if (source.charCodeAt(i) === 0x2a && source.charCodeAt(i + 1) === 0x2f) { + i += 2; + break; + } + i++; + } + continue; + } + if (ch === 0x28) depth++; + else if (ch === 0x29) { + depth--; + if (depth === 0) return i; + } + i++; + } + return -1; +} + +/** Match a whole-word identifier at `idx`. Returns the byte after the identifier, or -1 on miss. */ +function matchIdentifierAt(source: string, idx: number, name: string): number { + if (idx > 0) { + const prev = source.charCodeAt(idx - 1); + if ( + (prev >= 0x30 && prev <= 0x39) || + (prev >= 0x41 && prev <= 0x5a) || + (prev >= 0x61 && prev <= 0x7a) || + prev === 0x5f + ) { + return -1; + } + } + for (let k = 0; k < name.length; k++) { + if (source.charCodeAt(idx + k) !== name.charCodeAt(k)) return -1; + } + const after = idx + name.length; + if (after < source.length) { + const next = source.charCodeAt(after); + if ( + (next >= 0x30 && next <= 0x39) || + (next >= 0x41 && next <= 0x5a) || + (next >= 0x61 && next <= 0x7a) || + next === 0x5f + ) { + return -1; + } + } + return after; +} + +/** Skip ASCII whitespace forward from `idx`. Returns the next non-whitespace byte index. */ +function skipWhitespace(source: string, idx: number): number { + const len = source.length; + while (idx < len) { + const ch = source.charCodeAt(idx); + if (ch === 0x20 || ch === 0x09 || ch === 0x0a || ch === 0x0d) { + idx++; + continue; + } + break; + } + return idx; +} + +/** + * Strip Unreal Engine reflection macros from C++ source, length-preserving. + * + * Returns the original string unchanged if no strong UE marker is detected, + * so non-UE C++ files (including ones that contain `*_API`-suffixed + * identifiers like `REST_API` or `HTTP_API`) incur only a single regex test. + * + * The `_filePath` parameter is part of the `LanguageProvider.preprocessSource` + * contract but is unused — UE detection is purely content-based. Accepted and + * ignored here so the function matches the hook signature exactly. + */ +export function stripUeMacros(source: string, _filePath?: string): string { + if (!HAS_UE_HINT.test(source)) return source; + + const chars: string[] = source.split(''); + + for (const macro of PARENTHESIZED_MACROS) { + let searchFrom = 0; + while (true) { + const hit = source.indexOf(macro, searchFrom); + if (hit < 0) break; + searchFrom = hit + 1; + const after = matchIdentifierAt(source, hit, macro); + if (after < 0) continue; + const parenIdx = skipWhitespace(source, after); + if (source.charCodeAt(parenIdx) !== 0x28) continue; + const close = findMatchingParen(source, parenIdx); + if (close < 0) continue; + eraseRange(chars, hit, close + 1); + } + } + + for (const macro of SIMPLE_MACROS_NO_ARGS) { + let searchFrom = 0; + while (true) { + const hit = source.indexOf(macro, searchFrom); + if (hit < 0) break; + searchFrom = hit + 1; + const after = matchIdentifierAt(source, hit, macro); + if (after < 0) continue; + const parenIdx = skipWhitespace(source, after); + if (source.charCodeAt(parenIdx) === 0x28) { + const close = findMatchingParen(source, parenIdx); + if (close < 0) continue; + eraseRange(chars, hit, close + 1); + } else { + eraseRange(chars, hit, after); + } + } + } + + for (const re of [DELEGATE_MACRO_RE, API_MACRO_RE]) { + re.lastIndex = 0; + let match: RegExpExecArray | null; + while ((match = re.exec(source)) !== null) { + const start = match.index; + let end = start + match[0].length; + if (re === DELEGATE_MACRO_RE) { + const parenIdx = skipWhitespace(source, end); + if (source.charCodeAt(parenIdx) === 0x28) { + const close = findMatchingParen(source, parenIdx); + if (close >= 0) end = close + 1; + } + } + eraseRange(chars, start, end); + } + } + + return chars.join(''); +} diff --git a/gitnexus/src/core/ingestion/field-extractors/configs/csharp.ts b/gitnexus/src/core/ingestion/field-extractors/configs/csharp.ts index 4d83a6cd7..75032a9d7 100644 --- a/gitnexus/src/core/ingestion/field-extractors/configs/csharp.ts +++ b/gitnexus/src/core/ingestion/field-extractors/configs/csharp.ts @@ -9,6 +9,11 @@ import type { SyntaxNode } from '../../utils/ast-helpers.js'; const CSHARP_VIS = new Set(['public', 'private', 'protected', 'internal']); +const extractCsharpDeclaredType = (typeNode: SyntaxNode): string | undefined => { + if (typeNode.type === 'generic_name') return typeNode.text.trim(); + return extractSimpleTypeName(typeNode) ?? typeNode.text?.trim(); +}; + /** * C# field extraction config. * @@ -53,17 +58,17 @@ export const csharpConfig: FieldExtractionConfig = { const child = node.namedChild(i); if (child?.type === 'variable_declaration') { const typeNode = child.childForFieldName('type'); - if (typeNode) return extractSimpleTypeName(typeNode) ?? typeNode.text?.trim(); + if (typeNode) return extractCsharpDeclaredType(typeNode); // fallback: first child that is a type const first = child.firstNamedChild; if (first && first.type !== 'variable_declarator') { - return extractSimpleTypeName(first) ?? first.text?.trim(); + return extractCsharpDeclaredType(first); } } } // property_declaration: type is first named child const typeNode = node.childForFieldName('type'); - if (typeNode) return extractSimpleTypeName(typeNode) ?? typeNode.text?.trim(); + if (typeNode) return extractCsharpDeclaredType(typeNode); return undefined; }, diff --git a/gitnexus/src/core/ingestion/heritage-processor.ts b/gitnexus/src/core/ingestion/heritage-processor.ts index 2c973ad8e..f8628e651 100644 --- a/gitnexus/src/core/ingestion/heritage-processor.ts +++ b/gitnexus/src/core/ingestion/heritage-processor.ts @@ -219,9 +219,13 @@ export const processHeritage = async ( let tree = astCache.get(file.path); if (!tree) { // Use larger bufferSize for files > 32KB + // Per-language source preprocessor (length-preserving, e.g. UE macro + // stripping for C++). MUST mirror parsing-processor on cache miss so + // re-parses see the same input as the cached AST. + const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; try { - tree = parser.parse(file.content, undefined, { - bufferSize: getTreeSitterBufferSize(file.content), + tree = parser.parse(parseContent, undefined, { + bufferSize: getTreeSitterBufferSize(parseContent), }); } catch (parseError) { // Skip files that can't be parsed @@ -413,9 +417,10 @@ export async function extractExtractedHeritageFromFiles( let tree = astCache.get(file.path); if (!tree) { + const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; try { - tree = parser.parse(file.content, undefined, { - bufferSize: getTreeSitterBufferSize(file.content), + tree = parser.parse(parseContent, undefined, { + bufferSize: getTreeSitterBufferSize(parseContent), }); } catch { continue; diff --git a/gitnexus/src/core/ingestion/import-processor.ts b/gitnexus/src/core/ingestion/import-processor.ts index 03cbed5b7..6f0b40b6f 100644 --- a/gitnexus/src/core/ingestion/import-processor.ts +++ b/gitnexus/src/core/ingestion/import-processor.ts @@ -305,9 +305,10 @@ export const processImports = async ( let wasReparsed = false; if (!tree) { + const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content; try { - tree = parser.parse(file.content, undefined, { - bufferSize: getTreeSitterBufferSize(file.content), + tree = parser.parse(parseContent, undefined, { + bufferSize: getTreeSitterBufferSize(parseContent), }); } catch (parseError) { continue; diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index 12ce0839b..e139cf5f3 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -116,6 +116,39 @@ interface LanguageProviderConfig { * Required for tree-sitter languages; empty string for standalone processors. */ readonly treeSitterQueries: string; + /** + * Optional source-text transform that runs **before** tree-sitter parses the file. + * + * Used to elide language constructs that confuse the grammar without affecting + * source-position fidelity — e.g., Unreal Engine reflection macros (`UCLASS`, + * `UFUNCTION`, `MODULENAME_API`) in C++ headers that prevent the parser from + * recognising class/function names correctly. + * + * **Length / position preservation:** the returned string MUST have the same + * JavaScript `.length` as the input AND preserve every newline (`\n`/`\r`) + * position byte-for-byte. Implementations replace elided characters with + * ASCII spaces while leaving newlines untouched. With this contract: + * + * - tree-sitter's reported `startPosition.row`/`startPosition.column` + * match the original file exactly (line/column come from newline counts) + * - `startIndex`/`endIndex` byte offsets match the original file exactly + * **when the elided range is pure ASCII** (UTF-16 `.length` equals UTF-8 + * byte length only for ASCII). + * + * Implementations targeting languages where elided ranges may contain + * non-ASCII content must therefore preserve byte length, not just `.length`, + * if downstream code uses `startIndex` to slice the original UTF-8 bytes. + * The current C++ UE-macro preprocessor relies on the practical fact that + * UE reflection macros and module-export tokens are ASCII-only. + * + * Must be a pure function — same input always yields the same output. Called + * once per file, on every code path that re-parses (parsing-processor, import + * processor, heritage processor, call processor, parse worker). + * + * Default: undefined (no preprocessing — `file.content` is parsed verbatim). + */ + readonly preprocessSource?: (sourceText: string, filePath: string) => string; + // ── Core (required) ─────────────────────────────────────────────── /** Type extraction: declarations, initializers, for-loop bindings */ readonly typeConfig: LanguageTypeConfig; diff --git a/gitnexus/src/core/ingestion/languages/c-cpp.ts b/gitnexus/src/core/ingestion/languages/c-cpp.ts index a5b3e5729..693e8cec2 100644 --- a/gitnexus/src/core/ingestion/languages/c-cpp.ts +++ b/gitnexus/src/core/ingestion/languages/c-cpp.ts @@ -45,6 +45,7 @@ import { cVariableConfig, cppVariableConfig } from '../variable-extractors/confi import { createCallExtractor } from '../call-extractors/generic.js'; import { cCallConfig, cppCallConfig } from '../call-extractors/configs/c-cpp.js'; import { createHeritageExtractor } from '../heritage-extractors/generic.js'; +import { stripUeMacros } from '../cpp-ue-preprocessor.js'; const C_BUILT_INS: ReadonlySet = new Set([ 'printf', @@ -410,6 +411,7 @@ export const cppProvider = defineLanguage({ }, ] satisfies AstFrameworkPatternConfig[], treeSitterQueries: CPP_QUERIES, + preprocessSource: stripUeMacros, typeConfig: cCppConfig, exportChecker: cCppExportChecker, importResolver: createImportResolver(cppImportConfig), diff --git a/gitnexus/src/core/ingestion/languages/csharp/captures.ts b/gitnexus/src/core/ingestion/languages/csharp/captures.ts index 2f913a14e..e29552e78 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/captures.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/captures.ts @@ -42,6 +42,39 @@ const FUNCTION_NODE_TYPES = [ 'local_function_statement', ] as const; +const BUILTIN_TYPE_NAMES = new Set([ + 'bool', + 'byte', + 'char', + 'decimal', + 'double', + 'float', + 'int', + 'long', + 'object', + 'sbyte', + 'short', + 'string', + 'uint', + 'ulong', + 'ushort', + 'void', +]); + +function shouldEmitReadMember(memberNode: SyntaxNode): boolean { + const parent = memberNode.parent; + if (parent === null) return true; + + switch (parent.type) { + case 'invocation_expression': + return parent.childForFieldName('function')?.id !== memberNode.id; + case 'assignment_expression': + return parent.childForFieldName('left')?.id !== memberNode.id; + default: + return true; + } +} + export function emitCsharpScopeCaptures( sourceText: string, _filePath: string, @@ -94,6 +127,14 @@ export function emitCsharpScopeCaptures( continue; } + if (grouped['@reference.read.member'] !== undefined) { + const anchor = grouped['@reference.read.member']; + const memberNode = findNodeAtRange(tree.rootNode, anchor.range, 'member_access_expression'); + if (memberNode === null || !shouldEmitReadMember(memberNode)) { + continue; + } + } + // Synthesize `this` / `base` receiver type-bindings on every // instance method-like. Tree-sitter can't cleanly express "the // implicit receiver of a non-static member of a class/struct/ @@ -209,9 +250,63 @@ export function emitCsharpScopeCaptures( } } + out.push(...synthesizeGenericTypeArgumentReferences(tree.rootNode)); + return out; } +function synthesizeGenericTypeArgumentReferences(root: SyntaxNode): CaptureMatch[] { + const out: CaptureMatch[] = []; + // Treat all generic type arguments as static type references, including + // declaration signatures and call-site generic instantiations. + visit(root, (node) => { + if (node.type !== 'generic_name') return; + const args = findNamedChild(node, 'type_argument_list'); + if (args === null) return; + + for (const arg of args.namedChildren) { + if (arg === null) continue; + const nameNode = terminalTypeNameNode(arg); + if (nameNode === null) continue; + if (BUILTIN_TYPE_NAMES.has(nameNode.text)) continue; + out.push({ + '@reference.type': nodeToCapture('@reference.type', nameNode), + '@reference.name': nodeToCapture('@reference.name', nameNode), + }); + } + }); + return out; +} + +function terminalTypeNameNode(node: SyntaxNode): SyntaxNode | null { + switch (node.type) { + case 'identifier': + return node; + case 'nullable_type': + return node.firstNamedChild === null ? null : terminalTypeNameNode(node.firstNamedChild); + case 'qualified_name': + return node.lastNamedChild; + case 'generic_name': + return node.childForFieldName('name') ?? node.firstNamedChild; + default: + return null; + } +} + +function findNamedChild(node: SyntaxNode, type: string): SyntaxNode | null { + for (const child of node.namedChildren) { + if (child !== null && child.type === type) return child; + } + return null; +} + +function visit(node: SyntaxNode, cb: (node: SyntaxNode) => void): void { + cb(node); + for (const child of node.namedChildren) { + if (child !== null) visit(child, cb); + } +} + /** C# 12 primary constructor: `class X(a, b) { }` / `record X(a, b)`. * The parameters are a bare `parameter_list` named child of the type * declaration (no `constructor_declaration` node). Emit a synthetic diff --git a/gitnexus/src/core/ingestion/languages/csharp/query.ts b/gitnexus/src/core/ingestion/languages/csharp/query.ts index 11da2d8cd..f299aaafa 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/query.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/query.ts @@ -499,6 +499,12 @@ const CSHARP_SCOPE_QUERY = ` left: (member_access_expression expression: "base" @reference.receiver name: (identifier) @reference.name)) @reference.write.member + +;; References — field/property reads: \`obj.Name\` +;; Emit-side filtering drops call targets and assignment left-hand sides. +(member_access_expression + expression: (_) @reference.receiver + name: (identifier) @reference.name) @reference.read.member `; let _parser: Parser | null = null; diff --git a/gitnexus/src/core/ingestion/parsing-processor.ts b/gitnexus/src/core/ingestion/parsing-processor.ts index 8803ec023..98036fbe8 100644 --- a/gitnexus/src/core/ingestion/parsing-processor.ts +++ b/gitnexus/src/core/ingestion/parsing-processor.ts @@ -371,6 +371,11 @@ const processParsingSequential = async ( isVueSetup = extracted.isSetup; } + // Per-language source-text transform (e.g., UE macro stripping for C++). + // Length-preserving — see LanguageProvider.preprocessSource contract. + parseContent = + getProvider(language).preprocessSource?.(parseContent, file.path) ?? parseContent; + try { await loadLanguage(language, file.path); } catch { diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 5c6712562..4442b5a65 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -1407,6 +1407,11 @@ const processFileGroup = ( isVueSetup = extracted.isSetup; } + // Per-language source-text transform (e.g., UE macro stripping for C++). + // Length-preserving — see LanguageProvider.preprocessSource contract. + parseContent = + getProvider(language).preprocessSource?.(parseContent, file.path) ?? parseContent; + clearCaches(); // Reset memoization before each new file let tree; diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index 616a00cac..20df51ebd 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -301,11 +301,15 @@ export const streamAllCSVsToDisk = async ( 'Template', 'Module', ] as const; + const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType'; const multiLangWriters = new Map(); for (const t of MULTI_LANG_TYPES) { multiLangWriters.set( t, - new BufferedCSVWriter(path.join(csvDir, `${t.toLowerCase()}.csv`), multiLangHeader), + new BufferedCSVWriter( + path.join(csvDir, `${t.toLowerCase()}.csv`), + t === 'Property' ? propertyHeader : multiLangHeader, + ), ); } @@ -478,6 +482,9 @@ export const streamAllCSVsToDisk = async ( escapeCSVNumber(node.properties.endLine, -1), escapeCSVField(content), escapeCSVField(node.properties.description || ''), + ...(node.label === 'Property' + ? [escapeCSVField(node.properties.declaredType || '')] + : []), ].join(','), ); } diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 2fc12cf96..cb88986ca 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -19,7 +19,10 @@ import type { CachedEmbedding } from '../embeddings/types.js'; import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js'; import { closeLbugConnection, + isDbBusyError, + isOpenRetryExhausted, openLbugConnection, + waitForWindowsHandleRelease, type LbugConnectionHandle, } from './lbug-config.js'; import { isVectorExtensionSupportedByPlatform } from '../platform/capabilities.js'; @@ -185,21 +188,6 @@ const DB_LOCK_RETRY_ATTEMPTS = 3; /** Base back-off in ms between BUSY retries (multiplied by attempt number). */ const DB_LOCK_RETRY_DELAY_MS = 500; -/** - * Return true when the error message indicates that another process holds - * an exclusive lock on the LadybugDB file (e.g. `gitnexus analyze` or - * `gitnexus serve` running at the same time). - */ -export const isDbBusyError = (err: unknown): boolean => { - const msg = (err instanceof Error ? err.message : String(err)).toLowerCase(); - return ( - msg.includes('busy') || - msg.includes('lock') || - msg.includes('already in use') || - msg.includes('could not set lock') - ); -}; - /** * Return true when the error message indicates a write was attempted against * a read-only LadybugDB connection. The MCP query pool opens DBs read-only, @@ -252,7 +240,11 @@ export const withLbugDb = async (dbPath: string, operation: () => Promise) }); } catch (err) { lastError = err; - if (!isDbBusyError(err) || attempt === DB_LOCK_RETRY_ATTEMPTS) { + // Skip outer retry when the inner open-retry already exhausted: the + // ~1.5s open-time budget was just spent, repeating the full reset+ + // reopen cycle would only add 4-5s of tail latency without changing + // the outcome (both layers consult the same isDbBusyError matcher). + if (!isDbBusyError(err) || isOpenRetryExhausted(err) || attempt === DB_LOCK_RETRY_ATTEMPTS) { throw err; } // Close stale connection inside the session lock to prevent race conditions @@ -330,7 +322,16 @@ const doInitLbug = async (dbPath: string) => { await conn.query(schemaQuery); } catch (err) { const msg = err instanceof Error ? err.message : String(err); - if (!msg.includes('already exists')) { + // Suppression list: + // - "already exists": expected idempotent re-create on existing DBs + // - "could not set lock on file": LadybugDB v0.16.1 emits this on + // Windows when CREATE NODE TABLE runs against a path that was + // just opened (the WAL handle from a fresh Database briefly + // contests the table's first-write lock). The table is created + // anyway and any genuine cross-process lock contention surfaces + // on the next operation via withLbugDb's retry. Logging it here + // would just be noise in CI. + if (!msg.includes('already exists') && !isDbBusyError(err)) { logger.warn(`⚠️ Schema creation warning: ${msg.slice(0, 120)}`); } } @@ -607,6 +608,9 @@ const getCopyQuery = (table: NodeTableName, filePath: string): string => { if (table === 'Method') { return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content, description, parameterCount, returnType) FROM "${filePath}" ${COPY_CSV_OPTS}`; } + if (table === 'Property') { + return `COPY ${t}(id, name, filePath, startLine, endLine, content, description, declaredType) FROM "${filePath}" ${COPY_CSV_OPTS}`; + } // TypeScript/JS code element tables have isExported; multi-language tables do not if (TABLES_WITH_EXPORTED.has(table)) { return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content, description) FROM "${filePath}" ${COPY_CSV_OPTS}`; @@ -658,6 +662,11 @@ export const insertNodeToLbug = async ( ? `, description: ${escapeValue(properties.description)}` : ''; query = `CREATE (n:${t} {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, isExported: ${!!properties.isExported}, content: ${escapeValue(properties.content || '')}${descPart}})`; + } else if (label === 'Property') { + const descPart = properties.description + ? `, description: ${escapeValue(properties.description)}` + : ''; + query = `CREATE (n:${t} {id: ${escapeValue(properties.id)}, name: ${escapeValue(properties.name)}, filePath: ${escapeValue(properties.filePath)}, startLine: ${properties.startLine || 0}, endLine: ${properties.endLine || 0}, content: ${escapeValue(properties.content || '')}${descPart}, declaredType: ${escapeValue(properties.declaredType || '')}})`; } else { // Multi-language tables (Struct, Impl, Trait, Macro, etc.) — no isExported const descPart = properties.description @@ -736,6 +745,11 @@ export const batchInsertNodesToLbug = async ( ? `, n.description = ${escapeValue(properties.description)}` : ''; query = `MERGE (n:${t} {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.isExported = ${!!properties.isExported}, n.content = ${escapeValue(properties.content || '')}${descPart}`; + } else if (label === 'Property') { + const descPart = properties.description + ? `, n.description = ${escapeValue(properties.description)}` + : ''; + query = `MERGE (n:${t} {id: ${escapeValue(properties.id)}}) SET n.name = ${escapeValue(properties.name)}, n.filePath = ${escapeValue(properties.filePath)}, n.startLine = ${properties.startLine || 0}, n.endLine = ${properties.endLine || 0}, n.content = ${escapeValue(properties.content || '')}${descPart}, n.declaredType = ${escapeValue(properties.declaredType || '')}`; } else { const descPart = properties.description ? `, n.description = ${escapeValue(properties.description)}` @@ -1064,6 +1078,9 @@ export const flushWAL = async (): Promise => { */ export const safeClose = async (): Promise => { await flushWAL(); + // Capture before close — currentDbPath stays set so the Windows post-close + // probe below knows which file to wait on. + const closingDbPath = currentDbPath; if (conn) { try { // eslint-disable-next-line no-restricted-syntax -- sole authorised close site @@ -1082,6 +1099,24 @@ export const safeClose = async (): Promise => { } db = null; } + // Windows: libuv reports `db.close()` resolved before the kernel has + // released the file handle. A subsequent `new Database(samePath)` in + // the same process can race the release. The probe (lbug-config.ts) + // forces any residual lock to surface as EBUSY/EPERM/EACCES so the + // open-time retry absorbs the lag. + if (process.platform === 'win32' && closingDbPath) { + const released = await waitForWindowsHandleRelease(closingDbPath); + if (!released) { + // Probe exhausted with a lock code still in flight. The next + // openLbugConnection will absorb whatever residual lag remains, but + // a chronic warning helps operators spot AV interference (Windows + // Defender holding the file far past the 250ms budget). + logger.warn( + { dbPath: closingDbPath }, + '⚠️ LadybugDB file handle still locked after close (Windows). If this repeats, check antivirus/Defender exclusions for the GitNexus storage directory.', + ); + } + } }; export const closeLbug = async (): Promise => { diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index 5534f7d8b..ceb445693 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -1,3 +1,6 @@ +import fs from 'fs/promises'; +import os from 'os'; +import path from 'path'; import type lbug from '@ladybugdb/core'; /** @@ -42,10 +45,23 @@ export const LBUG_MAX_DB_SIZE: number = (() => { return 16 * 1024 * 1024 * 1024; })(); +/** Matches WAL corruption errors from the LadybugDB engine. */ +const WAL_CORRUPTION_RE = /corrupt(ed)?\s+wal|invalid\s+wal\s+record|wal.*corrupt|checksum.*wal/i; + +export const WAL_RECOVERY_SUGGESTION = + 'WAL corruption detected. Run `gitnexus analyze` to rebuild the index.'; + +export function isWalCorruptionError(err: unknown): boolean { + if (!err) return false; + const msg = err instanceof Error ? err.message : String(err); + return WAL_CORRUPTION_RE.test(msg); +} + type LbugModule = typeof lbug; export interface LbugDatabaseOptions { readOnly?: boolean; + throwOnWalReplayFailure?: boolean; } export interface LbugConnectionHandle { @@ -53,20 +69,200 @@ export interface LbugConnectionHandle { conn: lbug.Connection; } +/** + * Return true when the error message indicates that a LadybugDB file lock + * could not be acquired — either at construction time + * (`new lbug.Database(...)` raises from `local_file_system.cpp`) or during + * a query (another writer holds the exclusive lock). + * + * Lives here (not in `lbug-adapter.ts`) so both the construction-time + * retry (`openWithLockRetry` in this file) and the query-time retry + * (`withLbugDb` in `lbug-adapter.ts`) consult the same matcher. Callers + * import directly from this module — no re-export to keep in sync. + */ +export const isDbBusyError = (err: unknown): boolean => { + const msg = (err instanceof Error ? err.message : String(err)).toLowerCase(); + // `lock` already subsumes `could not set lock`; the broader term is kept + // because graph-DB transient errors include "deadlock", "lock contention", + // and the LadybugDB native module's "could not set lock on file" — all of + // which deserve a retry. If a non-transient lock-shaped error ever + // surfaces (e.g., "lock file missing" during recovery), tighten this + // matcher rather than raising the retry budget. + return msg.includes('busy') || msg.includes('lock') || msg.includes('already in use'); +}; + export function createLbugDatabase( lbugModule: LbugModule, databasePath: string, options: LbugDatabaseOptions = {}, ): lbug.Database { - return new lbugModule.Database( + // .d.ts declares fewer args than the native constructor accepts. + return new (lbugModule.Database as any)( databasePath, - 0, - false, + 0, // bufferManagerSize + false, // enableCompression (pinned for v0.16.0) options.readOnly ?? false, LBUG_MAX_DB_SIZE, - ); + true, // autoCheckpoint + -1, // checkpointThreshold + options.throwOnWalReplayFailure ?? true, + true, // enableChecksums + ) as lbug.Database; } +// ─── Lock-busy retry tuning knobs ─────────────────────────────────────────── +// +// All four GitNexus retry pairs that touch native LadybugDB locks live with +// a comment cross-reference here so an SRE tuning Windows flakes finds them +// in one grep: +// +// 1. OPEN_LOCK_RETRY_ATTEMPTS / OPEN_LOCK_RETRY_DELAY_MS (this file) +// → `new lbug.Database()` constructor lock failures +// 2. HANDLE_RELEASE_PROBE_ATTEMPTS / HANDLE_RELEASE_PROBE_DELAY_MS (this file) +// → post-close fs.open probe to absorb Windows handle-release lag +// 3. DB_LOCK_RETRY_ATTEMPTS / DB_LOCK_RETRY_DELAY_MS (lbug-adapter.ts withLbugDb) +// → query-time busy/lock retry around already-open connections +// +// `new lbug.Database()` calls into the native module which performs an +// OS-level exclusive lock on ``. On Windows that lock can fail +// for reasons specific to the OS (Defender briefly opens new files, +// libuv handle release lags the JS-side close). 5 attempts × 100ms +// linear back-off (max sleep 100+200+300+400 = 1s, plus 5 ctor RTTs +// of 10–50ms each = ~1.0–1.2s worst case) clears the typical +// AV-scanner hold without masking real cross-process conflicts. +// +// Source: https://github.com/LadybugDB/ladybug/blob/v0.16.1/src/common/file_system/local_file_system.cpp#L126 +const OPEN_LOCK_RETRY_ATTEMPTS = 5; +const OPEN_LOCK_RETRY_DELAY_MS = 100; + +const HANDLE_RELEASE_PROBE_ATTEMPTS = 5; +const HANDLE_RELEASE_PROBE_DELAY_MS = 50; +const HANDLE_RELEASE_LOCK_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']); + +/** + * Test-fixture directory prefixes recognized by `isTestFixturePath`. + * + * IMPORTANT: this list must stay in sync with the prefixes passed to + * `createTempDir` in `gitnexus/test/helpers/test-db.ts` and the prefixes + * used by `withTestLbugDB` (`gitnexus/test/helpers/test-indexed-db.ts`). + * If you add a new test that passes a custom prefix to `createTempDir`, + * add it here too — otherwise the stale-sidecar sweep silently won't + * fire for that fixture and CI flakes return. + * + * The default `createTempDir('gitnexus-test-')` and the lbug variant + * `'gitnexus-lbug-'` cover today's call sites. + */ +const TEST_FIXTURE_PREFIXES = ['gitnexus-lbug-', 'gitnexus-test-']; + +/** + * Marker symbol attached to lock errors after `openWithLockRetry` exhausts + * its budget. `withLbugDb`'s outer query-time retry consults this so it + * does not re-retry a path that just spent up to ~1.5s in the open-time + * loop — preventing 6s tail latencies (3× outer × 5× inner attempts). + * + * The symbol is internal to GitNexus; consumers should treat the underlying + * error message as the user-visible signal. + */ +export const LBUG_OPEN_RETRY_EXHAUSTED = Symbol.for('gitnexus.lbug.openRetryExhausted'); + +export const isOpenRetryExhausted = (err: unknown): boolean => { + if (err === null || err === undefined || typeof err !== 'object') return false; + return (err as { [LBUG_OPEN_RETRY_EXHAUSTED]?: boolean })[LBUG_OPEN_RETRY_EXHAUSTED] === true; +}; + +const tagOpenRetryExhausted = (err: unknown): unknown => { + if (err && typeof err === 'object') { + (err as { [LBUG_OPEN_RETRY_EXHAUSTED]?: boolean })[LBUG_OPEN_RETRY_EXHAUSTED] = true; + } + return err; +}; + +/** + * True when `dbPath` resolves to a recognized test fixture under the OS + * temp directory. Used to gate the stale-sidecar sweep so production + * paths never have their `.wal` / `.lock` files deleted. + * + * Defensive shape: + * - `path.resolve` normalizes `..` segments before the prefix check, so + * `/gitnexus-lbug-x/../../etc/passwd` is rejected. + * - The tmpRoot check trims any trailing separator returned by some + * Windows TMP configurations (`C:\Users\X\Temp\`) so the startsWith + * comparison stays correct. + * - Only the IMMEDIATE parent directory is matched against the prefix + * list. An ancestor walk would let a tmpdir whose own basename starts + * with `gitnexus-lbug-` accept arbitrary nested paths under it. + */ +const isTestFixturePath = (dbPath: string): boolean => { + const tmpRoot = os.tmpdir().replace(new RegExp(`${path.sep === '\\' ? '\\\\' : path.sep}+$`), ''); + const resolved = path.resolve(dbPath); + if (!resolved.startsWith(tmpRoot + path.sep) && resolved !== tmpRoot) return false; + const parentBase = path.basename(path.dirname(resolved)); + return TEST_FIXTURE_PREFIXES.some((p) => parentBase.startsWith(p)); +}; + +/** Exported only for direct unit testing — production callers use `openWithLockRetry`. */ +export const _isTestFixturePathForTest = isTestFixturePath; + +const sleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)); + +/** + * Attempt to remove stale `.wal` / `.lock` sidecars that a previous aborted + * test run may have left behind. Best-effort: ENOENT is normal, anything + * else is swallowed so the caller's retry can surface the original error. + */ +const sweepStaleSidecars = async (dbPath: string): Promise => { + for (const suffix of ['.wal', '.lock']) { + try { + await fs.unlink(dbPath + suffix); + } catch { + /* missing sidecar or permission error — let the open retry surface it */ + } + } +}; + +/** + * Run `construct` with bounded retries when `new lbug.Database(...)` throws + * a busy/lock error. The original (loop-captured) error is preferred over + * any post-sweep error so triage sees the real LadybugDB lock message. + * On exhaustion the rethrown error is tagged via + * `LBUG_OPEN_RETRY_EXHAUSTED` so the outer query-time retry in + * `withLbugDb` skips re-retrying a freshly-exhausted path. + */ +const openWithLockRetry = async ( + construct: () => lbug.Database, + dbPath: string, +): Promise => { + let originalLockError: unknown; + for (let attempt = 1; attempt <= OPEN_LOCK_RETRY_ATTEMPTS; attempt++) { + try { + return construct(); + } catch (err) { + if (!isDbBusyError(err)) throw err; + originalLockError = err; + if (attempt === OPEN_LOCK_RETRY_ATTEMPTS) break; + await sleep(OPEN_LOCK_RETRY_DELAY_MS * attempt); + } + } + + // Final defense: only for recognized test fixtures, sweep stale sidecars + // (a prior aborted test run can leave a `.wal` lock that survives the + // tmp dir cleanup). Production paths never reach this branch — the guard + // requires the immediate parent dir to match a test prefix AND the + // resolved path to live under the OS temp directory. + if (isTestFixturePath(dbPath)) { + await sweepStaleSidecars(dbPath); + try { + return construct(); + } catch { + // Intentionally do NOT overwrite originalLockError. The user-actionable + // signal is "we exhausted lock retries" — a different error from the + // post-sweep attempt is less useful than the lock failure that drove + // the sweep in the first place. + } + } + throw tagOpenRetryExhausted(originalLockError); +}; + export async function openLbugConnection( lbugModule: LbugModule, databasePath: string, @@ -74,7 +270,10 @@ export async function openLbugConnection( ): Promise { let db: lbug.Database | undefined; try { - db = createLbugDatabase(lbugModule, databasePath, options); + db = await openWithLockRetry( + () => createLbugDatabase(lbugModule, databasePath, options), + databasePath, + ); return { db, conn: new lbugModule.Connection(db) }; } catch (err) { if (db) await db.close().catch(() => {}); @@ -86,3 +285,60 @@ export async function closeLbugConnection(handle: LbugConnectionHandle): Promise await handle.conn.close().catch(() => {}); await handle.db.close().catch(() => {}); } + +/** + * Probe `dbPath` AND its `.wal` sidecar after `db.close()` so any + * residual native file handle surfaces as EBUSY/EPERM/EACCES and the + * bounded retry absorbs the release lag. Windows-only — Linux/macOS do + * not exhibit this race. + * + * Both files matter. Empirically, on rapid open→close→reopen cycles the + * main `dbPath` handle releases first; the `.wal` handle from the + * previous Database lingers and the new Database's first write (CREATE + * NODE TABLE during schema init) fails with "Could not set lock on + * file". Probing both makes safeClose actually return when the kernel + * is fully done with the path. + * + * Returns `true` when both probes succeeded (or skipped on non-lock + * errors / missing files). Returns `false` when either probe exhausted + * its budget with a lock code still in flight. + * + * Defensive shape: + * - Opens read+write (`'r+'`) so the probe actually surfaces exclusive + * locks held by the previous Database. A read-only probe (`'r'`) is + * insufficient — Windows will grant read access while the previous + * handle's exclusive write lock is still in flight, which lets + * `safeClose` return before the next CREATE NODE TABLE can lock the + * file. + * - `try/finally` around `handle.close()` guarantees no fd leak even + * if close itself throws. + */ +export const waitForWindowsHandleRelease = async (dbPath: string): Promise => { + const mainReleased = await probeSinglePath(dbPath); + const walReleased = await probeSinglePath(dbPath + '.wal'); + return mainReleased && walReleased; +}; + +const probeSinglePath = async (filePath: string): Promise => { + for (let attempt = 1; attempt <= HANDLE_RELEASE_PROBE_ATTEMPTS; attempt++) { + let handle: fs.FileHandle | undefined; + try { + handle = await fs.open(filePath, 'r+'); + return true; + } catch (err) { + const code = (err as NodeJS.ErrnoException | undefined)?.code; + if (!code || !HANDLE_RELEASE_LOCK_CODES.has(code)) return true; // ENOENT / unrelated → not our problem + if (attempt === HANDLE_RELEASE_PROBE_ATTEMPTS) return false; + await sleep(HANDLE_RELEASE_PROBE_DELAY_MS * attempt); + } finally { + if (handle) { + try { + await handle.close(); + } catch { + /* swallow — caller cannot do anything useful with a probe-close failure */ + } + } + } + } + return false; +}; diff --git a/gitnexus/src/core/lbug/pool-adapter.ts b/gitnexus/src/core/lbug/pool-adapter.ts index ca1c45611..ed999907e 100644 --- a/gitnexus/src/core/lbug/pool-adapter.ts +++ b/gitnexus/src/core/lbug/pool-adapter.ts @@ -18,7 +18,7 @@ import fs from 'fs/promises'; import lbug from '@ladybugdb/core'; import { loadFTSExtension } from './lbug-adapter.js'; -import { createLbugDatabase } from './lbug-config.js'; +import { createLbugDatabase, isWalCorruptionError } from './lbug-config.js'; /** Per-repo pool: one Database, many Connections */ interface PoolEntry { @@ -97,7 +97,7 @@ let idleTimer: ReturnType | null = null; // @ladybugdb/core), corrupting stdout in the pre-sentinel window. Routing // through the leaf breaks that chain. export { realStdoutWrite, realStderrWrite, setActiveStdoutWrite } from '../../mcp/stdio-capture.js'; -import { getActiveStdoutWrite } from '../../mcp/stdio-capture.js'; +import { getActiveStdoutWrite, realStderrWrite } from '../../mcp/stdio-capture.js'; let stdoutSilenceCount = 0; /** True while pre-warming connections — prevents watchdog from prematurely restoring stdout */ @@ -263,6 +263,46 @@ const WAITER_TIMEOUT_MS = 15_000; const LOCK_RETRY_ATTEMPTS = 3; const LOCK_RETRY_DELAY_MS = 2000; +async function openReadOnlyDatabase(dbPath: string): Promise { + let db: lbug.Database | undefined; + silenceStdout(); + try { + db = createLbugDatabase(lbug, dbPath, { + readOnly: true, + throwOnWalReplayFailure: false, + }); + await db.init(); + return db; + } catch (err) { + if (db) await db.close().catch(() => {}); + throw err; + } finally { + restoreStdout(); + } +} + +/** + * Quarantine the .wal file and retry opening the database. + * Used when the initial open fails with a WAL corruption error. + */ +async function tryQuarantineAndReopen(dbPath: string, repoId: string): Promise { + const walPath = dbPath + '.wal'; + const quarantineName = `${walPath}.corrupt.${Date.now()}-${Math.random().toString(36).slice(2)}`; + try { + await fs.rename(walPath, quarantineName); + } catch { + throw new Error( + `LadybugDB WAL corruption detected for ${repoId}. ` + + `Run \`gitnexus analyze\` to rebuild the index. (quarantine failed)`, + ); + } + realStderrWrite( + `GitNexus: LadybugDB WAL quarantined for ${repoId}; graph may be stale. ` + + `Run \`gitnexus analyze\` to rebuild the index.\n`, + ); + return await openReadOnlyDatabase(dbPath); +} + /** Deduplicates concurrent initLbug calls for the same repoId */ const initPromises = new Map>(); @@ -319,16 +359,29 @@ async function doInitLbug(repoId: string, dbPath: string): Promise { // avoids lock conflicts when `gitnexus analyze` is writing. let lastError: Error | null = null; for (let attempt = 1; attempt <= LOCK_RETRY_ATTEMPTS; attempt++) { - silenceStdout(); try { - const db = createLbugDatabase(lbug, dbPath, { readOnly: true }); - restoreStdout(); + const db = await openReadOnlyDatabase(dbPath); shared = { db, refCount: 0, ftsLoaded: false }; dbCache.set(dbPath, shared); break; } catch (err: any) { - restoreStdout(); lastError = err instanceof Error ? err : new Error(String(err)); + + if (isWalCorruptionError(lastError)) { + try { + const db = await tryQuarantineAndReopen(dbPath, repoId); + shared = { db, refCount: 0, ftsLoaded: false }; + dbCache.set(dbPath, shared); + break; + } catch (retryErr) { + throw new Error( + `LadybugDB WAL corruption detected for ${repoId}. ` + + `Run \`gitnexus analyze\` to rebuild the index. ` + + `(${retryErr instanceof Error ? retryErr.message : String(retryErr)})`, + ); + } + } + const isLockError = lastError.message.includes('Could not set lock') || lastError.message.includes('lock'); if (!isLockError || attempt === LOCK_RETRY_ATTEMPTS) break; diff --git a/gitnexus/src/core/lbug/schema.ts b/gitnexus/src/core/lbug/schema.ts index 5579036f1..cd6838dd8 100644 --- a/gitnexus/src/core/lbug/schema.ts +++ b/gitnexus/src/core/lbug/schema.ts @@ -167,7 +167,18 @@ export const TYPE_ALIAS_SCHEMA = CODE_ELEMENT_BASE('TypeAlias'); export const CONST_SCHEMA = CODE_ELEMENT_BASE('Const'); export const STATIC_SCHEMA = CODE_ELEMENT_BASE('Static'); export const VARIABLE_SCHEMA = CODE_ELEMENT_BASE('Variable'); -export const PROPERTY_SCHEMA = CODE_ELEMENT_BASE('Property'); +export const PROPERTY_SCHEMA = ` +CREATE NODE TABLE \`Property\` ( + id STRING, + name STRING, + filePath STRING, + startLine INT64, + endLine INT64, + content STRING, + description STRING, + declaredType STRING, + PRIMARY KEY (id) +)`; export const RECORD_SCHEMA = CODE_ELEMENT_BASE('Record'); export const DELEGATE_SCHEMA = CODE_ELEMENT_BASE('Delegate'); export const ANNOTATION_SCHEMA = CODE_ELEMENT_BASE('Annotation'); diff --git a/gitnexus/src/core/search/bm25-index.ts b/gitnexus/src/core/search/bm25-index.ts index 58c0343c9..27a7b9d8d 100644 --- a/gitnexus/src/core/search/bm25-index.ts +++ b/gitnexus/src/core/search/bm25-index.ts @@ -15,9 +15,16 @@ export interface BM25SearchResult { nodeIds?: string[]; } +export interface FTSSearchResponse { + results: BM25SearchResult[]; + /** True when at least one FTS index query succeeded (index exists). */ + ftsAvailable: boolean; +} + /** * Execute a single FTS query via a custom executor (for MCP connection pool). - * Returns the same shape as core queryFTS (from LadybugDB adapter). + * Returns `null` when the query fails (e.g. FTS index does not exist) so the + * caller can distinguish "zero matches" from "index missing". */ async function queryFTSViaExecutor( executor: (cypher: string) => Promise, @@ -25,7 +32,7 @@ async function queryFTSViaExecutor( indexName: string, query: string, limit: number, -): Promise> { +): Promise | null> { // Escape single quotes and backslashes to prevent Cypher injection const escapedQuery = query.replace(/\\/g, '\\\\').replace(/'/g, "''"); const cypher = ` @@ -46,7 +53,7 @@ async function queryFTSViaExecutor( }; }); } catch { - return []; + return null; } } @@ -65,8 +72,9 @@ export const searchFTSFromLbug = async ( query: string, limit: number = 20, repoId?: string, -): Promise => { +): Promise => { const resultsByIndex: any[][] = []; + let queriesSucceeded = 0; if (repoId) { // Use MCP connection pool via dynamic import @@ -77,15 +85,27 @@ export const searchFTSFromLbug = async ( const executor = (cypher: string) => executeQuery(repoId, cypher); for (const { table, indexName } of FTS_INDEXES) { - resultsByIndex.push(await queryFTSViaExecutor(executor, table, indexName, query, limit)); + const result = await queryFTSViaExecutor(executor, table, indexName, query, limit); + if (result !== null) { + queriesSucceeded++; + resultsByIndex.push(result); + } } } else { // Use core lbug adapter (CLI / pipeline context) — also sequential for safety. for (const { table, indexName } of FTS_INDEXES) { - resultsByIndex.push(await queryFTS(table, indexName, query, limit, false).catch(() => [])); + try { + const result = await queryFTS(table, indexName, query, limit, false); + queriesSucceeded++; + resultsByIndex.push(result); + } catch { + // FTS index may not exist — count as failed + } } } + const ftsAvailable = queriesSucceeded > 0; + // Collect all node scores per filePath to track which nodes actually matched const fileNodeScores = new Map>(); @@ -116,10 +136,13 @@ export const searchFTSFromLbug = async ( .sort((a, b) => b.score - a.score) .slice(0, limit); - return sorted.map((r, index) => ({ - filePath: r.filePath, - score: r.score, - rank: index + 1, - nodeIds: r.nodeIds, - })); + return { + results: sorted.map((r, index) => ({ + filePath: r.filePath, + score: r.score, + rank: index + 1, + nodeIds: r.nodeIds, + })), + ftsAvailable, + }; }; diff --git a/gitnexus/src/core/search/hybrid-search.ts b/gitnexus/src/core/search/hybrid-search.ts index 72dd1c9b5..b76a9f5e9 100644 --- a/gitnexus/src/core/search/hybrid-search.ts +++ b/gitnexus/src/core/search/hybrid-search.ts @@ -113,12 +113,13 @@ export const mergeWithRRF = ( }; /** - * Check if hybrid search is available - * LadybugDB FTS is always available once the database is initialized. - * Semantic search is optional - hybrid works with just FTS if embeddings aren't ready. + * Check if hybrid search is available. + * FTS indexes may be missing on read-only MCP connections (see #1403); + * callers should inspect `ftsAvailable` from searchFTSFromLbug for + * per-query availability. This helper is a coarse gate only. */ export const isHybridSearchReady = (): boolean => { - return true; // FTS is always available via LadybugDB when DB is open + return true; // FTS is attempted on every query; ftsAvailable signals actual availability }; /** @@ -160,7 +161,7 @@ export const hybridSearch = async ( ) => Promise, ): Promise => { // Use LadybugDB FTS for always-fresh BM25 results - const bm25Results = await searchFTSFromLbug(query, limit); + const { results: bm25Results } = await searchFTSFromLbug(query, limit); const semanticResults = await semanticSearch(executeQuery, query, limit); return mergeWithRRF(bm25Results, semanticResults, limit); }; diff --git a/gitnexus/src/mcp/core/embedder.ts b/gitnexus/src/mcp/core/embedder.ts index ff3a12fd8..f529f2805 100644 --- a/gitnexus/src/mcp/core/embedder.ts +++ b/gitnexus/src/mcp/core/embedder.ts @@ -12,7 +12,11 @@ import { httpEmbedQuery, } from '../../core/embeddings/http-client.js'; import { resolveEmbeddingConfig } from '../../core/embeddings/config.js'; -import { applyHfEnvOverrides } from '../../core/embeddings/hf-env.js'; +import { + applyHfEnvOverrides, + isHfDownloadFailure, + withHfDownloadRetry, +} from '../../core/embeddings/hf-env.js'; import { silenceStdout, restoreStdout, realStderrWrite } from '../../core/lbug/pool-adapter.js'; import { logger } from '../../core/logger.js'; @@ -69,23 +73,39 @@ export const initEmbedder = async (): Promise => { silenceStdout(); process.stderr.write = (() => true) as any; try { - embedderInstance = await (pipeline as any)('feature-extraction', MODEL_ID, { - device: device, - dtype: 'fp32', - session_options: { - logSeverityLevel: 3, - intraOpNumThreads: embeddingConfig.threads, - interOpNumThreads: 1, - executionMode: 'sequential', - }, - }); + embedderInstance = await withHfDownloadRetry(() => + pipeline('feature-extraction', MODEL_ID, { + device: device, + dtype: 'fp32', + session_options: { + logSeverityLevel: 3, + intraOpNumThreads: embeddingConfig.threads, + interOpNumThreads: 1, + executionMode: 'sequential', + }, + }), + ); } finally { restoreStdout(); process.stderr.write = realStderrWrite; } logger.info({ device }, 'GitNexus: Embedding model loaded'); return embedderInstance!; - } catch { + } catch (deviceError) { + // Network errors and circuit-open errors are not device-specific — + // they will fail the same way on every device. Rethrow immediately + // with actionable HF_ENDPOINT guidance rather than silently falling + // back to the next device. + const errMsg = deviceError instanceof Error ? deviceError.message : String(deviceError); + if (isHfDownloadFailure(errMsg)) { + const endpointHint = process.env.HF_ENDPOINT + ? `The configured endpoint (${process.env.HF_ENDPOINT}) may be unreachable.` + : `huggingface.co may be unreachable from your network.\n` + + ` Set HF_ENDPOINT to a mirror and retry:\n` + + ` HF_ENDPOINT=https://hf-mirror.com npx gitnexus analyze --embeddings\n` + + ` (Windows: set HF_ENDPOINT=https://hf-mirror.com && npx gitnexus analyze --embeddings)`; + throw new Error(`Failed to download embedding model: ${errMsg}\n ${endpointHint}`); + } if (device === 'cpu') throw new Error('Failed to load embedding model'); } } diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index 11f7e61f3..167c1db81 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -16,6 +16,7 @@ import { isLbugReady, isWriteQuery, } from '../../core/lbug/pool-adapter.js'; +import { isWalCorruptionError, WAL_RECOVERY_SUGGESTION } from '../../core/lbug/lbug-config.js'; export { isWriteQuery }; // Embedding imports are lazy (dynamic import) to avoid loading onnxruntime-node // at MCP server startup — crashes on unsupported Node ABI versions (#89) @@ -40,7 +41,7 @@ import { isVectorExtensionSupportedByPlatform, } from '../../core/platform/capabilities.js'; import { PhaseTimer } from '../../core/search/phase-timer.js'; -import { checkStaleness, checkCwdMatch } from '../../core/git-staleness.js'; +import { checkStalenessAsync, checkCwdMatch } from '../../core/git-staleness.js'; import { logger } from '../../core/logger.js'; // AI context generation is CLI-only (gitnexus analyze) // import { generateAIContextFiles } from '../../cli/ai-context.js'; @@ -554,8 +555,15 @@ export class LocalBackend { byRemote.set(h.remoteUrl, list); } - return handles.map((h) => { - const stale = checkStaleness(h.repoPath, h.lastCommit); + // Check staleness for all repos in parallel instead of sequentially. + // Each check spawns an async `git rev-list` — with 200 repos the sync + // variant took ~50 s; parallel async brings it under a second (#1363). + const stalenessResults = await Promise.all( + handles.map((h) => checkStalenessAsync(h.repoPath, h.lastCommit)), + ); + + return handles.map((h, i) => { + const stale = stalenessResults[i]; const selfNorm = norm(h.repoPath); const siblings = h.remoteUrl ? (byRemote.get(h.remoteUrl) ?? []).filter((e) => norm(e.repoPath) !== selfNorm) @@ -971,7 +979,7 @@ export class LocalBackend { timing, ...(!ftsUsed && { warning: - 'FTS extension unavailable - keyword search degraded. Run: gitnexus analyze --force to rebuild indexes.', + 'FTS indexes missing — keyword search degraded. Run: gitnexus analyze --force to rebuild indexes.', }), }; } @@ -985,9 +993,9 @@ export class LocalBackend { limit: number, ): Promise<{ results: any[]; ftsUsed: boolean }> { const { searchFTSFromLbug } = await import('../../core/search/bm25-index.js'); - let bm25Results; + let ftsResponse; try { - bm25Results = await searchFTSFromLbug(query, limit, repo.id); + ftsResponse = await searchFTSFromLbug(query, limit, repo.id); } catch (err: any) { logger.error( { err: err.message }, @@ -996,7 +1004,8 @@ export class LocalBackend { return { results: [], ftsUsed: false }; } - const ftsUsed = bm25Results.length === 0 || bm25Results[0]?.ftsUsed !== false; + const bm25Results = ftsResponse.results; + const ftsUsed = ftsResponse.ftsAvailable; const results: any[] = []; @@ -1218,7 +1227,14 @@ export class LocalBackend { const result = await executeQuery(repo.id, params.query); return result; } catch (err: any) { - return { error: err.message || 'Query failed' }; + const msg = err.message || 'Query failed'; + if (isWalCorruptionError(err)) { + return { + error: msg, + recoverySuggestion: WAL_RECOVERY_SUGGESTION, + }; + } + return { error: msg }; } } @@ -1672,6 +1688,30 @@ export class LocalBackend { kind?: string; include_content?: boolean; }, + ): Promise { + try { + return await this._contextImpl(repo, params); + } catch (err: any) { + const msg = (err instanceof Error ? err.message : String(err)) || 'Context query failed'; + if (isWalCorruptionError(err)) { + return { + error: msg, + recoverySuggestion: WAL_RECOVERY_SUGGESTION, + }; + } + throw err; + } + } + + private async _contextImpl( + repo: RepoHandle, + params: { + name?: string; + uid?: string; + file_path?: string; + kind?: string; + include_content?: boolean; + }, ): Promise { await this.ensureInitialized(repo.id); @@ -1716,12 +1756,13 @@ export class LocalBackend { repo.id, ` MATCH (caller)-[r:CodeRelation]->(n {id: $symId}) - WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'HAS_PROPERTY', 'METHOD_OVERRIDES', 'OVERRIDES', 'METHOD_IMPLEMENTS', 'ACCESSES'] + WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'USES', 'HAS_METHOD', 'HAS_PROPERTY', 'METHOD_OVERRIDES', 'OVERRIDES', 'METHOD_IMPLEMENTS', 'ACCESSES'] RETURN r.type AS relType, caller.id AS uid, caller.name AS name, caller.filePath AS filePath, labels(caller)[0] AS kind LIMIT 30 `, { symId }, ); + let typedPropertyRows: any[] = []; // Fix #480: Class/Interface nodes have no direct CALLS/IMPORTS edges — // those point to Constructor and File nodes respectively. Fetch those @@ -1755,23 +1796,24 @@ export class LocalBackend { if (isClassLike) { try { - // Run both incoming-ref queries in parallel — they are independent. - const [ctorIncoming, fileIncoming] = await Promise.all([ - executeParameterized( - repo.id, - ` + // Run incoming-ref queries in parallel — they are independent. + const [ctorIncoming, fileIncoming, typedPropertyIncoming, typedProperties] = + await Promise.all([ + executeParameterized( + repo.id, + ` MATCH (n)-[hm:CodeRelation]->(ctor:Constructor) WHERE n.id = $symId AND hm.type = 'HAS_METHOD' MATCH (caller)-[r:CodeRelation]->(ctor) - WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'ACCESSES'] + WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'USES', 'ACCESSES'] RETURN r.type AS relType, caller.id AS uid, caller.name AS name, caller.filePath AS filePath, labels(caller)[0] AS kind LIMIT 30 `, - { symId }, - ), - executeParameterized( - repo.id, - ` + { symId }, + ), + executeParameterized( + repo.id, + ` MATCH (f:File)-[rel:CodeRelation]->(n) WHERE n.id = $symId AND rel.type = 'DEFINES' MATCH (caller)-[r:CodeRelation]->(f) @@ -1779,9 +1821,45 @@ export class LocalBackend { RETURN r.type AS relType, caller.id AS uid, caller.name AS name, caller.filePath AS filePath, labels(caller)[0] AS kind LIMIT 30 `, - { symId }, - ), - ]); + { symId }, + ), + executeParameterized( + repo.id, + ` + MATCH (p:\`Property\`) + WHERE p.declaredType = $name + OR p.declaredType STARTS WITH $genericPrefix + OR p.declaredType CONTAINS $genericArg + MATCH (caller)-[r:CodeRelation]->(p) + WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'USES', 'ACCESSES'] + RETURN r.type AS relType, caller.id AS uid, caller.name AS name, caller.filePath AS filePath, labels(caller)[0] AS kind + LIMIT 30 + `, + { + name: sym.name, + genericPrefix: `${sym.name}<`, + genericArg: `<${sym.name}>`, + }, + ), + executeParameterized( + repo.id, + ` + MATCH (p:\`Property\`) + WHERE p.declaredType = $name + OR p.declaredType STARTS WITH $genericPrefix + OR p.declaredType CONTAINS $genericArg + RETURN p.id AS uid, p.name AS name, p.filePath AS filePath, labels(p)[0] AS kind, + p.declaredType AS declaredType + LIMIT 30 + `, + { + name: sym.name, + genericPrefix: `${sym.name}<`, + genericArg: `<${sym.name}>`, + }, + ), + ]); + typedPropertyRows = typedProperties; // Deduplicate by (relType, uid) — a caller can have multiple relation // types to the same target (e.g. both IMPORTS and CALLS), and each @@ -1789,7 +1867,7 @@ export class LocalBackend { const seenKeys = new Set( incomingRows.map((r: any) => `${r.relType || r[0]}:${r.uid || r[1]}`), ); - for (const r of [...ctorIncoming, ...fileIncoming]) { + for (const r of [...ctorIncoming, ...fileIncoming, ...typedPropertyIncoming]) { const key = `${r.relType || r[0]}:${r.uid || r[1]}`; if (!seenKeys.has(key)) { seenKeys.add(key); @@ -1806,7 +1884,7 @@ export class LocalBackend { repo.id, ` MATCH (n {id: $symId})-[r:CodeRelation]->(target) - WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'HAS_METHOD', 'HAS_PROPERTY', 'METHOD_OVERRIDES', 'OVERRIDES', 'METHOD_IMPLEMENTS', 'ACCESSES'] + WHERE r.type IN ['CALLS', 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', 'USES', 'HAS_METHOD', 'HAS_PROPERTY', 'METHOD_OVERRIDES', 'OVERRIDES', 'METHOD_IMPLEMENTS', 'ACCESSES'] RETURN r.type AS relType, target.id AS uid, target.name AS name, target.filePath AS filePath, labels(target)[0] AS kind LIMIT 30 `, @@ -1895,6 +1973,17 @@ export class LocalBackend { }, incoming: categorize(incomingRows), outgoing: categorize(outgoingRows), + ...(typedPropertyRows.length > 0 + ? { + typed_properties: typedPropertyRows.map((r: any) => ({ + uid: r.uid || r[0], + name: r.name || r[1], + filePath: r.filePath || r[2], + kind: r.kind || r[3], + declaredType: r.declaredType || r[4], + })), + } + : {}), processes: processRows.map((r: any) => ({ id: r.pid || r[0], name: r.label || r[1], @@ -2433,6 +2522,7 @@ export class LocalBackend { impactedCount: 0, risk: 'UNKNOWN', suggestion: 'The graph query failed — try gitnexus context as a fallback', + ...(isWalCorruptionError(err) ? { recoverySuggestion: WAL_RECOVERY_SUGGESTION } : {}), }; } } @@ -2459,6 +2549,7 @@ export class LocalBackend { const mappedRelTypes = params.relationTypes?.flatMap((t: string) => t === 'OVERRIDES' ? ['OVERRIDES', 'METHOD_OVERRIDES'] : [t], ); + const hasExplicitRelationTypes = mappedRelTypes !== undefined && mappedRelTypes.length > 0; const rawRelTypes = mappedRelTypes && mappedRelTypes.length > 0 ? mappedRelTypes.filter((t: string) => VALID_RELATION_TYPES.has(t)) @@ -2467,6 +2558,7 @@ export class LocalBackend { 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', + 'USES', 'METHOD_OVERRIDES', 'OVERRIDES', 'METHOD_IMPLEMENTS', @@ -2479,6 +2571,7 @@ export class LocalBackend { 'IMPORTS', 'EXTENDS', 'IMPLEMENTS', + 'USES', 'METHOD_OVERRIDES', 'OVERRIDES', 'METHOD_IMPLEMENTS', @@ -2538,9 +2631,16 @@ export class LocalBackend { }; const symType = outcome.resolvedLabel || outcome.symbol.type || ''; + const effectiveRelationTypes = + (symType === 'Class' || symType === 'Interface') && + !hasExplicitRelationTypes && + !relationTypes.includes('ACCESSES') + ? [...relationTypes, 'ACCESSES'] + : relationTypes; + return this._runImpactBFS(repo, sym, symType, direction, { maxDepth, - relationTypes, + relationTypes: effectiveRelationTypes, includeTests, minConfidence, }); @@ -2619,6 +2719,30 @@ export class LocalBackend { frontier.push(rid); } } + + const typedPropertyRows = await executeParameterized( + repo.id, + ` + MATCH (p:\`Property\`) + WHERE p.declaredType = $name + OR p.declaredType STARTS WITH $genericPrefix + OR p.declaredType CONTAINS $genericArg + RETURN p.id AS id, p.name AS name, labels(p)[0] AS type, p.filePath AS filePath + `, + { + name: sym.name, + genericPrefix: `${sym.name}<`, + genericArg: `<${sym.name}>`, + }, + ); + + for (const r of typedPropertyRows) { + const rid = r.id || r[0]; + if (rid && !visited.has(rid)) { + visited.add(rid); + frontier.push(rid); + } + } } catch (e) { logQueryError('impact:class-node-expansion', e); } @@ -2984,8 +3108,14 @@ export class LocalBackend { relationTypes: string[]; minConfidence: number; includeTests: boolean; + signal?: AbortSignal; }, ): Promise { + // Honor an already-aborted signal at the entry boundary as a fast + // path. Cooperative cancellation inside _runImpactBFS is out of + // scope — the caller's Promise.race against the same signal + // resolves the await regardless of how long this body runs. + if (opts.signal?.aborted) return null; try { await this.refreshRepos(); await this.ensureInitialized(repoId); diff --git a/gitnexus/src/server/api.ts b/gitnexus/src/server/api.ts index 773190785..cc65daa3d 100644 --- a/gitnexus/src/server/api.ts +++ b/gitnexus/src/server/api.ts @@ -1060,11 +1060,12 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => const results = await withLbugDb(lbugPath, async () => { let searchResults: any[]; + let ftsAvailable: boolean | undefined; if (mode === 'semantic') { const { isEmbedderReady } = await import('../core/embeddings/embedder.js'); if (!isEmbedderReady()) { - return [] as any[]; + return { searchResults: [] as any[], ftsAvailable: undefined }; } const { semanticSearch: semSearch } = await import('../core/embeddings/embedding-pipeline.js'); @@ -1077,8 +1078,9 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => sources: ['semantic'], })); } else if (mode === 'bm25') { - searchResults = await searchFTSFromLbug(query, limit); - searchResults = searchResults.map((r: any, i: number) => ({ + const ftsResponse = await searchFTSFromLbug(query, limit); + ftsAvailable = ftsResponse.ftsAvailable; + searchResults = ftsResponse.results.map((r: any, i: number) => ({ ...r, rank: i + 1, sources: ['bm25'], @@ -1091,11 +1093,13 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => await import('../core/embeddings/embedding-pipeline.js'); searchResults = await hybridSearch(query, limit, executeQuery, semSearch); } else { - searchResults = await searchFTSFromLbug(query, limit); + const ftsResponse = await searchFTSFromLbug(query, limit); + ftsAvailable = ftsResponse.ftsAvailable; + searchResults = ftsResponse.results; } } - if (!enrich) return searchResults; + if (!enrich) return { searchResults, ftsAvailable }; // Server-side enrichment: add connections, cluster, processes per result // Uses parameterized queries to prevent Cypher injection via nodeId @@ -1177,9 +1181,14 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => }), ); - return enriched; + return { searchResults: enriched, ftsAvailable }; }); - res.json({ results }); + const response: any = { results: results.searchResults ?? results }; + if (results.ftsAvailable === false) { + response.warning = + 'FTS indexes missing — keyword search degraded. Run: gitnexus analyze --force to rebuild indexes.'; + } + res.json(response); } catch (err: any) { res.status(500).json({ error: err.message || 'Search failed' }); } diff --git a/gitnexus/test/fixtures/lang-resolution/csharp-generic-type-refs/Program.cs b/gitnexus/test/fixtures/lang-resolution/csharp-generic-type-refs/Program.cs new file mode 100644 index 000000000..28674ecda --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/csharp-generic-type-refs/Program.cs @@ -0,0 +1,18 @@ +namespace App; + +public class USER_INFO +{ + public string? USER_ID { get; set; } +} + +public interface IEntityTypeConfiguration +{ +} + +public class UserInfoConfiguration : IEntityTypeConfiguration +{ + public Task> Load(List users) + { + return Task.FromResult(users); + } +} diff --git a/gitnexus/test/helpers/test-db.ts b/gitnexus/test/helpers/test-db.ts index 5818fdc8e..37032ebc8 100644 --- a/gitnexus/test/helpers/test-db.ts +++ b/gitnexus/test/helpers/test-db.ts @@ -37,6 +37,13 @@ export async function cleanupTempDir(tmpDir: string): Promise { /** * Create a temporary directory for LadybugDB tests. * Returns the path and a cleanup function. + * + * IMPORTANT: when adding a new test that passes a custom `prefix`, also add + * the prefix to `TEST_FIXTURE_PREFIXES` in + * `gitnexus/src/core/lbug/lbug-config.ts`. The stale-sidecar sweep relies + * on the prefix list to recognize test fixtures; an unknown prefix means + * the sweep silently won't fire for that fixture and Windows CI flakes + * return. */ export async function createTempDir(prefix: string = 'gitnexus-test-'): Promise { const tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), prefix)); diff --git a/gitnexus/test/integration/context-typed-property.test.ts b/gitnexus/test/integration/context-typed-property.test.ts new file mode 100644 index 000000000..aef442147 --- /dev/null +++ b/gitnexus/test/integration/context-typed-property.test.ts @@ -0,0 +1,77 @@ +/** + * Integration test: context() expands Class symbols through typed properties. + * + * Reproduces EF-style usage where code reads a DbContext property + * (`db.USER_INFO`) whose source type is `DbSet`. The direct + * graph edge is Method -> Property, not Method -> Class, so context() must + * use the same typed-property bridge that impact() uses. + */ +import { beforeAll, describe, expect, it, vi } from 'vitest'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', () => ({ + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), +})); + +const SEED = [ + `CREATE (c:Class {id:'Class:Models/USER_INFO.cs:USER_INFO', name:'USER_INFO', filePath:'Models/USER_INFO.cs', startLine:1, endLine:5, content:'public class USER_INFO {}', description:''})`, + `CREATE (p:\`Property\` {id:'Property:Data/UserDbContext.cs:UserDbContext.USER_INFO', name:'USER_INFO', filePath:'Data/UserDbContext.cs', startLine:10, endLine:10, content:'public DbSet USER_INFO { get; set; }', description:'', declaredType:'DbSet'})`, + `CREATE (m:Method {id:'Method:Services/UserService.cs:UserService.GetUserInfo#1', name:'GetUserInfo', filePath:'Services/UserService.cs', startLine:20, endLine:30, isExported:false, content:'db.USER_INFO.FirstOrDefault();', description:'', parameterCount:1, returnType:'USER_INFO'})`, + `MATCH (m:Method {id:'Method:Services/UserService.cs:UserService.GetUserInfo#1'}), (p:\`Property\` {id:'Property:Data/UserDbContext.cs:UserDbContext.USER_INFO'}) CREATE (m)-[:CodeRelation {type:'ACCESSES', confidence:1.0, reason:'read', step:1}]->(p)`, +]; + +withTestLbugDB( + 'context-typed-property', + (handle) => { + let backend: LocalBackend; + + beforeAll(async () => { + backend = (handle as any)._backend; + }); + + describe('context() typed-property expansion', () => { + it('surfaces property callers and explains the typed property bridge', async () => { + const result = await backend.callTool('context', { + uid: 'Class:Models/USER_INFO.cs:USER_INFO', + }); + + expect(result.status).toBe('found'); + expect(result.symbol.kind).toBe('Class'); + + const accesses = result.incoming.accesses || []; + expect(accesses.map((r: any) => r.name)).toContain('GetUserInfo'); + + expect(result.typed_properties).toEqual([ + expect.objectContaining({ + uid: 'Property:Data/UserDbContext.cs:UserDbContext.USER_INFO', + name: 'USER_INFO', + declaredType: 'DbSet', + }), + ]); + }); + }); + }, + { + seed: SEED, + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'test-repo', + path: '/test/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'abc123', + stats: { files: 3, nodes: 3, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/group/include-extractor-sync.test.ts b/gitnexus/test/integration/group/include-extractor-sync.test.ts new file mode 100644 index 000000000..908664bdc --- /dev/null +++ b/gitnexus/test/integration/group/include-extractor-sync.test.ts @@ -0,0 +1,195 @@ +/** + * Integration test: IncludeExtractor output → group matching → bridge DB. + * + * Covers PR #1156 review finding #7: verifies that the full runtime path + * (IncludeExtractor → StoredContract → runExactMatch → CrossLinks → writeBridge) + * stays wired up. A regression in either normalizeContractId or the include + * branch of ManifestExtractor.resolveSymbol would produce 0 cross-links and + * fail this test. + */ +import { describe, it, expect } from 'vitest'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { parseGroupConfig } from '../../../src/core/group/config-parser.js'; +import { syncGroup } from '../../../src/core/group/sync.js'; +import type { StoredContract } from '../../../src/core/group/types.js'; +import { IncludeExtractor } from '../../../src/core/group/extractors/include-extractor.js'; +import { normalizeContractId } from '../../../src/core/group/matching.js'; + +const GROUP_YAML = [ + 'version: 1', + 'name: include-test-group', + 'description: "IncludeExtractor integration test"', + '', + 'repos:', + ' app/provider: include-provider', + ' app/consumer: include-consumer', + '', + 'links: []', + 'packages: {}', + '', + 'detect:', + ' http: false', + ' grpc: false', + ' topics: false', + ' shared_libs: false', + ' includes: true', + ' embedding_fallback: false', + '', + 'matching:', + ' bm25_threshold: 0.7', + ' embedding_threshold: 0.65', + ' max_candidates_per_step: 3', +].join('\n'); + +describe('IncludeExtractor → syncGroup integration (finding #7)', () => { + it('produces a CrossLink when provider and consumer emit the same include contract-id', async () => { + const config = parseGroupConfig(GROUP_YAML); + + // Mock the IncludeExtractor output directly — a header provider in one + // repo and a quoted #include consumer in the other, both normalized to + // the same include::map/base/view.h contract-id. + const mockContracts: StoredContract[] = [ + { + contractId: 'include::map/base/view.h', + type: 'include', + role: 'provider', + symbolUid: 'File:map/base/view.h', + symbolRef: { filePath: 'map/base/view.h', name: 'view.h' }, + symbolName: 'view.h', + confidence: 0.95, + meta: { source: 'filesystem' }, + repo: 'app/provider', + }, + { + contractId: 'include::map/base/view.h', + type: 'include', + role: 'consumer', + symbolUid: 'File:src/controller.cpp', + symbolRef: { filePath: 'src/controller.cpp', name: 'map/base/view.h' }, + symbolName: 'map/base/view.h', + confidence: 0.85, + meta: { source: 'tree_sitter', includePath: 'map/base/view.h' }, + repo: 'app/consumer', + }, + ]; + + const result = await syncGroup(config, { + extractorOverride: async () => mockContracts, + skipWrite: true, + }); + + const includeLinks = result.crossLinks.filter((l) => l.type === 'include'); + expect(includeLinks.length).toBeGreaterThanOrEqual(1); + + const link = includeLinks[0]; + expect(link.contractId).toBe('include::map/base/view.h'); + expect(link.matchType).toBe('exact'); + expect(link.from.repo).toBe('app/consumer'); + expect(link.to.repo).toBe('app/provider'); + }); + + it('normalizes mixed-case / backslash include paths to the same contract-id end-to-end', async () => { + const config = parseGroupConfig(GROUP_YAML); + + // Provider writes the canonical form; consumer's include has mixed case + // and a backslash. After normalizeContractId they must still match. + const providerId = 'include::map/base/view.h'; + const rawConsumerId = 'include::Map\\Base\\View.h'; + + // Sanity — normalizeContractId must collapse them. + expect(normalizeContractId(rawConsumerId)).toBe(providerId); + + const mockContracts: StoredContract[] = [ + { + contractId: providerId, + type: 'include', + role: 'provider', + symbolUid: 'File:map/base/view.h', + symbolRef: { filePath: 'map/base/view.h', name: 'view.h' }, + symbolName: 'view.h', + confidence: 0.95, + meta: { source: 'filesystem' }, + repo: 'app/provider', + }, + { + contractId: rawConsumerId, + type: 'include', + role: 'consumer', + symbolUid: 'File:src/controller.cpp', + symbolRef: { filePath: 'src/controller.cpp', name: 'Map/Base/View.h' }, + symbolName: 'Map/Base/View.h', + confidence: 0.85, + meta: { source: 'tree_sitter', includePath: 'Map\\Base\\View.h' }, + repo: 'app/consumer', + }, + ]; + + const result = await syncGroup(config, { + extractorOverride: async () => mockContracts, + skipWrite: true, + }); + + const includeLinks = result.crossLinks.filter((l) => l.type === 'include'); + expect(includeLinks.length).toBeGreaterThanOrEqual(1); + }); + + it('round-trip: extractor output from two real temp repos produces matching contract-ids', async () => { + // Drives the extractor directly (no `syncGroup`) against two on-disk + // fixture repos, then hands the StoredContract-shaped output to + // syncGroup via extractorOverride. This exercises the real extraction + // code + the matching pipeline together. + const providerDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-include-int-provider-')); + const consumerDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-include-int-consumer-')); + try { + fs.mkdirSync(path.join(providerDir, 'shared/api'), { recursive: true }); + fs.writeFileSync( + path.join(providerDir, 'shared/api/client.h'), + '#pragma once\nstruct Client {};', + ); + fs.mkdirSync(path.join(consumerDir, 'src'), { recursive: true }); + fs.writeFileSync( + path.join(consumerDir, 'src/main.cpp'), + '#include "shared/api/client.h"\nint main(){return 0;}', + ); + + const extractor = new IncludeExtractor(); + const providerOutput = await extractor.extract(null, providerDir, { + id: 'provider', + path: 'app/provider', + repoPath: providerDir, + storagePath: path.join(providerDir, '.gitnexus'), + }); + const consumerOutput = await extractor.extract(null, consumerDir, { + id: 'consumer', + path: 'app/consumer', + repoPath: consumerDir, + storagePath: path.join(consumerDir, '.gitnexus'), + }); + + const stored: StoredContract[] = [ + ...providerOutput + .filter((c) => c.role === 'provider') + .map((c) => ({ ...c, repo: 'app/provider' })), + ...consumerOutput + .filter((c) => c.role === 'consumer') + .map((c) => ({ ...c, repo: 'app/consumer' })), + ]; + + const config = parseGroupConfig(GROUP_YAML); + const result = await syncGroup(config, { + extractorOverride: async () => stored, + skipWrite: true, + }); + + const includeLinks = result.crossLinks.filter((l) => l.type === 'include'); + expect(includeLinks.length).toBeGreaterThanOrEqual(1); + expect(includeLinks[0].contractId).toBe('include::shared/api/client.h'); + expect(includeLinks[0].matchType).toBe('exact'); + } finally { + fs.rmSync(providerDir, { recursive: true, force: true }); + fs.rmSync(consumerDir, { recursive: true, force: true }); + } + }); +}); diff --git a/gitnexus/test/integration/lbug-close-handle-release.test.ts b/gitnexus/test/integration/lbug-close-handle-release.test.ts new file mode 100644 index 000000000..c0a3b8758 --- /dev/null +++ b/gitnexus/test/integration/lbug-close-handle-release.test.ts @@ -0,0 +1,41 @@ +/** + * Integration test: safeClose's Windows post-close handle-release wait. + * + * On Windows, libuv reports `db.close()` resolved before the kernel has + * released the file handle. A subsequent open of the same path can then + * race the release and surface "Could not set lock on file". `safeClose` + * probes the file with `fs.open` to force the residual lock to surface, + * absorbed by the open-time retry in `lbug-config.ts`. + */ +import path from 'path'; +import { describe, it } from 'vitest'; +import { createTempDir } from '../helpers/test-db.js'; + +describe('safeClose — close + reopen does not surface lock errors', () => { + it('survives 10 sequential open/close/reopen cycles on the same path', async () => { + const tmp = await createTempDir('gitnexus-lbug-close-cycle-'); + const dbPath = path.join(tmp.dbPath, 'lbug'); + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + for (let i = 0; i < 10; i++) { + await adapter.initLbug(dbPath); + await adapter.closeLbug(); + } + } finally { + await tmp.cleanup(); + } + }); + + it('safeClose is idempotent — calling twice in a row does not throw', async () => { + const tmp = await createTempDir('gitnexus-lbug-idempotent-'); + const dbPath = path.join(tmp.dbPath, 'lbug'); + try { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + await adapter.initLbug(dbPath); + await adapter.closeLbug(); + await adapter.closeLbug(); + } finally { + await tmp.cleanup(); + } + }); +}); diff --git a/gitnexus/test/integration/lbug-lock-retry.test.ts b/gitnexus/test/integration/lbug-lock-retry.test.ts index 1279c77fb..b95092e2d 100644 --- a/gitnexus/test/integration/lbug-lock-retry.test.ts +++ b/gitnexus/test/integration/lbug-lock-retry.test.ts @@ -14,7 +14,7 @@ import { withTestLbugDB } from '../helpers/test-indexed-db.js'; // Pure-function tests — no DB needed, but grouped here for cohesion // with the retry logic they guard. -import { isDbBusyError } from '../../src/core/lbug/lbug-adapter.js'; +import { isDbBusyError } from '../../src/core/lbug/lbug-config.js'; describe('isDbBusyError', () => { it('returns true for "busy" errors (case-insensitive)', () => { @@ -46,6 +46,18 @@ describe('isDbBusyError', () => { expect(isDbBusyError(undefined)).toBe(false); }); + // Documented behavior for lock-shaped strings: the matcher is intentionally + // broad because in graph-DB contexts these are all transient. If LadybugDB + // ever surfaces a non-transient lock-shaped error (e.g., a recovery-time + // "lock file missing"), tighten the matcher and add a negative test here + // rather than raising the retry budget. + it('treats other lock-shaped errors as transient (current intentional behavior)', () => { + expect(isDbBusyError(new Error('deadlock detected'))).toBe(true); + expect(isDbBusyError(new Error('unlock failed'))).toBe(true); + expect(isDbBusyError(new Error('lock contention'))).toBe(true); + expect(isDbBusyError(new Error('Could not open lock file'))).toBe(true); + }); + it('handles non-Error values gracefully', () => { expect(isDbBusyError('BUSY error')).toBe(true); expect(isDbBusyError(42)).toBe(false); diff --git a/gitnexus/test/integration/lbug-open-retry.test.ts b/gitnexus/test/integration/lbug-open-retry.test.ts new file mode 100644 index 000000000..80e68617d --- /dev/null +++ b/gitnexus/test/integration/lbug-open-retry.test.ts @@ -0,0 +1,310 @@ +/** + * Integration tests: open-time lock-busy retry in `lbug-config.ts`. + * + * The lock IO exception raised by `local_file_system.cpp` happens + * synchronously inside `new lbug.Database(...)`, before any query is + * issued — so `withLbugDb`'s query-time retry cannot see it. These tests + * exercise the construction-time retry wrapper directly by stubbing the + * `Database` constructor. + * + * See: docs/plans/2026-05-08-002-fix-windows-lbug-lock-ci-flakes-plan.md + */ +import fs from 'fs/promises'; +import os from 'os'; +import path from 'path'; +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { + _isTestFixturePathForTest as isTestFixturePath, + isDbBusyError, + isOpenRetryExhausted, + openLbugConnection, + waitForWindowsHandleRelease, +} from '../../src/core/lbug/lbug-config.js'; + +// ─── Minimal stub of the `lbug` module surface used by openLbugConnection ── + +interface StubModuleControl { + /** Errors thrown by sequential `new Database(...)` calls. `null` = success. */ + databaseThrows: Array; + /** Number of times the `Database` constructor was invoked. */ + databaseCallCount: number; + /** Number of times `db.close()` was called. */ + closeCallCount: number; +} + +const makeStubLbug = (control: StubModuleControl) => { + class FakeDatabase { + constructor(_path: string, ..._rest: unknown[]) { + control.databaseCallCount++; + const next = control.databaseThrows.shift(); + if (next instanceof Error) throw next; + } + async close(): Promise { + control.closeCallCount++; + } + } + class FakeConnection { + constructor(_db: FakeDatabase) {} + async close(): Promise {} + } + return { Database: FakeDatabase, Connection: FakeConnection } as any; +}; + +describe('isDbBusyError', () => { + it('matches the documented Windows lock-error wording', () => { + expect(isDbBusyError(new Error('Could not set lock on file foo.lbug'))).toBe(true); + expect(isDbBusyError(new Error('database is locked'))).toBe(true); + }); + it('does not match unrelated errors', () => { + expect(isDbBusyError(new Error('Cypher syntax error'))).toBe(false); + expect(isDbBusyError(null)).toBe(false); + }); +}); + +describe('openLbugConnection — open-time lock-busy retry', () => { + it('returns a handle when the constructor succeeds on the first try', async () => { + const control: StubModuleControl = { + databaseThrows: [null], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + const handle = await openLbugConnection(stub, '/some/path/lbug'); + expect(handle.db).toBeDefined(); + expect(handle.conn).toBeDefined(); + expect(control.databaseCallCount).toBe(1); + }); + + it('retries on busy/lock errors and succeeds on a later attempt', async () => { + const control: StubModuleControl = { + databaseThrows: [new Error('Could not set lock on file'), null], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + const handle = await openLbugConnection(stub, '/some/path/lbug'); + expect(handle.db).toBeDefined(); + expect(control.databaseCallCount).toBe(2); + }); + + it('exhausts the retry budget and rethrows the last error preserving its message', async () => { + const lockErr = new Error('Could not set lock on file foo.lbug'); + const control: StubModuleControl = { + // 5 attempts + production paths get no sweep retry, so 5 throws total. + databaseThrows: [lockErr, lockErr, lockErr, lockErr, lockErr], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + await expect(openLbugConnection(stub, '/var/data/non-test/lbug')).rejects.toThrow( + 'Could not set lock on file foo.lbug', + ); + expect(control.databaseCallCount).toBe(5); + }); + + it('tags the exhausted error so withLbugDb skips its outer retry', async () => { + const lockErr = new Error('Could not set lock on file'); + const control: StubModuleControl = { + databaseThrows: [lockErr, lockErr, lockErr, lockErr, lockErr], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + let caught: unknown; + try { + await openLbugConnection(stub, '/var/data/non-test/lbug'); + } catch (err) { + caught = err; + } + expect(caught).toBeDefined(); + expect(isOpenRetryExhausted(caught)).toBe(true); + expect(isOpenRetryExhausted(new Error('plain error'))).toBe(false); + expect(isOpenRetryExhausted(null)).toBe(false); + expect(isOpenRetryExhausted(undefined)).toBe(false); + }); + + it('does not retry non-busy errors', async () => { + const syntaxErr = new Error('Cypher syntax error'); + const control: StubModuleControl = { + databaseThrows: [syntaxErr], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + await expect(openLbugConnection(stub, '/some/path/lbug')).rejects.toThrow( + 'Cypher syntax error', + ); + expect(control.databaseCallCount).toBe(1); + }); +}); + +describe('openLbugConnection — stale-sidecar sweep (test fixtures only)', () => { + let fixtureDir: string; + let dbPath: string; + + beforeEach(async () => { + fixtureDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-lbug-sweep-')); + dbPath = path.join(fixtureDir, 'lbug'); + }); + + afterEach(async () => { + await fs.rm(fixtureDir, { recursive: true, force: true }).catch(() => {}); + }); + + it('sweeps stale .wal/.lock for a recognized test fixture path and retries once', async () => { + await fs.writeFile(dbPath + '.wal', 'stale'); + await fs.writeFile(dbPath + '.lock', 'stale'); + + const lockErr = new Error('Could not set lock on file'); + const control: StubModuleControl = { + // 5 retries throw, then sweep + 1 final attempt succeeds (6 total). + databaseThrows: [lockErr, lockErr, lockErr, lockErr, lockErr, null], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + + const handle = await openLbugConnection(stub, dbPath); + expect(handle.db).toBeDefined(); + expect(control.databaseCallCount).toBe(6); + + // Sidecars removed by the sweep + await expect(fs.access(dbPath + '.wal')).rejects.toThrow(); + await expect(fs.access(dbPath + '.lock')).rejects.toThrow(); + }); + + it('does not sweep production paths even if they share the prefix', async () => { + // A non-tmp dir that *starts* with the prefix must still be rejected. + const lockErr = new Error('Could not set lock on file'); + const control: StubModuleControl = { + databaseThrows: [lockErr, lockErr, lockErr, lockErr, lockErr], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + + // Path is outside os.tmpdir() so the predicate must reject it. + await expect(openLbugConnection(stub, '/var/data/gitnexus-lbug-fake/lbug')).rejects.toThrow( + 'Could not set lock on file', + ); + expect(control.databaseCallCount).toBe(5); // no sweep retry + }); + + it('handles missing sidecars gracefully (ENOENT swallowed, retry runs)', async () => { + // No .wal or .lock pre-created — sweep ENOENTs both, then succeeds. + const lockErr = new Error('Could not set lock on file'); + const control: StubModuleControl = { + databaseThrows: [lockErr, lockErr, lockErr, lockErr, lockErr, null], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + + const handle = await openLbugConnection(stub, dbPath); + expect(handle.db).toBeDefined(); + expect(control.databaseCallCount).toBe(6); + }); + + it('sweep retry that throws a different error preserves the original lock error', async () => { + // 5 lock errors, then sweep fires, then post-sweep throws an unrelated + // error. The user-actionable signal is "lock retries exhausted" — the + // post-sweep error must NOT shadow the original lock message. + const lockErr = new Error('Could not set lock on file foo.lbug'); + const unrelatedErr = new Error('Schema validation error during open'); + const control: StubModuleControl = { + databaseThrows: [lockErr, lockErr, lockErr, lockErr, lockErr, unrelatedErr], + databaseCallCount: 0, + closeCallCount: 0, + }; + const stub = makeStubLbug(control); + + let caught: Error | undefined; + try { + await openLbugConnection(stub, dbPath); + } catch (err) { + caught = err as Error; + } + expect(caught?.message).toBe('Could not set lock on file foo.lbug'); + expect(control.databaseCallCount).toBe(6); // sweep retry did fire + }); +}); + +describe('isTestFixturePath — production-safety guard', () => { + it('accepts a fixture under os.tmpdir with a recognized prefix on the immediate parent', () => { + const tmp = os.tmpdir(); + expect(isTestFixturePath(path.join(tmp, 'gitnexus-lbug-XXX', 'lbug'))).toBe(true); + expect(isTestFixturePath(path.join(tmp, 'gitnexus-test-YYY', 'lbug'))).toBe(true); + }); + + it('rejects production paths even with a matching prefix', () => { + expect(isTestFixturePath('/var/data/gitnexus-lbug-fake/lbug')).toBe(false); + expect(isTestFixturePath('/home/user/gitnexus-test-foo/lbug')).toBe(false); + }); + + it('rejects path traversal attempts that resolve outside tmpdir', () => { + const tmp = os.tmpdir(); + const traversal = path.join(tmp, 'gitnexus-lbug-x', '..', '..', 'etc', 'passwd'); + expect(isTestFixturePath(traversal)).toBe(false); + }); + + it('rejects when the immediate parent does not match even if a deeper ancestor does', () => { + // Tightening: ancestor walk would have allowed nested paths under + // `/gitnexus-lbug-x/inner/lbug` to satisfy the predicate. We + // require the immediate parent to match. + const tmp = os.tmpdir(); + expect(isTestFixturePath(path.join(tmp, 'gitnexus-lbug-x', 'inner', 'lbug'))).toBe(false); + }); + + it('handles tmpdir trailing-separator gracefully', () => { + // Some Windows TMP configs return a trailing separator; the predicate + // strips it before the prefix check so fixtures still match. + const tmp = os.tmpdir(); + const fixture = path.join(tmp, 'gitnexus-lbug-trailing', 'lbug'); + // Whether or not os.tmpdir() itself has a trailing separator, + // the predicate must accept legit fixtures. + expect(isTestFixturePath(fixture)).toBe(true); + }); + + it('rejects unrelated prefixes in tmpdir', () => { + const tmp = os.tmpdir(); + expect(isTestFixturePath(path.join(tmp, 'random-dir', 'lbug'))).toBe(false); + expect(isTestFixturePath(path.join(tmp, 'malicious', 'lbug'))).toBe(false); + }); +}); + +describe('waitForWindowsHandleRelease', () => { + let fixtureDir: string; + let dbPath: string; + + beforeEach(async () => { + fixtureDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-lbug-probe-')); + dbPath = path.join(fixtureDir, 'lbug'); + }); + + afterEach(async () => { + await fs.rm(fixtureDir, { recursive: true, force: true }).catch(() => {}); + }); + + it('returns true when the file exists and is openable', async () => { + await fs.writeFile(dbPath, 'fake-db-content'); + const released = await waitForWindowsHandleRelease(dbPath); + expect(released).toBe(true); + }); + + it('returns true when the file does not exist (ENOENT is non-lock)', async () => { + // No fs.writeFile — path does not exist. Probe should bail to true, + // not retry, since ENOENT is not a lock code. + const released = await waitForWindowsHandleRelease(dbPath); + expect(released).toBe(true); + }); + + it('does not leak the file handle when close succeeds', async () => { + // Smoke test: 50 sequential probes with a real file. If close were + // skipped, fd usage would climb. We rely on test process not OOMing + // as the simplest indicator; fd table caps catch egregious leaks. + await fs.writeFile(dbPath, 'fake-db-content'); + for (let i = 0; i < 50; i++) { + await waitForWindowsHandleRelease(dbPath); + } + }); +}); diff --git a/gitnexus/test/integration/resolvers/csharp.test.ts b/gitnexus/test/integration/resolvers/csharp.test.ts index 613ff7ec2..073b2da16 100644 --- a/gitnexus/test/integration/resolvers/csharp.test.ts +++ b/gitnexus/test/integration/resolvers/csharp.test.ts @@ -1440,6 +1440,24 @@ describe('Write access tracking (C#)', () => { }); }); +// --------------------------------------------------------------------------- +// Generic type references: IEntityTypeConfiguration, List +// --------------------------------------------------------------------------- + +describe('C# generic type-reference tracking', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'csharp-generic-type-refs'), () => {}); + }, 60000); + + it('emits USES edges for generic type arguments', () => { + const uses = getRelationships(result, 'USES').filter((e) => e.target === 'USER_INFO'); + expect(edgeSet(uses)).toContain('UserInfoConfiguration → USER_INFO'); + expect(edgeSet(uses)).toContain('Load → USER_INFO'); + }); +}); + // --------------------------------------------------------------------------- // Call-result variable binding (Phase 9): var user = GetUser(); user.Save() // --------------------------------------------------------------------------- diff --git a/gitnexus/test/integration/resolvers/helpers.ts b/gitnexus/test/integration/resolvers/helpers.ts index 5368c1715..e92526b19 100644 --- a/gitnexus/test/integration/resolvers/helpers.ts +++ b/gitnexus/test/integration/resolvers/helpers.ts @@ -11,6 +11,9 @@ import type { GraphRelationship } from 'gitnexus-shared'; const LEGACY_RESOLVER_PARITY_EXPECTED_FAILURES: Readonly>> = { csharp: new Set([ 'emits the using-import edge App/Program.cs -> Models/User.cs through the scope-resolution path', + // Generic type-argument USES edges are emitted by the registry-primary + // resolver only; the legacy DAG path does not synthesize these references. + 'emits USES edges for generic type arguments', ]), go: new Set([ // The legacy DAG path does not resolve method calls when the method is diff --git a/gitnexus/test/integration/search-core.test.ts b/gitnexus/test/integration/search-core.test.ts index 2ccc14706..e49169b30 100644 --- a/gitnexus/test/integration/search-core.test.ts +++ b/gitnexus/test/integration/search-core.test.ts @@ -19,7 +19,7 @@ withTestLbugDB( (_handle) => { describe('searchFTSFromLbug — core adapter (no repoId)', () => { it('returns ranked results for a matching query', async () => { - const results = await searchFTSFromLbug('user authentication', 10); + const { results } = await searchFTSFromLbug('user authentication', 10); expect(results.length).toBeGreaterThan(0); @@ -40,7 +40,7 @@ withTestLbugDB( }); it('results are ordered by descending score', async () => { - const results = await searchFTSFromLbug('user authentication', 10); + const { results } = await searchFTSFromLbug('user authentication', 10); for (let i = 1; i < results.length; i++) { expect(results[i - 1].score).toBeGreaterThanOrEqual(results[i].score); @@ -48,7 +48,7 @@ withTestLbugDB( }); it('auth-related files rank higher than unrelated files', async () => { - const results = await searchFTSFromLbug('user authentication', 10); + const { results } = await searchFTSFromLbug('user authentication', 10); const filePaths = results.map((r) => r.filePath); expect(filePaths).toContain('src/auth.ts'); @@ -61,7 +61,7 @@ withTestLbugDB( }); it('merges scores from multiple node types for the same filePath', async () => { - const results = await searchFTSFromLbug('user authentication', 20); + const { results } = await searchFTSFromLbug('user authentication', 20); const authResult = results.find((r) => r.filePath === 'src/auth.ts'); expect(authResult).toBeDefined(); @@ -73,12 +73,12 @@ withTestLbugDB( }); it('respects limit parameter', async () => { - const results = await searchFTSFromLbug('user authentication', 2); + const { results } = await searchFTSFromLbug('user authentication', 2); expect(results.length).toBeLessThanOrEqual(2); }); it('returns empty array for a non-matching query', async () => { - const results = await searchFTSFromLbug('xyzzyplughtwisty', 10); + const { results } = await searchFTSFromLbug('xyzzyplughtwisty', 10); expect(results).toEqual([]); }); }); @@ -87,32 +87,32 @@ withTestLbugDB( describe('unhappy paths', () => { it('returns empty array for empty query string', async () => { - const results = await searchFTSFromLbug('', 10); + const { results } = await searchFTSFromLbug('', 10); expect(results).toEqual([]); }); it('returns empty array for whitespace-only query', async () => { - const results = await searchFTSFromLbug(' ', 10); + const { results } = await searchFTSFromLbug(' ', 10); expect(results).toEqual([]); }); it('handles special characters in query gracefully', async () => { - const results = await searchFTSFromLbug('user* OR auth+', 10); + const { results } = await searchFTSFromLbug('user* OR auth+', 10); expect(Array.isArray(results)).toBe(true); }); it('handles limit of 0', async () => { - const results = await searchFTSFromLbug('user authentication', 0); + const { results } = await searchFTSFromLbug('user authentication', 0); expect(results).toEqual([]); }); it('handles negative limit gracefully', async () => { - const results = await searchFTSFromLbug('user authentication', -1); + const { results } = await searchFTSFromLbug('user authentication', -1); expect(Array.isArray(results)).toBe(true); }); it('handles very large limit', async () => { - const results = await searchFTSFromLbug('user authentication', 100000); + const { results } = await searchFTSFromLbug('user authentication', 100000); expect(results.length).toBeLessThanOrEqual(100000); expect(results.length).toBeGreaterThan(0); }); diff --git a/gitnexus/test/integration/search-pool.test.ts b/gitnexus/test/integration/search-pool.test.ts index c0943483b..88128fcfa 100644 --- a/gitnexus/test/integration/search-pool.test.ts +++ b/gitnexus/test/integration/search-pool.test.ts @@ -19,7 +19,7 @@ withTestLbugDB( (handle) => { describe('searchFTSFromLbug — MCP pool adapter (with repoId)', () => { it('returns ranked results via pool adapter', async () => { - const results = await searchFTSFromLbug('user authentication', 10, handle.repoId); + const { results } = await searchFTSFromLbug('user authentication', 10, handle.repoId); expect(results.length).toBeGreaterThan(0); @@ -35,7 +35,7 @@ withTestLbugDB( }); it('results are ordered by descending score via pool adapter', async () => { - const results = await searchFTSFromLbug('user authentication', 10, handle.repoId); + const { results } = await searchFTSFromLbug('user authentication', 10, handle.repoId); for (let i = 1; i < results.length; i++) { expect(results[i - 1].score).toBeGreaterThanOrEqual(results[i].score); @@ -43,12 +43,12 @@ withTestLbugDB( }); it('returns empty array for non-matching query via pool adapter', async () => { - const results = await searchFTSFromLbug('xyzzyplughtwisty', 10, handle.repoId); + const { results } = await searchFTSFromLbug('xyzzyplughtwisty', 10, handle.repoId); expect(results).toEqual([]); }); it('respects limit parameter via pool adapter', async () => { - const results = await searchFTSFromLbug('user authentication', 1, handle.repoId); + const { results } = await searchFTSFromLbug('user authentication', 1, handle.repoId); expect(results.length).toBeLessThanOrEqual(1); }); }); @@ -57,22 +57,22 @@ withTestLbugDB( describe('unhappy paths', () => { it('returns empty array for empty query via pool', async () => { - const results = await searchFTSFromLbug('', 10, handle.repoId); + const { results } = await searchFTSFromLbug('', 10, handle.repoId); expect(results).toEqual([]); }); it('returns empty array for whitespace-only query via pool', async () => { - const results = await searchFTSFromLbug(' ', 10, handle.repoId); + const { results } = await searchFTSFromLbug(' ', 10, handle.repoId); expect(results).toEqual([]); }); it('handles special characters in query via pool', async () => { - const results = await searchFTSFromLbug('user* OR auth+', 10, handle.repoId); + const { results } = await searchFTSFromLbug('user* OR auth+', 10, handle.repoId); expect(Array.isArray(results)).toBe(true); }); it('handles limit of 0 via pool', async () => { - const results = await searchFTSFromLbug('user authentication', 0, handle.repoId); + const { results } = await searchFTSFromLbug('user authentication', 0, handle.repoId); expect(results).toEqual([]); }); }); diff --git a/gitnexus/test/unit/bm25-search.test.ts b/gitnexus/test/unit/bm25-search.test.ts index 9b232688f..03a591599 100644 --- a/gitnexus/test/unit/bm25-search.test.ts +++ b/gitnexus/test/unit/bm25-search.test.ts @@ -42,20 +42,24 @@ describe('BM25 search', () => { }); describe('searchFTSFromLbug', () => { - it('returns empty array when LadybugDB is not initialized', async () => { - // Without LadybugDB init, search should return empty (not crash) - const results = await searchFTSFromLbug('test query'); + it('returns empty results when LadybugDB is not initialized', async () => { + // Simulate an uninitialized DB: queryFTS throws instead of returning rows + const { queryFTS } = await import('../../src/core/lbug/lbug-adapter.js'); + vi.mocked(queryFTS).mockRejectedValue(new Error('DB not initialized')); + + const { results, ftsAvailable } = await searchFTSFromLbug('test query'); expect(Array.isArray(results)).toBe(true); expect(results).toHaveLength(0); + expect(ftsAvailable).toBe(false); }); it('handles empty query', async () => { - const results = await searchFTSFromLbug(''); + const { results } = await searchFTSFromLbug(''); expect(Array.isArray(results)).toBe(true); }); it('accepts custom limit parameter', async () => { - const results = await searchFTSFromLbug('test', 5); + const { results } = await searchFTSFromLbug('test', 5); expect(Array.isArray(results)).toBe(true); }); }); @@ -105,7 +109,7 @@ describe('BM25 search', () => { .mockResolvedValueOnce([]) // Method .mockResolvedValueOnce([]); // Interface - const results = await searchFTSFromLbug('queryset'); + const { results } = await searchFTSFromLbug('queryset'); expect(results).toHaveLength(1); expect(results[0].filePath).toBe('src/views.py'); @@ -127,7 +131,7 @@ describe('BM25 search', () => { .mockResolvedValueOnce([]) // Method .mockResolvedValueOnce([]); // Interface - const results = await searchFTSFromLbug('model'); + const { results } = await searchFTSFromLbug('model'); expect(results).toHaveLength(1); expect(results[0].score).toBe(8); // 5+3 @@ -147,7 +151,7 @@ describe('BM25 search', () => { .mockResolvedValueOnce([]) // Method .mockResolvedValueOnce([]); // Interface - const results = await searchFTSFromLbug('util'); + const { results } = await searchFTSFromLbug('util'); expect(results).toHaveLength(1); expect(results[0].nodeIds).toEqual([]); @@ -171,7 +175,7 @@ describe('BM25 search', () => { .mockResolvedValueOnce([]) // Method .mockResolvedValueOnce([]); // Interface - const results = await searchFTSFromLbug('auth'); + const { results } = await searchFTSFromLbug('auth'); expect(results).toHaveLength(1); // All 3 hits (scores 9+7+4=20) — each from a different table, all top-3 @@ -192,7 +196,7 @@ describe('BM25 search', () => { .mockResolvedValueOnce([]) // Method .mockResolvedValueOnce([]); // Interface - const results = await searchFTSFromLbug('fn'); + const { results } = await searchFTSFromLbug('fn'); expect(results[0].filePath).toBe('src/high.py'); expect(results[1].filePath).toBe('src/low.py'); @@ -220,7 +224,7 @@ describe('BM25 search', () => { return []; }); - const results = await searchFTSFromLbug('login', 5, REPO); + const { results } = await searchFTSFromLbug('login', 5, REPO); expect(results).toEqual([ { filePath: 'src/auth.ts', score: 8, rank: 1, nodeIds: ['func:login'] }, diff --git a/gitnexus/test/unit/calltool-dispatch.test.ts b/gitnexus/test/unit/calltool-dispatch.test.ts index f8d94d890..45e1b71d2 100644 --- a/gitnexus/test/unit/calltool-dispatch.test.ts +++ b/gitnexus/test/unit/calltool-dispatch.test.ts @@ -48,6 +48,7 @@ vi.mock('../../src/storage/repo-manager.js', () => ({ // tests don't shell out to git. vi.mock('../../src/core/git-staleness.js', () => ({ checkStaleness: vi.fn().mockReturnValue({ isStale: false, commitsBehind: 0 }), + checkStalenessAsync: vi.fn().mockResolvedValue({ isStale: false, commitsBehind: 0 }), checkCwdMatch: vi.fn().mockResolvedValue({ match: 'none' }), })); @@ -61,7 +62,7 @@ vi.mock('../../src/core/platform/capabilities.js', async (importOriginal) => { // Also mock the search modules to avoid loading onnxruntime vi.mock('../../src/core/search/bm25-index.js', () => ({ - searchFTSFromLbug: vi.fn().mockResolvedValue([]), + searchFTSFromLbug: vi.fn().mockResolvedValue({ results: [], ftsAvailable: true }), })); vi.mock('../../src/mcp/core/embedder.js', () => ({ @@ -194,6 +195,27 @@ describe('LocalBackend.callTool', () => { expect(result).toHaveProperty('definitions'); }); + it('includes FTS-unavailable warning when ftsAvailable is false (#1403)', async () => { + const { searchFTSFromLbug } = await import('../../src/core/search/bm25-index.js'); + vi.mocked(searchFTSFromLbug).mockResolvedValueOnce({ results: [], ftsAvailable: false }); + (executeParameterized as any).mockResolvedValue([]); + + const result = await backend.callTool('query', { query: 'ProcessActivity' }); + + expect(result).toHaveProperty('warning'); + expect((result as any).warning).toMatch(/gitnexus analyze --force/); + }); + + it('does not include warning when ftsAvailable is true with zero results', async () => { + const { searchFTSFromLbug } = await import('../../src/core/search/bm25-index.js'); + vi.mocked(searchFTSFromLbug).mockResolvedValueOnce({ results: [], ftsAvailable: true }); + (executeParameterized as any).mockResolvedValue([]); + + const result = await backend.callTool('query', { query: 'nonexistent' }); + + expect(result).not.toHaveProperty('warning'); + }); + it('skips vector index query when VECTOR is unsupported by the platform', async () => { const cap = _captureLogger(); platformMocks.isVectorExtensionSupportedByPlatform.mockReturnValue(false); diff --git a/gitnexus/test/unit/cli-commands.test.ts b/gitnexus/test/unit/cli-commands.test.ts index a059a37bc..a42afb25c 100644 --- a/gitnexus/test/unit/cli-commands.test.ts +++ b/gitnexus/test/unit/cli-commands.test.ts @@ -10,6 +10,9 @@ vi.mock('../../src/cli/mcp.js', () => ({ vi.mock('../../src/cli/setup.js', () => ({ setupCommand: vi.fn(), })); +vi.mock('../../src/cli/publish.js', () => ({ + publishCommand: vi.fn(), +})); describe('CLI commands', () => { describe('version', () => { @@ -84,4 +87,11 @@ describe('CLI commands', () => { expect(typeof setupCommand).toBe('function'); }); }); + + describe('publishCommand', () => { + it('is a function', async () => { + const { publishCommand } = await import('../../src/cli/publish.js'); + expect(typeof publishCommand).toBe('function'); + }); + }); }); diff --git a/gitnexus/test/unit/cli-index-help.test.ts b/gitnexus/test/unit/cli-index-help.test.ts index 59109c8d9..889a7473f 100644 --- a/gitnexus/test/unit/cli-index-help.test.ts +++ b/gitnexus/test/unit/cli-index-help.test.ts @@ -63,4 +63,17 @@ describe('CLI help surface', () => { expect(result.stdout).toContain('--model '); expect(result.stdout).toContain('--gist'); }); + + it('publish help names the registry, the token env var, and the opt-out behaviour', () => { + const result = runHelp('publish'); + + expect(result.status).toBe(0); + expect(result.stdout).toContain('--id '); + expect(result.stdout).toContain('--skip-git'); + // Discoverability contract: a contributor scanning `--help` must see + // (a) which registry this dispatches to, and (b) the env var that + // gates the opt-in. Both are part of the no-token contract. + expect(result.stdout).toContain('understand-quickly'); + expect(result.stdout).toContain('UNDERSTAND_QUICKLY_TOKEN'); + }); }); diff --git a/gitnexus/test/unit/cpp-ue-preprocessor.test.ts b/gitnexus/test/unit/cpp-ue-preprocessor.test.ts new file mode 100644 index 000000000..087c5ba2a --- /dev/null +++ b/gitnexus/test/unit/cpp-ue-preprocessor.test.ts @@ -0,0 +1,272 @@ +import { describe, it, expect } from 'vitest'; +import Parser from 'tree-sitter'; +import CPP from 'tree-sitter-cpp'; +import { stripUeMacros } from '../../src/core/ingestion/cpp-ue-preprocessor.js'; + +describe('stripUeMacros — detection guard', () => { + it('returns input unchanged when no UE markers are present', () => { + const src = `class Plain {\npublic:\n int Get() const;\n};`; + expect(stripUeMacros(src)).toBe(src); + }); + + it('returns input unchanged for STL-style code', () => { + const src = `#include \nstd::vector v;`; + expect(stripUeMacros(src)).toBe(src); + }); +}); + +describe('stripUeMacros — length preservation', () => { + const ueSamples: string[] = [ + `UCLASS()\nclass BRAWLUI_API UMyClass : public UObject { GENERATED_BODY() public: UFUNCTION() void Run(); };`, + `UPROPERTY(EditAnywhere, BlueprintReadOnly, Category = "Combat") int32 Health;`, + `USTRUCT(BlueprintType)\nstruct ENGINE_API FMyData { GENERATED_BODY() float Value; };`, + `DECLARE_DYNAMIC_MULTICAST_DELEGATE_TwoParams(FMyDelegate, int32, A, FString, B);`, + `UE_DEPRECATED(5.0, "Use NewThing instead") void OldThing();`, + ]; + + for (const src of ueSamples) { + it(`preserves byte length: ${src.slice(0, 40).replace(/\n/g, '\\n')}…`, () => { + const out = stripUeMacros(src); + expect(out.length).toBe(src.length); + }); + + it(`preserves newline positions: ${src.slice(0, 40).replace(/\n/g, '\\n')}…`, () => { + const out = stripUeMacros(src); + const inputNewlines: number[] = []; + const outputNewlines: number[] = []; + for (let i = 0; i < src.length; i++) { + if (src.charCodeAt(i) === 0x0a) inputNewlines.push(i); + if (out.charCodeAt(i) === 0x0a) outputNewlines.push(i); + } + expect(outputNewlines).toEqual(inputNewlines); + }); + } +}); + +describe('stripUeMacros — macro removal', () => { + it('elides UCLASS(...) with arguments', () => { + const src = `UCLASS(BlueprintType, Category="Foo")\nclass UFoo {};`; + const out = stripUeMacros(src); + expect(out).not.toContain('UCLASS'); + expect(out).not.toContain('BlueprintType'); + expect(out).toContain('class UFoo {};'); + }); + + it('elides UCLASS() with empty parens', () => { + const src = `UCLASS()\nclass UBar {};`; + const out = stripUeMacros(src); + expect(out).not.toContain('UCLASS'); + expect(out).toContain('class UBar {};'); + }); + + it('elides MODULE_API export macros (BRAWLUI_API style) when paired with a UE marker', () => { + const src = `UCLASS()\nclass BRAWLUI_API UMyClass : public UObject {};`; + const out = stripUeMacros(src); + expect(out).not.toContain('BRAWLUI_API'); + expect(out).toContain('class'); + expect(out).toContain('UMyClass'); + expect(out).toContain('public UObject'); + }); + + it('elides multiple distinct *_API tokens in same file when UE marker is present', () => { + const src = `UCLASS()\nclass CORE_API A {};\nUCLASS()\nclass UMG_API B : public A {};`; + const out = stripUeMacros(src); + expect(out).not.toContain('CORE_API'); + expect(out).not.toContain('UMG_API'); + expect(out).toContain('class'); + expect(out).toContain('A {};'); + }); + + it('elides GENERATED_BODY() inside class body', () => { + const src = `class UThing { GENERATED_BODY() public: void Foo(); };`; + const out = stripUeMacros(src); + expect(out).not.toContain('GENERATED_BODY'); + expect(out).toContain('public:'); + expect(out).toContain('void Foo();'); + }); + + it('elides UFUNCTION(...) before method declarations', () => { + const src = `class X { UFUNCTION(BlueprintCallable, Server, Reliable) void DoThing(); };`; + const out = stripUeMacros(src); + expect(out).not.toContain('UFUNCTION'); + expect(out).not.toContain('BlueprintCallable'); + expect(out).toContain('void DoThing();'); + }); + + it('elides UPROPERTY(...) before field declarations', () => { + const src = `class X { UPROPERTY(EditAnywhere) int32 Health; };`; + const out = stripUeMacros(src); + expect(out).not.toContain('UPROPERTY'); + expect(out).not.toContain('EditAnywhere'); + expect(out).toContain('int32 Health;'); + }); + + it('elides DECLARE_DYNAMIC_MULTICAST_DELEGATE_*Params(...)', () => { + const src = `DECLARE_DYNAMIC_MULTICAST_DELEGATE_OneParam(FMyDelegate, int32, Value);\nclass X {};`; + const out = stripUeMacros(src); + expect(out).not.toContain('DECLARE_DYNAMIC_MULTICAST_DELEGATE'); + expect(out).not.toContain('FMyDelegate'); + expect(out).toContain('class X {};'); + }); + + it('elides UE_DEPRECATED(...) before function declarations', () => { + const src = `UE_DEPRECATED(5.1, "Reason") void Old();`; + const out = stripUeMacros(src); + expect(out).not.toContain('UE_DEPRECATED'); + expect(out).not.toContain('5.1'); + expect(out).toContain('void Old();'); + }); +}); + +describe('stripUeMacros — non-UE files left alone', () => { + it('does NOT strip standalone *_API identifiers when no UE marker is present', () => { + const src = `enum class Status { REST_API = 1, HTTP_API = 2, MY_LIB_API = 3 };\nvoid handle(REST_API status);`; + expect(stripUeMacros(src)).toBe(src); + }); + + it('does NOT strip _API tokens in a file that only mentions DECLARE_DELEGATE-like macros from non-UE codebases', () => { + const src = `// Custom delegate framework, not UE\n#define DECLARE_HANDLER(x) void x()\nDECLARE_HANDLER(MyHandler);\nint REST_API = 0;`; + expect(stripUeMacros(src)).toBe(src); + }); +}); + +describe('stripUeMacros — non-ASCII content preservation', () => { + it('leaves non-ASCII content outside elided ranges intact and at the same .length offset', () => { + const src = `// Comment with non-ASCII: café résumé naïve\nUCLASS()\nclass UMyClass : public UObject\n{\n GENERATED_BODY()\n // Trailing: 日本語 αβγ\n};`; + const out = stripUeMacros(src); + expect(out.length).toBe(src.length); + expect(out).toContain('café résumé naïve'); + expect(out).toContain('日本語 αβγ'); + expect(out).toContain('class UMyClass : public UObject'); + expect(out).not.toContain('UCLASS'); + expect(out).not.toContain('GENERATED_BODY'); + }); + + it('preserves newline positions when the file contains non-ASCII characters', () => { + const src = `// café\nUPROPERTY()\nint32 Health;\n// résumé\nUFUNCTION()\nvoid Run();`; + const out = stripUeMacros(src); + const inputNewlines: number[] = []; + const outputNewlines: number[] = []; + for (let i = 0; i < src.length; i++) { + if (src.charCodeAt(i) === 0x0a) inputNewlines.push(i); + if (out.charCodeAt(i) === 0x0a) outputNewlines.push(i); + } + expect(outputNewlines).toEqual(inputNewlines); + }); +}); + +describe('stripUeMacros — false-positive guards', () => { + it('does NOT strip identifiers that merely contain UCLASS as a substring', () => { + const src = `void NotUCLASSAtAll(); int MyUCLASS = 0;`; + const out = stripUeMacros(src); + expect(out).toBe(src); + }); + + it('does NOT strip _API substrings inside larger identifiers', () => { + const src = `class MY_APIName {};\nint not_my_API_thing = 0;`; + const out = stripUeMacros(src); + expect(out).toContain('MY_APIName'); + expect(out).toContain('not_my_API_thing'); + }); + + it('does not eat parens balanced inside string literals', () => { + const src = `UFUNCTION(meta=(DisplayName="Foo (Bar)")) void Z();`; + const out = stripUeMacros(src); + expect(out).not.toContain('UFUNCTION'); + expect(out).not.toContain('DisplayName'); + expect(out).toContain('void Z();'); + }); + + it('handles UCLASS with deeply nested parens in arguments', () => { + const src = `UCLASS(meta=(Categories=("A.B", "C.D")), Within=Foo) class UDeep {};`; + const out = stripUeMacros(src); + expect(out).not.toContain('UCLASS'); + expect(out).not.toContain('Categories'); + expect(out).toContain('class UDeep {};'); + }); + + it('leaves Qt macros alone (only UE markers stripped)', () => { + const src = `class QFoo { Q_OBJECT public: void Bar(); };`; + const out = stripUeMacros(src); + expect(out).toContain('Q_OBJECT'); + }); +}); + +describe('stripUeMacros — class-name extraction sanity', () => { + it('after stripping, "class UMyClass" appears immediately after "class "', () => { + const src = `UCLASS(BlueprintType)\nclass BRAWLUI_API UMyClass : public UObject\n{\n GENERATED_BODY()\n};`; + const out = stripUeMacros(src); + const classIdx = out.indexOf('class '); + expect(classIdx).toBeGreaterThanOrEqual(0); + const tail = out.slice(classIdx + 'class '.length).trimStart(); + expect(tail.startsWith('UMyClass')).toBe(true); + }); +}); + +describe('stripUeMacros — tree-sitter extraction (end-to-end)', () => { + /** + * Walk the parse tree and return the captured class name(s). Works against + * the actual tree-sitter-cpp grammar so this is a true integration check + * for the core PR claim: the indexer now sees `UMyClass`, not `BRAWLUI_API`. + */ + function extractClassNames(source: string): string[] { + const parser = new Parser(); + parser.setLanguage(CPP as unknown as Parser.Language); + const tree = parser.parse(source); + const names: string[] = []; + const stack: Parser.SyntaxNode[] = [tree.rootNode]; + while (stack.length > 0) { + const node = stack.pop()!; + if (node.type === 'class_specifier' || node.type === 'struct_specifier') { + const nameNode = node.childForFieldName('name'); + if (nameNode) names.push(nameNode.text); + } + for (let i = node.namedChildCount - 1; i >= 0; i--) { + const child = node.namedChild(i); + if (child) stack.push(child); + } + } + return names; + } + + it('tree-sitter-cpp captures UMyClass as the class name (not BRAWLUI_API)', () => { + const src = `UCLASS(BlueprintType)\nclass BRAWLUI_API UMyClass : public UObject\n{\n GENERATED_BODY()\n public:\n UFUNCTION()\n void Run();\n};`; + const out = stripUeMacros(src); + const names = extractClassNames(out); + expect(names).toContain('UMyClass'); + expect(names).not.toContain('BRAWLUI_API'); + }); + + it('tree-sitter-cpp captures struct name correctly through USTRUCT + MODULE_API', () => { + const src = `USTRUCT(BlueprintType)\nstruct ENGINE_API FMyData : public FBase\n{\n GENERATED_BODY()\n float Value;\n};`; + const out = stripUeMacros(src); + const names = extractClassNames(out); + expect(names).toContain('FMyData'); + expect(names).not.toContain('ENGINE_API'); + }); + + it('tree-sitter-cpp source positions are preserved across stripping (line numbers match)', () => { + const src = `UCLASS()\nclass BRAWLUI_API UMyClass : public UObject\n{\n GENERATED_BODY()\n public:\n void Run();\n};`; + const out = stripUeMacros(src); + const parser = new Parser(); + parser.setLanguage(CPP as unknown as Parser.Language); + const tree = parser.parse(out); + const stack: Parser.SyntaxNode[] = [tree.rootNode]; + let runLine: number | undefined; + while (stack.length > 0) { + const node = stack.pop()!; + if (node.type === 'function_declarator') { + const declarator = node.childForFieldName('declarator'); + if (declarator?.text === 'Run') { + runLine = node.startPosition.row; + break; + } + } + for (let i = node.namedChildCount - 1; i >= 0; i--) { + const child = node.namedChild(i); + if (child) stack.push(child); + } + } + expect(runLine).toBe(5); // 0-indexed: "void Run();" is on line 6 (index 5) + }); +}); diff --git a/gitnexus/test/unit/group/config-parser.test.ts b/gitnexus/test/unit/group/config-parser.test.ts index e1d3b540f..22bb2ad26 100644 --- a/gitnexus/test/unit/group/config-parser.test.ts +++ b/gitnexus/test/unit/group/config-parser.test.ts @@ -75,6 +75,62 @@ repos: expect(config.detect.thrift).toBe(true); }); + // PR #1156 Codex follow-up: include extraction is opt-in. Existing + // group.yaml files that do not declare `detect.includes` must not gain + // a wave of new include::* contracts on the next sync after upgrade. + describe('detect.includes opt-in default', () => { + it('defaults includes detection to false when detect block omits it', () => { + const minimal = ` +version: 1 +name: test +repos: + app: my-app +`; + const config = parseGroupConfig(minimal); + expect(config.detect.includes).toBe(false); + }); + + it('defaults includes detection to false when detect block is present but omits the key', () => { + const yaml = ` +version: 1 +name: test +repos: + app: my-app +detect: + http: true + grpc: false +`; + const config = parseGroupConfig(yaml); + expect(config.detect.includes).toBe(false); + }); + + it('honors explicit detect.includes: true (opt-in works)', () => { + const yaml = ` +version: 1 +name: test +repos: + app: my-app +detect: + includes: true +`; + const config = parseGroupConfig(yaml); + expect(config.detect.includes).toBe(true); + }); + + it('honors explicit detect.includes: false', () => { + const yaml = ` +version: 1 +name: test +repos: + app: my-app +detect: + includes: false +`; + const config = parseGroupConfig(yaml); + expect(config.detect.includes).toBe(false); + }); + }); + it('parses thrift manifest links', () => { const yaml = ` version: 1 diff --git a/gitnexus/test/unit/group/cross-impact-phase2-timeout.test.ts b/gitnexus/test/unit/group/cross-impact-phase2-timeout.test.ts new file mode 100644 index 000000000..90b8a80a2 --- /dev/null +++ b/gitnexus/test/unit/group/cross-impact-phase2-timeout.test.ts @@ -0,0 +1,121 @@ +/** + * Phase-2 fanout timeout regression test. + * + * Codex adversarial review on PR #1331 surfaced that `validateGroupImpactParams` + * clamps `timeoutMs` and `safeLocalImpact` enforces it on the local leg, but + * the Phase-2 cross-repo fanout (`cross-impact.ts:521-526`) awaits each + * `port.impactByUid(...)` call without a per-call timeout. A single hung + * neighbor pins the request indefinitely; multiple slow neighbors compound + * past the clamped budget because each starts before `Date.now() > deadline`. + * + * This test pins the contract of the mitigation: a `safeNeighborImpact` + * helper that races `port.impactByUid` against a remaining-budget timer + * and returns `{ value: null, timedOut: true }` when the call cannot + * complete in time. + * + * Direct import + named symbol so this is a real regression net — no + * `??`-fallback or dynamic-import dance (the U8 false-green pattern). + */ +import { describe, expect, it } from 'vitest'; +import { safeNeighborImpact } from '../../../src/core/group/cross-impact.js'; +import type { GroupToolPort } from '../../../src/core/group/service.js'; + +const minimalOpts = { + maxDepth: 3, + relationTypes: [] as string[], + minConfidence: 0, + includeTests: false, +}; + +function makePort(impactByUid: GroupToolPort['impactByUid']): GroupToolPort { + return { + resolveRepo: async () => { + throw new Error('not used'); + }, + impact: async () => { + throw new Error('not used'); + }, + query: async () => { + throw new Error('not used'); + }, + context: async () => { + throw new Error('not used'); + }, + impactByUid, + }; +} + +describe('safeNeighborImpact — Phase-2 fanout per-call timeout', () => { + it('returns timedOut=true when impactByUid never resolves, within ~remainingMs', async () => { + // Hung neighbor: the promise never resolves. Without the timeout wrap + // this would hang the test runner. + const port = makePort(() => new Promise(() => {})); + const start = performance.now(); + const result = await safeNeighborImpact(port, 'repo-id', 'uid:1', 'upstream', minimalOpts, 150); + const elapsedMs = performance.now() - start; + expect(result.timedOut).toBe(true); + expect(result.value).toBeNull(); + // Allow generous slack for slow CI; the contract is "bounded", not + // "exactly remainingMs". A regression that drops the timeout entirely + // would hang far past 1500ms; a regression that uses the wrong unit + // (seconds vs ms) would fire much faster. + expect(elapsedMs).toBeGreaterThanOrEqual(140); + expect(elapsedMs).toBeLessThan(1500); + }); + + it('returns the resolved value and timedOut=false on a fast happy path', async () => { + const fakeFan = { byDepth: { 1: [{ id: 'u1' }] } }; + const port = makePort(async () => fakeFan); + const result = await safeNeighborImpact( + port, + 'repo-id', + 'uid:1', + 'upstream', + minimalOpts, + 1000, + ); + expect(result.timedOut).toBe(false); + expect(result.value).toBe(fakeFan); + }); + + it('returns timedOut=true immediately when remainingMs is 0 and the call still hangs', async () => { + // Defensive: even if the caller passes 0, the helper must not block. + const port = makePort(() => new Promise(() => {})); + const start = performance.now(); + const result = await safeNeighborImpact(port, 'repo-id', 'uid:1', 'upstream', minimalOpts, 0); + const elapsedMs = performance.now() - start; + expect(result.timedOut).toBe(true); + expect(result.value).toBeNull(); + // 0ms timeout fires on the next tick — should be well under 50ms even on slow CI. + expect(elapsedMs).toBeLessThan(50); + }); + + it('does not compound across calls — two hung neighbors complete within ~2× remainingMs total', async () => { + // The contract is per-call timeout. Two sequential hung calls should + // total ~2× remainingMs, not (numNeighbors × remainingMs² / 2) or + // anything compounding. A regression that shares one timer across + // calls would pass the first test but fail this one. + const port = makePort(() => new Promise(() => {})); + const start = performance.now(); + const r1 = await safeNeighborImpact(port, 'repo', 'u1', 'upstream', minimalOpts, 100); + const r2 = await safeNeighborImpact(port, 'repo', 'u2', 'upstream', minimalOpts, 100); + const elapsedMs = performance.now() - start; + expect(r1.timedOut).toBe(true); + expect(r2.timedOut).toBe(true); + expect(elapsedMs).toBeGreaterThanOrEqual(180); + expect(elapsedMs).toBeLessThan(1000); + }); + + it('propagates an immediate rejection from impactByUid as timedOut=false with null value', async () => { + // If the port itself rejects (rather than hangs), the helper should + // surface that as a non-timeout failure — the existing fanout block + // already handles `if (fan == null)` truncation, so returning null + // here keeps that path intact. + const port = makePort(async () => { + throw new Error('connection refused'); + }); + const result = await safeNeighborImpact(port, 'repo', 'u1', 'upstream', minimalOpts, 1000); + expect(result.timedOut).toBe(false); + expect(result.value).toBeNull(); + }); +}); diff --git a/gitnexus/test/unit/group/include-extractor.test.ts b/gitnexus/test/unit/group/include-extractor.test.ts new file mode 100644 index 000000000..3956cd5f2 --- /dev/null +++ b/gitnexus/test/unit/group/include-extractor.test.ts @@ -0,0 +1,563 @@ +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import * as fs from 'node:fs'; +import * as path from 'node:path'; +import * as os from 'node:os'; +import { IncludeExtractor } from '../../../src/core/group/extractors/include-extractor.js'; +import type { RepoHandle } from '../../../src/core/group/types.js'; +import { normalizeContractId } from '../../../src/core/group/matching.js'; + +describe('IncludeExtractor', () => { + let tmpDir: string; + let extractor: IncludeExtractor; + + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-include-')); + extractor = new IncludeExtractor(); + }); + + afterEach(() => { + fs.rmSync(tmpDir, { recursive: true, force: true }); + }); + + function writeFile(relPath: string, content: string): void { + const full = path.join(tmpDir, relPath); + fs.mkdirSync(path.dirname(full), { recursive: true }); + fs.writeFileSync(full, content); + } + + const makeRepo = (repoPath: string): RepoHandle => ({ + id: 'test-repo', + path: 'test/app', + repoPath, + storagePath: path.join(repoPath, '.gitnexus'), + }); + + // ---- Provider detection ---- + + describe('provider extraction', () => { + it('registers .h files as providers', async () => { + writeFile('map/base/view.h', '#pragma once\nclass View {};'); + writeFile('map/base/types.h', '#pragma once\nstruct Point {};'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providers = contracts.filter((c) => c.role === 'provider'); + + expect(providers).toHaveLength(2); + const ids = providers.map((p) => p.contractId).sort(); + expect(ids).toEqual(['include::map/base/types.h', 'include::map/base/view.h']); + expect(providers[0].type).toBe('include'); + expect(providers[0].confidence).toBeGreaterThanOrEqual(0.95); + }); + + it('registers .hpp files as providers', async () => { + writeFile('utils/helper.hpp', '#pragma once\ntemplate T id(T x) { return x; }'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providers = contracts.filter((c) => c.role === 'provider'); + + expect(providers).toHaveLength(1); + expect(providers[0].contractId).toBe('include::utils/helper.hpp'); + }); + + it('does not register .cpp files as providers', async () => { + writeFile('src/main.cpp', 'int main() { return 0; }'); + writeFile('src/utils.h', '#pragma once'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providers = contracts.filter((c) => c.role === 'provider'); + + expect(providers).toHaveLength(1); + expect(providers[0].contractId).toBe('include::src/utils.h'); + }); + }); + + // ---- Consumer detection ---- + + describe('consumer extraction', () => { + it('emits unresolved includes as consumers', async () => { + writeFile( + 'src/main.cpp', + `#include "map/base/view.h" +#include "map/base/types.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(2); + const ids = consumers.map((c) => c.contractId).sort(); + expect(ids).toEqual(['include::map/base/types.h', 'include::map/base/view.h']); + expect(consumers[0].type).toBe('include'); + expect(consumers[0].confidence).toBe(0.85); + }); + + it('skips locally resolved includes', async () => { + writeFile('map/base/view.h', '#pragma once\nclass View {};'); + writeFile( + 'src/main.cpp', + `#include "map/base/view.h" +#include "external/lib.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + // Only external/lib.h should be a consumer — map/base/view.h resolves locally + expect(consumers).toHaveLength(1); + expect(consumers[0].contractId).toBe('include::external/lib.h'); + }); + + it('skips angle-bracket includes', async () => { + writeFile( + 'src/main.cpp', + `#include +#include +#include "app/interface.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(1); + expect(consumers[0].contractId).toBe('include::app/interface.h'); + }); + + it('skips well-known system headers in quotes', async () => { + writeFile( + 'src/main.cpp', + `#include "stdio.h" +#include "stdlib.h" +#include "app/config.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(1); + expect(consumers[0].contractId).toBe('include::app/config.h'); + }); + + it('skips system path prefixes', async () => { + writeFile( + 'src/main.c', + `#include "sys/types.h" +#include "linux/input.h" +#include "mylib/types.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(1); + expect(consumers[0].contractId).toBe('include::mylib/types.h'); + }); + }); + + // ---- Cross-repo matching scenario ---- + + describe('cross-repo matching', () => { + it('provider and consumer produce matching contractIds', async () => { + // Simulate provider repo (header-only) + const providerDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-include-provider-')); + const providerFile = path.join(providerDir, 'map/base/dice_map_view.h'); + fs.mkdirSync(path.dirname(providerFile), { recursive: true }); + fs.writeFileSync(providerFile, '#pragma once\nclass DiceMapView {};'); + + // Simulate consumer repo + const consumerDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-include-consumer-')); + const consumerFile = path.join(consumerDir, 'src/controller.cpp'); + fs.mkdirSync(path.dirname(consumerFile), { recursive: true }); + fs.writeFileSync(consumerFile, '#include "map/base/dice_map_view.h"\nvoid init() {}'); + + try { + const providerContracts = await extractor.extract(null, providerDir, makeRepo(providerDir)); + const consumerContracts = await extractor.extract(null, consumerDir, makeRepo(consumerDir)); + + const providers = providerContracts.filter((c) => c.role === 'provider'); + const consumers = consumerContracts.filter((c) => c.role === 'consumer'); + + expect(providers.length).toBeGreaterThanOrEqual(1); + expect(consumers.length).toBeGreaterThanOrEqual(1); + + const providerIds = new Set(providers.map((p) => normalizeContractId(p.contractId))); + const consumerIds = consumers.map((c) => normalizeContractId(c.contractId)); + + // The consumer's include path should match a provider's file path + expect(providerIds.has(consumerIds[0])).toBe(true); + } finally { + fs.rmSync(providerDir, { recursive: true, force: true }); + fs.rmSync(consumerDir, { recursive: true, force: true }); + } + }); + }); + + // ---- Review finding #4: suffixResolve ambiguity ---- + + describe('finding #4: suffix-ambiguity does not silently suppress cross-repo include', () => { + it('emits a cross-repo contract when the include path does not match any local file (even if a shorter suffix does)', async () => { + // local repo has `internal/api.h` but NOT `ext/api.h` + writeFile('internal/api.h', '#pragma once'); + writeFile( + 'src/main.cpp', + `#include "ext/api.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + // Previously suffixResolve would match `api.h` against `internal/api.h` + // and drop the cross-repo contract. After finding #4 fix, we only + // accept exact full-path matches — so `ext/api.h` must still be + // emitted as a consumer contract. + expect(consumers).toHaveLength(1); + expect(consumers[0].contractId).toBe('include::ext/api.h'); + }); + + it('still suppresses a local include when the FULL path matches', async () => { + writeFile('ext/api.h', '#pragma once'); + writeFile('src/main.cpp', '#include "ext/api.h"\nint main(){return 0;}'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(0); + }); + + it('resolves locally when include omits extension and a matching .h exists', async () => { + writeFile('foo/bar.h', '#pragma once'); + writeFile('src/main.cpp', '#include "foo/bar"\nint main(){return 0;}'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(0); + }); + }); + + // ---- Review finding #5: regex fallback must strip block comments ---- + + describe('finding #5: regex fallback ignores block-commented includes', () => { + it('does not emit a contract for an #include inside /* ... */', async () => { + // Force regex fallback by producing a file larger than tree-sitter's + // 32 KB hard cap. The include we care about lives inside a block + // comment that spans the file. + const filler = 'int dummy_' + 'x'.repeat(32) + ' = 0;\n'.repeat(1200); + const content = `/* + * Historical include, kept for reference only: + * #include "legacy/old-api.h" + */ +${filler} +#include "real/api.h" +int main(){return 0;}`; + writeFile('src/huge.cpp', content); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + const ids = consumers.map((c) => c.contractId); + + // The live include should appear; the commented-out one must NOT. + expect(ids).toContain('include::real/api.h'); + expect(ids).not.toContain('include::legacy/old-api.h'); + }); + }); + + // ---- Review finding #6: meta.source must reflect which extraction path ran ---- + + describe('finding #6: meta.source reflects extraction path', () => { + it('stamps `tree_sitter` on contracts produced via AST walking', async () => { + writeFile('src/main.cpp', '#include "app/small.h"\nint main(){return 0;}'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(1); + expect((consumers[0].meta as { source?: string } | undefined)?.source).toBe('tree_sitter'); + }); + + it('meta.source is one of the two documented values (tree_sitter | regex_fallback)', async () => { + // Regex fallback is a defensive branch that only fires if + // parser.setLanguage() or parser.parse() throws. In practice + // tree-sitter-c/cpp handles realistic inputs, so we only assert + // the meta.source contract: it is always present and always one of + // the two documented values. This guards against future regressions + // that might hard-code the wrong string. + writeFile('src/main.cpp', '#include "ext/whatever.h"\nint main(){return 0;}'); + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumer = contracts.find((c) => c.role === 'consumer'); + expect(consumer).toBeDefined(); + const src = (consumer?.meta as { source?: string } | undefined)?.source; + expect(['tree_sitter', 'regex_fallback']).toContain(src); + }); + }); + + // ---- Review finding #3: provider id collision on case-sensitive FS ---- + + describe('finding #3: case-folding is documented and deterministic', () => { + it('collapses `Foo.h` and `foo.h` onto the same provider contract-id (documented trade-off)', async () => { + writeFile('Foo.h', '#pragma once\n// Capital Foo'); + // On case-insensitive filesystems (macOS default) the second writeFile + // will overwrite the first, so we only create this when distinct files + // can coexist (case-sensitive FS, e.g. Linux CI). + try { + fs.writeFileSync(path.join(tmpDir, 'foo.h'), '#pragma once\n// lowercase foo'); + } catch { + // Ignore — some FS won't allow both names to coexist. + } + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providers = contracts.filter((c) => c.role === 'provider'); + const ids = providers.map((p) => p.contractId); + + // Both files (if they coexist) must normalize to the same id. + // dedupe() keeps only one; caller code must be aware of this. + expect(ids).toContain('include::foo.h'); + // Never see a mixed-case contract-id leak out. + expect(ids.every((id) => id === id.toLowerCase())).toBe(true); + }); + }); + + // ---- Deduplication ---- + + describe('deduplication', () => { + it('deduplicates same include from multiple source files', async () => { + writeFile('src/a.cpp', '#include "ext/api.h"\nvoid a() {}'); + writeFile('src/b.cpp', '#include "ext/api.h"\nvoid b() {}'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + // Both files include "ext/api.h" — each should produce a separate + // consumer contract (different symbolRef.filePath) + expect(consumers).toHaveLength(2); + const files = consumers.map((c) => c.symbolRef.filePath).sort(); + expect(files).toEqual(['src/a.cpp', 'src/b.cpp']); + }); + }); + + // ---- normalizeContractId ---- + + describe('normalizeContractId for include', () => { + it('lowercases the path', () => { + expect(normalizeContractId('include::Map/Base/Foo.h')).toBe('include::map/base/foo.h'); + }); + + it('normalizes backslashes', () => { + expect(normalizeContractId('include::map\\base\\foo.h')).toBe('include::map/base/foo.h'); + }); + + it('strips leading ./', () => { + expect(normalizeContractId('include::./foo.h')).toBe('include::foo.h'); + }); + + it('collapses consecutive slashes', () => { + expect(normalizeContractId('include::map//base///foo.h')).toBe('include::map/base/foo.h'); + }); + }); + + // ---- PR #1156 follow-up: `../` relative includes ---- + + describe('follow-up: `../` relative includes are skipped', () => { + it('does not emit a consumer contract for `#include "../foo.h"`', async () => { + // Producer: a header that exists locally but only via parent reference + writeFile('include/foo.h', '#pragma once'); + writeFile( + 'src/sub/main.cpp', + `#include "../../include/foo.h" +#include "real/cross_repo.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + // Only `real/cross_repo.h` should remain — the `..`-prefixed include + // is intra-repo noise that no provider can ever satisfy. + expect(consumers.map((c) => c.contractId)).toEqual(['include::real/cross_repo.h']); + }); + + it('skips backslash-form `..\\` for completeness', async () => { + writeFile( + 'src/main.cpp', + `#include "..\\\\sibling\\\\foo.h" +#include "remote/header.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + const ids = consumers.map((c) => c.contractId); + expect(ids).toContain('include::remote/header.h'); + expect(ids.some((id) => id.includes('..'))).toBe(false); + }); + }); + + // ---- PR #1156 follow-up: macro-style includes ---- + + describe('follow-up: macro-style #include emits no consumer contract', () => { + it('does not emit a consumer contract for `#include PLATFORM_HEADER` (no separator, no dot)', async () => { + // `#include PLATFORM_HEADER` parses under tree-sitter as an identifier + // node, slips past the existing system-header / `..` filters, and used + // to leak through as a permanently orphaned consumer contract because + // no file is ever named `PLATFORM_HEADER`. Verify the macro guard + // suppresses it while preserving the real cross-repo include. + writeFile( + 'src/main.cpp', + `#include PLATFORM_HEADER +#include "real/api.h" +int main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers.map((c) => c.contractId)).toEqual(['include::real/api.h']); + }); + + it('skips multiple macro identifiers in the same translation unit', async () => { + writeFile( + 'src/cfg.cpp', + `#include CONFIG_HEADER +#include PLATFORM_HEADER +#include ASSERT_H_ +int main(){return 0;}`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumers = contracts.filter((c) => c.role === 'consumer'); + + expect(consumers).toHaveLength(0); + }); + }); + + // ---- PR #1156 follow-up: graph provider absolute paths ---- + + describe('follow-up: extractProvidersGraph strips repo root from absolute paths', () => { + it('produces repo-relative contract IDs when the graph returns absolute paths', async () => { + writeFile('map/base/view.h', '#pragma once\nclass View {};'); + writeFile('utils/types.hpp', '#pragma once'); + + // Stub the Cypher executor to return absolute paths the way + // gitnexus analyze actually persists them. + const absolute1 = path.join(tmpDir, 'map/base/view.h'); + const absolute2 = path.join(tmpDir, 'utils/types.hpp'); + const stubDb = async () => [ + { filePath: absolute1, fileId: 'File:abs:1' }, + { filePath: absolute2, fileId: 'File:abs:2' }, + ]; + + const contracts = await extractor.extract(stubDb, tmpDir, makeRepo(tmpDir)); + const providers = contracts.filter((c) => c.role === 'provider'); + + const ids = providers.map((p) => p.contractId).sort(); + expect(ids).toEqual(['include::map/base/view.h', 'include::utils/types.hpp']); + expect(providers.every((p) => p.meta?.source === 'graph')).toBe(true); + }); + + it('drops graph rows whose path resolves outside the repo root', async () => { + writeFile('local/header.h', '#pragma once'); + const absoluteLocal = path.join(tmpDir, 'local/header.h'); + const stubDb = async () => [ + { filePath: absoluteLocal, fileId: 'File:1' }, + // Stale absolute path from a different machine — must be skipped. + { filePath: '/some/other/repo/foreign.h', fileId: 'File:2' }, + ]; + + const contracts = await extractor.extract(stubDb, tmpDir, makeRepo(tmpDir)); + const providers = contracts.filter((c) => c.role === 'provider'); + + expect(providers.map((p) => p.contractId)).toEqual(['include::local/header.h']); + }); + }); + + // ---- PR #1156 Codex follow-up: discovery aligned with ingestion ---- + + describe('follow-up: file discovery honors createIgnoreFilter and getMaxFileSizeBytes', () => { + it('does not emit a provider contract for a header excluded by .gitignore', async () => { + writeFile('.gitignore', 'vendor-headers/\n'); + writeFile('vendor-headers/blocked.h', '#pragma once'); + writeFile('src/wanted.h', '#pragma once'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providerIds = contracts.filter((c) => c.role === 'provider').map((p) => p.contractId); + + expect(providerIds).toContain('include::src/wanted.h'); + expect(providerIds).not.toContain('include::vendor-headers/blocked.h'); + }); + + it('does not emit a provider contract for a header excluded by .gitnexusignore', async () => { + writeFile('.gitnexusignore', 'legacy/\n'); + writeFile('legacy/old.h', '#pragma once'); + writeFile('src/current.h', '#pragma once'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providerIds = contracts.filter((c) => c.role === 'provider').map((p) => p.contractId); + + expect(providerIds).toContain('include::src/current.h'); + expect(providerIds).not.toContain('include::legacy/old.h'); + }); + + it('does not parse #include directives in a source file excluded by .gitignore', async () => { + // The ignored source file references a header that would otherwise be + // a cross-repo consumer. After alignment, the ignored file is invisible + // to the consumer scan — no consumer contract should appear. + writeFile('.gitignore', 'generated/\n'); + writeFile( + 'generated/auto.cpp', + `#include "remote/should_not_appear.h" +int auto_main() { return 0; }`, + ); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumerIds = contracts.filter((c) => c.role === 'consumer').map((c) => c.contractId); + + expect(consumerIds).not.toContain('include::remote/should_not_appear.h'); + }); + + it('skips a provider header whose size exceeds GITNEXUS_MAX_FILE_SIZE', async () => { + const previous = process.env.GITNEXUS_MAX_FILE_SIZE; + process.env.GITNEXUS_MAX_FILE_SIZE = '1'; // 1 KB cap + try { + // 4 KB header — comfortably exceeds the cap. + const oversized = '#pragma once\n' + 'x'.repeat(4 * 1024); + writeFile('huge/big.h', oversized); + writeFile('small/tiny.h', '#pragma once'); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const providerIds = contracts.filter((c) => c.role === 'provider').map((p) => p.contractId); + + expect(providerIds).toContain('include::small/tiny.h'); + expect(providerIds).not.toContain('include::huge/big.h'); + } finally { + if (previous === undefined) delete process.env.GITNEXUS_MAX_FILE_SIZE; + else process.env.GITNEXUS_MAX_FILE_SIZE = previous; + } + }); + + it('skips parsing #include directives in source files exceeding GITNEXUS_MAX_FILE_SIZE', async () => { + const previous = process.env.GITNEXUS_MAX_FILE_SIZE; + process.env.GITNEXUS_MAX_FILE_SIZE = '1'; + try { + const oversized = + '#include "remote/should_not_appear.h"\n' + + '// padding to push the file past 1 KB\n' + + 'x'.repeat(4 * 1024); + writeFile('big/main.cpp', oversized); + + const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir)); + const consumerIds = contracts.filter((c) => c.role === 'consumer').map((c) => c.contractId); + + expect(consumerIds).not.toContain('include::remote/should_not_appear.h'); + } finally { + if (previous === undefined) delete process.env.GITNEXUS_MAX_FILE_SIZE; + else process.env.GITNEXUS_MAX_FILE_SIZE = previous; + } + }); + }); +}); diff --git a/gitnexus/test/unit/group/sync.test.ts b/gitnexus/test/unit/group/sync.test.ts index 88bc3af2d..9f14db231 100644 --- a/gitnexus/test/unit/group/sync.test.ts +++ b/gitnexus/test/unit/group/sync.test.ts @@ -3,6 +3,7 @@ import * as fs from 'node:fs'; import * as path from 'node:path'; import * as os from 'node:os'; import { syncGroup, stableRepoPoolId } from '../../../src/core/group/sync.js'; +import { cleanupTempDir } from '../../helpers/test-db.js'; import { _captureLogger } from '../../../src/core/logger.js'; import type { GroupConfig, @@ -583,6 +584,47 @@ service OrderService { } }); + it('does not extract include contracts during real sync when includes detection is disabled', async () => { + // PR #1156 Codex follow-up: ce-code-review T1 — verifies the gate at + // sync.ts:174 honors `detect.includes: false`. Mirrors the existing + // thrift-off pattern at sync.test.ts:545. + const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-sync-includes-off-')); + const storageDir = path.join(tmpDir, '.gitnexus'); + fs.mkdirSync(path.join(tmpDir, 'src'), { recursive: true }); + fs.mkdirSync(storageDir, { recursive: true }); + fs.writeFileSync(path.join(tmpDir, 'src', 'view.h'), '#pragma once\nclass View {};'); + + const config = makeConfig({ 'app/cpp-lib': 'cpp-lib-repo' }); + config.detect.http = false; + config.detect.grpc = false; + config.detect.thrift = false; + config.detect.topics = false; + config.detect.includes = false; + + const poolAdapter = await import('../../../src/core/lbug/pool-adapter.js'); + const initSpy = vi.spyOn(poolAdapter, 'initLbug').mockResolvedValue(undefined); + const closeSpy = vi.spyOn(poolAdapter, 'closeLbug').mockResolvedValue(undefined); + + try { + const result = await syncGroup(config, { + resolveRepoHandle: async (_name, groupPath) => ({ + id: 'cpp-lib-repo', + path: groupPath, + repoPath: tmpDir, + storagePath: storageDir, + }), + skipWrite: true, + }); + + expect(result.missingRepos).toHaveLength(0); + expect(result.contracts.filter((c) => c.type === 'include')).toHaveLength(0); + } finally { + initSpy.mockRestore(); + closeSpy.mockRestore(); + await cleanupTempDir(tmpDir); + } + }); + it('dedupes duplicate wildcard cross-links during sync', async () => { const config = makeConfig({ 'app/provider': 'provider-repo', 'app/consumer': 'consumer-repo' }); const provider: StoredContract = { @@ -689,7 +731,12 @@ service OrderService { expect(registry.version).toBe(1); expect(registry.contracts).toHaveLength(0); } finally { - fs.rmSync(tmpDir, { recursive: true, force: true }); + // syncGroup now writes bridge.lbug + WAL/shadow sidecars when + // skipWrite is false. On Windows, LadybugDB's checkpoint thread can + // briefly outlive closeBridgeDb, holding a Win32 lock on the file. + // cleanupTempDir tolerates the documented Windows-native lock codes + // (EBUSY/EPERM/EACCES/ENOTEMPTY) with bounded retries. + await cleanupTempDir(tmpDir); } }); diff --git a/gitnexus/test/unit/hf-env.test.ts b/gitnexus/test/unit/hf-env.test.ts index 6a5697059..fa8f23bd3 100644 --- a/gitnexus/test/unit/hf-env.test.ts +++ b/gitnexus/test/unit/hf-env.test.ts @@ -1,7 +1,20 @@ -import { describe, it, expect, beforeEach, afterEach } from 'vitest'; +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'; import os from 'node:os'; import { join } from 'node:path'; -import { applyHfEnvOverrides, type HfEnvSubset } from '../../src/core/embeddings/hf-env.js'; +import { + applyHfEnvOverrides, + isNetworkFetchError, + isHfDownloadFailure, + isHfCircuitOpenError, + HfDownloadCircuitBreaker, + withDownloadTimeout, + withHfDownloadRetry, + CIRCUIT_OPEN_TAG, + HF_MAX_ATTEMPTS, + HF_MAX_TIMEOUT_MS, + HF_MAX_ATTEMPTS_CAP, + type HfEnvSubset, +} from '../../src/core/embeddings/hf-env.js'; describe('applyHfEnvOverrides', () => { let envStub: HfEnvSubset; @@ -82,3 +95,399 @@ describe('applyHfEnvOverrides', () => { expect(envStub.remoteHost).toBe('https://hf-mirror.com/'); }); }); + +describe('isNetworkFetchError', () => { + it('returns true for "fetch failed" (the undici error seen on macOS/Node 24)', () => { + expect(isNetworkFetchError('fetch failed')).toBe(true); + }); + + it('returns true for ECONNREFUSED', () => { + expect(isNetworkFetchError('connect ECONNREFUSED 13.45.67.89:443')).toBe(true); + }); + + it('returns true for ENOTFOUND (DNS failure)', () => { + expect(isNetworkFetchError('getaddrinfo ENOTFOUND huggingface.co')).toBe(true); + }); + + it('returns true for ETIMEDOUT', () => { + expect(isNetworkFetchError('connect ETIMEDOUT 13.45.67.89:443')).toBe(true); + }); + + it('returns true for ECONNRESET', () => { + expect(isNetworkFetchError('read ECONNRESET')).toBe(true); + }); + + it('returns false for generic model-load errors (ONNX device failure)', () => { + expect(isNetworkFetchError('Failed to initialize CUDA backend')).toBe(false); + }); + + it('returns false for empty string', () => { + expect(isNetworkFetchError('')).toBe(false); + }); + + it('returns false for module-not-found errors', () => { + expect(isNetworkFetchError('Cannot find module onnxruntime-node')).toBe(false); + }); +}); + +describe('isHfCircuitOpenError', () => { + it('returns true for a circuit-open tag message', () => { + expect(isHfCircuitOpenError(`${CIRCUIT_OPEN_TAG}: circuit is open`)).toBe(true); + }); + + it('returns false for a plain network error', () => { + expect(isHfCircuitOpenError('fetch failed')).toBe(false); + }); +}); + +describe('isHfDownloadFailure', () => { + it('returns true for network fetch errors', () => { + expect(isHfDownloadFailure('ECONNREFUSED 127.0.0.1:443')).toBe(true); + }); + + it('returns true for circuit-open errors', () => { + expect(isHfDownloadFailure(`${CIRCUIT_OPEN_TAG}: open`)).toBe(true); + }); + + it('returns false for ONNX device errors', () => { + expect(isHfDownloadFailure('Failed to initialize CUDA')).toBe(false); + }); +}); + +describe('HfDownloadCircuitBreaker', () => { + it('starts in closed state', () => { + const cb = new HfDownloadCircuitBreaker(); + expect(cb.isOpen()).toBe(false); + expect(cb.state).toBe('closed'); + }); + + it('opens after reaching the failure threshold', () => { + const cb = new HfDownloadCircuitBreaker(3); + cb.recordFailure(); + cb.recordFailure(); + expect(cb.isOpen()).toBe(false); + cb.recordFailure(); // threshold reached + expect(cb.isOpen()).toBe(true); + expect(cb.state).toBe('open'); + }); + + it('closes on recordSuccess after being open', () => { + const cb = new HfDownloadCircuitBreaker(1); + cb.recordFailure(); + expect(cb.isOpen()).toBe(true); + cb.recordSuccess(); + expect(cb.isOpen()).toBe(false); + expect(cb.state).toBe('closed'); + }); + + it('transitions to half-open after the reset timeout', () => { + vi.useFakeTimers(); + try { + const cb = new HfDownloadCircuitBreaker(1, 100 /* 100ms */); + cb.recordFailure(); + expect(cb.isOpen()).toBe(true); + vi.advanceTimersByTime(200); + expect(cb.isOpen()).toBe(false); + expect(cb.state).toBe('half-open'); + } finally { + vi.useRealTimers(); + } + }); + + it('reset() restores closed state', () => { + const cb = new HfDownloadCircuitBreaker(1); + cb.recordFailure(); + expect(cb.isOpen()).toBe(true); + cb.reset(); + expect(cb.isOpen()).toBe(false); + expect(cb.state).toBe('closed'); + }); + + it('re-opens when a failure is recorded in half-open state', () => { + vi.useFakeTimers(); + try { + const cb = new HfDownloadCircuitBreaker(1, 100 /* 100ms */); + cb.recordFailure(); // opens the circuit + vi.advanceTimersByTime(200); // advance past reset timeout + expect(cb.state).toBe('half-open'); // getter transitions _state to half-open + cb.recordFailure(); // failure in half-open → re-opens + expect(cb.isOpen()).toBe(true); + expect(cb.state).toBe('open'); + } finally { + vi.useRealTimers(); + } + }); + + it('closes the circuit when success is recorded in half-open state', () => { + vi.useFakeTimers(); + try { + const cb = new HfDownloadCircuitBreaker(1, 100 /* 100ms */); + cb.recordFailure(); // opens the circuit + vi.advanceTimersByTime(200); // advance past reset timeout + expect(cb.state).toBe('half-open'); + cb.recordSuccess(); // success in half-open → closes + expect(cb.isOpen()).toBe(false); + expect(cb.state).toBe('closed'); + } finally { + vi.useRealTimers(); + } + }); +}); + +describe('withDownloadTimeout', () => { + it('resolves when fn completes before the timeout', async () => { + const result = await withDownloadTimeout(() => Promise.resolve(42), 1_000); + expect(result).toBe(42); + }); + + it('rejects with ETIMEDOUT when fn takes too long', async () => { + vi.useFakeTimers(); + try { + const neverResolves = () => new Promise(() => {}); + const promise = withDownloadTimeout(neverResolves, 20); + vi.advanceTimersByTime(30); + await expect(promise).rejects.toThrow('ETIMEDOUT'); + } finally { + vi.useRealTimers(); + } + }); + + it('propagates non-timeout errors from fn', async () => { + await expect( + withDownloadTimeout(() => Promise.reject(new Error('download error')), 1_000), + ).rejects.toThrow('download error'); + }); +}); + +describe('withHfDownloadRetry', () => { + it('returns the result on first success', async () => { + const fn = vi.fn().mockResolvedValue('ok'); + const cb = new HfDownloadCircuitBreaker(); + const result = await withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 }); + expect(result).toBe('ok'); + expect(fn).toHaveBeenCalledTimes(1); + }); + + it('retries on network errors and succeeds on second attempt', async () => { + const fn = vi.fn().mockRejectedValueOnce(new Error('fetch failed')).mockResolvedValue('ok'); + const cb = new HfDownloadCircuitBreaker(); + const result = await withHfDownloadRetry(fn, { + circuit: cb, + maxAttempts: 3, + baseDelayMs: 0, + }); + expect(result).toBe('ok'); + expect(fn).toHaveBeenCalledTimes(2); + }); + + it('throws the last network error after all attempts are exhausted', async () => { + const fn = vi.fn().mockRejectedValue(new Error('ECONNREFUSED 127.0.0.1:443')); + const cb = new HfDownloadCircuitBreaker(99 /* high threshold */); + await expect( + withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 3, baseDelayMs: 0 }), + ).rejects.toThrow('ECONNREFUSED'); + expect(fn).toHaveBeenCalledTimes(3); + }); + + it('does not retry non-network errors', async () => { + const fn = vi.fn().mockRejectedValue(new Error('Failed to initialize CUDA backend')); + const cb = new HfDownloadCircuitBreaker(); + await expect( + withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 3, baseDelayMs: 0 }), + ).rejects.toThrow('Failed to initialize CUDA backend'); + expect(fn).toHaveBeenCalledTimes(1); + }); + + it('fails immediately when the circuit is already open', async () => { + const fn = vi.fn().mockResolvedValue('ok'); + const cb = new HfDownloadCircuitBreaker(1); + cb.recordFailure(); // open the circuit + await expect(withHfDownloadRetry(fn, { circuit: cb })).rejects.toThrow(CIRCUIT_OPEN_TAG); + expect(fn).not.toHaveBeenCalled(); + }); + + it('opens the circuit after failureThreshold failures and throws a circuit-open error', async () => { + const fn = vi.fn().mockRejectedValue(new Error('ENOTFOUND huggingface.co')); + const cb = new HfDownloadCircuitBreaker(2 /* threshold */, 60_000); + // First call: 2 attempts, threshold=2 → circuit opens on 2nd failure + await expect( + withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 2, baseDelayMs: 0 }), + ).rejects.toThrow(CIRCUIT_OPEN_TAG); + expect(cb.isOpen()).toBe(true); + }); + + it('calls onRetry with correct arguments on each retry', async () => { + const fn = vi + .fn() + .mockRejectedValueOnce(new Error('fetch failed')) + .mockRejectedValueOnce(new Error('fetch failed')) + .mockResolvedValue('ok'); + const cb = new HfDownloadCircuitBreaker(99); + const onRetry = vi.fn(); + await withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 3, baseDelayMs: 0, onRetry }); + expect(onRetry).toHaveBeenCalledTimes(2); + expect(onRetry).toHaveBeenNthCalledWith( + 1, + 1, + 3, + expect.objectContaining({ message: 'fetch failed' }), + ); + expect(onRetry).toHaveBeenNthCalledWith( + 2, + 2, + 3, + expect.objectContaining({ message: 'fetch failed' }), + ); + }); + + it('resets the circuit on success', async () => { + const fn = vi.fn().mockResolvedValue('value'); + const cb = new HfDownloadCircuitBreaker(5); + cb.recordFailure(); + cb.recordFailure(); // 2 failures, circuit still closed + await withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 }); + expect(cb.state).toBe('closed'); + }); +}); + +describe('withHfDownloadRetry env overrides', () => { + let originalTimeout: string | undefined; + let originalMaxAttempts: string | undefined; + + beforeEach(() => { + originalTimeout = process.env.HF_DOWNLOAD_TIMEOUT_MS; + originalMaxAttempts = process.env.HF_MAX_ATTEMPTS; + delete process.env.HF_DOWNLOAD_TIMEOUT_MS; + delete process.env.HF_MAX_ATTEMPTS; + }); + + afterEach(() => { + if (originalTimeout === undefined) delete process.env.HF_DOWNLOAD_TIMEOUT_MS; + else process.env.HF_DOWNLOAD_TIMEOUT_MS = originalTimeout; + if (originalMaxAttempts === undefined) delete process.env.HF_MAX_ATTEMPTS; + else process.env.HF_MAX_ATTEMPTS = originalMaxAttempts; + }); + + it('HF_MAX_ATTEMPTS=1 gives exactly 1 attempt', async () => { + process.env.HF_MAX_ATTEMPTS = '1'; + const fn = vi.fn().mockRejectedValue(new Error('ECONNREFUSED 127.0.0.1:443')); + const cb = new HfDownloadCircuitBreaker(99_999 /* high threshold */); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'ECONNREFUSED', + ); + expect(fn).toHaveBeenCalledTimes(1); + }); + + it('HF_MAX_ATTEMPTS=2 gives exactly 2 attempts', async () => { + process.env.HF_MAX_ATTEMPTS = '2'; + const fn = vi.fn().mockRejectedValue(new Error('ENOTFOUND huggingface.co')); + const cb = new HfDownloadCircuitBreaker(99_999); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'ENOTFOUND', + ); + expect(fn).toHaveBeenCalledTimes(2); + }); + + it('HF_MAX_ATTEMPTS=abc falls back to the built-in default', async () => { + process.env.HF_MAX_ATTEMPTS = 'abc'; + const fn = vi.fn().mockRejectedValue(new Error('fetch failed')); + const cb = new HfDownloadCircuitBreaker(99_999); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'fetch failed', + ); + expect(fn).toHaveBeenCalledTimes(HF_MAX_ATTEMPTS); + }); + + it('HF_MAX_ATTEMPTS=0 falls back to the built-in default', async () => { + process.env.HF_MAX_ATTEMPTS = '0'; + const fn = vi.fn().mockRejectedValue(new Error('fetch failed')); + const cb = new HfDownloadCircuitBreaker(99_999); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'fetch failed', + ); + expect(fn).toHaveBeenCalledTimes(HF_MAX_ATTEMPTS); + }); + + it('HF_MAX_ATTEMPTS=-1 falls back to the built-in default', async () => { + process.env.HF_MAX_ATTEMPTS = '-1'; + const fn = vi.fn().mockRejectedValue(new Error('fetch failed')); + const cb = new HfDownloadCircuitBreaker(99_999); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'fetch failed', + ); + expect(fn).toHaveBeenCalledTimes(HF_MAX_ATTEMPTS); + }); + + it('HF_MAX_ATTEMPTS is clamped to HF_MAX_ATTEMPTS_CAP', async () => { + process.env.HF_MAX_ATTEMPTS = '9999'; + const fn = vi.fn().mockRejectedValue(new Error('fetch failed')); + const cb = new HfDownloadCircuitBreaker(99_999 /* very high threshold */); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'fetch failed', + ); + expect(fn).toHaveBeenCalledTimes(HF_MAX_ATTEMPTS_CAP); + }); + + it('HF_MAX_ATTEMPTS=2.9 is floored to 2', async () => { + process.env.HF_MAX_ATTEMPTS = '2.9'; + const fn = vi.fn().mockRejectedValue(new Error('fetch failed')); + const cb = new HfDownloadCircuitBreaker(99_999); + await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow( + 'fetch failed', + ); + expect(fn).toHaveBeenCalledTimes(2); + }); + + it('HF_DOWNLOAD_TIMEOUT_MS is used as the per-attempt timeout when valid', async () => { + vi.useFakeTimers(); + try { + process.env.HF_DOWNLOAD_TIMEOUT_MS = '50'; + const neverResolves = () => new Promise(() => {}); + const cb = new HfDownloadCircuitBreaker(99); + const promise = withHfDownloadRetry(neverResolves, { circuit: cb, maxAttempts: 1 }); + vi.advanceTimersByTime(100); + await expect(promise).rejects.toThrow('ETIMEDOUT'); + } finally { + vi.useRealTimers(); + } + }); + + it('HF_DOWNLOAD_TIMEOUT_MS=-1 falls back to the built-in default', async () => { + process.env.HF_DOWNLOAD_TIMEOUT_MS = '-1'; + // Passing explicit timeoutMs=0 (no real wait) so the test doesn't block; + // we just verify that the env var rejection causes options.timeoutMs to be + // the default constant (not -1) by confirming the resolved value is used. + const fn = vi.fn().mockResolvedValue('ok'); + const cb = new HfDownloadCircuitBreaker(99); + // Provide explicit timeoutMs to avoid the default 5-minute wait + const result = await withHfDownloadRetry(fn, { circuit: cb, timeoutMs: 100 }); + expect(result).toBe('ok'); + }); + + it('HF_DOWNLOAD_TIMEOUT_MS is clamped to HF_MAX_TIMEOUT_MS', async () => { + vi.useFakeTimers(); + try { + // Set an env value exceeding the 30-minute cap + process.env.HF_DOWNLOAD_TIMEOUT_MS = String(HF_MAX_TIMEOUT_MS + 60_000); + const neverResolves = () => new Promise(() => {}); + const cb = new HfDownloadCircuitBreaker(99); + const promise = withHfDownloadRetry(neverResolves, { circuit: cb, maxAttempts: 1 }); + // Advance just past the 30-minute cap + vi.advanceTimersByTime(HF_MAX_TIMEOUT_MS + 1); + await expect(promise).rejects.toThrow('ETIMEDOUT'); + } finally { + vi.useRealTimers(); + } + }); + + it('explicit options override env vars', async () => { + process.env.HF_MAX_ATTEMPTS = '5'; + const fn = vi.fn().mockRejectedValue(new Error('fetch failed')); + const cb = new HfDownloadCircuitBreaker(99); + // explicit maxAttempts: 2 must win over HF_MAX_ATTEMPTS=5 + await expect( + withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 2, baseDelayMs: 0 }), + ).rejects.toThrow('fetch failed'); + expect(fn).toHaveBeenCalledTimes(2); + }); +}); diff --git a/gitnexus/test/unit/ignore-service.test.ts b/gitnexus/test/unit/ignore-service.test.ts index b4e5cdca1..1e1908137 100644 --- a/gitnexus/test/unit/ignore-service.test.ts +++ b/gitnexus/test/unit/ignore-service.test.ts @@ -28,6 +28,8 @@ describe('shouldIgnorePath', () => { it.each([ 'node_modules', 'vendor', + 'third_party', + '3rdparty', 'venv', '.venv', '__pycache__', diff --git a/gitnexus/test/unit/lbug-config-wal.test.ts b/gitnexus/test/unit/lbug-config-wal.test.ts new file mode 100644 index 000000000..6baea3621 --- /dev/null +++ b/gitnexus/test/unit/lbug-config-wal.test.ts @@ -0,0 +1,59 @@ +import { describe, expect, it, vi } from 'vitest'; +import { createLbugDatabase, isWalCorruptionError } from '../../src/core/lbug/lbug-config.js'; + +describe('isWalCorruptionError', () => { + it.each([ + [ + 'Corrupted wal file', + 'Runtime exception: Corrupted wal file. Read out invalid WAL record type.', + ], + ['invalid WAL record', 'Error: invalid WAL record type'], + ['WAL checksum', 'Checksum verification failed, the WAL file is corrupted.'], + ['WAL + corrupt', 'the WAL file is corrupted'], + ])('matches WAL corruption: %s', (_label, msg) => { + expect(isWalCorruptionError(msg)).toBe(true); + expect(isWalCorruptionError(new Error(msg))).toBe(true); + }); + + it.each([ + ['lock error', 'Could not set lock on file : /path/to/db'], + ['generic', 'Query failed'], + ['not found', 'LadybugDB not found at /path'], + ['checksum without WAL', 'Checksum verification failed for parquet file'], + ['permission path with WAL', "EACCES: permission denied '/path/to/wal'"], + ['schema mismatch WAL', 'schema version mismatch in WAL'], + ])('does not match non-WAL error: %s', (_label, msg) => { + expect(isWalCorruptionError(msg)).toBe(false); + }); + + it('handles non-string input', () => { + expect(isWalCorruptionError(undefined)).toBe(false); + expect(isWalCorruptionError(null)).toBe(false); + expect(isWalCorruptionError(42)).toBe(false); + expect(isWalCorruptionError(new Error('ok'))).toBe(false); + }); +}); + +describe('createLbugDatabase WAL replay option', () => { + it('passes throwOnWalReplayFailure and checksum constructor args explicitly', () => { + const Database = vi.fn(function (this: any) {}); + const lbugModule = { Database } as any; + + createLbugDatabase(lbugModule, '/tmp/lbug', { + readOnly: true, + throwOnWalReplayFailure: false, + }); + + expect(Database).toHaveBeenCalledWith( + '/tmp/lbug', + 0, + false, + true, + expect.any(Number), + true, + -1, + false, + true, + ); + }); +}); diff --git a/gitnexus/test/unit/mcp-wal-feedback.test.ts b/gitnexus/test/unit/mcp-wal-feedback.test.ts new file mode 100644 index 000000000..e387e0b97 --- /dev/null +++ b/gitnexus/test/unit/mcp-wal-feedback.test.ts @@ -0,0 +1,160 @@ +/** + * Tests for WAL corruption feedback in MCP error responses (#1402). + */ +import { beforeEach, describe, expect, it, vi } from 'vitest'; + +const { lbugMocks, platformMocks, repoMocks } = vi.hoisted(() => ({ + lbugMocks: { + initLbug: vi.fn().mockResolvedValue(undefined), + executeQuery: vi.fn(), + executeParameterized: vi.fn(), + closeLbug: vi.fn().mockResolvedValue(undefined), + isLbugReady: vi.fn().mockReturnValue(true), + isWriteQuery: vi.fn().mockReturnValue(false), + }, + platformMocks: { + isVectorExtensionSupportedByPlatform: vi.fn().mockReturnValue(true), + }, + repoMocks: { + listRegisteredRepos: vi.fn(), + }, +})); + +vi.mock('../../src/core/lbug/pool-adapter.js', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, ...lbugMocks }; +}); + +vi.mock('../../src/mcp/core/lbug-adapter.js', async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, ...lbugMocks }; +}); + +vi.mock('../../src/storage/repo-manager.js', () => ({ + listRegisteredRepos: repoMocks.listRegisteredRepos, + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), +})); + +vi.mock('../../src/core/git-staleness.js', () => ({ + checkStaleness: vi.fn().mockReturnValue({ isStale: false, commitsBehind: 0 }), + checkCwdMatch: vi.fn().mockResolvedValue({ match: 'none' }), +})); + +vi.mock('../../src/core/platform/capabilities.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + isVectorExtensionSupportedByPlatform: platformMocks.isVectorExtensionSupportedByPlatform, + }; +}); + +vi.mock('../../src/core/search/bm25-index.js', () => ({ + searchFTSFromLbug: vi.fn().mockResolvedValue([]), +})); + +vi.mock('../../src/mcp/core/embedder.js', () => ({ + embedQuery: vi.fn().mockResolvedValue([]), + getEmbeddingDims: vi.fn().mockReturnValue(384), +})); + +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; + +const MOCK_REPO_ENTRY = { + name: 'test-repo', + path: '/tmp/test', + storagePath: '/tmp/test/.gitnexus', + indexedAt: '2026-05-01T00:00:00Z', + lastCommit: 'abc1234', +}; + +async function makeBackend(): Promise { + const backend = new LocalBackend(); + await backend.init(); + return backend; +} + +describe('WAL corruption feedback in MCP responses (#1402)', () => { + beforeEach(() => { + vi.clearAllMocks(); + lbugMocks.initLbug.mockResolvedValue(undefined); + lbugMocks.executeQuery.mockResolvedValue([]); + lbugMocks.executeParameterized.mockResolvedValue([]); + lbugMocks.isLbugReady.mockReturnValue(true); + lbugMocks.isWriteQuery.mockReturnValue(false); + repoMocks.listRegisteredRepos.mockResolvedValue([MOCK_REPO_ENTRY]); + }); + + it('impact returns WAL suggestion on corrupted WAL error', async () => { + const backend = await makeBackend(); + lbugMocks.executeParameterized.mockRejectedValueOnce( + new Error('Runtime exception: Corrupted wal file. Read out invalid WAL record type.'), + ); + + const result = await backend.callTool('impact', { + repo: 'test-repo', + target: 'MyClass', + direction: 'upstream', + }); + + expect(result.error).toBeDefined(); + expect(result.suggestion).toBe( + 'The graph query failed — try gitnexus context as a fallback', + ); + expect(result.recoverySuggestion).toBeDefined(); + }); + + it('cypher returns WAL recoverySuggestion on corrupted WAL error', async () => { + const backend = await makeBackend(); + lbugMocks.executeQuery.mockRejectedValueOnce(new Error('Corrupted wal file')); + + const result = await backend.callTool('cypher', { + repo: 'test-repo', + query: 'MATCH (n) RETURN n LIMIT 1', + }); + + expect(result.error).toBe('Corrupted wal file'); + expect(result.recoverySuggestion).toBeDefined(); + }); + + it('context returns WAL recoverySuggestion on corrupted WAL error', async () => { + const backend = await makeBackend(); + lbugMocks.executeParameterized.mockRejectedValueOnce(new Error('Corrupted wal file')); + + const result = await backend.callTool('context', { + repo: 'test-repo', + name: 'MyClass', + }); + + expect(result.error).toBe('Corrupted wal file'); + expect(result.recoverySuggestion).toBeDefined(); + }); + + it('non-WAL errors do not include WAL suggestion', async () => { + const backend = await makeBackend(); + lbugMocks.executeParameterized.mockRejectedValueOnce(new Error('Some other error')); + + const result = await backend.callTool('impact', { + repo: 'test-repo', + target: 'MyClass', + direction: 'upstream', + }); + + expect(result.error).toBeDefined(); + expect(result.suggestion).toBe( + 'The graph query failed — try gitnexus context as a fallback', + ); + }); + + it('context preserves non-WAL throw behavior', async () => { + const backend = await makeBackend(); + lbugMocks.executeParameterized.mockRejectedValueOnce(new Error('Some other error')); + + await expect( + backend.callTool('context', { + repo: 'test-repo', + name: 'MyClass', + }), + ).rejects.toThrow('Some other error'); + }); +}); diff --git a/gitnexus/test/unit/mcp/group-repo-routing.test.ts b/gitnexus/test/unit/mcp/group-repo-routing.test.ts index ddf7d03a5..474abeaa8 100644 --- a/gitnexus/test/unit/mcp/group-repo-routing.test.ts +++ b/gitnexus/test/unit/mcp/group-repo-routing.test.ts @@ -32,7 +32,7 @@ vi.mock('../../../src/storage/repo-manager.js', () => ({ })); vi.mock('../../../src/core/search/bm25-index.js', () => ({ - searchFTSFromLbug: vi.fn().mockResolvedValue([]), + searchFTSFromLbug: vi.fn().mockResolvedValue({ results: [], ftsAvailable: true }), })); vi.mock('../../../src/mcp/core/embedder.js', () => ({ diff --git a/gitnexus/test/unit/pool-wal-recovery.test.ts b/gitnexus/test/unit/pool-wal-recovery.test.ts new file mode 100644 index 000000000..19b24c583 --- /dev/null +++ b/gitnexus/test/unit/pool-wal-recovery.test.ts @@ -0,0 +1,177 @@ +/** + * Tests for WAL corruption recovery in the connection pool (#1402). + * + * Mocks createLbugDatabase and fs to verify quarantine + retry behavior + * without needing a real LadybugDB instance or corrupted WAL file. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +const { stderrWriteMock } = vi.hoisted(() => ({ + stderrWriteMock: vi.fn(), +})); + +vi.mock('fs/promises', () => ({ + default: { + stat: vi.fn().mockResolvedValue({}), + unlink: vi.fn().mockResolvedValue(undefined), + rename: vi.fn().mockResolvedValue(undefined), + }, +})); + +vi.mock('@ladybugdb/core', () => ({ + default: { + Database: vi.fn(), + Connection: vi.fn(function (this: any) { + this.close = vi.fn().mockResolvedValue(undefined); + }), + }, +})); + +vi.mock('../../src/core/lbug/lbug-adapter.js', () => ({ + loadFTSExtension: vi.fn().mockResolvedValue(true), +})); + +vi.mock('../../src/core/lbug/lbug-config.js', () => ({ + createLbugDatabase: vi.fn(), + LBUG_MAX_DB_SIZE: 1024, + isWalCorruptionError: vi.fn((err: unknown) => { + const msg = err instanceof Error ? err.message : String(err ?? ''); + return /corrupt(ed)?\s+wal|invalid\s+wal\s+record/i.test(msg); + }), +})); + +vi.mock('../../src/mcp/stdio-capture.js', () => ({ + realStdoutWrite: vi.fn(), + realStderrWrite: stderrWriteMock, + setActiveStdoutWrite: vi.fn(), + getActiveStdoutWrite: vi.fn(() => vi.fn()), +})); + +import fs from 'fs/promises'; +import { createLbugDatabase } from '../../src/core/lbug/lbug-config.js'; + +const { closeLbug } = await import('../../src/core/lbug/pool-adapter.js'); + +const mockInit = vi.fn().mockResolvedValue(undefined); +const mockClose = vi.fn().mockResolvedValue(undefined); + +function makeMockDb() { + return { init: mockInit, close: mockClose, _isClosed: false } as any; +} + +describe('WAL corruption recovery in doInitLbug (#1402)', () => { + beforeEach(() => { + (createLbugDatabase as any).mockReset(); + (fs.stat as any).mockReset(); + (fs.rename as any).mockReset(); + mockInit.mockReset(); + mockClose.mockReset(); + mockInit.mockResolvedValue(undefined); + mockClose.mockResolvedValue(undefined); + (fs.stat as any).mockResolvedValue({}); + (fs.rename as any).mockResolvedValue(undefined); + }); + + afterEach(async () => { + vi.useRealTimers(); + await closeLbug().catch(() => {}); + vi.clearAllMocks(); + }); + + it('retries with WAL quarantine on corrupted WAL init error', async () => { + const { initLbug } = await import('../../src/core/lbug/pool-adapter.js'); + const dbPath = '/tmp/test-wal-recovery/lbug'; + + const badDb = makeMockDb(); + const goodDb = makeMockDb(); + badDb.init = vi.fn().mockRejectedValueOnce(new Error('Corrupted wal file')); + (createLbugDatabase as any).mockReturnValueOnce(badDb).mockReturnValueOnce(goodDb); + + await initLbug('test-repo-init', dbPath); + + expect(badDb.init).toHaveBeenCalledTimes(1); + expect(createLbugDatabase).toHaveBeenCalledTimes(2); + expect(createLbugDatabase).toHaveBeenCalledWith( + expect.anything(), + dbPath, + expect.objectContaining({ + readOnly: true, + throwOnWalReplayFailure: false, + }), + ); + expect(fs.rename).toHaveBeenCalledWith( + dbPath + '.wal', + expect.stringContaining('.wal.corrupt.'), + ); + expect(stderrWriteMock).toHaveBeenCalledWith( + expect.stringContaining('WAL quarantined for test-repo-init'), + ); + }); + + it('does not quarantine on lock error (preserves existing lock retry)', async () => { + const { initLbug } = await import('../../src/core/lbug/pool-adapter.js'); + const setTimeoutSpy = vi.spyOn(global, 'setTimeout').mockImplementation((callback: any) => { + callback(); + return 0 as any; + }); + const dbPath = '/tmp/test-wal-recovery/lbug'; + + (createLbugDatabase as any).mockImplementation(() => { + throw new Error('Could not set lock on file'); + }); + + try { + await expect(initLbug('test-repo-lock', dbPath)).rejects.toThrow(); + } finally { + setTimeoutSpy.mockRestore(); + } + + expect(fs.rename).not.toHaveBeenCalled(); + }); + + it('throws with analyze suggestion after retry also fails', async () => { + const { initLbug } = await import('../../src/core/lbug/pool-adapter.js'); + const dbPath = '/tmp/test-wal-recovery/lbug'; + + (createLbugDatabase as any) + .mockImplementationOnce(() => { + throw new Error('Corrupted wal file'); + }) + .mockImplementationOnce(() => { + throw new Error('Still broken'); + }); + + await expect(initLbug('test-repo-fail', dbPath)).rejects.toThrow(/gitnexus analyze/); + expect(createLbugDatabase).toHaveBeenCalledTimes(2); + }); + + it('does not reuse poisoned state after WAL failure', async () => { + const { initLbug, isLbugReady: ready } = await import('../../src/core/lbug/pool-adapter.js'); + const dbPath = '/tmp/test-wal-recovery/lbug'; + + (createLbugDatabase as any) + .mockImplementationOnce(() => { + throw new Error('Corrupted wal file'); + }) + .mockImplementationOnce(() => { + throw new Error('Still broken'); + }); + + await expect(initLbug('test-repo-nocache', dbPath)).rejects.toThrow(); + + expect(ready('test-repo-nocache')).toBe(false); + }); + + it('handles quarantine gracefully when .wal file does not exist', async () => { + const { initLbug } = await import('../../src/core/lbug/pool-adapter.js'); + const dbPath = '/tmp/test-wal-recovery/lbug'; + + (fs.rename as any).mockRejectedValueOnce(new Error('ENOENT: no such file')); + + (createLbugDatabase as any).mockImplementationOnce(() => { + throw new Error('Corrupted wal file'); + }); + + await expect(initLbug('test-repo-enoent', dbPath)).rejects.toThrow(/gitnexus analyze/); + }); +}); diff --git a/gitnexus/test/unit/publish.test.ts b/gitnexus/test/unit/publish.test.ts new file mode 100644 index 000000000..2ba767594 --- /dev/null +++ b/gitnexus/test/unit/publish.test.ts @@ -0,0 +1,316 @@ +import { afterEach, beforeEach, describe, expect, it, test, vi } from 'vitest'; +import fs from 'fs/promises'; +import os from 'os'; +import path from 'path'; +import { performance } from 'node:perf_hooks'; +import { + buildUqDispatchPayload, + isValidOwnerRepo, + parseOwnerRepoFromRemote, + stripGitSuffix, + UNDERSTAND_QUICKLY_TOKEN_ENV, +} from 'gitnexus-shared'; + +describe('understand-quickly helpers (gitnexus-shared)', () => { + describe('isValidOwnerRepo', () => { + it.each([ + ['looptech-ai/understand-quickly', true], + ['abhigyanpatwari/GitNexus', true], + // LOW 8: GitHub user/org slugs are alnum/hyphen only — no underscore. + ['Some_Org/Some.Repo-2', false], + ['', false], + ['just-a-name', false], + ['/Users/me/code/repo', false], + ['org/with spaces', false], + ['org//double', false], + // LOW 8 additions: + ['some_org/repo', false], // underscore in owner — invalid + ['-org/repo', false], // leading hyphen — invalid + ['org-/repo', false], // trailing hyphen — GitHub rejects at account creation; we mirror that here + ['org/repo_with_underscore', true], + ['org/.dotfile', true], // repos may start with dot + ])('returns %s for %j', (id, expected) => { + expect(isValidOwnerRepo(id as string)).toBe(expected); + }); + }); + + describe('stripGitSuffix (BLOCKER 1 — ReDoS-safe)', () => { + it.each([ + ['https://github.com/o/r.git', 'https://github.com/o/r'], + ['https://github.com/o/r.git/', 'https://github.com/o/r'], + ['https://github.com/o/r/', 'https://github.com/o/r'], + ['https://github.com/o/r', 'https://github.com/o/r'], + ['https://github.com/o/r.GIT', 'https://github.com/o/r'], + ['https://github.com/o/r//', 'https://github.com/o/r'], + ['', ''], + ['/', ''], + ])('strips %j -> %j', (input, expected) => { + expect(stripGitSuffix(input)).toBe(expected); + }); + + test('linear time on adversarial trailing slashes (regression for ReDoS)', () => { + const adversarial = 'https://github.com/o/r' + '/'.repeat(10_000); + const start = performance.now(); + const result = stripGitSuffix(adversarial); + const elapsed = performance.now() - start; + expect(result).toBe('https://github.com/o/r'); + expect(elapsed).toBeLessThan(50); // generous; should be sub-millisecond + }); + + test('parseOwnerRepoFromRemote terminates quickly on adversarial input', () => { + const adversarial = 'https://github.com/o/r.git' + '/'.repeat(10_000); + const start = performance.now(); + const result = parseOwnerRepoFromRemote(adversarial); + const elapsed = performance.now() - start; + expect(result).toBe('o/r'); + expect(elapsed).toBeLessThan(50); + }); + }); + + describe('parseOwnerRepoFromRemote', () => { + it.each([ + ['git@github.com:looptech-ai/understand-quickly.git', 'looptech-ai/understand-quickly'], + ['https://github.com/looptech-ai/understand-quickly', 'looptech-ai/understand-quickly'], + ['https://github.com/looptech-ai/understand-quickly.git', 'looptech-ai/understand-quickly'], + ['ssh://git@github.com/abhigyanpatwari/GitNexus.git', 'abhigyanpatwari/GitNexus'], + ])('parses %s -> %s', (url, expected) => { + expect(parseOwnerRepoFromRemote(url)).toBe(expected); + }); + + // LOW 9: non-GitHub remotes must be rejected — a wrong id is worse + // than no id, since the user can always pass --id explicitly. + it.each([ + ['https://gitlab.example.com/group/sub/project.git'], + ['git@gitlab.example.com:group/sub/project.git'], + ['https://bitbucket.org/team/repo.git'], + ])('returns null for non-GitHub host %j', (input) => { + expect(parseOwnerRepoFromRemote(input)).toBeNull(); + }); + + it.each([null, undefined, '', ' ', 'not-a-url', 'https://github.com/'])( + 'returns null for %j', + (input) => { + expect(parseOwnerRepoFromRemote(input as string | null | undefined)).toBeNull(); + }, + ); + }); + + describe('buildUqDispatchPayload', () => { + it('wraps the id in the registry-expected event shape', () => { + expect(buildUqDispatchPayload('looptech-ai/understand-quickly')).toEqual({ + event_type: 'sync-entry', + client_payload: { id: 'looptech-ai/understand-quickly' }, + }); + }); + + it('throws on a malformed id rather than building an invalid payload', () => { + expect(() => buildUqDispatchPayload('just-a-name')).toThrow(/owner\/repo/); + expect(() => buildUqDispatchPayload('/Users/me/repo')).toThrow(/owner\/repo/); + }); + }); +}); + +describe('publishCommand (no-token no-op)', () => { + let tempDir: string; + let originalToken: string | undefined; + let exitCodeBefore: number | undefined; + + beforeEach(async () => { + vi.resetModules(); + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-publish-test-')); + // Simulate an existing index so hasIndex() returns true. + await fs.mkdir(path.join(tempDir, '.gitnexus'), { recursive: true }); + await fs.writeFile( + path.join(tempDir, '.gitnexus', 'meta.json'), + JSON.stringify({ repoPath: tempDir, lastCommit: '', indexedAt: '' }), + 'utf-8', + ); + originalToken = process.env[UNDERSTAND_QUICKLY_TOKEN_ENV]; + delete process.env[UNDERSTAND_QUICKLY_TOKEN_ENV]; + exitCodeBefore = process.exitCode; + process.exitCode = 0; + }); + + afterEach(async () => { + if (originalToken !== undefined) { + process.env[UNDERSTAND_QUICKLY_TOKEN_ENV] = originalToken; + } else { + delete process.env[UNDERSTAND_QUICKLY_TOKEN_ENV]; + } + process.exitCode = exitCodeBefore; + await fs.rm(tempDir, { recursive: true, force: true }); + }); + + it('exits 0 without firing a network call when the token is unset', async () => { + const fetchSpy = vi.spyOn(globalThis, 'fetch').mockImplementation(() => { + throw new Error('publishCommand should NOT call fetch when the token is missing'); + }); + + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { id: 'looptech-ai/understand-quickly', skipGit: true }); + + expect(fetchSpy).not.toHaveBeenCalled(); + expect(process.exitCode ?? 0).toBe(0); + fetchSpy.mockRestore(); + }); + + it('exits 0 with no token even when no index/repo exists (BLOCKER 2)', async () => { + // Per the README, CLI --help, and PR body: without a token, the + // command must be a no-op even if the repo lacks `.gitnexus/`. + const noIndexDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-publish-noidx-')); + try { + const fetchSpy = vi.spyOn(globalThis, 'fetch').mockImplementation(() => { + throw new Error('publishCommand should NOT call fetch when the token is missing'); + }); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(noIndexDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(fetchSpy).not.toHaveBeenCalled(); + expect(process.exitCode ?? 0).toBe(0); + fetchSpy.mockRestore(); + } finally { + await fs.rm(noIndexDir, { recursive: true, force: true }); + } + }); +}); + +describe('publishCommand response branches (MEDIUM 5)', () => { + let tempDir: string; + let originalToken: string | undefined; + let exitCodeBefore: number | undefined; + let fetchSpy: ReturnType; + + beforeEach(async () => { + vi.resetModules(); + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-publish-resp-')); + await fs.mkdir(path.join(tempDir, '.gitnexus'), { recursive: true }); + await fs.writeFile( + path.join(tempDir, '.gitnexus', 'meta.json'), + JSON.stringify({ repoPath: tempDir, lastCommit: '', indexedAt: '' }), + 'utf-8', + ); + originalToken = process.env[UNDERSTAND_QUICKLY_TOKEN_ENV]; + process.env[UNDERSTAND_QUICKLY_TOKEN_ENV] = 'pat_test'; + exitCodeBefore = process.exitCode; + process.exitCode = 0; + fetchSpy = vi.spyOn(globalThis, 'fetch'); + }); + + afterEach(async () => { + if (originalToken !== undefined) { + process.env[UNDERSTAND_QUICKLY_TOKEN_ENV] = originalToken; + } else { + delete process.env[UNDERSTAND_QUICKLY_TOKEN_ENV]; + } + process.exitCode = exitCodeBefore; + vi.restoreAllMocks(); + await fs.rm(tempDir, { recursive: true, force: true }); + }); + + function mockResponse(status: number, body = '') { + fetchSpy.mockResolvedValueOnce({ + status, + ok: status >= 200 && status < 300, + text: async () => body, + body: { cancel: async () => {} }, + headers: new Headers(), + } as unknown as Response); + } + + it('204 → exit 0 with success message', async () => { + mockResponse(204); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(fetchSpy).toHaveBeenCalledTimes(1); + expect(process.exitCode ?? 0).toBe(0); + }); + + it('401 → exit 1 with PAT-invalid hint', async () => { + mockResponse(401, '{"message":"Bad credentials"}'); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(process.exitCode).toBe(1); + }); + + it('403 → exit 1 with scope-missing hint', async () => { + mockResponse(403, '{"message":"Resource not accessible"}'); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(process.exitCode).toBe(1); + }); + + it('404 → exit 1 with repo-access hint', async () => { + mockResponse(404, '{"message":"Not Found"}'); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(process.exitCode).toBe(1); + }); + + it('5xx → exit 1 with raw body', async () => { + mockResponse(503, 'gateway timeout'); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(process.exitCode).toBe(1); + }); + + it('network throw → exit 1', async () => { + fetchSpy.mockRejectedValueOnce(new Error('ECONNRESET')); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(process.exitCode).toBe(1); + }); + + it('TimeoutError (HIGH 4 — fetch timeout) → exit 1 with timed-out message', async () => { + // `AbortSignal.timeout()` throws a real `DOMException` with + // `name === 'TimeoutError'`. Faking it as `Error{name:'AbortError'}` + // (the previous shape of this test) hid a mismatch in publish.ts — + // the catch branch only matched 'AbortError' and the user-facing + // "timed out" message never fired in production. + const abort = new DOMException('The operation was aborted due to timeout', 'TimeoutError'); + fetchSpy.mockRejectedValueOnce(abort); + const errSpy = vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + expect(process.exitCode).toBe(1); + const written = errSpy.mock.calls.map((c) => String(c[0])).join(''); + expect(written).toMatch(/timed out/i); + errSpy.mockRestore(); + }); + + it('token never appears in any logged output', async () => { + process.env[UNDERSTAND_QUICKLY_TOKEN_ENV] = 'pat_secret_value'; + mockResponse(401, ''); + const errSpy = vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + const { publishCommand } = await import('../../src/cli/publish.js'); + await publishCommand(tempDir, { + id: 'looptech-ai/understand-quickly', + skipGit: true, + }); + const written = errSpy.mock.calls.map((c) => String(c[0])).join(''); + expect(written).not.toContain('pat_secret_value'); + errSpy.mockRestore(); + }); +}); diff --git a/gitnexus/test/unit/schema.test.ts b/gitnexus/test/unit/schema.test.ts index cab80984a..17db9bdf9 100644 --- a/gitnexus/test/unit/schema.test.ts +++ b/gitnexus/test/unit/schema.test.ts @@ -13,6 +13,7 @@ import { CLASS_SCHEMA, INTERFACE_SCHEMA, METHOD_SCHEMA, + PROPERTY_SCHEMA, CODE_ELEMENT_SCHEMA, COMMUNITY_SCHEMA, PROCESS_SCHEMA, @@ -117,6 +118,11 @@ describe('LadybugDB Schema', () => { expect(FUNCTION_SCHEMA).toContain('isExported BOOLEAN'); }); + it('Property schema preserves declaredType', () => { + expect(SCHEMA_QUERIES).toContain(PROPERTY_SCHEMA); + expect(PROPERTY_SCHEMA).toContain('declaredType STRING'); + }); + it('Community schema has heuristicLabel and cohesion', () => { expect(COMMUNITY_SCHEMA).toContain('heuristicLabel STRING'); expect(COMMUNITY_SCHEMA).toContain('cohesion DOUBLE'); diff --git a/gitnexus/test/unit/scope-resolution/csharp/csharp-captures.test.ts b/gitnexus/test/unit/scope-resolution/csharp/csharp-captures.test.ts index c8111d203..9fa9bf181 100644 --- a/gitnexus/test/unit/scope-resolution/csharp/csharp-captures.test.ts +++ b/gitnexus/test/unit/scope-resolution/csharp/csharp-captures.test.ts @@ -422,4 +422,47 @@ describe('emitCsharpScopeCaptures — references', () => { expect(m!['@reference.receiver'].text).toBe('obj'); expect(m!['@reference.name'].text).toBe('Name'); }); + + it('captures member reads `obj.Name`', () => { + const m = findMatch('class A { void M(User obj) { var name = obj.Name; } }', (t) => + t.includes('@reference.read.member'), + ); + expect(m).toBeDefined(); + expect(m!['@reference.receiver'].text).toBe('obj'); + expect(m!['@reference.name'].text).toBe('Name'); + }); + + it('does not capture member calls as member reads', () => { + const matches = emitCsharpScopeCaptures( + 'class A { void M(User obj) { obj.Save(); } }', + 'test.cs', + ); + expect(matches.some((m) => '@reference.call.member' in m)).toBe(true); + expect(matches.some((m) => '@reference.read.member' in m)).toBe(false); + }); + + it('captures generic type arguments as type references', () => { + const matches = emitCsharpScopeCaptures( + 'class A : IEntityTypeConfiguration { public Task> Load(List users) => null!; }', + 'test.cs', + ); + const names = matches + .filter((m) => '@reference.type' in m) + .map((m) => m['@reference.name'].text); + + expect(names).toContain('USER_INFO'); + expect(names).not.toContain('string'); + }); + + it('captures call-site generic type arguments as type references', () => { + const matches = emitCsharpScopeCaptures( + 'class A { void M(IRepo repo) { repo.Get(); } }', + 'test.cs', + ); + const names = matches + .filter((m) => '@reference.type' in m) + .map((m) => m['@reference.name'].text); + + expect(names).toContain('USER_INFO'); + }); }); diff --git a/gitnexus/test/unit/staleness.test.ts b/gitnexus/test/unit/staleness.test.ts index b8ea398ff..1645d9987 100644 --- a/gitnexus/test/unit/staleness.test.ts +++ b/gitnexus/test/unit/staleness.test.ts @@ -8,7 +8,7 @@ */ import { describe, it, expect } from 'vitest'; import { execFileSync } from 'child_process'; -import { checkStaleness } from '../../src/core/git-staleness.js'; +import { checkStaleness, checkStalenessAsync } from '../../src/core/git-staleness.js'; // We test checkStaleness with a real git repo (the project itself) // since mocking execFileSync across ESM modules is complex. @@ -65,3 +65,82 @@ describe('checkStaleness', () => { expect(result.commitsBehind).toBe(0); }); }); + +describe('checkStalenessAsync', () => { + it('returns not stale when HEAD matches lastCommit', async () => { + let headCommit: string; + try { + headCommit = execFileSync('git', ['rev-parse', 'HEAD'], { + encoding: 'utf-8', + stdio: ['pipe', 'pipe', 'pipe'], + }).trim(); + } catch { + return; + } + + const result = await checkStalenessAsync(process.cwd(), headCommit); + expect(result.isStale).toBe(false); + expect(result.commitsBehind).toBe(0); + expect(result.hint).toBeUndefined(); + }); + + it('returns stale when lastCommit is behind HEAD', async () => { + let previousCommit: string; + try { + previousCommit = execFileSync('git', ['rev-parse', 'HEAD~1'], { + encoding: 'utf-8', + stdio: ['pipe', 'pipe', 'pipe'], + }).trim(); + } catch { + return; + } + + if (!previousCommit) return; + + const result = await checkStalenessAsync(process.cwd(), previousCommit); + expect(result.isStale).toBe(true); + expect(result.commitsBehind).toBeGreaterThan(0); + expect(result.hint).toContain('behind HEAD'); + }); + + it('fails open when git command fails (e.g., invalid path)', async () => { + const result = await checkStalenessAsync('/nonexistent/path', 'abc123'); + expect(result.isStale).toBe(false); + expect(result.commitsBehind).toBe(0); + }); + + it('fails open with invalid commit hash', async () => { + const result = await checkStalenessAsync(process.cwd(), 'not-a-real-commit-hash'); + expect(result.isStale).toBe(false); + expect(result.commitsBehind).toBe(0); + }); + + it('parallel calls complete faster than sequential', async () => { + let headCommit: string; + try { + headCommit = execFileSync('git', ['rev-parse', 'HEAD'], { + encoding: 'utf-8', + stdio: ['pipe', 'pipe', 'pipe'], + }).trim(); + } catch { + return; + } + + const cwd = process.cwd(); + const N = 10; + + // Parallel + const t0 = performance.now(); + await Promise.all(Array.from({ length: N }, () => checkStalenessAsync(cwd, headCommit))); + const parallelMs = performance.now() - t0; + + // Sequential sync + const t1 = performance.now(); + for (let i = 0; i < N; i++) checkStaleness(cwd, headCommit); + const sequentialMs = performance.now() - t1; + + // Parallel should be meaningfully faster than sequential. + // Use a generous ratio to avoid flakiness on slow CI machines. + expect(parallelMs).toBeLessThan(sequentialMs * 1.5); + }); +}); diff --git a/gitnexus/test/unit/u8-redos-resource-exhaustion.test.ts b/gitnexus/test/unit/u8-redos-resource-exhaustion.test.ts new file mode 100644 index 000000000..6dd6d9955 --- /dev/null +++ b/gitnexus/test/unit/u8-redos-resource-exhaustion.test.ts @@ -0,0 +1,196 @@ +/** + * Regression tests for U8 — closes: + * #186 js/redos rust-workspace-extractor.ts + * #187 js/redos cobol-preprocessor.ts + * #184 js/resource-exhaustion cross-impact.ts + * + * These tests import the production symbols directly. A previous shape + * dynamic-imported names that did not exist (`extractRustWorkspace` vs. + * the real `extractRustWorkspaceLinks`) and `??`-fell-back to inline + * regex copies, so the tests stayed green even when the production + * fixes regressed. Static imports + named symbols make a regression in + * any of the three sites a hard test failure. + */ +import { describe, expect, it } from 'vitest'; +import { RE_SET_TO_TRUE, RE_SET_INDEX } from '../../src/core/ingestion/cobol/cobol-preprocessor.js'; +import { parseCargoPackageName } from '../../src/core/group/extractors/rust-workspace-extractor.js'; +import { + clampTimeout, + IMPACT_TIMEOUT_MIN_MS, + IMPACT_TIMEOUT_MAX_MS, +} from '../../src/core/group/cross-impact.js'; + +/** + * Time a single regex.exec call. Used by the linearity tests below to + * compute a 10k/5k ratio in addition to the absolute <500ms bound. + * + * Ratio assertions catch sub-exponential O(n²) regressions that fit + * inside the absolute cap on warm CI; the absolute cap catches + * catastrophic backtracking on cold CI. Two complementary signals. + */ +function timeRegex(re: RegExp, input: string): number { + // Reset regex.lastIndex for global/sticky regexes — ours are not, but + // be defensive in case future shape changes add the `g` flag. + re.lastIndex = 0; + const start = performance.now(); + re.exec(input); + return performance.now() - start; +} + +function timeFn(fn: () => T): number { + const start = performance.now(); + fn(); + return performance.now() - start; +} + +// Linear scaling is ~2.0× when input doubles; 3.0× allows generous +// slack for CI-runner GC and tier-up jitter. An O(n²) regression on a +// 2× input takes ~4× as long, well outside this bound. +const LINEAR_RATIO_BOUND = 3.0; + +/** + * Minimum elapsed time (in ms) below which `performance.now()` ratios + * are dominated by scheduler jitter and become meaningless. When both + * timed runs come in below this floor, we skip the ratio assertion — + * the absolute <500ms bound still catches catastrophic backtracking, + * and the next CI run will measure higher absolute times that the + * ratio assertion can evaluate reliably. + * + * Calibrated empirically: a flake on macOS reported ratio 5.29× + * between two sub-millisecond measurements (~0.5ms vs ~2.6ms), both + * genuinely linear but indistinguishable from noise. 5ms is a + * comfortable floor where individual measurements are well-separated + * from the ~10-100µs `performance.now()` resolution band. + */ +const RATIO_MEASUREMENT_FLOOR_MS = 5; + +/** + * Assert linear scaling between two timed runs on inputs that differ + * by 2×. When measurements are too small to be reliable, the ratio + * assertion is skipped (the absolute bound still fires elsewhere). + */ +function assertSubLinearRatio(elapsedSmall: number, elapsedLarge: number, label: string): void { + if (elapsedSmall < RATIO_MEASUREMENT_FLOOR_MS && elapsedLarge < RATIO_MEASUREMENT_FLOOR_MS) { + // Both runs completed faster than the noise floor — the ratio is + // not meaningful. The absolute <500ms bound elsewhere in this + // describe block still pins linearity; we skip rather than risk a + // flake on a genuinely-linear implementation. + return; + } + const ratio = elapsedLarge / Math.max(elapsedSmall, 0.001); + if (ratio >= LINEAR_RATIO_BOUND) { + throw new Error( + `${label}: ratio ${ratio.toFixed(2)}× exceeds bound ${LINEAR_RATIO_BOUND}× ` + + `(small=${elapsedSmall.toFixed(2)}ms, large=${elapsedLarge.toFixed(2)}ms)`, + ); + } +} + +describe('cobol-preprocessor RE_SET_TO_TRUE — linear time on pathological input', () => { + it('matches in <500ms on 50k repetitions of "A OF A " AND 100k/50k ratio is sub-linear when measurable', () => { + // 50k/100k repetitions chosen so timings exceed the + // RATIO_MEASUREMENT_FLOOR_MS noise floor on typical CI hardware. + // Pre-fix nested-quantifier shape would be exponential here; the + // post-fix `.+?` shape is linear (~2× when input doubles). + const inputSmall = 'SET ' + 'A OF A '.repeat(50_000) + 'TO TRUE'; + const inputLarge = 'SET ' + 'A OF A '.repeat(100_000) + 'TO TRUE'; + const elapsedSmall = timeRegex(RE_SET_TO_TRUE, inputSmall); + const elapsedLarge = timeRegex(RE_SET_TO_TRUE, inputLarge); + expect(RE_SET_TO_TRUE.exec(inputSmall)).not.toBeNull(); + expect(elapsedSmall).toBeLessThan(500); + expect(elapsedLarge).toBeLessThan(500); + assertSubLinearRatio(elapsedSmall, elapsedLarge, 'RE_SET_TO_TRUE'); + }); + + it('still matches a normal SET ... TO TRUE statement', () => { + const m = RE_SET_TO_TRUE.exec('SET WS-FLAG TO TRUE'); + expect(m).not.toBeNull(); + expect(m?.[1]).toBe('WS-FLAG'); + }); +}); + +describe('cobol-preprocessor RE_SET_INDEX — linear time on pathological input', () => { + it('rejects in <500ms on 50k tokens with no valid suffix AND 100k/50k ratio is sub-linear when measurable', () => { + // Forces backtracking against the (TO|UP\s+BY|DOWN\s+BY) alternation + // — the richer pathological surface of the two regexes. + const inputSmall = 'SET ' + 'A '.repeat(50_000) + 'X'; + const inputLarge = 'SET ' + 'A '.repeat(100_000) + 'X'; + const elapsedSmall = timeRegex(RE_SET_INDEX, inputSmall); + const elapsedLarge = timeRegex(RE_SET_INDEX, inputLarge); + expect(RE_SET_INDEX.exec(inputSmall)).toBeNull(); + expect(elapsedSmall).toBeLessThan(500); + expect(elapsedLarge).toBeLessThan(500); + assertSubLinearRatio(elapsedSmall, elapsedLarge, 'RE_SET_INDEX'); + }); + + it('still matches a normal SET INDEX statement', () => { + const m = RE_SET_INDEX.exec('SET WS-IDX TO 5'); + expect(m).not.toBeNull(); + expect(m?.[1]).toBe('WS-IDX'); + expect(m?.[2]).toBe('TO'); + expect(m?.[3]).toBe('5'); + }); +}); + +describe('rust-workspace parseCargoPackageName — linear-time line walk', () => { + it('extracts the package name in <500ms on 100k blank lines AND 200k/100k ratio is sub-linear when measurable', () => { + // 100k/200k blank lines chosen so timings exceed the + // RATIO_MEASUREMENT_FLOOR_MS noise floor. Earlier 10k/20k pairing + // produced sub-millisecond measurements where scheduler jitter + // dominated and the ratio became meaningless (a real macOS run + // saw 5.29× between two genuinely-linear sub-ms measurements). + const cargoTomlSmall = + '[package]\n' + '\n'.repeat(100_000) + 'name = "myrepo"\nversion = "0.1.0"\n'; + const cargoTomlLarge = + '[package]\n' + '\n'.repeat(200_000) + 'name = "myrepo"\nversion = "0.1.0"\n'; + const elapsedSmall = timeFn(() => parseCargoPackageName(cargoTomlSmall)); + const elapsedLarge = timeFn(() => parseCargoPackageName(cargoTomlLarge)); + expect(parseCargoPackageName(cargoTomlSmall)).toBe('myrepo'); + expect(elapsedSmall).toBeLessThan(500); + expect(elapsedLarge).toBeLessThan(500); + assertSubLinearRatio(elapsedSmall, elapsedLarge, 'parseCargoPackageName'); + }); + + it('returns null when [package] section is absent', () => { + expect(parseCargoPackageName('[workspace]\nmembers = ["a"]\n')).toBeNull(); + }); + + it('stops at the next section header (does not pick up a name= from a later section)', () => { + const toml = '[package]\nversion = "1.0"\n[other]\nname = "wrong"\n'; + expect(parseCargoPackageName(toml)).toBeNull(); + }); + + it('extracts the name from a normal [package] section', () => { + const toml = '[package]\nname = "real-crate"\nversion = "0.1.0"\n'; + expect(parseCargoPackageName(toml)).toBe('real-crate'); + }); +}); + +describe('cross-impact clampTimeout — bounds user-supplied impact timeouts', () => { + it('rejects negative and zero timeouts, returning MIN', () => { + expect(clampTimeout(0)).toBe(IMPACT_TIMEOUT_MIN_MS); + expect(clampTimeout(-1)).toBe(IMPACT_TIMEOUT_MIN_MS); + expect(clampTimeout(-999_999)).toBe(IMPACT_TIMEOUT_MIN_MS); + }); + + it('rejects NaN/Infinity, returning MIN', () => { + expect(clampTimeout(NaN)).toBe(IMPACT_TIMEOUT_MIN_MS); + expect(clampTimeout(Infinity)).toBe(IMPACT_TIMEOUT_MIN_MS); + expect(clampTimeout(-Infinity)).toBe(IMPACT_TIMEOUT_MIN_MS); + }); + + it('caps very large timeouts at MAX (5 minutes)', () => { + expect(clampTimeout(999_999_999)).toBe(IMPACT_TIMEOUT_MAX_MS); + expect(clampTimeout(IMPACT_TIMEOUT_MAX_MS + 1)).toBe(IMPACT_TIMEOUT_MAX_MS); + }); + + it('passes through a reasonable timeout unchanged (truncated to integer)', () => { + expect(clampTimeout(30_000)).toBe(30_000); + expect(clampTimeout(30_500.7)).toBe(30_500); + }); + + it('floors below-MIN positive values to MIN', () => { + expect(clampTimeout(50)).toBe(IMPACT_TIMEOUT_MIN_MS); + expect(clampTimeout(0.1)).toBe(IMPACT_TIMEOUT_MIN_MS); + }); +}); diff --git a/gitnexus/vitest.config.ts b/gitnexus/vitest.config.ts index 9330e86a5..862357668 100644 --- a/gitnexus/vitest.config.ts +++ b/gitnexus/vitest.config.ts @@ -60,6 +60,8 @@ export default defineConfig({ 'test/integration/augmentation.test.ts', 'test/integration/staleness-and-stability.test.ts', 'test/integration/lbug-lock-retry.test.ts', + 'test/integration/lbug-open-retry.test.ts', + 'test/integration/lbug-close-handle-release.test.ts', 'test/integration/api-impact-e2e.test.ts', 'test/integration/shape-check-regression.test.ts', 'test/integration/java-class-impact.test.ts', @@ -87,6 +89,8 @@ export default defineConfig({ 'test/integration/augmentation.test.ts', 'test/integration/staleness-and-stability.test.ts', 'test/integration/lbug-lock-retry.test.ts', + 'test/integration/lbug-open-retry.test.ts', + 'test/integration/lbug-close-handle-release.test.ts', 'test/integration/api-impact-e2e.test.ts', 'test/integration/shape-check-regression.test.ts', 'test/integration/java-class-impact.test.ts',