mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-05 02:43:32 +00:00
Merge branch 'main' into fix/sanitize-repo-name-rebased-v2
This commit is contained in:
commit
5d55e8f85e
21 changed files with 2967 additions and 446 deletions
595
.github/workflows/pr-autofix-apply.yml
vendored
Normal file
595
.github/workflows/pr-autofix-apply.yml
vendored
Normal file
|
|
@ -0,0 +1,595 @@
|
|||
name: PR Autofix (apply)
|
||||
|
||||
# CHATOPS HALF of the autofix pipeline.
|
||||
#
|
||||
# Triggered when a contributor comments `/autofix` on a PR. Validates
|
||||
# permission, locates the most recent successful `pr-autofix.yml`
|
||||
# artifact for the PR's current head SHA, applies the patch to the PR
|
||||
# head, and pushes a commit back to the PR branch.
|
||||
#
|
||||
# This workflow runs from the default branch's copy of the file
|
||||
# regardless of where the comment originates -- that's the trust
|
||||
# anchor. Comment body and author login are untrusted; both flow
|
||||
# through env vars and pattern-matched, never interpolated into shell.
|
||||
#
|
||||
# Fork PR support: `git push` with the GITHUB_TOKEN succeeds against
|
||||
# fork branches only when the contributor enabled "Allow edits by
|
||||
# maintainers" on the PR (the default). When they disabled it, we
|
||||
# fail loud with a 👎 reaction and an explanation comment.
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
concurrency:
|
||||
# Per-PR scope. issue_comment events expose `github.event.issue.number`
|
||||
# for both PR and Issue comments; the `pull_request != null` guard on
|
||||
# the job ensures we only run on PRs, so this number is the PR number.
|
||||
# cancel-in-progress: false — a second `/autofix` should wait for the
|
||||
# first to finish (idempotency check on the second invocation handles
|
||||
# the no-op case).
|
||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.event.issue.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions: {}
|
||||
|
||||
jobs:
|
||||
apply:
|
||||
name: apply-autofix
|
||||
# Pre-filter at the workflow level so non-PR comments and unrelated
|
||||
# comments don't even spawn a runner. The job-level body re-check
|
||||
# below (Step 1) is the strict gate.
|
||||
if: >-
|
||||
github.event.issue.pull_request != null
|
||||
&& startsWith(github.event.comment.body, '/autofix')
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
# React on the triggering comment + post reply comments.
|
||||
pull-requests: write
|
||||
# Push the apply commit to the PR head branch.
|
||||
contents: write
|
||||
# Required by actions/download-artifact to fetch artifacts produced
|
||||
# by a different workflow run.
|
||||
actions: read
|
||||
steps:
|
||||
- name: Validate comment body precisely
|
||||
id: body
|
||||
env:
|
||||
BODY: ${{ github.event.comment.body }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# Whole-line, case-sensitive match: `^/autofix\s*$`. The
|
||||
# workflow-level startsWith guard is coarse — `please don't
|
||||
# /autofix this code` would pass that filter but fail this one.
|
||||
# We exit silently (no reaction) on body mismatch so quoted
|
||||
# text in unrelated discussions doesn't get a visible response.
|
||||
if [[ ! "${BODY}" =~ ^/autofix[[:space:]]*$ ]]; then
|
||||
echo "Body did not match strict /autofix regex — exiting silently."
|
||||
echo "match=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "match=true" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Validate commenter permission
|
||||
id: perm
|
||||
if: steps.body.outputs.match == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
COMMENTER: ${{ github.event.comment.user.login }}
|
||||
PR_AUTHOR: ${{ github.event.issue.user.login }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# Retry wrapper for transient 5xx / 429 / network blips.
|
||||
# Mirrors the helper in pr-autofix-publish.yml. Used on
|
||||
# idempotent GETs only; reactions/comment-POSTs are NOT
|
||||
# wrapped (retrying a POST would dupe the resource).
|
||||
gh_retry() {
|
||||
local n=0 max=3
|
||||
while true; do
|
||||
if gh "$@"; then return 0; fi
|
||||
n=$((n+1))
|
||||
if [ "$n" -ge "$max" ]; then return 1; fi
|
||||
sleep $((n * 2))
|
||||
done
|
||||
}
|
||||
|
||||
# Allowlist the commenter login before it flows into a URL.
|
||||
# GitHub usernames: alphanumeric + dashes, max 39 chars.
|
||||
if ! [[ "${COMMENTER}" =~ ^[A-Za-z0-9-]{1,39}$ ]]; then
|
||||
echo "::error::Invalid commenter login format: $(printf '%q' "${COMMENTER}")"
|
||||
echo "allowed=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Self-comparison: PR author can always /autofix their own PR.
|
||||
if [ "${COMMENTER}" = "${PR_AUTHOR}" ]; then
|
||||
echo "Commenter is PR author — granting access."
|
||||
echo "allowed=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Repo permission lookup. admin/write/maintain are sufficient.
|
||||
# Distinguish API failure (5xx, 429, network) from genuine
|
||||
# permission denial (404 = not a collaborator). Conflating them
|
||||
# would silently refuse a legitimate maintainer with a public
|
||||
# 👎 every time GitHub blips. gh_retry handles transient blips;
|
||||
# the stderr-grep distinguishes 404 from persistent failure.
|
||||
perm_stderr=$(mktemp)
|
||||
if permission=$(gh_retry api "repos/${GH_REPO}/collaborators/${COMMENTER}/permission" \
|
||||
--jq '.permission' 2>"$perm_stderr"); then
|
||||
echo "Commenter permission: ${permission}"
|
||||
case "${permission}" in
|
||||
admin|write|maintain)
|
||||
echo "allowed=true" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
*)
|
||||
echo "allowed=false" >> "$GITHUB_OUTPUT"
|
||||
;;
|
||||
esac
|
||||
else
|
||||
err=$(cat "$perm_stderr")
|
||||
echo "Permission lookup stderr: ${err}" >&2
|
||||
# 404 (not a collaborator) is a genuine deny.
|
||||
# Anything else is a transient API/network failure.
|
||||
if grep -qE "HTTP 404|Not Found" "$perm_stderr"; then
|
||||
echo "allowed=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "::error::Permission lookup failed transiently — refusing to act."
|
||||
echo "allowed=api-failed" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
fi
|
||||
|
||||
- name: React 😕 on transient permission-API failure
|
||||
if: steps.body.outputs.match == 'true' && steps.perm.outputs.allowed == 'api-failed'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
COMMENT_ID: ${{ github.event.comment.id }}
|
||||
PR: ${{ github.event.issue.number }}
|
||||
RUN_ID: ${{ github.run_id }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="confused" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⚠️ Couldn't verify your repo permission (transient GitHub API failure). Please comment \`/autofix\` again. ([apply run](https://github.com/${GH_REPO}/actions/runs/${RUN_ID}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
|
||||
- name: React 👎 on permission denial
|
||||
if: steps.body.outputs.match == 'true' && steps.perm.outputs.allowed == 'false'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
COMMENT_ID: ${{ github.event.comment.id }}
|
||||
PR: ${{ github.event.issue.number }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="🚫 \`/autofix\` is restricted to users with write access or the PR author. Comment ignored." \
|
||||
>/dev/null
|
||||
# Hard exit so the rest of the job is skipped.
|
||||
exit 1
|
||||
|
||||
- name: React 👀 to acknowledge
|
||||
if: steps.perm.outputs.allowed == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
COMMENT_ID: ${{ github.event.comment.id }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="eyes" >/dev/null
|
||||
|
||||
- name: Resolve PR head and locate autofix run
|
||||
id: locate
|
||||
if: steps.perm.outputs.allowed == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
PR: ${{ github.event.issue.number }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# Same retry wrapper used in the permission step, repeated
|
||||
# because each YAML `run:` block is a fresh bash session.
|
||||
gh_retry() {
|
||||
local n=0 max=3
|
||||
while true; do
|
||||
if gh "$@"; then return 0; fi
|
||||
n=$((n+1))
|
||||
if [ "$n" -ge "$max" ]; then return 1; fi
|
||||
sleep $((n * 2))
|
||||
done
|
||||
}
|
||||
|
||||
# Fetch PR metadata. All fields here are server-controlled API
|
||||
# output, but we still allowlist before exporting so anything
|
||||
# weird short-circuits before $GITHUB_OUTPUT. Wrapped in
|
||||
# gh_retry so transient blips don't surface as "no autofix run
|
||||
# found" with a wrong remediation.
|
||||
if ! pr_json=$(gh_retry api "repos/${GH_REPO}/pulls/${PR}"); then
|
||||
echo "::error::PR metadata fetch failed after retries."
|
||||
echo "found_status=api-failed" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
head_sha=$(jq -r '.head.sha' <<< "${pr_json}")
|
||||
head_ref=$(jq -r '.head.ref' <<< "${pr_json}")
|
||||
head_repo=$(jq -r '.head.repo.full_name' <<< "${pr_json}")
|
||||
|
||||
[[ "${head_sha}" =~ ^[0-9a-f]{40}$ ]] || { echo "::error::Bad head_sha"; exit 1; }
|
||||
[[ "${head_ref}" =~ ^[A-Za-z0-9._/-]+$ ]] || { echo "::error::Bad head_ref"; exit 1; }
|
||||
[[ "${head_repo}" =~ ^[A-Za-z0-9._-]+/[A-Za-z0-9._-]+$ ]] || { echo "::error::Bad head_repo"; exit 1; }
|
||||
|
||||
# Find the latest successful pr-autofix.yml run for this head SHA.
|
||||
if ! runs_json=$(gh_retry api "repos/${GH_REPO}/actions/workflows/pr-autofix.yml/runs?head_sha=${head_sha}&per_page=10"); then
|
||||
echo "::error::Workflow run lookup failed after retries."
|
||||
echo "found_status=api-failed" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
run_id=$(jq -r '[.workflow_runs[] | select(.conclusion == "success")] | .[0].id // empty' <<< "${runs_json}")
|
||||
|
||||
if [ -n "${run_id}" ] && [[ "${run_id}" =~ ^[0-9]+$ ]]; then
|
||||
echo "found_status=success" >> "$GITHUB_OUTPUT"
|
||||
{
|
||||
echo "found=true"
|
||||
echo "head_sha=${head_sha}"
|
||||
echo "head_ref=${head_ref}"
|
||||
echo "head_repo=${head_repo}"
|
||||
echo "run_id=${run_id}"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# No successful run. Distinguish "still running" (producer in
|
||||
# flight after a recent push) from "never ran / all failed".
|
||||
# in_progress / queued / pending / waiting cover the GitHub
|
||||
# workflow-run lifecycle states that precede success/failure.
|
||||
in_progress=$(jq -r '[.workflow_runs[] | select(.status == "in_progress" or .status == "queued" or .status == "pending" or .status == "waiting")] | length' <<< "${runs_json}")
|
||||
if [ "${in_progress:-0}" -gt 0 ]; then
|
||||
echo "::warning::pr-autofix run is still in progress for head ${head_sha}."
|
||||
echo "found_status=in-progress" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "::warning::No successful pr-autofix run found for head ${head_sha}."
|
||||
echo "found_status=not-found" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
# Existing `found` boolean is preserved so downstream gates
|
||||
# (`steps.locate.outputs.found == 'true'`) still work.
|
||||
echo "found=false" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Reply when locate did not yield a usable run
|
||||
if: steps.perm.outputs.allowed == 'true' && steps.locate.outputs.found != 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
COMMENT_ID: ${{ github.event.comment.id }}
|
||||
PR: ${{ github.event.issue.number }}
|
||||
FOUND_STATUS: ${{ steps.locate.outputs.found_status }}
|
||||
RUN_ID: ${{ github.run_id }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
run_url="https://github.com/${GH_REPO}/actions/runs/${RUN_ID}"
|
||||
case "${FOUND_STATUS}" in
|
||||
in-progress)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="confused" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⏳ A pr-autofix run is still in progress for this PR's current head SHA. Wait for it to finish, then comment \`/autofix\` again. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
;;
|
||||
api-failed)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="confused" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⚠️ Couldn't reach the GitHub API to look up the autofix run (transient failure after retries). Please comment \`/autofix\` again. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
;;
|
||||
*)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="🤔 No successful autofix run found for this PR's current head SHA. Push a new commit to trigger one, then comment \`/autofix\` again." \
|
||||
>/dev/null
|
||||
;;
|
||||
esac
|
||||
exit 1
|
||||
|
||||
# Pinned to v8.0.1. Same SHA as pr-autofix-publish.yml.
|
||||
# `continue-on-error: true` lets the workflow proceed when the
|
||||
# artifact is expired or pruned (1-day retention). The apply
|
||||
# step distinguishes "patch file missing entirely" (artifact-
|
||||
# expired) from "patch file zero bytes" (genuinely empty patch).
|
||||
- name: Download autofix artifact
|
||||
if: steps.locate.outputs.found == 'true'
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
continue-on-error: true
|
||||
with:
|
||||
name: autofix
|
||||
run-id: ${{ steps.locate.outputs.run_id }}
|
||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
||||
path: autofix-in
|
||||
|
||||
# Pinned to v5.0.4. Verify SHA via:
|
||||
# gh api repos/actions/checkout/git/refs/tags/v5.0.4
|
||||
#
|
||||
# `persist-credentials: false` disables the default behavior where
|
||||
# actions/checkout writes the GITHUB_TOKEN into `.git/config` as an
|
||||
# extraheader. That default is convenient (subsequent git commands
|
||||
# auth automatically) but it means the token is sitting on disk in
|
||||
# the checkout directory — an `actions/upload-artifact` step on
|
||||
# this directory would leak the token. We don't upload, but
|
||||
# zizmor's `credential-persistence` lint flags it defensively.
|
||||
# Push auth is provided inline at push time via the URL.
|
||||
- name: Checkout PR head
|
||||
if: steps.locate.outputs.found == 'true'
|
||||
uses: actions/checkout@08c6903cd8c0fde910a37f88322edcfb5dd907a8 # v5.0.4
|
||||
with:
|
||||
repository: ${{ steps.locate.outputs.head_repo }}
|
||||
ref: ${{ steps.locate.outputs.head_sha }}
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
persist-credentials: false
|
||||
# Fetch full history so the push doesn't hit shallow-clone errors.
|
||||
fetch-depth: 0
|
||||
path: pr-checkout
|
||||
|
||||
- name: Apply patch and push
|
||||
id: apply
|
||||
if: steps.locate.outputs.found == 'true'
|
||||
env:
|
||||
HEAD_REF: ${{ steps.locate.outputs.head_ref }}
|
||||
HEAD_REPO: ${{ steps.locate.outputs.head_repo }}
|
||||
# The SHA we resolved earlier in `locate` — this is what the
|
||||
# remote ref MUST still equal at push time. If the contributor
|
||||
# force-pushed between resolve and now, the lease fails and
|
||||
# we surface that distinctly from a fork-without-maintainer
|
||||
# -edit push failure.
|
||||
HEAD_SHA: ${{ steps.locate.outputs.head_sha }}
|
||||
# Auth for the push only — never persisted to disk. Provided
|
||||
# via env to avoid interpolating into the shell command line.
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
shell: bash
|
||||
working-directory: pr-checkout
|
||||
run: |
|
||||
set -euo pipefail
|
||||
patch="../autofix-in/autofix.patch"
|
||||
|
||||
# Distinguish artifact-expired (file missing entirely, because
|
||||
# actions/download-artifact ran with continue-on-error and the
|
||||
# 1-day retention had elapsed) from genuinely empty patch
|
||||
# (file present, zero bytes, formatter found nothing).
|
||||
if [ ! -e "$patch" ]; then
|
||||
echo "::warning::Patch file does not exist — autofix artifact likely expired."
|
||||
echo "result=artifact-expired" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [ ! -s "$patch" ]; then
|
||||
echo "::warning::Empty patch — nothing to apply."
|
||||
echo "result=empty-patch" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Sensitive-paths guard: refuse to apply patches that touch
|
||||
# `.github/` — workflow files, action definitions, CODEOWNERS,
|
||||
# dependabot config, etc. A malicious PR could ship a custom
|
||||
# prettier/ESLint config that reformats workflow YAML; the
|
||||
# producer would then capture those edits in autofix.patch,
|
||||
# and a maintainer running `/autofix` would push them under
|
||||
# `contents: write`. The default GITHUB_TOKEN lacks `workflows`
|
||||
# scope so the platform would reject workflow-file pushes
|
||||
# anyway, but that surfaces as a generic `push-failed` and
|
||||
# misleads users into enabling maintainer-edit. Reject early
|
||||
# with a specific reason. CODEOWNERS and dependabot.yml live
|
||||
# under .github/ but outside .github/workflows/ — the broader
|
||||
# match is intentional (they all govern trust boundaries).
|
||||
if grep -qE '^(diff --git|---|\+\+\+) [ab]?/?\.github/' "$patch"; then
|
||||
echo "::warning::Patch touches .github/ — refusing to apply (sensitive paths)."
|
||||
echo "result=sensitive-paths" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Re-entrancy guard: if HEAD itself is an autofix bot commit,
|
||||
# refuse to apply again. Without this, lint/formatter config
|
||||
# drift between runs could pump arbitrary apply commits into
|
||||
# the same PR if an automated agent watches the sticky and
|
||||
# re-fires `/autofix` on each new "fixes-available" surface.
|
||||
# The contributor can still get out by force-pushing a
|
||||
# human-authored commit to revert the autofix and re-trigger.
|
||||
head_author=$(git log -1 --format='%ae' HEAD)
|
||||
head_subject=$(git log -1 --format='%s' HEAD)
|
||||
if [ "${head_author}" = "41898282+github-actions[bot]@users.noreply.github.com" ] \
|
||||
&& [[ "${head_subject}" =~ ^chore\(autofix\) ]]; then
|
||||
echo "::warning::HEAD is an autofix bot commit — refusing to re-apply (loop guard)."
|
||||
echo "result=loop-prevented" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Idempotency probe: does the forward apply work?
|
||||
if git apply --check "$patch" 2>/dev/null; then
|
||||
echo "Patch applies cleanly — proceeding."
|
||||
elif git apply --check --reverse "$patch" 2>/dev/null; then
|
||||
# Reverse-check passes => the patch is already applied to
|
||||
# the current tree. Treat as success no-op.
|
||||
echo "Patch is already applied (reverse-check passed) — no-op."
|
||||
echo "result=already-applied" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
else
|
||||
echo "::error::Patch does not apply (stale or conflicting)."
|
||||
echo "result=stale" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Wrap the apply/commit phase so any non-zero exit sets a
|
||||
# meaningful `result=` instead of leaving it unset (which would
|
||||
# send the user to the `*` "unexpected state" arm with a
|
||||
# non-actionable confused-emoji reply).
|
||||
if ! {
|
||||
git config user.email "41898282+github-actions[bot]@users.noreply.github.com" &&
|
||||
git config user.name "github-actions[bot]" &&
|
||||
git apply "$patch" &&
|
||||
git add -A &&
|
||||
git commit -m "chore(autofix): apply prettier + eslint fixes via /autofix command"
|
||||
}; then
|
||||
echo "::error::git apply / config / commit failed after idempotency probe passed."
|
||||
echo "result=apply-failed" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Push to the PR head branch with a lease against the resolved
|
||||
# SHA. The lease ensures the remote ref still points at HEAD_SHA
|
||||
# when the push lands — if the contributor force-pushed in the
|
||||
# window between resolve and now, the lease fails and we return
|
||||
# `lease-failed` (NOT `push-failed`, which would mislead users
|
||||
# into enabling maintainer-edit). For fork PRs, the push still
|
||||
# requires "Allow edits by maintainers" to be enabled.
|
||||
#
|
||||
# Auth is supplied inline via `-c http.<base>.extraheader` (NOT
|
||||
# via a `https://x-access-token:TOKEN@…` URL — those leak into
|
||||
# process listings and `git remote -v` output). The header is
|
||||
# set per-invocation; it never lands in `.git/config` on disk.
|
||||
# The token is base64-encoded for the Basic auth header per
|
||||
# GitHub's documented pattern for this scope.
|
||||
push_url="https://github.com/${HEAD_REPO}.git"
|
||||
auth_header="Authorization: Basic $(printf 'x-access-token:%s' "${GITHUB_TOKEN}" | base64 -w0)"
|
||||
# GitHub's secret-masker only masks the raw token, not its
|
||||
# base64-encoded form. Mask the encoded value so any subsequent
|
||||
# log line (set -x, GIT_TRACE, error spew) gets ***-redacted.
|
||||
echo "::add-mask::${auth_header}"
|
||||
push_stderr=$(mktemp)
|
||||
if git -c http.extraheader="${auth_header}" \
|
||||
push --force-with-lease="refs/heads/${HEAD_REF}:${HEAD_SHA}" \
|
||||
"${push_url}" "HEAD:${HEAD_REF}" 2>"$push_stderr"; then
|
||||
echo "result=applied" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
cat "$push_stderr" >&2
|
||||
# `--force-with-lease` reports "stale info" when the remote
|
||||
# ref has moved past the expected SHA. Other lease-failure
|
||||
# phrases git emits include "remote rejected" (server-side
|
||||
# reject), "non-fast-forward", and the literal flag name. Match
|
||||
# any of those to distinguish from auth/network/maintainer-
|
||||
# edit failures.
|
||||
if grep -qE "stale info|force-with-lease|rejected.*non-fast-forward|remote rejected|! \[rejected\]" "$push_stderr"; then
|
||||
echo "::error::git push lease failed — branch moved during apply."
|
||||
echo "result=lease-failed" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "::error::git push failed — likely fork without maintainer-edit enabled."
|
||||
echo "result=push-failed" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
- name: React and reply on outcome
|
||||
if: always() && steps.locate.outputs.found == 'true' && steps.apply.outcome != 'skipped'
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
COMMENT_ID: ${{ github.event.comment.id }}
|
||||
PR: ${{ github.event.issue.number }}
|
||||
RESULT: ${{ steps.apply.outputs.result }}
|
||||
RUN_ID: ${{ github.run_id }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
run_url="https://github.com/${GH_REPO}/actions/runs/${RUN_ID}"
|
||||
|
||||
case "${RESULT}" in
|
||||
applied)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="+1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="✅ Applied autofix and pushed a commit. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
;;
|
||||
already-applied)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="+1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="✅ Autofix is already applied — no changes needed." \
|
||||
>/dev/null
|
||||
;;
|
||||
empty-patch)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="+1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="✅ No autofix to apply — formatter found nothing." \
|
||||
>/dev/null
|
||||
;;
|
||||
artifact-expired)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="confused" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⏳ The autofix artifact for this PR's head SHA has expired (1-day retention). Push a new commit to regenerate it, then comment \`/autofix\` again. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
loop-prevented)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="confused" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="🔁 Refusing to re-apply autofix on top of an existing autofix commit. If formatter rules drifted and you genuinely need another pass, push a human-authored commit (or revert the existing autofix commit) before commenting \`/autofix\` again. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
sensitive-paths)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="🛑 Refusing to apply: the autofix patch touches files under \`.github/\` (workflow / CODEOWNERS / dependabot config). Apply formatter changes to those files manually in a regular commit so they get human review. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
stale)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⚠️ The autofix patch is stale or conflicts with the current head — push a new commit to regenerate, then comment \`/autofix\` again. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
apply-failed)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⚠️ Autofix applied cleanly in the dry run, but \`git apply\` / \`git commit\` failed when actually landing the patch. This usually means a race with concurrent edits or a corrupt patch. See logs: ${run_url}" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
push-failed)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⚠️ Couldn't push the autofix commit. If this is a fork PR, please tick **Allow edits by maintainers** in the PR sidebar, then comment \`/autofix\` again. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
lease-failed)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="-1" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="⚠️ The PR head moved while autofix was applying — a new commit landed in the window between resolve and push. Comment \`/autofix\` again to retry against the latest head. ([apply run](${run_url}))" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
*)
|
||||
gh api -X POST "repos/${GH_REPO}/issues/comments/${COMMENT_ID}/reactions" \
|
||||
-f content="confused" >/dev/null
|
||||
gh api -X POST "repos/${GH_REPO}/issues/${PR}/comments" \
|
||||
-f body="❓ Autofix run finished in an unexpected state (\`${RESULT:-unknown}\`). See logs: ${run_url}" \
|
||||
>/dev/null
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
178
.github/workflows/pr-autofix-publish.yml
vendored
178
.github/workflows/pr-autofix-publish.yml
vendored
|
|
@ -3,20 +3,19 @@ name: PR Autofix (publish)
|
|||
# TRUSTED HALF of the autofix pipeline.
|
||||
#
|
||||
# Triggered by `pr-autofix.yml` completing on a PR (including fork PRs).
|
||||
# Downloads the diff artifact produced by the untrusted job and posts
|
||||
# inline review-comment suggestions to the PR using `reviewdog`. This
|
||||
# job NEVER checks out fork code — it only consumes the diff (data) and
|
||||
# calls the GitHub API. That isolation is what makes it safe to run
|
||||
# under `pull-requests: write` on fork-triggered events.
|
||||
# Downloads the diff artifact produced by the untrusted job, verifies
|
||||
# its claimed PR identity against the workflow_run authority, then
|
||||
# posts (or edits) a single sticky summary comment plus a
|
||||
# `gitnexus/autofix` Check Run. This job NEVER checks out fork code —
|
||||
# it only consumes the diff (data) and calls the GitHub API. That
|
||||
# isolation is what makes it safe to run under `pull-requests: write`
|
||||
# on fork-triggered events.
|
||||
#
|
||||
# Also posts (or edits) a single sticky summary comment so contributors
|
||||
# and AI agents have one stable, machine-readable signal that says
|
||||
# whether autofix had anything to suggest. Look for the heading
|
||||
# "## :sparkles: PR Autofix" in the PR's top-level comments.
|
||||
#
|
||||
# Reviewdog reporter: `github-pr-review` reads $REVIEWDOG_GITHUB_API_TOKEN
|
||||
# and posts via the GraphQL/REST PR-review API. It does not need a
|
||||
# checkout because the diff itself encodes file paths + line numbers.
|
||||
# The sticky comment is the contributor signal: heading
|
||||
# "## :sparkles: PR Autofix" in the PR's top-level comments, with a
|
||||
# fenced `gitnexus-autofix` JSON block carrying machine-readable state
|
||||
# for AI agents. Contributors apply the patch by commenting `/autofix`
|
||||
# on the PR — handled by the separate `pr-autofix-apply.yml` workflow.
|
||||
|
||||
on:
|
||||
workflow_run:
|
||||
|
|
@ -49,9 +48,9 @@ jobs:
|
|||
# by a different workflow run.
|
||||
actions: read
|
||||
# Required to create the `gitnexus/autofix` Check Run that reports
|
||||
# the outcome (clean / suggestions-posted / skipped-too-large) to
|
||||
# the PR's Checks tab. Branch protection or agents can grep the
|
||||
# conclusion + output title without parsing the sticky comment.
|
||||
# the outcome (clean / fixes-available) to the PR's Checks tab.
|
||||
# Branch protection or agents can grep the conclusion + output
|
||||
# title without parsing the sticky comment.
|
||||
checks: write
|
||||
steps:
|
||||
# Pinned to v8.0.1. Verify SHA via:
|
||||
|
|
@ -114,71 +113,78 @@ jobs:
|
|||
echo "changed_lines=${CHANGED}"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
# Pinned to v1.5.0. Verify SHA via:
|
||||
# gh api repos/reviewdog/action-setup/git/refs/tags/v1.5.0
|
||||
# (annotated tag — resolve via .../git/tags/<sha> --jq .object)
|
||||
- name: Install reviewdog
|
||||
if: steps.meta.outputs.changed_lines != '0'
|
||||
uses: reviewdog/action-setup@d8a7baabd7f3e8544ee4dbde3ee41d0011c3a93f # v1.5.0
|
||||
with:
|
||||
# Pin the binary, not just the action SHA — a bad reviewdog
|
||||
# release otherwise breaks every PR with no rollback. Bump
|
||||
# this knob deliberately when validating a new release.
|
||||
reviewdog_version: v0.21.0
|
||||
|
||||
- name: Post inline suggestions
|
||||
id: suggest
|
||||
# Cross-verify the artifact's claimed identity against the
|
||||
# GitHub-controlled workflow_run event. The previous step's
|
||||
# allowlist only proves the fields are well-formed — not that
|
||||
# they refer to the PR/SHA that actually triggered this run.
|
||||
# A fork-controlled `npm run lint:fix` could plausibly mutate
|
||||
# metadata.json to reference another PR or SHA, redirecting our
|
||||
# write-scoped sticky/check-run onto an attacker-chosen target.
|
||||
#
|
||||
# Authority sources are all server-controlled GitHub event fields:
|
||||
# - workflow_run.head_sha
|
||||
# - workflow_run.head_repository.full_name
|
||||
# - workflow_run.pull_requests[].number (within-repo PRs only;
|
||||
# empty array on fork PRs — fall back to commits/{sha}/pulls)
|
||||
#
|
||||
# Mismatch => fail loud BEFORE any sticky/check-run side effect.
|
||||
- name: Verify metadata against workflow_run authority
|
||||
id: verify
|
||||
if: steps.meta.outputs.changed_lines != '0'
|
||||
env:
|
||||
REVIEWDOG_GITHUB_API_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
CI_REPO_OWNER: ${{ github.repository_owner }}
|
||||
CI_REPO_NAME: ${{ github.event.repository.name }}
|
||||
CI_PULL_REQUEST: ${{ steps.meta.outputs.pr_number }}
|
||||
CI_COMMIT: ${{ steps.meta.outputs.head_sha }}
|
||||
# Pull `changed_lines` through env so bash gets a real
|
||||
# variable (and shellcheck SC2170 doesn't fire on `-gt` against
|
||||
# a `${{ }}`-interpolated literal).
|
||||
CHANGED_LINES: ${{ steps.meta.outputs.changed_lines }}
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GH_REPO: ${{ github.repository }}
|
||||
META_PR_NUMBER: ${{ steps.meta.outputs.pr_number }}
|
||||
META_HEAD_SHA: ${{ steps.meta.outputs.head_sha }}
|
||||
META_HEAD_REPO: ${{ steps.meta.outputs.head_repo }}
|
||||
WF_HEAD_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
WF_HEAD_REPO: ${{ github.event.workflow_run.head_repository.full_name }}
|
||||
WF_PR_NUMBERS: ${{ toJSON(github.event.workflow_run.pull_requests.*.number) }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
patch=autofix-in/autofix.patch
|
||||
if [ ! -s "$patch" ]; then
|
||||
echo "Empty patch — nothing to suggest."
|
||||
echo "posted=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
|
||||
# 1) head_sha must match exactly. workflow_run.head_sha is the
|
||||
# commit GitHub actually ran the producer against — definitive.
|
||||
if [ "${META_HEAD_SHA}" != "${WF_HEAD_SHA}" ]; then
|
||||
echo "::error::Artifact head_sha (${META_HEAD_SHA}) does not match workflow_run.head_sha (${WF_HEAD_SHA}) — refusing to publish."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# GitHub's review-comment API returns 406 on diffs above ~3k
|
||||
# changed lines. Bail out gracefully and let the summary
|
||||
# comment carry the signal instead.
|
||||
if [ "$CHANGED_LINES" -gt 3000 ]; then
|
||||
echo "Diff too large ($CHANGED_LINES lines) — skipping inline suggestions."
|
||||
echo "posted=skipped-too-large" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
# 2) head_repo must match exactly. Same authority anchor.
|
||||
if [ "${META_HEAD_REPO}" != "${WF_HEAD_REPO}" ]; then
|
||||
echo "::error::Artifact head_repo (${META_HEAD_REPO}) does not match workflow_run.head_repository (${WF_HEAD_REPO}) — refusing to publish."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# `-f.diff.strip=1` matches `git diff` output (a/foo b/foo).
|
||||
# `-filter-mode=added` only suggests on lines the PR added,
|
||||
# which avoids re-suggesting on already-resolved threads when
|
||||
# the contributor re-adds the autoformat label.
|
||||
reviewdog \
|
||||
-f=diff -f.diff.strip=1 \
|
||||
-name="prettier+eslint" \
|
||||
-reporter=github-pr-review \
|
||||
-filter-mode=added \
|
||||
-level=warning \
|
||||
-fail-on-error=false < "$patch"
|
||||
# 3) pr_number must reference an open PR with this head SHA.
|
||||
# Within-repo PRs: workflow_run.pull_requests[] is populated.
|
||||
# Fork PRs: that array is empty by GitHub design — fall back
|
||||
# to the REST commit-to-PRs lookup. Fail closed if the lookup
|
||||
# finds no matching open PR (avoids attacker-forged PR ids).
|
||||
allowed_numbers=$(jq -c '.' <<< "${WF_PR_NUMBERS}")
|
||||
if [ "${allowed_numbers}" = "[]" ]; then
|
||||
echo "workflow_run.pull_requests is empty (fork PR) — falling back to commits/{sha}/pulls."
|
||||
allowed_numbers=$(gh api "repos/${GH_REPO}/commits/${WF_HEAD_SHA}/pulls" \
|
||||
--jq '[.[] | select(.state == "open") | .number]' 2>/dev/null || echo "[]")
|
||||
if [ "${allowed_numbers}" = "[]" ]; then
|
||||
echo "::error::No open PR found for head ${WF_HEAD_SHA} via commits/{sha}/pulls — refusing to publish."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "posted=true" >> "$GITHUB_OUTPUT"
|
||||
if ! jq -e --argjson n "${META_PR_NUMBER}" 'index($n) != null' <<< "${allowed_numbers}" >/dev/null; then
|
||||
echo "::error::Artifact pr_number (${META_PR_NUMBER}) is not in the authoritative PR list (${allowed_numbers}) — refusing to publish."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Verified: metadata identity matches workflow_run authority (PR=${META_PR_NUMBER}, head_sha=${META_HEAD_SHA}, head_repo=${META_HEAD_REPO})."
|
||||
|
||||
- name: Upsert sticky summary comment
|
||||
# Only post when ci-quality found something fixable (= the
|
||||
# autofix patch is non-empty). When prettier/eslint are clean
|
||||
# the patch is zero bytes and the sticky comment is pure noise,
|
||||
# so we skip it. When the diff was too large for inline
|
||||
# suggestions, the sticky is the only signal the contributor
|
||||
# gets, so we still post in that case.
|
||||
# so we skip it.
|
||||
if: >-
|
||||
always()
|
||||
&& steps.meta.outputs.pr_number != ''
|
||||
|
|
@ -189,8 +195,6 @@ jobs:
|
|||
PR: ${{ steps.meta.outputs.pr_number }}
|
||||
CHANGED: ${{ steps.meta.outputs.changed_lines }}
|
||||
HEAD_SHA: ${{ steps.meta.outputs.head_sha }}
|
||||
SCHEMA: ${{ steps.meta.outputs.schema }}
|
||||
POSTED: ${{ steps.suggest.outputs.posted }}
|
||||
RUN_ID: ${{ github.run_id }}
|
||||
shell: bash
|
||||
run: |
|
||||
|
|
@ -200,25 +204,29 @@ jobs:
|
|||
marker="<!-- gitnexus:pr-autofix-summary -->"
|
||||
heading="## :sparkles: PR Autofix"
|
||||
|
||||
if [ "${POSTED}" = "skipped-too-large" ]; then
|
||||
ui_state="skipped-too-large"
|
||||
prose="Diff is **${CHANGED}** lines — too large for inline suggestions (GitHub caps the review-comment API at ~3000). Run locally: \`npm run lint:fix && npm run format\`."
|
||||
else
|
||||
ui_state="suggestions-posted"
|
||||
prose="Posted formatting / unused-import suggestions inline. Click **Apply suggestion** on each, or run locally: \`npm run lint:fix && npm run format\`."
|
||||
fi
|
||||
# Single state. The /autofix slash command works for any diff
|
||||
# size — there's no 3K cap and no no-overlap dead-end because
|
||||
# the apply workflow uses `git apply` + push, not the GitHub
|
||||
# review-comment API.
|
||||
ui_state="fixes-available"
|
||||
prose="Found fixable formatting / unused-import issues across **${CHANGED}** changed lines. **Comment \`/autofix\` on this PR to apply them**, or run \`npm run lint:fix && npm run format\` locally."
|
||||
|
||||
# Machine-readable JSON block — agents parse this instead of
|
||||
# regexing English. Fenced code-block info string is
|
||||
# `gitnexus-autofix` so agents can locate it without ambiguity.
|
||||
# Schema bumped from v1 -> v2: adds `apply_command`. The v1
|
||||
# field set is preserved as a superset, but the `state` enum
|
||||
# is redefined (v1: suggestions-posted | skipped-too-large |
|
||||
# diff-no-overlap; v2: fixes-available). v1 readers checking
|
||||
# `schema == 'gitnexus.pr-autofix/v1'` see an unfamiliar version
|
||||
# and fall back to prose, which is the intended migration path.
|
||||
json=$(jq -n -c \
|
||||
--arg schema "${SCHEMA}" \
|
||||
--arg state "${ui_state}" \
|
||||
--argjson pr_number "${PR}" \
|
||||
--argjson changed_lines "${CHANGED}" \
|
||||
--arg head_sha "${HEAD_SHA}" \
|
||||
--arg run_id "${RUN_ID}" \
|
||||
'{schema:$schema, state:$state, pr_number:$pr_number, changed_lines:$changed_lines, head_sha:$head_sha, run_id:$run_id}')
|
||||
'{schema:"gitnexus.pr-autofix/v2", state:$state, pr_number:$pr_number, changed_lines:$changed_lines, head_sha:$head_sha, run_id:$run_id, apply_command:"/autofix"}')
|
||||
|
||||
# Multi-line quoted string instead of a column-0 heredoc — YAML's
|
||||
# `run: |` block ends as soon as a content line dedents below the
|
||||
|
|
@ -272,10 +280,9 @@ jobs:
|
|||
- name: Emit gitnexus/autofix Check Run
|
||||
# Stable check name `gitnexus/autofix` so PR-watching agents can
|
||||
# `gh pr checks <pr>` and read the conclusion + title without
|
||||
# parsing the sticky comment. Three outcomes:
|
||||
# clean → conclusion: success
|
||||
# suggestions-posted → conclusion: neutral (review suggestions)
|
||||
# skipped-too-large → conclusion: neutral (diff > 3000 lines)
|
||||
# parsing the sticky comment. Two outcomes:
|
||||
# clean → conclusion: success
|
||||
# fixes-available → conclusion: neutral
|
||||
# `neutral` does not block branch-protection required-checks but
|
||||
# is visually distinct from a green pass.
|
||||
if: always() && steps.meta.outputs.head_sha != ''
|
||||
|
|
@ -284,7 +291,6 @@ jobs:
|
|||
GH_REPO: ${{ github.repository }}
|
||||
HEAD_SHA: ${{ steps.meta.outputs.head_sha }}
|
||||
CHANGED: ${{ steps.meta.outputs.changed_lines }}
|
||||
POSTED: ${{ steps.suggest.outputs.posted }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
|
@ -293,14 +299,10 @@ jobs:
|
|||
conclusion="success"
|
||||
title="Formatting clean"
|
||||
summary="Prettier and ESLint --fix produced no changes."
|
||||
elif [ "${POSTED}" = "skipped-too-large" ]; then
|
||||
conclusion="neutral"
|
||||
title="Diff too large for inline suggestions (${CHANGED} lines)"
|
||||
summary="GitHub caps the review-comment API at ~3000 lines. Run \`npm run lint:fix && npm run format\` locally."
|
||||
else
|
||||
conclusion="neutral"
|
||||
title="Suggestions posted"
|
||||
summary="Inline review-comment suggestions posted. Click **Apply suggestion** on each, or run \`npm run lint:fix && npm run format\` locally."
|
||||
title="Autofix available — comment /autofix to apply"
|
||||
summary="Comment \`/autofix\` on this PR to apply formatter + unused-import fixes (works at any diff size). Or run \`npm run lint:fix && npm run format\` locally."
|
||||
fi
|
||||
|
||||
gh api -X POST "repos/${GH_REPO}/check-runs" \
|
||||
|
|
|
|||
16
.github/workflows/pr-autofix.yml
vendored
16
.github/workflows/pr-autofix.yml
vendored
|
|
@ -6,7 +6,9 @@ name: PR Autofix
|
|||
# (including fork heads) and uploads the resulting diff as an artifact.
|
||||
# This job has NO privileged token and CANNOT post to the PR. The trusted
|
||||
# `pr-autofix-publish.yml` workflow downloads the artifact via
|
||||
# `workflow_run` and posts the inline review-comment suggestions.
|
||||
# `workflow_run` and posts a sticky summary comment + Check Run.
|
||||
# Contributors apply the patch by commenting `/autofix` on the PR —
|
||||
# handled by the separate `pr-autofix-apply.yml` ChatOps workflow.
|
||||
#
|
||||
# Why the split:
|
||||
# ESLint loads plugins from fork-controlled `node_modules`, so running
|
||||
|
|
@ -14,8 +16,8 @@ name: PR Autofix
|
|||
# ship a poisoned eslint plugin and execute arbitrary code under that
|
||||
# token. By keeping fork code execution in this job (token: read-only)
|
||||
# and posting from a separate trusted job that never touches fork
|
||||
# code, we get the inline-suggestion UX for fork PRs without the
|
||||
# supply-chain hole. (See autofix.ci for the same pattern.)
|
||||
# code, we get the autofix UX for fork PRs without the supply-chain
|
||||
# hole. (See autofix.ci for the same pattern.)
|
||||
#
|
||||
# Removes unused imports via `eslint-plugin-unused-imports`, already in
|
||||
# devDependencies and wired into the `lint` config.
|
||||
|
|
@ -105,11 +107,9 @@ jobs:
|
|||
git diff --no-color > autofix-out/autofix.patch
|
||||
|
||||
# NOTE: `changed_lines` is the line-count of the patch file,
|
||||
# which includes hunk headers and context lines — NOT the
|
||||
# added/removed source-line count. The 3000-line cap in
|
||||
# pr-autofix-publish.yml is therefore conservative (fires
|
||||
# before reviewdog hits GitHub's ~3k review-comment API
|
||||
# ceiling). That bias is intentional.
|
||||
# (hunk headers + context lines + added/removed). Surfaced in
|
||||
# the sticky comment so contributors and AI agents have a
|
||||
# quick size hint before invoking `/autofix`.
|
||||
changed_lines=$(wc -l < autofix-out/autofix.patch | tr -d ' ')
|
||||
echo "changed_lines=${changed_lines}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
|
|
|
|||
|
|
@ -30,17 +30,17 @@ Format: `<type>[(scope)][!]: <subject>`
|
|||
|
||||
Allowed types and the release-notes section each one lands in (defined in `.github/release.yml`):
|
||||
|
||||
| Type | Label applied | Release-notes section |
|
||||
|------|---------------|-----------------------|
|
||||
| `feat` | `enhancement` | 🚀 Features |
|
||||
| `fix` | `bug` | 🐛 Bug Fixes |
|
||||
| `perf` | `performance` | 🏎️ Performance |
|
||||
| `refactor` | `refactor` | 🔄 Refactoring |
|
||||
| `test` | `test` | 🧪 Tests |
|
||||
| `ci` | `ci` | 👷 CI/CD |
|
||||
| `build` / `deps` | `dependencies` | 📦 Dependencies |
|
||||
| `docs` | `documentation` | (grouped under Other Changes unless a Docs section is added) |
|
||||
| `chore` / `revert` | `chore` | (excluded from release notes) |
|
||||
| Type | Label applied | Release-notes section |
|
||||
| ------------------ | --------------- | ------------------------------------------------------------ |
|
||||
| `feat` | `enhancement` | 🚀 Features |
|
||||
| `fix` | `bug` | 🐛 Bug Fixes |
|
||||
| `perf` | `performance` | 🏎️ Performance |
|
||||
| `refactor` | `refactor` | 🔄 Refactoring |
|
||||
| `test` | `test` | 🧪 Tests |
|
||||
| `ci` | `ci` | 👷 CI/CD |
|
||||
| `build` / `deps` | `dependencies` | 📦 Dependencies |
|
||||
| `docs` | `documentation` | (grouped under Other Changes unless a Docs section is added) |
|
||||
| `chore` / `revert` | `chore` | (excluded from release notes) |
|
||||
|
||||
Append `!` to the type (e.g. `feat(api)!: drop /v1 endpoint`) or include `BREAKING CHANGE:` in the PR body to flag a breaking change — the labeler then adds the `breaking` label and the 💥 Breaking Changes section is rendered first.
|
||||
|
||||
|
|
@ -81,17 +81,17 @@ Every workflow under `.github/workflows/` MUST declare a top-level `concurrency:
|
|||
- **Merge queue (`merge_group`)**: when this event is added, use `${{ github.workflow }}-${{ github.event.merge_group.head_ref }}` with `cancel-in-progress: false` (every queue entry is a distinct ref; never cancel).
|
||||
- **`cancel-in-progress` policy:**
|
||||
|
||||
| Event | `cancel-in-progress` | Why |
|
||||
|-------|----------------------|-----|
|
||||
| `pull_request` CI run | `true` | New push supersedes old run |
|
||||
| `push` to `main` | `false` | Every main commit gets validated |
|
||||
| Tag push (`v*` publish) | `false` | Never cancel mid-publish |
|
||||
| `push` to `main` for release-candidate | `false` | Never cancel mid-RC publish |
|
||||
| `workflow_dispatch` (release/publish) | `false` | Manual runs are intentional |
|
||||
| `workflow_run` (sticky-comment reports) | `false` | Serialize, don't race |
|
||||
| Per-PR bot workflows (`@claude`, review) | `false` | Serialize comments per PR |
|
||||
| PR-meta re-checks (pr-description-check) | `true` | Cheap, latest wins |
|
||||
| Single-slot utilities (triage sweep) | `true` | Latest dispatch supersedes |
|
||||
| Event | `cancel-in-progress` | Why |
|
||||
| ---------------------------------------- | -------------------- | -------------------------------- |
|
||||
| `pull_request` CI run | `true` | New push supersedes old run |
|
||||
| `push` to `main` | `false` | Every main commit gets validated |
|
||||
| Tag push (`v*` publish) | `false` | Never cancel mid-publish |
|
||||
| `push` to `main` for release-candidate | `false` | Never cancel mid-RC publish |
|
||||
| `workflow_dispatch` (release/publish) | `false` | Manual runs are intentional |
|
||||
| `workflow_run` (sticky-comment reports) | `false` | Serialize, don't race |
|
||||
| Per-PR bot workflows (`@claude`, review) | `false` | Serialize comments per PR |
|
||||
| PR-meta re-checks (pr-description-check) | `true` | Cheap, latest wins |
|
||||
| Single-slot utilities (triage sweep) | `true` | Latest dispatch supersedes |
|
||||
|
||||
- For workflows that serve multiple events at once (e.g. `ci.yml` handles `pull_request`, `push`, and `workflow_call`), make `cancel-in-progress` event-aware:
|
||||
|
||||
|
|
@ -109,18 +109,35 @@ Two workflows produce machine-readable signals on every PR. Coding agents and hu
|
|||
|
||||
### `gitnexus/autofix`
|
||||
|
||||
`pr-autofix.yml` (untrusted) + `pr-autofix-publish.yml` (trusted) run `prettier --write` and `eslint --fix` against the PR head and surface the diff as inline review-comment suggestions. Three signals are emitted:
|
||||
`pr-autofix.yml` (untrusted) + `pr-autofix-publish.yml` (trusted) run `prettier --write` and `eslint --fix` against the PR head and surface a single ChatOps button on the PR. Three signals are emitted:
|
||||
|
||||
| Surface | Where | Notes |
|
||||
|---|---|---|
|
||||
| Sticky PR comment | Top-level comment with the HTML marker `<!-- gitnexus:pr-autofix-summary -->` and heading `## :sparkles: PR Autofix`. Only posted when there is something to fix; clean PRs stay silent. | Edit-in-place via marker; one comment per PR. |
|
||||
| Fenced JSON block | Inside the sticky, fenced as `gitnexus-autofix`. Schema `gitnexus.pr-autofix/v1` with fields `state` (`suggestions-posted` \| `skipped-too-large`), `pr_number`, `head_sha`, `changed_lines`, `run_id`. | Parseable signal — preferred over regexing prose. |
|
||||
| Check Run | Stable name `gitnexus/autofix` on the PR head SHA. Conclusion: `success` (clean) or `neutral` (suggestions-posted / skipped-too-large). The output title disambiguates the two `neutral` cases. | Surfaced under PR Checks; readable via `gh pr checks <pr>`. |
|
||||
| Surface | Where | Notes |
|
||||
| ----------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------- |
|
||||
| Sticky PR comment | Top-level comment with the HTML marker `<!-- gitnexus:pr-autofix-summary -->` and heading `## :sparkles: PR Autofix`. Only posted when there is something to fix; clean PRs stay silent. | Edit-in-place via marker; one comment per PR. |
|
||||
| Fenced JSON block | Inside the sticky, fenced as `gitnexus-autofix`. Schema `gitnexus.pr-autofix/v2` with fields `state` (`fixes-available`), `pr_number`, `head_sha`, `changed_lines`, `run_id`, and `apply_command` (literal `/autofix`). | Parseable signal — preferred over regexing prose. v1 fields preserved as a superset. |
|
||||
| Check Run | Stable name `gitnexus/autofix` on the PR head SHA. Conclusion: `success` (clean) or `neutral` (`fixes-available`). The neutral title is `Autofix available — comment /autofix to apply`. | Surfaced under PR Checks; readable via `gh pr checks <pr>`. |
|
||||
|
||||
To detect outcome from an agent: `gh pr checks <pr> --json name,conclusion,output | jq '.[] | select(.name == "gitnexus/autofix")'`.
|
||||
|
||||
Forks are supported. The untrusted half runs fork code with `permissions: {}` and ships the diff as an artifact; the trusted publish job consumes only the diff (data, not code) and posts the comment + check run.
|
||||
|
||||
#### Applying autofix
|
||||
|
||||
Comment `/autofix` on the PR (whole-line, no arguments). The `pr-autofix-apply.yml` workflow:
|
||||
|
||||
1. Validates the comment body matches `^/autofix\s*$` exactly. Quoted or inline mentions are silently ignored.
|
||||
2. Validates the commenter has `admin`, `write`, or `maintain` permission on the repo, OR is the PR author. Other commenters get a 👎 reaction and a refusal reply.
|
||||
3. Locates the most recent successful `pr-autofix.yml` run for the PR's current head SHA, downloads its `autofix` artifact, applies the patch, and pushes a `chore(autofix): ...` commit back to the PR head branch.
|
||||
4. Reacts ✅ on success, 👎 on stale-patch / push-failure, and posts a short reply with the apply-run URL in either case.
|
||||
|
||||
The apply workflow runs from the default branch's copy of the file regardless of where the comment originates — that's the trust anchor. There is no diff-size cap (the apply workflow uses `git apply` + push, not the GitHub review-comment API).
|
||||
|
||||
For fork PRs, the push succeeds only when the contributor has **Allow edits by maintainers** enabled on the PR (the default). When they have disabled it, the workflow fails loud with a 👎 reaction and an explanation comment.
|
||||
|
||||
Re-invoking `/autofix` after a successful apply is a safe no-op — the workflow detects the already-applied state via `git apply --check --reverse` and reacts ✅ without pushing.
|
||||
|
||||
**Sensitive paths.** The apply workflow refuses any patch that touches `.github/` (workflow files, CODEOWNERS, dependabot config). A malicious PR could ship a custom prettier or ESLint config that reformats workflow YAML; if accepted, those edits would be pushed under `contents: write` without human review. Apply formatter changes to files under `.github/` manually in a normal commit so they get the same review every other workflow change gets.
|
||||
|
||||
## AI-assisted contributions
|
||||
|
||||
If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes.
|
||||
|
|
@ -182,8 +199,7 @@ Two publish workflows ship `gitnexus` to npm:
|
|||
the Docker build.
|
||||
- Manually run `docker build` + `docker push` locally and sign with Cosign
|
||||
against the same digest.
|
||||
- Delete `rc/<HEAD_SHA>` and `v<RC>` tags, then redispatch with `force:
|
||||
true` to re-run the full RC pipeline (cuts a new RC number).
|
||||
- Delete `rc/<HEAD_SHA>` and `v<RC>` tags, then redispatch with `force: true` to re-run the full RC pipeline (cuts a new RC number).
|
||||
|
||||
The rc workflow never moves `latest`. To verify after a change, inspect dist-tags:
|
||||
|
||||
|
|
|
|||
|
|
@ -10,6 +10,10 @@
|
|||
".": {
|
||||
"types": "./dist/index.d.ts",
|
||||
"default": "./dist/index.js"
|
||||
},
|
||||
"./test-helpers": {
|
||||
"types": "./dist/test-helpers.d.ts",
|
||||
"default": "./dist/test-helpers.js"
|
||||
}
|
||||
},
|
||||
"scripts": {
|
||||
|
|
|
|||
|
|
@ -143,6 +143,22 @@ export type { ScopeTree } from './scope-resolution/scope-tree.js';
|
|||
export { buildPositionIndex } from './scope-resolution/position-index.js';
|
||||
export type { PositionIndex } from './scope-resolution/position-index.js';
|
||||
|
||||
// Resilient fetch primitives — bounded retries + per-process circuit breaker.
|
||||
// Test-only helpers (`__resetBreakerRegistry__`, `classifyOutcome`) are
|
||||
// reachable via the separate `gitnexus-shared/test-helpers` subpath; do
|
||||
// NOT add them here. Production consumers must not call them.
|
||||
export { withRetry, computeBackoffMs } from './integrations/retry.js';
|
||||
export type { RetryOptions, RetryDecision } from './integrations/retry.js';
|
||||
export { CircuitBreaker, CircuitOpenError, getBreaker } from './integrations/circuit-breaker.js';
|
||||
export type { CircuitBreakerOptions } from './integrations/circuit-breaker.js';
|
||||
export {
|
||||
resilientFetch,
|
||||
ResilientFetchExhaustedError,
|
||||
RETRY_AFTER_CAP_MS,
|
||||
parseRetryAfter,
|
||||
} from './integrations/resilient-fetch.js';
|
||||
export type { ResilientFetchOptions } from './integrations/resilient-fetch.js';
|
||||
|
||||
// Understand-Quickly registry integration (opt-in)
|
||||
export {
|
||||
UNDERSTAND_QUICKLY_DISPATCH_URL,
|
||||
|
|
|
|||
273
gitnexus-shared/src/integrations/circuit-breaker.ts
Normal file
273
gitnexus-shared/src/integrations/circuit-breaker.ts
Normal file
|
|
@ -0,0 +1,273 @@
|
|||
/**
|
||||
* Per-process circuit breaker.
|
||||
*
|
||||
* Closed -> Open transition fires after `failureThreshold` consecutive
|
||||
* failures. While Open, `check` throws `CircuitOpenError` until
|
||||
* `cooldownMs` has elapsed since the breaker tripped. The first call
|
||||
* after the cooldown enters Half-Open and consumes the *probe permit*:
|
||||
* a recorded success returns to Closed; a recorded failure flips back
|
||||
* to Open with a fresh timestamp.
|
||||
*
|
||||
* Half-open admits exactly one in-flight probe at a time. Concurrent
|
||||
* callers attempting `check()` while a probe is outstanding receive
|
||||
* `CircuitOpenError` with `retryAfterMs = halfOpenRetryAfterMs` (default
|
||||
* 1000ms; configurable). This prevents the recovery-time thundering
|
||||
* herd that defeats the breaker's "fail fast" promise.
|
||||
*
|
||||
* Outcome reporting splits permit-release from state-resolution:
|
||||
* - `recordSuccess` — releases the probe permit, resets the failure
|
||||
* counter, transitions to Closed. Reserved for true 2xx/3xx outcomes.
|
||||
* - `recordFailure` — releases the probe permit, increments the
|
||||
* consecutive-failure counter, transitions to Open with a fresh
|
||||
* `openedAt` (when called from Half-Open or when the threshold
|
||||
* trips from Closed).
|
||||
* - `recordNeutral` — releases the probe permit, BUT leaves state and
|
||||
* counter untouched. Used for outcomes that are neither evidence of
|
||||
* backend health nor evidence of backend failure (caller-driven
|
||||
* cancellation, local timeout, terminal 4xx client errors). Critical
|
||||
* design point: if `recordNeutral` did not release the permit, a
|
||||
* single `TimeoutError` from per-attempt `AbortSignal.timeout` would
|
||||
* route through `recordNeutral` and permanently park the breaker in
|
||||
* half-open until process restart. Releasing the permit while leaving
|
||||
* state half-open keeps the "neutral doesn't claim health" semantic
|
||||
* without creating that wedge.
|
||||
*
|
||||
* Pairing invariant: every successful `check()` MUST be paired with
|
||||
* exactly one `record*()` on every code path including throws. Direct
|
||||
* consumers should wrap the protected operation in `try/finally`:
|
||||
*
|
||||
* breaker.check();
|
||||
* try {
|
||||
* const result = await operation();
|
||||
* breaker.recordSuccess();
|
||||
* return result;
|
||||
* } catch (err) {
|
||||
* // classify err and call recordFailure / recordNeutral / etc.
|
||||
* throw err;
|
||||
* }
|
||||
*
|
||||
* `resilientFetch`'s catch-all on `fetchImpl` already satisfies this
|
||||
* for that consumer.
|
||||
*
|
||||
* Atomicity model: the half-open gate relies on JavaScript event-loop
|
||||
* single-threadedness within a synchronous `check()` body. There is no
|
||||
* `await` inside `check()`; concurrent callers serialize on microtask
|
||||
* order, and exactly one observes `probeInFlight === false`. Do not
|
||||
* introduce `await` inside `check()` without revisiting the gate. If
|
||||
* this code is ever ported to a runtime with shared-memory threads
|
||||
* (Node `worker_threads` with `SharedArrayBuffer`, Web Workers with
|
||||
* shared registries), the boolean must become an atomic CAS — Resilience4j
|
||||
* and Hystrix use atomic permits *because* they run in JVM thread pools.
|
||||
*
|
||||
* Runtime-agnostic: depends only on a `now()` clock and standard JS —
|
||||
* no Node-only imports. Tests inject `now` to advance the clock
|
||||
* deterministically without `vi.useFakeTimers()`.
|
||||
*/
|
||||
|
||||
export class CircuitOpenError extends Error {
|
||||
override readonly name = 'CircuitOpenError';
|
||||
/** Approximate wait time before the breaker may transition to Half-Open
|
||||
* (or before the in-flight probe is expected to resolve). */
|
||||
readonly retryAfterMs: number;
|
||||
|
||||
constructor(retryAfterMs: number, key?: string) {
|
||||
super(
|
||||
key
|
||||
? `Circuit '${key}' is open; retry in ${Math.ceil(retryAfterMs / 1000)}s`
|
||||
: `Circuit is open; retry in ${Math.ceil(retryAfterMs / 1000)}s`,
|
||||
);
|
||||
this.retryAfterMs = retryAfterMs;
|
||||
}
|
||||
}
|
||||
|
||||
export interface CircuitBreakerOptions {
|
||||
/** Consecutive failures required to trip Closed -> Open. */
|
||||
failureThreshold?: number;
|
||||
/** Milliseconds Open before the next call may probe (Half-Open). */
|
||||
cooldownMs?: number;
|
||||
/**
|
||||
* Milliseconds to suggest in `CircuitOpenError.retryAfterMs` when the
|
||||
* breaker is Half-Open with the probe permit consumed. Default 1000ms.
|
||||
* Consumers with long-running protected ops (LLM streaming, large
|
||||
* uploads) should raise this — the cooldown clock is no longer the
|
||||
* right answer because cooldown has elapsed. Returning 0 invites
|
||||
* retry storms; returning the full cooldown misleads about wait.
|
||||
*/
|
||||
halfOpenRetryAfterMs?: number;
|
||||
/** Optional key for error messages and registry lookups. */
|
||||
key?: string;
|
||||
/** Clock override — defaults to `Date.now`. Tests inject deterministic time. */
|
||||
now?: () => number;
|
||||
}
|
||||
|
||||
type State = 'closed' | 'open' | 'half-open';
|
||||
|
||||
export class CircuitBreaker {
|
||||
private readonly failureThreshold: number;
|
||||
private readonly cooldownMs: number;
|
||||
private readonly halfOpenRetryAfterMs: number;
|
||||
private readonly key: string | undefined;
|
||||
private readonly now: () => number;
|
||||
|
||||
private state: State = 'closed';
|
||||
private consecutiveFailures = 0;
|
||||
private openedAt: number | null = null;
|
||||
/**
|
||||
* True between a successful `check()` and the next `record*()` call
|
||||
* during Half-Open. Gates concurrent callers from stampeding a still-
|
||||
* recovering dependency. Boolean rather than counter — single-permit
|
||||
* is the conservative end of the Hystrix/Resilience4j spectrum.
|
||||
*/
|
||||
private probeInFlight = false;
|
||||
|
||||
constructor(opts: CircuitBreakerOptions = {}) {
|
||||
this.failureThreshold = opts.failureThreshold ?? 3;
|
||||
this.cooldownMs = opts.cooldownMs ?? 30_000;
|
||||
this.halfOpenRetryAfterMs = opts.halfOpenRetryAfterMs ?? 1_000;
|
||||
this.key = opts.key;
|
||||
this.now = opts.now ?? (() => Date.now());
|
||||
}
|
||||
|
||||
/**
|
||||
* Throw `CircuitOpenError` if the breaker won't admit this call.
|
||||
* Otherwise consume the half-open probe permit (if applicable) and
|
||||
* return so the caller can attempt the protected work.
|
||||
*
|
||||
* Three rejection paths:
|
||||
* 1. Open and still in cooldown → throws with `retryAfterMs` =
|
||||
* remaining cooldown.
|
||||
* 2. Open with cooldown elapsed AND a probe is already in flight
|
||||
* (race: another caller transitioned to half-open and grabbed
|
||||
* the permit on a microtask before us) → throws with
|
||||
* `halfOpenRetryAfterMs`.
|
||||
* 3. Half-Open with probe in flight → throws with `halfOpenRetryAfterMs`.
|
||||
*
|
||||
* **Pairing invariant**: every successful return from `check()` MUST
|
||||
* be paired with exactly one `recordSuccess` / `recordFailure` /
|
||||
* `recordNeutral` on every code path including thrown exceptions.
|
||||
* Failing to pair leaves the probe permit consumed forever and
|
||||
* wedges the breaker. See file-header JSDoc for the canonical
|
||||
* try/finally pattern.
|
||||
*/
|
||||
check(): void {
|
||||
if (this.state === 'open' && this.openedAt !== null) {
|
||||
const elapsed = this.now() - this.openedAt;
|
||||
if (elapsed < this.cooldownMs) {
|
||||
throw new CircuitOpenError(this.cooldownMs - elapsed, this.key);
|
||||
}
|
||||
// Cooldown elapsed — transition to Half-Open. The very next
|
||||
// `probeInFlight` check below decides whether THIS caller gets
|
||||
// the permit or hits the gate.
|
||||
this.state = 'half-open';
|
||||
}
|
||||
|
||||
if (this.state === 'half-open') {
|
||||
if (this.probeInFlight) {
|
||||
throw new CircuitOpenError(this.halfOpenRetryAfterMs, this.key);
|
||||
}
|
||||
this.probeInFlight = true;
|
||||
}
|
||||
// Closed state falls through silently.
|
||||
}
|
||||
|
||||
recordSuccess(): void {
|
||||
this.probeInFlight = false;
|
||||
this.consecutiveFailures = 0;
|
||||
this.state = 'closed';
|
||||
this.openedAt = null;
|
||||
}
|
||||
|
||||
recordFailure(): void {
|
||||
this.probeInFlight = false;
|
||||
this.consecutiveFailures += 1;
|
||||
if (this.state === 'half-open' || this.consecutiveFailures >= this.failureThreshold) {
|
||||
this.state = 'open';
|
||||
this.openedAt = this.now();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Releases the probe permit BUT leaves state and counter untouched.
|
||||
* Use when an attempt produced a response or error that should not
|
||||
* influence breaker health in either direction — caller-driven aborts,
|
||||
* local AbortSignal timeouts, terminal 4xx client errors.
|
||||
*
|
||||
* Why permit-release-without-state-resolution: if `recordNeutral` did
|
||||
* not clear `probeInFlight`, a single `TimeoutError` from per-attempt
|
||||
* `AbortSignal.timeout` (which routes through neutral classification)
|
||||
* would permanently park the breaker in half-open. Since timeouts are
|
||||
* an *expected* outcome under flaky-dependency conditions, the cited
|
||||
* "per-attempt timeout bounds the stuck state" mitigation would itself
|
||||
* be the trigger for a permanent wedge. Releasing the permit closes
|
||||
* that loop while keeping the "neutral doesn't claim dependency
|
||||
* health" semantic.
|
||||
*
|
||||
* Calling `recordSuccess` for these would erase legitimate prior
|
||||
* failure signal; calling `recordFailure` would trip the breaker for
|
||||
* outcomes the backend isn't responsible for.
|
||||
*/
|
||||
recordNeutral(): void {
|
||||
this.probeInFlight = false;
|
||||
// State and consecutiveFailures are preserved by design.
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure read — no state mutation, no permit accounting. Returns the
|
||||
* *would-be* state at the current instant: 'half-open' if the breaker
|
||||
* is open with cooldown elapsed (regardless of whether a probe is in
|
||||
* flight), 'open' if open and still in cooldown, 'closed' otherwise.
|
||||
*
|
||||
* Inspection-only; safe to call from tests without consuming a probe
|
||||
* permit. The implicit Open -> Half-Open transition that mutates
|
||||
* `state` lives in `check()` only.
|
||||
*/
|
||||
getState(): State {
|
||||
if (this.state === 'open' && this.openedAt !== null) {
|
||||
const elapsed = this.now() - this.openedAt;
|
||||
if (elapsed >= this.cooldownMs) return 'half-open';
|
||||
}
|
||||
return this.state;
|
||||
}
|
||||
getConsecutiveFailures(): number {
|
||||
return this.consecutiveFailures;
|
||||
}
|
||||
/** Inspection-only test accessor for the half-open probe permit. */
|
||||
isProbeInFlight(): boolean {
|
||||
return this.probeInFlight;
|
||||
}
|
||||
/** Timestamp (ms since epoch) when the breaker last transitioned to Open,
|
||||
* or `null` if it's currently Closed. Useful for computing remaining
|
||||
* cooldown without consuming a probe permit via `check()`. */
|
||||
getOpenedAt(): number | null {
|
||||
return this.openedAt;
|
||||
}
|
||||
/** Configured cooldown duration in milliseconds. */
|
||||
getCooldownMs(): number {
|
||||
return this.cooldownMs;
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Per-process registry ────────────────────────────────────────────
|
||||
//
|
||||
// Single shared map keyed on caller-chosen strings. Used by
|
||||
// `resilient-fetch.ts` so multiple call sites targeting the same logical
|
||||
// endpoint share breaker state. Per-process only — not persisted.
|
||||
|
||||
const registry = new Map<string, CircuitBreaker>();
|
||||
|
||||
export function getBreaker(key: string, opts?: CircuitBreakerOptions): CircuitBreaker {
|
||||
let breaker = registry.get(key);
|
||||
if (!breaker) {
|
||||
breaker = new CircuitBreaker({ ...opts, key });
|
||||
registry.set(key, breaker);
|
||||
}
|
||||
return breaker;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test-only: clear all registered breakers. Tests must call this in
|
||||
* `beforeEach` to prevent breaker state from leaking across test cases.
|
||||
*/
|
||||
export function __resetBreakerRegistry__(): void {
|
||||
registry.clear();
|
||||
}
|
||||
279
gitnexus-shared/src/integrations/resilient-fetch.ts
Normal file
279
gitnexus-shared/src/integrations/resilient-fetch.ts
Normal file
|
|
@ -0,0 +1,279 @@
|
|||
/**
|
||||
* `resilientFetch` — fetch wrapped in retry + circuit breaker, with
|
||||
* GitHub-flavoured retry classification baked in (Retry-After parsing,
|
||||
* 401/403/404/422 treated as terminal client errors).
|
||||
*
|
||||
* Designed for the `gitnexus publish` GitHub `repository_dispatch`
|
||||
* call, but the classification rules apply to any GitHub REST endpoint.
|
||||
* Runtime-agnostic — no Node-only imports.
|
||||
*/
|
||||
|
||||
import {
|
||||
CircuitBreaker,
|
||||
CircuitOpenError,
|
||||
getBreaker,
|
||||
type CircuitBreakerOptions,
|
||||
} from './circuit-breaker.js';
|
||||
import { computeBackoffMs, type RetryOptions } from './retry.js';
|
||||
|
||||
export { CircuitOpenError };
|
||||
|
||||
export interface ResilientFetchOptions {
|
||||
/** Optional fetch implementation override. Defaults to `globalThis.fetch`. */
|
||||
fetchImpl?: typeof fetch;
|
||||
/**
|
||||
* Logical key for the breaker. Defaults to `<host><pathname>` of the
|
||||
* request URL — call sites targeting the same endpoint share breaker
|
||||
* state regardless of query-string differences.
|
||||
*/
|
||||
breakerKey?: string;
|
||||
/** Per-call breaker override. Used for tests and one-off configuration. */
|
||||
breaker?: CircuitBreaker;
|
||||
/** Tuning knobs for the breaker registered under `breakerKey`. */
|
||||
breakerOptions?: CircuitBreakerOptions;
|
||||
/** Tuning knobs for the retry helper. */
|
||||
retry?: Partial<Pick<RetryOptions, 'maxAttempts' | 'baseDelayMs' | 'capDelayMs'>> & {
|
||||
sleep?: RetryOptions['sleep'];
|
||||
random?: RetryOptions['random'];
|
||||
};
|
||||
/** Clock override propagated into Retry-After HTTP-date math and breaker. */
|
||||
now?: () => number;
|
||||
}
|
||||
|
||||
/** Cap on any single Retry-After wait — protects CLI from a buggy registry. */
|
||||
export const RETRY_AFTER_CAP_MS = 30_000;
|
||||
|
||||
const DEFAULT_RETRY = {
|
||||
maxAttempts: 3,
|
||||
baseDelayMs: 500,
|
||||
capDelayMs: 5_000,
|
||||
};
|
||||
|
||||
/**
|
||||
* Parse a `Retry-After` header value into milliseconds.
|
||||
* Accepts either a delta-seconds integer (`"30"`) or an HTTP-date.
|
||||
* Returns null on parse failure or negative deltas.
|
||||
*/
|
||||
export function parseRetryAfter(value: string | null, now: () => number = Date.now): number | null {
|
||||
if (!value) return null;
|
||||
const trimmed = value.trim();
|
||||
if (trimmed === '') return null;
|
||||
|
||||
if (/^[0-9]+$/.test(trimmed)) {
|
||||
const seconds = parseInt(trimmed, 10);
|
||||
if (Number.isNaN(seconds) || seconds < 0) return null;
|
||||
return seconds * 1000;
|
||||
}
|
||||
|
||||
const target = Date.parse(trimmed);
|
||||
if (Number.isNaN(target)) return null;
|
||||
const delta = target - now();
|
||||
return delta >= 0 ? delta : 0;
|
||||
}
|
||||
|
||||
/** Internal: outcome classification used by the resilientFetch loop. */
|
||||
type Outcome =
|
||||
| { kind: 'success'; resp: Response }
|
||||
| { kind: 'terminal-client'; resp: Response } // 4xx other than 429: no retry, breaker neutral
|
||||
| { kind: 'retryable-status'; resp: Response; afterMs: number | undefined } // 5xx, 429
|
||||
| { kind: 'terminal-network'; err: unknown } // TimeoutError or AbortError: no retry, breaker neutral
|
||||
| { kind: 'retryable-network'; err: unknown }; // DNS, ECONNRESET, etc.
|
||||
|
||||
/** Exported for unit tests. */
|
||||
export function classifyOutcome(
|
||||
result: { kind: 'error'; err: unknown } | { kind: 'response'; resp: Response },
|
||||
now: () => number,
|
||||
): Outcome {
|
||||
if (result.kind === 'error') {
|
||||
// Both timer-fired aborts (`AbortSignal.timeout()` → `TimeoutError`)
|
||||
// and caller-driven aborts (`AbortController.abort()` → `AbortError`)
|
||||
// are terminal: retrying against an already-aborted signal would
|
||||
// fail again immediately, and neither outcome reflects backend
|
||||
// health. They route through the breaker's neutral path.
|
||||
if (
|
||||
result.err instanceof DOMException &&
|
||||
(result.err.name === 'TimeoutError' || result.err.name === 'AbortError')
|
||||
) {
|
||||
return { kind: 'terminal-network', err: result.err };
|
||||
}
|
||||
return { kind: 'retryable-network', err: result.err };
|
||||
}
|
||||
const resp = result.resp;
|
||||
if (resp.status >= 200 && resp.status < 400) return { kind: 'success', resp };
|
||||
if (resp.status === 429) {
|
||||
// `resp.headers` is always present on a real `Response`, but tests
|
||||
// sometimes stub `fetch` with a plain `{ ok, status }` object. Be
|
||||
// defensive — a missing `Retry-After` falls through to exponential
|
||||
// backoff, which is the correct behaviour anyway.
|
||||
const retryAfterHeader =
|
||||
typeof resp.headers?.get === 'function' ? resp.headers.get('Retry-After') : null;
|
||||
const parsed = parseRetryAfter(retryAfterHeader, now);
|
||||
return {
|
||||
kind: 'retryable-status',
|
||||
resp,
|
||||
afterMs: parsed !== null ? Math.min(parsed, RETRY_AFTER_CAP_MS) : undefined,
|
||||
};
|
||||
}
|
||||
if (resp.status >= 500) return { kind: 'retryable-status', resp, afterMs: undefined };
|
||||
return { kind: 'terminal-client', resp };
|
||||
}
|
||||
|
||||
const defaultSleep = (ms: number): Promise<void> =>
|
||||
new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
function defaultBreakerKey(input: string | URL): string {
|
||||
try {
|
||||
const url = typeof input === 'string' ? new URL(input) : input;
|
||||
return `${url.host}${url.pathname}`;
|
||||
} catch {
|
||||
return String(input);
|
||||
}
|
||||
}
|
||||
|
||||
/** Final error thrown when retries are exhausted on a 5xx / 429. */
|
||||
export class ResilientFetchExhaustedError extends Error {
|
||||
override readonly name = 'ResilientFetchExhaustedError';
|
||||
constructor(public readonly response: Response) {
|
||||
super(`Request failed after retries (HTTP ${response.status})`);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wrap `fetch` with bounded retries and a per-process circuit breaker.
|
||||
*
|
||||
* Semantics:
|
||||
* - 5xx and 429 responses are retried; 429 honors `Retry-After` (capped).
|
||||
* - Network throws are retried unless they are `TimeoutError` DOMExceptions.
|
||||
* - Timeouts and 4xx (other than 429) are returned/thrown without retry
|
||||
* AND without incrementing the breaker — they reflect caller config
|
||||
* or local network state, not registry health.
|
||||
* - Each `fetch` call carries the caller-supplied `signal` (e.g. an
|
||||
* `AbortSignal.timeout()`) — that timeout bounds each individual
|
||||
* attempt, not the whole retry sequence.
|
||||
* - When the breaker is open, throws `CircuitOpenError` synchronously
|
||||
* without invoking `fetch`.
|
||||
* - When retries are exhausted on a 5xx / 429, throws
|
||||
* `ResilientFetchExhaustedError` carrying the last response.
|
||||
*
|
||||
* Cumulative wall-clock budget:
|
||||
* maxAttempts × (per-attempt-timeout + capDelayMs)
|
||||
* With defaults (3, 500ms base, 5000ms cap) and a typical 15s per-attempt
|
||||
* timeout from the caller's signal, worst case is ~3 × (15s + 5s) = 60s.
|
||||
* Callers that want a tighter total bound should reduce `maxAttempts` or
|
||||
* wrap `resilientFetch` in their own outer `AbortSignal.timeout()`.
|
||||
*/
|
||||
export async function resilientFetch(
|
||||
input: string | URL,
|
||||
init: RequestInit | undefined,
|
||||
opts: ResilientFetchOptions = {},
|
||||
): Promise<Response> {
|
||||
const fetchImpl = opts.fetchImpl ?? globalThis.fetch;
|
||||
const now = opts.now ?? (() => Date.now());
|
||||
const breaker =
|
||||
opts.breaker ?? getBreaker(opts.breakerKey ?? defaultBreakerKey(input), opts.breakerOptions);
|
||||
|
||||
const retryConfig = {
|
||||
maxAttempts: opts.retry?.maxAttempts ?? DEFAULT_RETRY.maxAttempts,
|
||||
baseDelayMs: opts.retry?.baseDelayMs ?? DEFAULT_RETRY.baseDelayMs,
|
||||
capDelayMs: opts.retry?.capDelayMs ?? DEFAULT_RETRY.capDelayMs,
|
||||
};
|
||||
const sleep = opts.retry?.sleep ?? defaultSleep;
|
||||
const random = opts.retry?.random ?? Math.random;
|
||||
|
||||
// Fail fast on an open breaker, before invoking fetch.
|
||||
breaker.check();
|
||||
|
||||
for (let attempt = 0; attempt < retryConfig.maxAttempts; attempt++) {
|
||||
let result: { kind: 'error'; err: unknown } | { kind: 'response'; resp: Response };
|
||||
try {
|
||||
// CodeQL js/server-side-request-forgery — flagged because `input`
|
||||
// is caller-supplied. Suppressed: every concrete caller passes
|
||||
// either a hardcoded URL constant (UNDERSTAND_QUICKLY_DISPATCH_URL,
|
||||
// OpenRouter base URL) or a value derived from configuration
|
||||
// (env vars, saved settings, the local backend URL). User-input
|
||||
// request fields (e.g. PR title, repo name) never flow into
|
||||
// `input`. Validating URL shape here would push false-positive
|
||||
// rejection onto every caller — wrong layer for the check.
|
||||
// lgtm[js/server-side-request-forgery]
|
||||
// codeql[js/server-side-request-forgery]
|
||||
const resp = await fetchImpl(input, init);
|
||||
result = { kind: 'response', resp };
|
||||
} catch (err) {
|
||||
result = { kind: 'error', err };
|
||||
}
|
||||
|
||||
const outcome = classifyOutcome(result, now);
|
||||
|
||||
switch (outcome.kind) {
|
||||
case 'success':
|
||||
breaker.recordSuccess();
|
||||
return outcome.resp;
|
||||
|
||||
case 'terminal-client':
|
||||
// 4xx: do not count as breaker failure (the server is healthy
|
||||
// and rejecting our request — auth, scope, or routing). But
|
||||
// also do NOT call recordSuccess: a 401 sandwiched between
|
||||
// 5xx responses would otherwise erase the running outage
|
||||
// signal. The breaker's neutral path leaves state untouched.
|
||||
breaker.recordNeutral();
|
||||
return outcome.resp;
|
||||
|
||||
case 'terminal-network':
|
||||
// Either `AbortSignal.timeout()` fired locally OR an external
|
||||
// caller cancelled the request via AbortController. The server
|
||||
// never had a chance to answer; this reflects the user's
|
||||
// network or an explicit cancel, not registry health. Don't
|
||||
// punish the breaker AND don't reset its outage signal.
|
||||
breaker.recordNeutral();
|
||||
throw outcome.err;
|
||||
|
||||
case 'retryable-status':
|
||||
if (attempt + 1 >= retryConfig.maxAttempts) {
|
||||
breaker.recordFailure();
|
||||
throw new ResilientFetchExhaustedError(outcome.resp);
|
||||
}
|
||||
await sleep(
|
||||
computeBackoffMs(
|
||||
attempt,
|
||||
retryConfig.baseDelayMs,
|
||||
retryConfig.capDelayMs,
|
||||
outcome.afterMs,
|
||||
random,
|
||||
),
|
||||
);
|
||||
break;
|
||||
|
||||
case 'retryable-network':
|
||||
if (attempt + 1 >= retryConfig.maxAttempts) {
|
||||
breaker.recordFailure();
|
||||
throw outcome.err;
|
||||
}
|
||||
await sleep(
|
||||
computeBackoffMs(
|
||||
attempt,
|
||||
retryConfig.baseDelayMs,
|
||||
retryConfig.capDelayMs,
|
||||
undefined,
|
||||
random,
|
||||
),
|
||||
);
|
||||
break;
|
||||
|
||||
default: {
|
||||
// Exhaustiveness guard. If a sixth `Outcome` kind is added in
|
||||
// future, TypeScript will refuse to assign it to `never` and
|
||||
// this line forces the maintainer to add an explicit arm
|
||||
// rather than silently fall through to retry/no-retry behaviour.
|
||||
const _exhaustive: never = outcome;
|
||||
throw new Error(`resilientFetch: unhandled outcome ${JSON.stringify(_exhaustive)}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Unreachable: every iteration of the loop either returns (success
|
||||
// / terminal-client) or throws (terminal-network / retry exhaustion).
|
||||
// The throw is here purely so TypeScript's control-flow analysis sees
|
||||
// the function never falls off the end without producing `Promise<Response>`.
|
||||
/* c8 ignore next 2 */
|
||||
throw new Error('resilientFetch: retry loop terminated unexpectedly');
|
||||
}
|
||||
105
gitnexus-shared/src/integrations/retry.ts
Normal file
105
gitnexus-shared/src/integrations/retry.ts
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
/**
|
||||
* Bounded retry helper with full-jitter exponential backoff.
|
||||
*
|
||||
* Runtime-agnostic: depends only on `setTimeout`, `Math.random`, and the
|
||||
* Promise machinery — no Node-only imports. Safe to consume from CLI,
|
||||
* server, or browser callers.
|
||||
*
|
||||
* Pattern reference: gitnexus/src/core/embeddings/http-client.ts. This
|
||||
* helper is the upgraded form: classification is caller-supplied (so
|
||||
* 4xx-vs-5xx-vs-timeout decisions live with the protocol that knows
|
||||
* them), backoff is exponential with full jitter, and an optional
|
||||
* `afterMs` lets callers honor `Retry-After` headers.
|
||||
*/
|
||||
|
||||
export interface RetryOptions {
|
||||
/** Initial delay before the first retry attempt, in milliseconds. */
|
||||
baseDelayMs: number;
|
||||
/** Upper bound on any single delay, in milliseconds. */
|
||||
capDelayMs: number;
|
||||
/** Total attempts including the first call. Must be >= 1. */
|
||||
maxAttempts: number;
|
||||
/**
|
||||
* Decide whether to retry after a thrown error.
|
||||
* Return `{retry:false}` to terminate immediately and rethrow.
|
||||
* Return `{retry:true}` to retry with exponential-backoff jitter.
|
||||
* Return `{retry:true, afterMs}` to wait at least `afterMs` (still
|
||||
* subject to `capDelayMs`) — used by callers parsing `Retry-After`.
|
||||
*/
|
||||
isRetryable: (err: unknown, attempt: number) => RetryDecision;
|
||||
/** Sleep override — defaults to `setTimeout`. Tests inject fake timers. */
|
||||
sleep?: (ms: number) => Promise<void>;
|
||||
/** Random override — defaults to `Math.random`. Tests inject seeded values. */
|
||||
random?: () => number;
|
||||
}
|
||||
|
||||
export type RetryDecision = { retry: false } | { retry: true; afterMs?: number };
|
||||
|
||||
const defaultSleep = (ms: number): Promise<void> =>
|
||||
new Promise((resolve) => setTimeout(resolve, ms));
|
||||
|
||||
/**
|
||||
* Compute the delay before the next retry attempt.
|
||||
*
|
||||
* - When the caller specifies `afterMs` (e.g., from `Retry-After`), use
|
||||
* `min(afterMs, capDelayMs)` so a misbehaving server can't pin the
|
||||
* client for an arbitrarily long wait.
|
||||
* - Otherwise compute full-jitter exponential backoff:
|
||||
* `random() * min(cap, base * 2^attempt)`. Full jitter (rather than
|
||||
* "equal jitter") avoids retry-storm thundering herd, per AWS
|
||||
* guidance on backoff strategies.
|
||||
*/
|
||||
export function computeBackoffMs(
|
||||
attempt: number,
|
||||
baseDelayMs: number,
|
||||
capDelayMs: number,
|
||||
afterMs: number | undefined,
|
||||
random: () => number,
|
||||
): number {
|
||||
if (afterMs !== undefined) {
|
||||
return Math.min(Math.max(0, afterMs), capDelayMs);
|
||||
}
|
||||
const exponential = baseDelayMs * Math.pow(2, attempt);
|
||||
const upper = Math.min(capDelayMs, exponential);
|
||||
return Math.floor(random() * upper);
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute `fn` with bounded retries.
|
||||
*
|
||||
* The classification of "retryable" is the caller's responsibility — see
|
||||
* `resilient-fetch.ts` for the GitHub-dispatch-specific rules. This
|
||||
* helper is the mechanical retry loop only.
|
||||
*/
|
||||
export async function withRetry<T>(
|
||||
fn: (attempt: number) => Promise<T>,
|
||||
opts: RetryOptions,
|
||||
): Promise<T> {
|
||||
if (opts.maxAttempts < 1) {
|
||||
throw new Error(`withRetry: maxAttempts must be >= 1, got ${opts.maxAttempts}`);
|
||||
}
|
||||
const sleep = opts.sleep ?? defaultSleep;
|
||||
const random = opts.random ?? Math.random;
|
||||
|
||||
let lastError: unknown;
|
||||
for (let attempt = 0; attempt < opts.maxAttempts; attempt++) {
|
||||
try {
|
||||
return await fn(attempt);
|
||||
} catch (err) {
|
||||
lastError = err;
|
||||
const decision = opts.isRetryable(err, attempt);
|
||||
if (!decision.retry) throw err;
|
||||
// Don't sleep after the final attempt.
|
||||
if (attempt + 1 >= opts.maxAttempts) break;
|
||||
const delayMs = computeBackoffMs(
|
||||
attempt,
|
||||
opts.baseDelayMs,
|
||||
opts.capDelayMs,
|
||||
decision.afterMs,
|
||||
random,
|
||||
);
|
||||
if (delayMs > 0) await sleep(delayMs);
|
||||
}
|
||||
}
|
||||
throw lastError;
|
||||
}
|
||||
13
gitnexus-shared/src/test-helpers.ts
Normal file
13
gitnexus-shared/src/test-helpers.ts
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
/**
|
||||
* Test-only helpers.
|
||||
*
|
||||
* Symbols here are reachable from `gitnexus-shared/test-helpers` so test
|
||||
* suites can reset shared registries or exercise internal classifiers,
|
||||
* but they are deliberately NOT re-exported from the main `gitnexus-shared`
|
||||
* barrel. Production consumers should never import this module — calling
|
||||
* `__resetBreakerRegistry__()` from a tool implementation would silently
|
||||
* nuke every circuit breaker process-wide.
|
||||
*/
|
||||
|
||||
export { __resetBreakerRegistry__ } from './integrations/circuit-breaker.js';
|
||||
export { classifyOutcome } from './integrations/resilient-fetch.js';
|
||||
|
|
@ -20,6 +20,7 @@ import {
|
|||
ProviderConfig,
|
||||
} from './types';
|
||||
import { DEFAULT_OPENROUTER_BASE_URL, DEFAULT_OLLAMA_BASE_URL } from '../../config/ui-constants';
|
||||
import { resilientFetch } from 'gitnexus-shared';
|
||||
|
||||
const STORAGE_KEY = 'gitnexus-llm-settings';
|
||||
|
||||
|
|
@ -407,7 +408,10 @@ export const getAvailableModels = (provider: LLMProvider): string[] => {
|
|||
*/
|
||||
export const fetchOpenRouterModels = async (): Promise<Array<{ id: string; name: string }>> => {
|
||||
try {
|
||||
const response = await fetch(`${DEFAULT_OPENROUTER_BASE_URL}/models`);
|
||||
const response = await resilientFetch(`${DEFAULT_OPENROUTER_BASE_URL}/models`, undefined, {
|
||||
breakerKey: 'openrouter-models',
|
||||
retry: { maxAttempts: 2, baseDelayMs: 500, capDelayMs: 2_000 },
|
||||
});
|
||||
if (!response.ok) throw new Error('Failed to fetch models');
|
||||
const data = await response.json();
|
||||
return data.data.map((model: any) => ({
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@
|
|||
*/
|
||||
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import { CircuitOpenError, ResilientFetchExhaustedError, resilientFetch } from 'gitnexus-shared';
|
||||
|
||||
// ── Types ──────────────────────────────────────────────────────────────────
|
||||
|
||||
|
|
@ -237,29 +238,91 @@ export function normalizeServerUrl(input: string): string {
|
|||
const DEFAULT_TIMEOUT_MS = 30_000;
|
||||
const PROBE_TIMEOUT_MS = 2_000;
|
||||
|
||||
/** Idempotent HTTP methods. Other verbs (POST, PATCH, PUT, DELETE) get
|
||||
* a single-attempt retry budget by default to avoid duplicate side
|
||||
* effects on retry — a POST that 5xx'd may have already executed
|
||||
* server-side. Callers that have idempotency keys or otherwise know
|
||||
* their mutation is safe to retry can opt in via `forceRetry`. */
|
||||
const IDEMPOTENT_METHODS = new Set(['GET', 'HEAD', 'OPTIONS']);
|
||||
|
||||
const fetchWithTimeout = async (
|
||||
url: string,
|
||||
init: RequestInit = {},
|
||||
timeoutMs: number = DEFAULT_TIMEOUT_MS,
|
||||
/**
|
||||
* Force a retry budget on non-idempotent methods. Default false.
|
||||
* Pass true only when the endpoint is known-idempotent (e.g. DELETE
|
||||
* of a known-deleted resource — second call is a 404 / no-op) AND
|
||||
* the duplicate-side-effect window is acceptable.
|
||||
*/
|
||||
forceRetry = false,
|
||||
): Promise<Response> => {
|
||||
const controller = new AbortController();
|
||||
// Merge external signal if provided
|
||||
// Merge the external caller signal (if any) with an
|
||||
// `AbortSignal.timeout()` so a timer-fired abort produces a
|
||||
// `DOMException` with `name === 'TimeoutError'` — which
|
||||
// `resilientFetch` correctly classifies as terminal-network (no
|
||||
// retry, no breaker hit). A manual `AbortController.abort()` would
|
||||
// produce `name === 'AbortError'` and route through the
|
||||
// retryable-network branch, which mis-penalizes the breaker for
|
||||
// user-side network slowness.
|
||||
const timeoutSignal = AbortSignal.timeout(timeoutMs);
|
||||
const externalSignal = init.signal;
|
||||
if (externalSignal) {
|
||||
externalSignal.addEventListener('abort', () => controller.abort());
|
||||
const signal = externalSignal ? AbortSignal.any([timeoutSignal, externalSignal]) : timeoutSignal;
|
||||
|
||||
const method = (init.method ?? 'GET').toUpperCase();
|
||||
const isIdempotent = IDEMPOTENT_METHODS.has(method);
|
||||
const maxAttempts = isIdempotent || forceRetry ? 2 : 1;
|
||||
|
||||
// Key the breaker by the current backend origin so switching backend
|
||||
// URLs (e.g. recovering from a flapping local server by pointing at
|
||||
// a different host) gives the new origin a fresh breaker state. A
|
||||
// single shared `'web-backend'` key would otherwise leave a user
|
||||
// locked out for the full cooldown after one bad host trips the
|
||||
// circuit. The malformed-URL fallback is defensive — `setBackendUrl`
|
||||
// normalizes input, so this branch shouldn't fire in practice.
|
||||
let breakerKey: string;
|
||||
try {
|
||||
breakerKey = `web-backend:${new URL(_backendUrl).origin}`;
|
||||
} catch {
|
||||
breakerKey = 'web-backend:invalid';
|
||||
}
|
||||
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
||||
|
||||
try {
|
||||
const response = await fetch(url, { ...init, signal: controller.signal });
|
||||
// Bounded retries + 5xx/429 handling are delegated to resilientFetch.
|
||||
// Method-aware budget: idempotent verbs retry once on transient
|
||||
// backend failures; mutations (POST/PATCH/PUT/DELETE) default to
|
||||
// single-attempt to avoid duplicate side effects.
|
||||
const response = await resilientFetch(
|
||||
url,
|
||||
{ ...init, signal },
|
||||
{
|
||||
breakerKey,
|
||||
retry: { maxAttempts, baseDelayMs: 250, capDelayMs: 1500 },
|
||||
},
|
||||
);
|
||||
return response;
|
||||
} catch (error: unknown) {
|
||||
if (error instanceof DOMException && error.name === 'AbortError') {
|
||||
if (externalSignal?.aborted) {
|
||||
throw new BackendError('Request aborted', 0, 'network');
|
||||
}
|
||||
if (error instanceof CircuitOpenError) {
|
||||
throw new BackendError(
|
||||
`GitNexus backend at ${_backendUrl} is unhealthy; retry in ${Math.ceil(error.retryAfterMs / 1000)}s`,
|
||||
0,
|
||||
'network',
|
||||
);
|
||||
}
|
||||
if (error instanceof ResilientFetchExhaustedError) {
|
||||
// Fall through to caller — surface the raw response so assertOk
|
||||
// can craft the BackendError with the right code.
|
||||
return error.response;
|
||||
}
|
||||
if (error instanceof DOMException && error.name === 'TimeoutError') {
|
||||
throw new BackendError(`Request to ${url} timed out after ${timeoutMs}ms`, 0, 'timeout');
|
||||
}
|
||||
if (error instanceof DOMException && error.name === 'AbortError') {
|
||||
// External caller-driven cancellation — `timeoutSignal` would
|
||||
// have surfaced as TimeoutError above, so this branch covers
|
||||
// only the externally-aborted case.
|
||||
throw new BackendError('Request aborted', 0, 'network');
|
||||
}
|
||||
if (error instanceof TypeError) {
|
||||
throw new BackendError(
|
||||
`Network error reaching GitNexus backend at ${_backendUrl}: ${error.message}`,
|
||||
|
|
@ -268,8 +331,6 @@ const fetchWithTimeout = async (
|
|||
);
|
||||
}
|
||||
throw error;
|
||||
} finally {
|
||||
clearTimeout(timer);
|
||||
}
|
||||
};
|
||||
|
||||
|
|
|
|||
110
gitnexus-web/test/unit/backend-client-retry.test.ts
Normal file
110
gitnexus-web/test/unit/backend-client-retry.test.ts
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
/**
|
||||
* Method-aware retry budget + timeout-as-TimeoutError verification for
|
||||
* backend-client's `fetchWithTimeout`.
|
||||
*
|
||||
* Closes review findings on PR #1448:
|
||||
* - Non-idempotent POST/DELETE must NOT be retried by default —
|
||||
* a 5xx on `startAnalyze` could otherwise start a duplicate job.
|
||||
* - Timer-fired timeout must surface as `DOMException(name='TimeoutError')`,
|
||||
* not `AbortError`, so resilientFetch routes it through the
|
||||
* terminal-network branch (no retry, no breaker hit).
|
||||
*/
|
||||
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { getBreaker } from 'gitnexus-shared';
|
||||
import { __resetBreakerRegistry__ } from 'gitnexus-shared/test-helpers';
|
||||
import { fetchRepos, setBackendUrl, startAnalyze } from '../../src/services/backend-client';
|
||||
|
||||
const BASE = 'http://localhost:4747';
|
||||
|
||||
describe('backend-client retry budget (method-aware)', () => {
|
||||
beforeEach(() => {
|
||||
__resetBreakerRegistry__();
|
||||
setBackendUrl(BASE);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('GET retries once on transient 503 (idempotent verb)', async () => {
|
||||
let n = 0;
|
||||
const fetchMock = vi.fn(async () => {
|
||||
n += 1;
|
||||
if (n === 1) return new Response('boom', { status: 503 });
|
||||
return new Response('[]', {
|
||||
status: 200,
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
});
|
||||
});
|
||||
vi.stubGlobal('fetch', fetchMock);
|
||||
|
||||
const repos = await fetchRepos();
|
||||
expect(repos).toEqual([]);
|
||||
// 1 retry budget on idempotent GET → 2 total fetch calls.
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('POST does NOT retry on 503 by default (non-idempotent verb)', async () => {
|
||||
const fetchMock = vi.fn(async () => new Response('boom', { status: 503 }));
|
||||
vi.stubGlobal('fetch', fetchMock);
|
||||
|
||||
await expect(startAnalyze({ path: '/tmp/repo' })).rejects.toBeTruthy();
|
||||
// Single attempt — never duplicates a job-start POST.
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('switching backend URL after a circuit opens reaches a fresh breaker (U3)', async () => {
|
||||
// Pre-open the breaker for host-A by directly recording 3 failures.
|
||||
setBackendUrl('http://host-a.test:4747');
|
||||
const aKey = 'web-backend:http://host-a.test:4747';
|
||||
const breakerA = getBreaker(aKey);
|
||||
breakerA.recordFailure();
|
||||
breakerA.recordFailure();
|
||||
breakerA.recordFailure();
|
||||
expect(breakerA.getState()).toBe('open');
|
||||
|
||||
// Switch to host-B and make a request — must succeed against the
|
||||
// new origin without tripping the host-A circuit. Under the old
|
||||
// single-key behaviour the call would throw CircuitOpenError.
|
||||
setBackendUrl('http://host-b.test:4747');
|
||||
const fetchMock = vi.fn(
|
||||
async () =>
|
||||
new Response('[]', {
|
||||
status: 200,
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
}),
|
||||
);
|
||||
vi.stubGlobal('fetch', fetchMock);
|
||||
|
||||
const repos = await fetchRepos();
|
||||
expect(repos).toEqual([]);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
|
||||
// Host-A's breaker is still open in cooldown.
|
||||
expect(breakerA.getState()).toBe('open');
|
||||
// Host-B has its own (fresh) breaker.
|
||||
const bKey = 'web-backend:http://host-b.test:4747';
|
||||
expect(getBreaker(bKey).getState()).toBe('closed');
|
||||
expect(getBreaker(bKey).getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('breaker not incremented when timeout fires (TimeoutError, not AbortError)', async () => {
|
||||
// Reject directly with a TimeoutError DOMException, mimicking what
|
||||
// `fetch` produces when its `AbortSignal.timeout()`-wired signal
|
||||
// fires. The real-fetch path goes signal.reason → reject(reason);
|
||||
// we shortcut that here so the test doesn't have to wait the
|
||||
// 30-second default timeout.
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new DOMException('aborted by timeout', 'TimeoutError');
|
||||
});
|
||||
vi.stubGlobal('fetch', fetchMock);
|
||||
|
||||
await expect(fetchRepos()).rejects.toMatchObject({ code: 'timeout' });
|
||||
|
||||
// The breaker must not have been penalized for a local timeout.
|
||||
expect(getBreaker(`web-backend:${BASE}`).getConsecutiveFailures()).toBe(0);
|
||||
// Timeout is terminal — no retry attempted.
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
|
|
@ -1,6 +1,8 @@
|
|||
import os from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
|
||||
import { CircuitBreaker, withRetry } from 'gitnexus-shared';
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Download resilience defaults
|
||||
// ---------------------------------------------------------------------------
|
||||
|
|
@ -108,70 +110,19 @@ export function isNetworkFetchError(message: string): boolean {
|
|||
/** @internal Used by `withHfDownloadRetry` to mark a circuit-open rejection. */
|
||||
export const CIRCUIT_OPEN_TAG = 'hf-circuit-open';
|
||||
|
||||
/** Circuit-breaker states. */
|
||||
type CircuitState = 'closed' | 'open' | 'half-open';
|
||||
|
||||
/**
|
||||
* Circuit breaker for HuggingFace model downloads.
|
||||
*
|
||||
* After `failureThreshold` consecutive network failures the circuit opens and
|
||||
* all subsequent calls to `withHfDownloadRetry` fail immediately without
|
||||
* issuing any network requests. After `resetTimeoutMs` the circuit enters the
|
||||
* half-open state and the next call is attempted — if it succeeds the circuit
|
||||
* closes again; if it fails the circuit re-opens.
|
||||
*
|
||||
* Exported for unit-testing; production code should use the module-level
|
||||
* `hfDownloadCircuit` singleton.
|
||||
* Module-level singleton shared by both embedder entry points
|
||||
* (`core/embeddings/embedder.ts` + `mcp/core/embedder.ts`). Per-process
|
||||
* only — not persisted across restarts. Backed by the shared
|
||||
* `CircuitBreaker` from `gitnexus-shared` (same state machine, same
|
||||
* semantics, plus the single-permit half-open gate that prevents
|
||||
* recovery-time stampedes).
|
||||
*/
|
||||
export class HfDownloadCircuitBreaker {
|
||||
private _state: CircuitState = 'closed';
|
||||
private _failures = 0;
|
||||
/** Timestamp of the last recorded failure (ms since epoch). */
|
||||
lastFailureAt = 0;
|
||||
|
||||
constructor(
|
||||
readonly failureThreshold: number = CB_FAILURE_THRESHOLD,
|
||||
readonly resetTimeoutMs: number = CB_RESET_TIMEOUT_MS,
|
||||
) {}
|
||||
|
||||
/** Effective state, factoring in the reset-timeout transition. */
|
||||
get state(): CircuitState {
|
||||
if (this._state === 'open' && Date.now() - this.lastFailureAt > this.resetTimeoutMs) {
|
||||
this._state = 'half-open';
|
||||
}
|
||||
return this._state;
|
||||
}
|
||||
|
||||
/** Returns true when the circuit is open and calls should be rejected. */
|
||||
isOpen(): boolean {
|
||||
return this.state === 'open';
|
||||
}
|
||||
|
||||
/** Record a successful call — resets the failure counter and closes the circuit. */
|
||||
recordSuccess(): void {
|
||||
this._failures = 0;
|
||||
this._state = 'closed';
|
||||
}
|
||||
|
||||
/** Record a failed call — increments the counter and opens the circuit when the threshold is reached. */
|
||||
recordFailure(): void {
|
||||
this._failures++;
|
||||
this.lastFailureAt = Date.now();
|
||||
if (this._failures >= this.failureThreshold) {
|
||||
this._state = 'open';
|
||||
}
|
||||
}
|
||||
|
||||
/** @internal Reset to initial state (used in tests). */
|
||||
reset(): void {
|
||||
this._failures = 0;
|
||||
this._state = 'closed';
|
||||
this.lastFailureAt = 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Module-level singleton shared by both embedder entry points. */
|
||||
export const hfDownloadCircuit = new HfDownloadCircuitBreaker();
|
||||
export const hfDownloadCircuit = new CircuitBreaker({
|
||||
failureThreshold: CB_FAILURE_THRESHOLD,
|
||||
cooldownMs: CB_RESET_TIMEOUT_MS,
|
||||
key: 'hf-download',
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Retry + timeout wrapper
|
||||
|
|
@ -219,11 +170,6 @@ export function withDownloadTimeout<T>(fn: () => Promise<T>, timeoutMs: number):
|
|||
});
|
||||
}
|
||||
|
||||
/** @internal Async sleep (exposed for testing). */
|
||||
export function sleep(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
export interface HfRetryOptions {
|
||||
/** Maximum total attempts including the initial one (default: `HF_MAX_ATTEMPTS`). */
|
||||
maxAttempts?: number;
|
||||
|
|
@ -235,7 +181,7 @@ export interface HfRetryOptions {
|
|||
* Circuit-breaker instance to use. Defaults to the module-level
|
||||
* `hfDownloadCircuit` singleton. Pass a fresh instance in tests.
|
||||
*/
|
||||
circuit?: HfDownloadCircuitBreaker;
|
||||
circuit?: CircuitBreaker;
|
||||
/**
|
||||
* Optional callback invoked before each retry (not the initial attempt).
|
||||
* @param attempt - 1-based retry number
|
||||
|
|
@ -295,49 +241,74 @@ export async function withHfDownloadRetry<T>(
|
|||
circuit = hfDownloadCircuit,
|
||||
onRetry,
|
||||
} = options;
|
||||
if (circuit.isOpen()) {
|
||||
const secsUntilReset = Math.ceil(
|
||||
(circuit.resetTimeoutMs - (Date.now() - circuit.lastFailureAt)) / 1000,
|
||||
);
|
||||
if (circuit.getState() === 'open') {
|
||||
// Compute remaining cooldown without consuming a probe permit.
|
||||
const openedAt = circuit.getOpenedAt();
|
||||
const secsUntilReset =
|
||||
openedAt !== null ? Math.ceil((circuit.getCooldownMs() - (Date.now() - openedAt)) / 1000) : 0;
|
||||
throw new Error(
|
||||
`${CIRCUIT_OPEN_TAG}: HuggingFace download circuit is open after repeated network failures` +
|
||||
(secsUntilReset > 0 ? ` — will reset in ~${secsUntilReset}s` : ''),
|
||||
);
|
||||
}
|
||||
|
||||
let lastError: Error = new Error('unknown error');
|
||||
// Retry budget delegated to `withRetry` from gitnexus-shared. The
|
||||
// HF-specific bits — per-attempt timeout, network-vs-non-network
|
||||
// classification, circuit-breaker recording, onRetry callback — wire
|
||||
// through the `isRetryable` callback. `circuitTripped` is the
|
||||
// sentinel that lets us replace the final thrown error with a
|
||||
// CIRCUIT_OPEN_TAG message when the breaker tripped mid-loop.
|
||||
let circuitTripped = false;
|
||||
|
||||
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
try {
|
||||
const result = await withDownloadTimeout(fn, timeoutMs);
|
||||
circuit.recordSuccess();
|
||||
return result;
|
||||
} catch (err) {
|
||||
lastError = err instanceof Error ? err : new Error(String(err));
|
||||
|
||||
if (!isNetworkFetchError(lastError.message)) {
|
||||
// Non-network error (e.g. CUDA unavailable) — propagate without retry
|
||||
throw lastError;
|
||||
}
|
||||
|
||||
circuit.recordFailure();
|
||||
|
||||
if (circuit.isOpen()) {
|
||||
// Circuit just tripped — fail fast, no more retries
|
||||
throw new Error(
|
||||
`${CIRCUIT_OPEN_TAG}: HuggingFace download circuit opened after ${circuit.failureThreshold} consecutive failures`,
|
||||
);
|
||||
}
|
||||
|
||||
if (attempt < maxAttempts - 1) {
|
||||
const delay = baseDelayMs * Math.pow(2, attempt);
|
||||
onRetry?.(attempt + 1, maxAttempts, lastError);
|
||||
await sleep(delay);
|
||||
}
|
||||
try {
|
||||
return await withRetry(
|
||||
async () => {
|
||||
const result = await withDownloadTimeout(fn, timeoutMs);
|
||||
circuit.recordSuccess();
|
||||
return result;
|
||||
},
|
||||
{
|
||||
maxAttempts,
|
||||
baseDelayMs,
|
||||
// Disable the cap to match the bespoke pure-exponential
|
||||
// progression. With the default `HF_MAX_ATTEMPTS_CAP = 10` and
|
||||
// `baseDelayMs = 2000`, the largest possible delay is
|
||||
// `2000 * 2^9 = ~17 minutes` — bounded enough not to need a cap.
|
||||
capDelayMs: Number.MAX_SAFE_INTEGER,
|
||||
isRetryable: (err, attempt) => {
|
||||
const error = err instanceof Error ? err : new Error(String(err));
|
||||
if (!isNetworkFetchError(error.message)) {
|
||||
// Non-network error (e.g. CUDA unavailable) — propagate
|
||||
// without retry. Use recordNeutral so the breaker's existing
|
||||
// failure-count progress isn't reset by a non-network failure
|
||||
// that says nothing about the CDN's health.
|
||||
circuit.recordNeutral();
|
||||
return { retry: false };
|
||||
}
|
||||
circuit.recordFailure();
|
||||
if (circuit.getState() === 'open') {
|
||||
// Circuit just tripped — fail fast, no more retries.
|
||||
circuitTripped = true;
|
||||
return { retry: false };
|
||||
}
|
||||
// Mirror the bespoke onRetry contract: fire only when there's
|
||||
// actually a next attempt.
|
||||
if (attempt + 1 < maxAttempts) {
|
||||
onRetry?.(attempt + 1, maxAttempts, error);
|
||||
}
|
||||
return { retry: true };
|
||||
},
|
||||
},
|
||||
);
|
||||
} catch (err) {
|
||||
if (circuitTripped) {
|
||||
throw new Error(
|
||||
`${CIRCUIT_OPEN_TAG}: HuggingFace download circuit opened after ${CB_FAILURE_THRESHOLD} consecutive failures`,
|
||||
);
|
||||
}
|
||||
// All retries exhausted — rethrow the last network error so
|
||||
// isNetworkFetchError patterns in the calling code still match and
|
||||
// surface HF_ENDPOINT guidance.
|
||||
throw err;
|
||||
}
|
||||
|
||||
// All retries exhausted — throw the last network error so isNetworkFetchError
|
||||
// patterns in the calling code still match and surface HF_ENDPOINT guidance.
|
||||
throw lastError;
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,13 +3,22 @@
|
|||
*
|
||||
* Shared fetch+retry logic for OpenAI-compatible /v1/embeddings endpoints.
|
||||
* Imported by both the core embedder (batch) and MCP embedder (query).
|
||||
*
|
||||
* Network resilience is delegated to `resilientFetch` from
|
||||
* `gitnexus-shared` — bounded retries with exponential-backoff jitter,
|
||||
* `Retry-After` honored on 429, and an in-process circuit breaker that
|
||||
* fails fast on a flapping endpoint. Per-attempt timeout is enforced
|
||||
* via `AbortSignal.timeout` on the underlying fetch.
|
||||
*/
|
||||
|
||||
import { CircuitOpenError, ResilientFetchExhaustedError, resilientFetch } from 'gitnexus-shared';
|
||||
|
||||
const HTTP_TIMEOUT_MS = 30_000;
|
||||
const HTTP_MAX_RETRIES = 2;
|
||||
const HTTP_RETRY_BACKOFF_MS = 1_000;
|
||||
const HTTP_BATCH_SIZE = 64;
|
||||
const DEFAULT_DIMS = 384;
|
||||
const HTTP_BREAKER_KEY = 'embeddings-http';
|
||||
|
||||
interface HttpConfig {
|
||||
baseUrl: string;
|
||||
|
|
@ -90,46 +99,51 @@ const httpEmbedBatch = async (
|
|||
model: string,
|
||||
apiKey: string,
|
||||
batchIndex = 0,
|
||||
attempt = 0,
|
||||
): Promise<EmbeddingItem[]> => {
|
||||
let resp: Response;
|
||||
try {
|
||||
resp = await fetch(url, {
|
||||
method: 'POST',
|
||||
signal: AbortSignal.timeout(HTTP_TIMEOUT_MS),
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
resp = await resilientFetch(
|
||||
url,
|
||||
{
|
||||
method: 'POST',
|
||||
signal: AbortSignal.timeout(HTTP_TIMEOUT_MS),
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({ input: batch, model }),
|
||||
},
|
||||
body: JSON.stringify({ input: batch, model }),
|
||||
});
|
||||
{
|
||||
breakerKey: HTTP_BREAKER_KEY,
|
||||
retry: { maxAttempts: HTTP_MAX_RETRIES + 1, baseDelayMs: HTTP_RETRY_BACKOFF_MS },
|
||||
},
|
||||
);
|
||||
} catch (err) {
|
||||
// Timeouts should not be retried — the server is unresponsive.
|
||||
// AbortSignal.timeout() throws DOMException with name 'TimeoutError'.
|
||||
const isTimeout = err instanceof DOMException && err.name === 'TimeoutError';
|
||||
if (isTimeout) {
|
||||
if (err instanceof CircuitOpenError) {
|
||||
throw new Error(
|
||||
`Embedding endpoint circuit open (${safeUrl(url)}, batch ${batchIndex}): retry in ${Math.ceil(err.retryAfterMs / 1000)}s`,
|
||||
);
|
||||
}
|
||||
if (err instanceof DOMException && err.name === 'TimeoutError') {
|
||||
throw new Error(
|
||||
`Embedding request timed out after ${HTTP_TIMEOUT_MS}ms (${safeUrl(url)}, batch ${batchIndex})`,
|
||||
);
|
||||
}
|
||||
// DNS, connection errors — retry with backoff
|
||||
if (attempt < HTTP_MAX_RETRIES) {
|
||||
const delay = HTTP_RETRY_BACKOFF_MS * (attempt + 1);
|
||||
await new Promise((r) => setTimeout(r, delay));
|
||||
return httpEmbedBatch(url, batch, model, apiKey, batchIndex, attempt + 1);
|
||||
if (err instanceof ResilientFetchExhaustedError) {
|
||||
throw new Error(
|
||||
`Embedding endpoint returned ${err.response.status} (${safeUrl(url)}, batch ${batchIndex})`,
|
||||
);
|
||||
}
|
||||
const reason = err instanceof Error ? err.message : String(err);
|
||||
throw new Error(`Embedding request failed (${safeUrl(url)}, batch ${batchIndex}): ${reason}`);
|
||||
}
|
||||
|
||||
if (!resp.ok) {
|
||||
const status = resp.status;
|
||||
if ((status === 429 || status >= 500) && attempt < HTTP_MAX_RETRIES) {
|
||||
const delay = HTTP_RETRY_BACKOFF_MS * (attempt + 1);
|
||||
await new Promise((r) => setTimeout(r, delay));
|
||||
return httpEmbedBatch(url, batch, model, apiKey, batchIndex, attempt + 1);
|
||||
}
|
||||
throw new Error(`Embedding endpoint returned ${status} (${safeUrl(url)}, batch ${batchIndex})`);
|
||||
// resilientFetch already retried 5xx/429; any non-OK response here is
|
||||
// a terminal client error (4xx other than 429).
|
||||
throw new Error(
|
||||
`Embedding endpoint returned ${resp.status} (${safeUrl(url)}, batch ${batchIndex})`,
|
||||
);
|
||||
}
|
||||
|
||||
const data = (await resp.json()) as { data: EmbeddingItem[] };
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
import { logger } from '../logger.js';
|
||||
import { CircuitOpenError, ResilientFetchExhaustedError, resilientFetch } from 'gitnexus-shared';
|
||||
/**
|
||||
* LLM Client for Wiki Generation
|
||||
*
|
||||
|
|
@ -170,86 +171,85 @@ export async function callLLM(
|
|||
? { 'api-key': config.apiKey }
|
||||
: { Authorization: `Bearer ${config.apiKey}` };
|
||||
|
||||
const MAX_RETRIES = 3;
|
||||
let lastError: Error | null = null;
|
||||
|
||||
for (let attempt = 0; attempt < MAX_RETRIES; attempt++) {
|
||||
try {
|
||||
const response = await fetch(url, {
|
||||
// Network resilience (bounded retries with exponential-backoff jitter,
|
||||
// 5xx + 429 + Retry-After handling, in-process circuit breaker on the
|
||||
// LLM endpoint) is delegated to resilientFetch. Provider-specific
|
||||
// error parsing (Azure content filter, empty-content checks) stays
|
||||
// here since it requires response-body inspection.
|
||||
let response: Response;
|
||||
try {
|
||||
response = await resilientFetch(
|
||||
url,
|
||||
{
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
...authHeaders,
|
||||
},
|
||||
body: JSON.stringify(body),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text().catch(() => 'unknown error');
|
||||
|
||||
// Azure content filter — surface a clear message instead of a generic API error
|
||||
if (
|
||||
azure &&
|
||||
response.status === 400 &&
|
||||
(errorText.includes('content_filter') ||
|
||||
errorText.includes('ResponsibleAIPolicyViolation'))
|
||||
) {
|
||||
throw new Error(
|
||||
`Azure content filter blocked this request. The prompt triggered content policy. Details: ${errorText.slice(0, 300)}`,
|
||||
);
|
||||
}
|
||||
|
||||
// Rate limit — wait with exponential backoff and retry
|
||||
if (response.status === 429 && attempt < MAX_RETRIES - 1) {
|
||||
const retryAfter = parseInt(response.headers.get('retry-after') || '0', 10);
|
||||
const delay = retryAfter > 0 ? retryAfter * 1000 : 2 ** attempt * 3000;
|
||||
await sleep(delay);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Server error — retry with backoff
|
||||
if (response.status >= 500 && attempt < MAX_RETRIES - 1) {
|
||||
await sleep((attempt + 1) * 2000);
|
||||
continue;
|
||||
}
|
||||
|
||||
throw new Error(`LLM API error (${response.status}): ${errorText.slice(0, 500)}`);
|
||||
}
|
||||
|
||||
// Streaming path
|
||||
if (useStream && response.body) {
|
||||
return await readSSEStream(response.body, options!.onChunk!);
|
||||
}
|
||||
|
||||
// Non-streaming path
|
||||
const json = (await response.json()) as any;
|
||||
const choice = json.choices?.[0];
|
||||
if (!choice?.message?.content) {
|
||||
throw new Error('LLM returned empty response');
|
||||
}
|
||||
|
||||
return {
|
||||
content: choice.message.content,
|
||||
promptTokens: json.usage?.prompt_tokens,
|
||||
completionTokens: json.usage?.completion_tokens,
|
||||
};
|
||||
} catch (err: any) {
|
||||
lastError = err;
|
||||
|
||||
// Network error — retry with backoff
|
||||
if (
|
||||
attempt < MAX_RETRIES - 1 &&
|
||||
(err.code === 'ECONNREFUSED' || err.code === 'ETIMEDOUT' || err.message?.includes('fetch'))
|
||||
) {
|
||||
await sleep((attempt + 1) * 3000);
|
||||
continue;
|
||||
}
|
||||
|
||||
throw err;
|
||||
// Per-attempt timeout. Without this each retry can hang
|
||||
// indefinitely on a frozen TCP connection — the per-call
|
||||
// signal is the only timeout `resilientFetch` honors;
|
||||
// `capDelayMs` only bounds the *backoff* between attempts.
|
||||
// 60s matches typical LLM completion budgets.
|
||||
signal: AbortSignal.timeout(60_000),
|
||||
},
|
||||
{
|
||||
breakerKey: `wiki-llm-${new URL(url).host}`,
|
||||
retry: { maxAttempts: 3, baseDelayMs: 2_000, capDelayMs: 30_000 },
|
||||
},
|
||||
);
|
||||
} catch (err) {
|
||||
if (err instanceof CircuitOpenError) {
|
||||
throw new Error(
|
||||
`LLM endpoint circuit open: retry in ${Math.ceil(err.retryAfterMs / 1000)}s. ${err.message}`,
|
||||
);
|
||||
}
|
||||
if (err instanceof ResilientFetchExhaustedError) {
|
||||
const errorText = await err.response.text().catch(() => 'unknown error');
|
||||
throw new Error(
|
||||
`LLM API error (${err.response.status} after retries): ${errorText.slice(0, 500)}`,
|
||||
);
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
|
||||
throw lastError || new Error('LLM call failed after retries');
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text().catch(() => 'unknown error');
|
||||
|
||||
// Azure content filter — surface a clear message instead of a generic API error.
|
||||
if (
|
||||
azure &&
|
||||
response.status === 400 &&
|
||||
(errorText.includes('content_filter') || errorText.includes('ResponsibleAIPolicyViolation'))
|
||||
) {
|
||||
throw new Error(
|
||||
`Azure content filter blocked this request. The prompt triggered content policy. Details: ${errorText.slice(0, 300)}`,
|
||||
);
|
||||
}
|
||||
|
||||
// Any other non-OK response here is a terminal 4xx — resilientFetch
|
||||
// already retried 5xx/429 to exhaustion and would have thrown above.
|
||||
throw new Error(`LLM API error (${response.status}): ${errorText.slice(0, 500)}`);
|
||||
}
|
||||
|
||||
// Streaming path
|
||||
if (useStream && response.body) {
|
||||
return await readSSEStream(response.body, options!.onChunk!);
|
||||
}
|
||||
|
||||
// Non-streaming path
|
||||
const json = (await response.json()) as any;
|
||||
const choice = json.choices?.[0];
|
||||
if (!choice?.message?.content) {
|
||||
throw new Error('LLM returned empty response');
|
||||
}
|
||||
|
||||
return {
|
||||
content: choice.message.content,
|
||||
promptTokens: json.usage?.prompt_tokens,
|
||||
completionTokens: json.usage?.completion_tokens,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -312,7 +312,3 @@ async function readSSEStream(
|
|||
|
||||
return { content };
|
||||
}
|
||||
|
||||
function sleep(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,12 +1,12 @@
|
|||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import os from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { CircuitBreaker } from 'gitnexus-shared';
|
||||
import {
|
||||
applyHfEnvOverrides,
|
||||
isNetworkFetchError,
|
||||
isHfDownloadFailure,
|
||||
isHfCircuitOpenError,
|
||||
HfDownloadCircuitBreaker,
|
||||
withDownloadTimeout,
|
||||
withHfDownloadRetry,
|
||||
CIRCUIT_OPEN_TAG,
|
||||
|
|
@ -154,85 +154,13 @@ describe('isHfDownloadFailure', () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe('HfDownloadCircuitBreaker', () => {
|
||||
it('starts in closed state', () => {
|
||||
const cb = new HfDownloadCircuitBreaker();
|
||||
expect(cb.isOpen()).toBe(false);
|
||||
expect(cb.state).toBe('closed');
|
||||
});
|
||||
|
||||
it('opens after reaching the failure threshold', () => {
|
||||
const cb = new HfDownloadCircuitBreaker(3);
|
||||
cb.recordFailure();
|
||||
cb.recordFailure();
|
||||
expect(cb.isOpen()).toBe(false);
|
||||
cb.recordFailure(); // threshold reached
|
||||
expect(cb.isOpen()).toBe(true);
|
||||
expect(cb.state).toBe('open');
|
||||
});
|
||||
|
||||
it('closes on recordSuccess after being open', () => {
|
||||
const cb = new HfDownloadCircuitBreaker(1);
|
||||
cb.recordFailure();
|
||||
expect(cb.isOpen()).toBe(true);
|
||||
cb.recordSuccess();
|
||||
expect(cb.isOpen()).toBe(false);
|
||||
expect(cb.state).toBe('closed');
|
||||
});
|
||||
|
||||
it('transitions to half-open after the reset timeout', () => {
|
||||
vi.useFakeTimers();
|
||||
try {
|
||||
const cb = new HfDownloadCircuitBreaker(1, 100 /* 100ms */);
|
||||
cb.recordFailure();
|
||||
expect(cb.isOpen()).toBe(true);
|
||||
vi.advanceTimersByTime(200);
|
||||
expect(cb.isOpen()).toBe(false);
|
||||
expect(cb.state).toBe('half-open');
|
||||
} finally {
|
||||
vi.useRealTimers();
|
||||
}
|
||||
});
|
||||
|
||||
it('reset() restores closed state', () => {
|
||||
const cb = new HfDownloadCircuitBreaker(1);
|
||||
cb.recordFailure();
|
||||
expect(cb.isOpen()).toBe(true);
|
||||
cb.reset();
|
||||
expect(cb.isOpen()).toBe(false);
|
||||
expect(cb.state).toBe('closed');
|
||||
});
|
||||
|
||||
it('re-opens when a failure is recorded in half-open state', () => {
|
||||
vi.useFakeTimers();
|
||||
try {
|
||||
const cb = new HfDownloadCircuitBreaker(1, 100 /* 100ms */);
|
||||
cb.recordFailure(); // opens the circuit
|
||||
vi.advanceTimersByTime(200); // advance past reset timeout
|
||||
expect(cb.state).toBe('half-open'); // getter transitions _state to half-open
|
||||
cb.recordFailure(); // failure in half-open → re-opens
|
||||
expect(cb.isOpen()).toBe(true);
|
||||
expect(cb.state).toBe('open');
|
||||
} finally {
|
||||
vi.useRealTimers();
|
||||
}
|
||||
});
|
||||
|
||||
it('closes the circuit when success is recorded in half-open state', () => {
|
||||
vi.useFakeTimers();
|
||||
try {
|
||||
const cb = new HfDownloadCircuitBreaker(1, 100 /* 100ms */);
|
||||
cb.recordFailure(); // opens the circuit
|
||||
vi.advanceTimersByTime(200); // advance past reset timeout
|
||||
expect(cb.state).toBe('half-open');
|
||||
cb.recordSuccess(); // success in half-open → closes
|
||||
expect(cb.isOpen()).toBe(false);
|
||||
expect(cb.state).toBe('closed');
|
||||
} finally {
|
||||
vi.useRealTimers();
|
||||
}
|
||||
});
|
||||
});
|
||||
// CircuitBreaker state-machine tests live in
|
||||
// `gitnexus/test/unit/integrations/circuit-breaker.test.ts` — that suite
|
||||
// already covers the closed/open/half-open transitions, recordSuccess/
|
||||
// recordFailure semantics, half-open probe gating, and configurable
|
||||
// thresholds. No need to duplicate here; this file's remaining tests
|
||||
// focus on HF-specific composition (withHfDownloadRetry, env-var
|
||||
// overrides, error classification).
|
||||
|
||||
describe('withDownloadTimeout', () => {
|
||||
it('resolves when fn completes before the timeout', async () => {
|
||||
|
|
@ -262,7 +190,7 @@ describe('withDownloadTimeout', () => {
|
|||
describe('withHfDownloadRetry', () => {
|
||||
it('returns the result on first success', async () => {
|
||||
const fn = vi.fn().mockResolvedValue('ok');
|
||||
const cb = new HfDownloadCircuitBreaker();
|
||||
const cb = new CircuitBreaker();
|
||||
const result = await withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 });
|
||||
expect(result).toBe('ok');
|
||||
expect(fn).toHaveBeenCalledTimes(1);
|
||||
|
|
@ -270,7 +198,7 @@ describe('withHfDownloadRetry', () => {
|
|||
|
||||
it('retries on network errors and succeeds on second attempt', async () => {
|
||||
const fn = vi.fn().mockRejectedValueOnce(new Error('fetch failed')).mockResolvedValue('ok');
|
||||
const cb = new HfDownloadCircuitBreaker();
|
||||
const cb = new CircuitBreaker();
|
||||
const result = await withHfDownloadRetry(fn, {
|
||||
circuit: cb,
|
||||
maxAttempts: 3,
|
||||
|
|
@ -282,7 +210,7 @@ describe('withHfDownloadRetry', () => {
|
|||
|
||||
it('throws the last network error after all attempts are exhausted', async () => {
|
||||
const fn = vi.fn().mockRejectedValue(new Error('ECONNREFUSED 127.0.0.1:443'));
|
||||
const cb = new HfDownloadCircuitBreaker(99 /* high threshold */);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99 });
|
||||
await expect(
|
||||
withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 3, baseDelayMs: 0 }),
|
||||
).rejects.toThrow('ECONNREFUSED');
|
||||
|
|
@ -291,7 +219,7 @@ describe('withHfDownloadRetry', () => {
|
|||
|
||||
it('does not retry non-network errors', async () => {
|
||||
const fn = vi.fn().mockRejectedValue(new Error('Failed to initialize CUDA backend'));
|
||||
const cb = new HfDownloadCircuitBreaker();
|
||||
const cb = new CircuitBreaker();
|
||||
await expect(
|
||||
withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 3, baseDelayMs: 0 }),
|
||||
).rejects.toThrow('Failed to initialize CUDA backend');
|
||||
|
|
@ -300,7 +228,7 @@ describe('withHfDownloadRetry', () => {
|
|||
|
||||
it('fails immediately when the circuit is already open', async () => {
|
||||
const fn = vi.fn().mockResolvedValue('ok');
|
||||
const cb = new HfDownloadCircuitBreaker(1);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 1 });
|
||||
cb.recordFailure(); // open the circuit
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb })).rejects.toThrow(CIRCUIT_OPEN_TAG);
|
||||
expect(fn).not.toHaveBeenCalled();
|
||||
|
|
@ -308,12 +236,12 @@ describe('withHfDownloadRetry', () => {
|
|||
|
||||
it('opens the circuit after failureThreshold failures and throws a circuit-open error', async () => {
|
||||
const fn = vi.fn().mockRejectedValue(new Error('ENOTFOUND huggingface.co'));
|
||||
const cb = new HfDownloadCircuitBreaker(2 /* threshold */, 60_000);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 2, cooldownMs: 60_000 });
|
||||
// First call: 2 attempts, threshold=2 → circuit opens on 2nd failure
|
||||
await expect(
|
||||
withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 2, baseDelayMs: 0 }),
|
||||
).rejects.toThrow(CIRCUIT_OPEN_TAG);
|
||||
expect(cb.isOpen()).toBe(true);
|
||||
expect(cb.getState()).toBe('open');
|
||||
});
|
||||
|
||||
it('calls onRetry with correct arguments on each retry', async () => {
|
||||
|
|
@ -322,7 +250,7 @@ describe('withHfDownloadRetry', () => {
|
|||
.mockRejectedValueOnce(new Error('fetch failed'))
|
||||
.mockRejectedValueOnce(new Error('fetch failed'))
|
||||
.mockResolvedValue('ok');
|
||||
const cb = new HfDownloadCircuitBreaker(99);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99 });
|
||||
const onRetry = vi.fn();
|
||||
await withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 3, baseDelayMs: 0, onRetry });
|
||||
expect(onRetry).toHaveBeenCalledTimes(2);
|
||||
|
|
@ -342,11 +270,11 @@ describe('withHfDownloadRetry', () => {
|
|||
|
||||
it('resets the circuit on success', async () => {
|
||||
const fn = vi.fn().mockResolvedValue('value');
|
||||
const cb = new HfDownloadCircuitBreaker(5);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 5 });
|
||||
cb.recordFailure();
|
||||
cb.recordFailure(); // 2 failures, circuit still closed
|
||||
await withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 });
|
||||
expect(cb.state).toBe('closed');
|
||||
expect(cb.getState()).toBe('closed');
|
||||
});
|
||||
});
|
||||
|
||||
|
|
@ -371,7 +299,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS=1 gives exactly 1 attempt', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '1';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('ECONNREFUSED 127.0.0.1:443'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999 /* high threshold */);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'ECONNREFUSED',
|
||||
);
|
||||
|
|
@ -381,7 +309,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS=2 gives exactly 2 attempts', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '2';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('ENOTFOUND huggingface.co'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'ENOTFOUND',
|
||||
);
|
||||
|
|
@ -391,7 +319,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS=abc falls back to the built-in default', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = 'abc';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('fetch failed'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'fetch failed',
|
||||
);
|
||||
|
|
@ -401,7 +329,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS=0 falls back to the built-in default', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '0';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('fetch failed'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'fetch failed',
|
||||
);
|
||||
|
|
@ -411,7 +339,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS=-1 falls back to the built-in default', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '-1';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('fetch failed'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'fetch failed',
|
||||
);
|
||||
|
|
@ -421,7 +349,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS is clamped to HF_MAX_ATTEMPTS_CAP', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '9999';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('fetch failed'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999 /* very high threshold */);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'fetch failed',
|
||||
);
|
||||
|
|
@ -431,7 +359,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('HF_MAX_ATTEMPTS=2.9 is floored to 2', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '2.9';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('fetch failed'));
|
||||
const cb = new HfDownloadCircuitBreaker(99_999);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99_999 });
|
||||
await expect(withHfDownloadRetry(fn, { circuit: cb, baseDelayMs: 0 })).rejects.toThrow(
|
||||
'fetch failed',
|
||||
);
|
||||
|
|
@ -443,7 +371,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
try {
|
||||
process.env.HF_DOWNLOAD_TIMEOUT_MS = '50';
|
||||
const neverResolves = () => new Promise<never>(() => {});
|
||||
const cb = new HfDownloadCircuitBreaker(99);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99 });
|
||||
const promise = withHfDownloadRetry(neverResolves, { circuit: cb, maxAttempts: 1 });
|
||||
vi.advanceTimersByTime(100);
|
||||
await expect(promise).rejects.toThrow('ETIMEDOUT');
|
||||
|
|
@ -458,7 +386,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
// we just verify that the env var rejection causes options.timeoutMs to be
|
||||
// the default constant (not -1) by confirming the resolved value is used.
|
||||
const fn = vi.fn().mockResolvedValue('ok');
|
||||
const cb = new HfDownloadCircuitBreaker(99);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99 });
|
||||
// Provide explicit timeoutMs to avoid the default 5-minute wait
|
||||
const result = await withHfDownloadRetry(fn, { circuit: cb, timeoutMs: 100 });
|
||||
expect(result).toBe('ok');
|
||||
|
|
@ -470,7 +398,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
// Set an env value exceeding the 30-minute cap
|
||||
process.env.HF_DOWNLOAD_TIMEOUT_MS = String(HF_MAX_TIMEOUT_MS + 60_000);
|
||||
const neverResolves = () => new Promise<never>(() => {});
|
||||
const cb = new HfDownloadCircuitBreaker(99);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99 });
|
||||
const promise = withHfDownloadRetry(neverResolves, { circuit: cb, maxAttempts: 1 });
|
||||
// Advance just past the 30-minute cap
|
||||
vi.advanceTimersByTime(HF_MAX_TIMEOUT_MS + 1);
|
||||
|
|
@ -483,7 +411,7 @@ describe('withHfDownloadRetry env overrides', () => {
|
|||
it('explicit options override env vars', async () => {
|
||||
process.env.HF_MAX_ATTEMPTS = '5';
|
||||
const fn = vi.fn().mockRejectedValue(new Error('fetch failed'));
|
||||
const cb = new HfDownloadCircuitBreaker(99);
|
||||
const cb = new CircuitBreaker({ failureThreshold: 99 });
|
||||
// explicit maxAttempts: 2 must win over HF_MAX_ATTEMPTS=5
|
||||
await expect(
|
||||
withHfDownloadRetry(fn, { circuit: cb, maxAttempts: 2, baseDelayMs: 0 }),
|
||||
|
|
|
|||
395
gitnexus/test/unit/integrations/circuit-breaker.test.ts
Normal file
395
gitnexus/test/unit/integrations/circuit-breaker.test.ts
Normal file
|
|
@ -0,0 +1,395 @@
|
|||
import { describe, it, expect, beforeEach } from 'vitest';
|
||||
import { CircuitBreaker, CircuitOpenError, getBreaker } from 'gitnexus-shared';
|
||||
import { __resetBreakerRegistry__ } from 'gitnexus-shared/test-helpers';
|
||||
|
||||
describe('CircuitBreaker', () => {
|
||||
beforeEach(() => __resetBreakerRegistry__());
|
||||
|
||||
function makeClock(start = 1_700_000_000_000) {
|
||||
let t = start;
|
||||
return {
|
||||
now: () => t,
|
||||
advance: (ms: number) => {
|
||||
t += ms;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
it('runs through check/recordSuccess in closed state', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 3, cooldownMs: 30_000, now: clock.now });
|
||||
expect(b.getState()).toBe('closed');
|
||||
b.check(); // does not throw
|
||||
b.recordSuccess();
|
||||
expect(b.getState()).toBe('closed');
|
||||
expect(b.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('stays closed below the failure threshold', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 3, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
b.recordFailure();
|
||||
expect(b.getState()).toBe('closed');
|
||||
expect(b.getConsecutiveFailures()).toBe(2);
|
||||
});
|
||||
|
||||
it('opens after failureThreshold consecutive failures and check throws', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 3, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
b.recordFailure();
|
||||
b.recordFailure();
|
||||
expect(b.getState()).toBe('open');
|
||||
expect(() => b.check()).toThrow(CircuitOpenError);
|
||||
});
|
||||
|
||||
it('CircuitOpenError.retryAfterMs decreases as time advances', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught?.retryAfterMs).toBe(30_000);
|
||||
|
||||
clock.advance(10_000);
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught?.retryAfterMs).toBe(20_000);
|
||||
});
|
||||
|
||||
it('transitions Open -> Half-Open after cooldown elapses (via check)', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
expect(b.getState()).toBe('open');
|
||||
clock.advance(31_000);
|
||||
b.check(); // should not throw
|
||||
// After check, internal state is half-open (next call probes).
|
||||
expect(b.getConsecutiveFailures()).toBe(1); // unchanged until next outcome
|
||||
});
|
||||
|
||||
it('half-open + recordSuccess -> closed and counter reset', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
b.check();
|
||||
b.recordSuccess();
|
||||
expect(b.getState()).toBe('closed');
|
||||
expect(b.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('half-open + recordFailure -> open with fresh openedAt', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
const firstOpen = clock.now();
|
||||
clock.advance(31_000); // cooldown expired
|
||||
b.check(); // half-open
|
||||
b.recordFailure();
|
||||
// Open with fresh timestamp — full cooldown again.
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught).toBeInstanceOf(CircuitOpenError);
|
||||
expect(caught?.retryAfterMs).toBe(30_000);
|
||||
// Sanity: not the original openedAt (would be negative remaining).
|
||||
expect(clock.now()).toBeGreaterThan(firstOpen);
|
||||
});
|
||||
|
||||
it('recordSuccess from closed state with prior partial failures resets counter', () => {
|
||||
const b = new CircuitBreaker({ failureThreshold: 5 });
|
||||
b.recordFailure();
|
||||
b.recordFailure();
|
||||
expect(b.getConsecutiveFailures()).toBe(2);
|
||||
b.recordSuccess();
|
||||
expect(b.getConsecutiveFailures()).toBe(0);
|
||||
expect(b.getState()).toBe('closed');
|
||||
});
|
||||
|
||||
describe('recordNeutral (U1)', () => {
|
||||
it('is a no-op from closed state with zero prior failures', () => {
|
||||
const b = new CircuitBreaker({ failureThreshold: 3 });
|
||||
b.recordNeutral();
|
||||
expect(b.getState()).toBe('closed');
|
||||
expect(b.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('preserves partial-failure progress (does not reset counter)', () => {
|
||||
const b = new CircuitBreaker({ failureThreshold: 3 });
|
||||
b.recordFailure();
|
||||
b.recordFailure();
|
||||
b.recordNeutral();
|
||||
expect(b.getConsecutiveFailures()).toBe(2);
|
||||
expect(b.getState()).toBe('closed');
|
||||
// Real third failure still trips the breaker — neutrals didn't
|
||||
// erase the running count toward the threshold.
|
||||
b.recordFailure();
|
||||
expect(b.getState()).toBe('open');
|
||||
});
|
||||
|
||||
it('does not reset openedAt or transition out of open state', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
expect(b.getState()).toBe('open');
|
||||
b.recordNeutral();
|
||||
// Still open; cooldown clock unchanged.
|
||||
expect(() => b.check()).toThrow(CircuitOpenError);
|
||||
});
|
||||
|
||||
it('leaves half-open state alone (next true outcome decides)', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
b.check(); // half-open
|
||||
b.recordNeutral();
|
||||
// Still half-open; a subsequent recordFailure flips to open.
|
||||
b.recordFailure();
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught).toBeInstanceOf(CircuitOpenError);
|
||||
});
|
||||
|
||||
it('integration: 2 failures + 5 neutrals + 1 failure → opens on third real failure', () => {
|
||||
const b = new CircuitBreaker({ failureThreshold: 3 });
|
||||
b.recordFailure();
|
||||
b.recordFailure();
|
||||
for (let i = 0; i < 5; i++) b.recordNeutral();
|
||||
expect(b.getConsecutiveFailures()).toBe(2);
|
||||
expect(b.getState()).toBe('closed');
|
||||
b.recordFailure();
|
||||
expect(b.getState()).toBe('open');
|
||||
});
|
||||
});
|
||||
|
||||
describe('half-open probe permit gate (U1)', () => {
|
||||
it('admits exactly one caller after cooldown; subsequent check() throws halfOpenRetryAfterMs', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({
|
||||
failureThreshold: 1,
|
||||
cooldownMs: 30_000,
|
||||
halfOpenRetryAfterMs: 1_000,
|
||||
now: clock.now,
|
||||
});
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
|
||||
// Caller A: gets the probe permit.
|
||||
b.check();
|
||||
expect(b.isProbeInFlight()).toBe(true);
|
||||
|
||||
// Caller B: blocked.
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught).toBeInstanceOf(CircuitOpenError);
|
||||
expect(caught?.retryAfterMs).toBe(1_000);
|
||||
|
||||
// Caller A's recordSuccess clears the breaker.
|
||||
b.recordSuccess();
|
||||
expect(b.isProbeInFlight()).toBe(false);
|
||||
expect(b.getState()).toBe('closed');
|
||||
|
||||
// Caller C: succeeds in closed state.
|
||||
b.check();
|
||||
expect(b.getState()).toBe('closed');
|
||||
});
|
||||
|
||||
it('recordFailure on probe re-opens with fresh cooldown (NOT halfOpenRetryAfterMs)', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({
|
||||
failureThreshold: 1,
|
||||
cooldownMs: 30_000,
|
||||
halfOpenRetryAfterMs: 1_000,
|
||||
now: clock.now,
|
||||
});
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
|
||||
b.check(); // A: probe
|
||||
expect(() => b.check()).toThrow(CircuitOpenError); // B: blocked
|
||||
|
||||
b.recordFailure(); // A reports failure → reopens with fresh openedAt
|
||||
|
||||
// C: should see the fresh cooldown remaining, not the probe-in-flight 1s default.
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught).toBeInstanceOf(CircuitOpenError);
|
||||
// Fresh openedAt = current clock; cooldown is 30s; retryAfter ≈ 30s.
|
||||
expect(caught?.retryAfterMs).toBe(30_000);
|
||||
});
|
||||
|
||||
it('recordNeutral releases the probe permit but leaves state half-open', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
|
||||
b.check(); // A: probe
|
||||
expect(b.isProbeInFlight()).toBe(true);
|
||||
|
||||
b.recordNeutral(); // A: neutral — permit released, state untouched
|
||||
expect(b.isProbeInFlight()).toBe(false);
|
||||
expect(b.getState()).toBe('half-open');
|
||||
|
||||
// B: succeeds (becomes the new probe), no longer blocked.
|
||||
b.check();
|
||||
expect(b.isProbeInFlight()).toBe(true);
|
||||
|
||||
// B's recordSuccess clears the breaker.
|
||||
b.recordSuccess();
|
||||
expect(b.getState()).toBe('closed');
|
||||
});
|
||||
|
||||
it('three sequential probes via neutrals: A → A.neutral → B → B.neutral → C', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
const initialFailures = b.getConsecutiveFailures();
|
||||
clock.advance(31_000);
|
||||
|
||||
for (let i = 0; i < 3; i++) {
|
||||
b.check();
|
||||
b.recordNeutral();
|
||||
}
|
||||
// Counter unchanged; state still half-open; permit released.
|
||||
expect(b.getConsecutiveFailures()).toBe(initialFailures);
|
||||
expect(b.getState()).toBe('half-open');
|
||||
expect(b.isProbeInFlight()).toBe(false);
|
||||
});
|
||||
|
||||
it('5 same-tick sequential callers: exactly one passes, the other 4 throw', () => {
|
||||
// `check()` is synchronous — these calls execute on a single
|
||||
// microtask in declaration order. The first mutates probeInFlight
|
||||
// = true; the next four observe the mutation and throw. This
|
||||
// tests mutation ordering, not true concurrency (the actual
|
||||
// interleaved-async-microtask scenario lives in U2).
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
|
||||
const results: Array<'pass' | 'throw'> = [];
|
||||
for (let i = 0; i < 5; i++) {
|
||||
try {
|
||||
b.check();
|
||||
results.push('pass');
|
||||
} catch {
|
||||
results.push('throw');
|
||||
}
|
||||
}
|
||||
expect(results.filter((r) => r === 'pass').length).toBe(1);
|
||||
expect(results.filter((r) => r === 'throw').length).toBe(4);
|
||||
});
|
||||
|
||||
it('probe permit consumed; clock advances another full cooldown without record*; still throws', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
|
||||
b.check(); // probe permit consumed
|
||||
clock.advance(60_000); // another full cooldown elapses, no record*
|
||||
|
||||
// Half-open semantics: wait for an outcome, not a timer. The
|
||||
// permit-consumed state doesn't auto-resolve on time.
|
||||
expect(() => b.check()).toThrow(CircuitOpenError);
|
||||
});
|
||||
|
||||
it('halfOpenRetryAfterMs default is 1000 when not configured', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
b.check();
|
||||
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught?.retryAfterMs).toBe(1_000);
|
||||
});
|
||||
|
||||
it('halfOpenRetryAfterMs is configurable for long-running protected ops', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({
|
||||
failureThreshold: 1,
|
||||
cooldownMs: 30_000,
|
||||
halfOpenRetryAfterMs: 10_000, // LLM-streaming-friendly
|
||||
now: clock.now,
|
||||
});
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
b.check();
|
||||
|
||||
let caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
b.check();
|
||||
} catch (err) {
|
||||
caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(caught?.retryAfterMs).toBe(10_000);
|
||||
});
|
||||
|
||||
it('getState() is a pure read — does not consume the probe permit', () => {
|
||||
const clock = makeClock();
|
||||
const b = new CircuitBreaker({ failureThreshold: 1, cooldownMs: 30_000, now: clock.now });
|
||||
b.recordFailure();
|
||||
clock.advance(31_000);
|
||||
|
||||
// Test calls getState() to inspect — must not consume the permit.
|
||||
expect(b.getState()).toBe('half-open');
|
||||
expect(b.isProbeInFlight()).toBe(false);
|
||||
// First check() still gets the permit.
|
||||
b.check();
|
||||
expect(b.isProbeInFlight()).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('getBreaker registry', () => {
|
||||
it('returns the same instance for the same key', () => {
|
||||
const a = getBreaker('endpoint-a');
|
||||
const b = getBreaker('endpoint-a');
|
||||
expect(a).toBe(b);
|
||||
});
|
||||
|
||||
it('returns different instances for different keys', () => {
|
||||
const a = getBreaker('endpoint-a');
|
||||
const b = getBreaker('endpoint-b');
|
||||
expect(a).not.toBe(b);
|
||||
});
|
||||
|
||||
it('__resetBreakerRegistry__ clears all instances', () => {
|
||||
const a = getBreaker('endpoint-a');
|
||||
__resetBreakerRegistry__();
|
||||
const a2 = getBreaker('endpoint-a');
|
||||
expect(a2).not.toBe(a);
|
||||
});
|
||||
});
|
||||
});
|
||||
558
gitnexus/test/unit/integrations/resilient-fetch.test.ts
Normal file
558
gitnexus/test/unit/integrations/resilient-fetch.test.ts
Normal file
|
|
@ -0,0 +1,558 @@
|
|||
import { describe, it, expect, beforeEach, vi } from 'vitest';
|
||||
import {
|
||||
CircuitBreaker,
|
||||
CircuitOpenError,
|
||||
parseRetryAfter,
|
||||
resilientFetch,
|
||||
ResilientFetchExhaustedError,
|
||||
RETRY_AFTER_CAP_MS,
|
||||
} from 'gitnexus-shared';
|
||||
import { __resetBreakerRegistry__, classifyOutcome } from 'gitnexus-shared/test-helpers';
|
||||
|
||||
describe('parseRetryAfter', () => {
|
||||
it('parses delta-seconds form', () => {
|
||||
expect(parseRetryAfter('30')).toBe(30_000);
|
||||
expect(parseRetryAfter('0')).toBe(0);
|
||||
});
|
||||
it('returns null on negative or non-numeric garbage', () => {
|
||||
expect(parseRetryAfter(null)).toBeNull();
|
||||
expect(parseRetryAfter('')).toBeNull();
|
||||
expect(parseRetryAfter(' ')).toBeNull();
|
||||
expect(parseRetryAfter('not-a-number')).toBeNull();
|
||||
});
|
||||
it('parses HTTP-date form against an injected clock', () => {
|
||||
const now = () => Date.parse('Wed, 21 Oct 2025 07:28:00 GMT');
|
||||
expect(parseRetryAfter('Wed, 21 Oct 2025 07:28:30 GMT', now)).toBe(30_000);
|
||||
});
|
||||
it('returns 0 (not negative) on past HTTP-date', () => {
|
||||
const now = () => Date.parse('Wed, 21 Oct 2025 08:00:00 GMT');
|
||||
expect(parseRetryAfter('Wed, 21 Oct 2025 07:28:00 GMT', now)).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('classifyOutcome', () => {
|
||||
const now = () => 1_700_000_000_000;
|
||||
|
||||
it('classifies 2xx as success', () => {
|
||||
const resp = new Response(null, { status: 204 });
|
||||
const out = classifyOutcome({ kind: 'response', resp }, now);
|
||||
expect(out.kind).toBe('success');
|
||||
});
|
||||
it('classifies 5xx as retryable-status without afterMs', () => {
|
||||
const resp = new Response(null, { status: 503 });
|
||||
const out = classifyOutcome({ kind: 'response', resp }, now);
|
||||
expect(out.kind).toBe('retryable-status');
|
||||
if (out.kind === 'retryable-status') expect(out.afterMs).toBeUndefined();
|
||||
});
|
||||
it('classifies 429 with Retry-After (capped) as retryable-status', () => {
|
||||
const resp = new Response(null, { status: 429, headers: { 'Retry-After': '99999' } });
|
||||
const out = classifyOutcome({ kind: 'response', resp }, now);
|
||||
expect(out.kind).toBe('retryable-status');
|
||||
if (out.kind === 'retryable-status') expect(out.afterMs).toBe(RETRY_AFTER_CAP_MS);
|
||||
});
|
||||
it('classifies 429 from a header-less fetch mock without throwing', () => {
|
||||
// Tests sometimes stub `fetch` with a plain `{ ok, status }` object
|
||||
// (e.g. http-embedder.test.ts). Real `Response` always carries
|
||||
// `Headers`, but the helper must not crash when the stub does not.
|
||||
// Falls through to exponential-backoff retry like a 429 with no
|
||||
// Retry-After header.
|
||||
const resp = { ok: false, status: 429 } as unknown as Response;
|
||||
const out = classifyOutcome({ kind: 'response', resp }, now);
|
||||
expect(out.kind).toBe('retryable-status');
|
||||
if (out.kind === 'retryable-status') expect(out.afterMs).toBeUndefined();
|
||||
});
|
||||
it('classifies 401/403/404/422 as terminal-client', () => {
|
||||
for (const status of [401, 403, 404, 422, 400]) {
|
||||
const resp = new Response(null, { status });
|
||||
const out = classifyOutcome({ kind: 'response', resp }, now);
|
||||
expect(out.kind).toBe('terminal-client');
|
||||
}
|
||||
});
|
||||
it('classifies TimeoutError as terminal-network', () => {
|
||||
const err = new DOMException('aborted', 'TimeoutError');
|
||||
const out = classifyOutcome({ kind: 'error', err }, now);
|
||||
expect(out.kind).toBe('terminal-network');
|
||||
});
|
||||
it('classifies generic network throw as retryable-network', () => {
|
||||
const err = new TypeError('fetch failed');
|
||||
const out = classifyOutcome({ kind: 'error', err }, now);
|
||||
expect(out.kind).toBe('retryable-network');
|
||||
});
|
||||
});
|
||||
|
||||
describe('resilientFetch', () => {
|
||||
const URL_STR = 'https://example.test/api/dispatch';
|
||||
|
||||
beforeEach(() => __resetBreakerRegistry__());
|
||||
|
||||
function jsonResp(status: number, headers?: Record<string, string>): Response {
|
||||
return new Response(null, { status, headers });
|
||||
}
|
||||
|
||||
function makeBreaker(opts: Partial<ConstructorParameters<typeof CircuitBreaker>[0]> = {}) {
|
||||
let t = 1_700_000_000_000;
|
||||
const breaker = new CircuitBreaker({
|
||||
failureThreshold: 3,
|
||||
cooldownMs: 30_000,
|
||||
key: 'test',
|
||||
now: () => t,
|
||||
...opts,
|
||||
});
|
||||
return { breaker, advance: (ms: number) => (t += ms) };
|
||||
}
|
||||
|
||||
it('204 returns immediately, no retries, breaker stays closed', async () => {
|
||||
const fetchImpl = vi.fn(async () => jsonResp(204));
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
const resp = await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep },
|
||||
});
|
||||
expect(resp.status).toBe(204);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(1);
|
||||
expect(sleep).not.toHaveBeenCalled();
|
||||
expect(breaker.getState()).toBe('closed');
|
||||
expect(breaker.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('one 503 then 204 → retried once, returns 204, breaker stays closed', async () => {
|
||||
let n = 0;
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
n += 1;
|
||||
return n === 1 ? jsonResp(503) : jsonResp(204);
|
||||
});
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
const resp = await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep, random: () => 0.5, baseDelayMs: 100, capDelayMs: 1000 },
|
||||
});
|
||||
expect(resp.status).toBe(204);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(2);
|
||||
expect(sleep).toHaveBeenCalledTimes(1);
|
||||
expect(breaker.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('429 with Retry-After honored (capped at RETRY_AFTER_CAP_MS)', async () => {
|
||||
let n = 0;
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
n += 1;
|
||||
return n === 1 ? jsonResp(429, { 'Retry-After': '1' }) : jsonResp(204);
|
||||
});
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep },
|
||||
});
|
||||
expect(sleep).toHaveBeenCalledWith(1000); // 1s
|
||||
});
|
||||
|
||||
it('429 with absurd Retry-After is capped to RETRY_AFTER_CAP_MS', async () => {
|
||||
let n = 0;
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
n += 1;
|
||||
return n === 1 ? jsonResp(429, { 'Retry-After': '99999' }) : jsonResp(204);
|
||||
});
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep, capDelayMs: 999_999 }, // ensure cap comes from RETRY_AFTER_CAP_MS, not retry config
|
||||
});
|
||||
expect(sleep).toHaveBeenCalledWith(RETRY_AFTER_CAP_MS);
|
||||
});
|
||||
|
||||
it('429 without Retry-After falls back to exponential-backoff delay', async () => {
|
||||
let n = 0;
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
n += 1;
|
||||
return n === 1 ? jsonResp(429) : jsonResp(204);
|
||||
});
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep, baseDelayMs: 100, capDelayMs: 1000, random: () => 0.5 },
|
||||
});
|
||||
// attempt 0: full-jitter upper = min(1000, 100*1) = 100; floor(0.5*100) = 50
|
||||
expect(sleep).toHaveBeenCalledWith(50);
|
||||
});
|
||||
|
||||
it('401 returned as Response, no retry, breaker not incremented', async () => {
|
||||
const fetchImpl = vi.fn(async () => jsonResp(401));
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
const resp = await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep },
|
||||
});
|
||||
expect(resp.status).toBe(401);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(1);
|
||||
expect(breaker.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('422 returned as Response, no retry', async () => {
|
||||
const fetchImpl = vi.fn(async () => jsonResp(422));
|
||||
const { breaker } = makeBreaker();
|
||||
const resp = await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {} },
|
||||
});
|
||||
expect(resp.status).toBe(422);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('TimeoutError rethrown immediately, no retry, breaker not incremented', async () => {
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
throw new DOMException('aborted', 'TimeoutError');
|
||||
});
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep },
|
||||
}),
|
||||
).rejects.toThrow(DOMException);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(1);
|
||||
expect(sleep).not.toHaveBeenCalled();
|
||||
expect(breaker.getConsecutiveFailures()).toBe(0);
|
||||
});
|
||||
|
||||
it('three consecutive 503 throws ResilientFetchExhaustedError; breaker increments by 1', async () => {
|
||||
const fetchImpl = vi.fn(async () => jsonResp(503));
|
||||
const sleep = vi.fn(async () => {});
|
||||
const { breaker } = makeBreaker();
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep, maxAttempts: 3 },
|
||||
}),
|
||||
).rejects.toBeInstanceOf(ResilientFetchExhaustedError);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(3);
|
||||
expect(breaker.getConsecutiveFailures()).toBe(1);
|
||||
});
|
||||
|
||||
it('after three exhausted 503 batches, breaker opens and fails fast', async () => {
|
||||
const fetchImpl = vi.fn(async () => jsonResp(503));
|
||||
const { breaker } = makeBreaker({ failureThreshold: 3, cooldownMs: 60_000 });
|
||||
for (let i = 0; i < 3; i++) {
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 3 },
|
||||
}),
|
||||
).rejects.toBeInstanceOf(ResilientFetchExhaustedError);
|
||||
}
|
||||
expect(breaker.getState()).toBe('open');
|
||||
// 4th call: breaker open, no fetch invoked.
|
||||
const fetchCallsBefore = fetchImpl.mock.calls.length;
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 3 },
|
||||
}),
|
||||
).rejects.toBeInstanceOf(CircuitOpenError);
|
||||
expect(fetchImpl.mock.calls.length).toBe(fetchCallsBefore);
|
||||
});
|
||||
|
||||
it('retryable-network error retries, breaker counts only on exhaustion', async () => {
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
throw new TypeError('fetch failed');
|
||||
});
|
||||
const { breaker } = makeBreaker();
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 3 },
|
||||
}),
|
||||
).rejects.toBeInstanceOf(TypeError);
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(3);
|
||||
expect(breaker.getConsecutiveFailures()).toBe(1);
|
||||
});
|
||||
|
||||
describe('U2: terminal outcomes route through recordNeutral', () => {
|
||||
it('401 does not erase prior partial-failure progress on the breaker', async () => {
|
||||
const { breaker } = makeBreaker();
|
||||
// Pre-seed the breaker with 2 failures (still closed; threshold 3).
|
||||
breaker.recordFailure();
|
||||
breaker.recordFailure();
|
||||
expect(breaker.getConsecutiveFailures()).toBe(2);
|
||||
|
||||
const fetchImpl = vi.fn(async () => jsonResp(401));
|
||||
await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {} },
|
||||
});
|
||||
|
||||
// Counter MUST stay at 2 — under the old behaviour recordSuccess
|
||||
// would have reset to 0 and the next 5xx batch would have started
|
||||
// from scratch instead of tipping over the threshold.
|
||||
expect(breaker.getConsecutiveFailures()).toBe(2);
|
||||
expect(breaker.getState()).toBe('closed');
|
||||
});
|
||||
|
||||
it('TimeoutError does not erase prior partial-failure progress', async () => {
|
||||
const { breaker } = makeBreaker();
|
||||
breaker.recordFailure();
|
||||
breaker.recordFailure();
|
||||
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
throw new DOMException('aborted by timeout', 'TimeoutError');
|
||||
});
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {} },
|
||||
}),
|
||||
).rejects.toBeInstanceOf(DOMException);
|
||||
|
||||
expect(breaker.getConsecutiveFailures()).toBe(2);
|
||||
});
|
||||
|
||||
it('external AbortError is terminal: no retry, breaker untouched', async () => {
|
||||
const { breaker } = makeBreaker();
|
||||
breaker.recordFailure();
|
||||
|
||||
const fetchImpl = vi.fn(async () => {
|
||||
throw new DOMException('aborted by caller', 'AbortError');
|
||||
});
|
||||
const sleep = vi.fn(async () => {});
|
||||
|
||||
await expect(
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep, maxAttempts: 3 },
|
||||
}),
|
||||
).rejects.toMatchObject({ name: 'AbortError' });
|
||||
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(1);
|
||||
expect(sleep).not.toHaveBeenCalled();
|
||||
// Counter unchanged — neither incremented (no failure) nor reset
|
||||
// (no synthetic success).
|
||||
expect(breaker.getConsecutiveFailures()).toBe(1);
|
||||
});
|
||||
|
||||
it('interleaved 5xx + 401 + 5xx + 401 + 5xx opens breaker on third real failure', async () => {
|
||||
const { breaker } = makeBreaker({ failureThreshold: 3 });
|
||||
const sequence = [503, 401, 503, 401, 503];
|
||||
let i = 0;
|
||||
const fetchImpl = vi.fn(async () => jsonResp(sequence[i++]));
|
||||
|
||||
// Each call uses maxAttempts:1 so each surfaces a single response
|
||||
// (5xx → ResilientFetchExhaustedError; 4xx → returned Response).
|
||||
const driveOne = () =>
|
||||
resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
|
||||
await expect(driveOne()).rejects.toBeInstanceOf(ResilientFetchExhaustedError); // 5xx fail #1
|
||||
await driveOne(); // 401 neutral
|
||||
await expect(driveOne()).rejects.toBeInstanceOf(ResilientFetchExhaustedError); // 5xx fail #2
|
||||
await driveOne(); // 401 neutral
|
||||
await expect(driveOne()).rejects.toBeInstanceOf(ResilientFetchExhaustedError); // 5xx fail #3 → opens
|
||||
|
||||
expect(breaker.getState()).toBe('open');
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(5);
|
||||
});
|
||||
});
|
||||
|
||||
describe('half-open single-probe gating (U2)', () => {
|
||||
/** Test helper: a fetch mock whose Response is controlled by the test. */
|
||||
function deferredFetch(): {
|
||||
promise: Promise<Response>;
|
||||
resolve: (resp: Response) => void;
|
||||
reject: (err: unknown) => void;
|
||||
} {
|
||||
let resolve!: (resp: Response) => void;
|
||||
let reject!: (err: unknown) => void;
|
||||
const promise = new Promise<Response>((res, rej) => {
|
||||
resolve = res;
|
||||
reject = rej;
|
||||
});
|
||||
return { promise, resolve, reject };
|
||||
}
|
||||
|
||||
/** Builds a clock-injected breaker pre-opened with cooldown elapsed. */
|
||||
function preOpenedBreaker(opts: { cooldownMs: number; halfOpenRetryAfterMs?: number }): {
|
||||
breaker: CircuitBreaker;
|
||||
advance: (ms: number) => void;
|
||||
} {
|
||||
let t = 1_700_000_000_000;
|
||||
const breaker = new CircuitBreaker({
|
||||
failureThreshold: 1,
|
||||
cooldownMs: opts.cooldownMs,
|
||||
halfOpenRetryAfterMs: opts.halfOpenRetryAfterMs ?? 1_000,
|
||||
key: 'test',
|
||||
now: () => t,
|
||||
});
|
||||
breaker.recordFailure();
|
||||
t += opts.cooldownMs + 1; // cooldown elapsed
|
||||
return { breaker, advance: (ms) => (t += ms) };
|
||||
}
|
||||
|
||||
it('happy: 3 concurrent calls — exactly 1 hits fetch, others throw CircuitOpenError', async () => {
|
||||
const { breaker } = preOpenedBreaker({ cooldownMs: 10 });
|
||||
const deferred = deferredFetch();
|
||||
const fetchImpl = vi.fn(() => deferred.promise);
|
||||
|
||||
// Synchronous portion of each `resilientFetch` runs eagerly up to
|
||||
// the first await, so by the time r2/r3 are constructed the probe
|
||||
// permit is already consumed by r1 and they reject synchronously.
|
||||
const r1 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
const r2 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
const r3 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
|
||||
// Resolve the probe with 200; r1 should now settle.
|
||||
deferred.resolve(new Response(null, { status: 200 }));
|
||||
|
||||
const results = await Promise.allSettled([r1, r2, r3]);
|
||||
|
||||
expect(results[0].status).toBe('fulfilled');
|
||||
if (results[0].status === 'fulfilled') {
|
||||
expect(results[0].value.status).toBe(200);
|
||||
}
|
||||
expect(results[1].status).toBe('rejected');
|
||||
if (results[1].status === 'rejected') {
|
||||
expect(results[1].reason).toBeInstanceOf(CircuitOpenError);
|
||||
}
|
||||
expect(results[2].status).toBe('rejected');
|
||||
if (results[2].status === 'rejected') {
|
||||
expect(results[2].reason).toBeInstanceOf(CircuitOpenError);
|
||||
}
|
||||
|
||||
// Only ONE underlying fetch was invoked.
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(1);
|
||||
// Breaker closed after the probe's success.
|
||||
expect(breaker.getState()).toBe('closed');
|
||||
});
|
||||
|
||||
it('error: probe gets 503 — exhausted error; subsequent caller sees fresh full cooldown', async () => {
|
||||
const { breaker } = preOpenedBreaker({ cooldownMs: 10_000, halfOpenRetryAfterMs: 1_000 });
|
||||
const deferred = deferredFetch();
|
||||
const fetchImpl = vi.fn(() => deferred.promise);
|
||||
|
||||
const r1 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
const r2 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
const r3 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
|
||||
// Probe fails with 503 → exhausted (maxAttempts: 1) → recordFailure → reopen.
|
||||
deferred.resolve(new Response(null, { status: 503 }));
|
||||
|
||||
const results = await Promise.allSettled([r1, r2, r3]);
|
||||
|
||||
expect(results[0].status).toBe('rejected');
|
||||
if (results[0].status === 'rejected') {
|
||||
expect(results[0].reason).toBeInstanceOf(ResilientFetchExhaustedError);
|
||||
}
|
||||
expect(results[1].status).toBe('rejected');
|
||||
if (results[1].status === 'rejected') {
|
||||
expect(results[1].reason).toBeInstanceOf(CircuitOpenError);
|
||||
// Blocked-while-half-open used the halfOpenRetryAfterMs default.
|
||||
expect((results[1].reason as CircuitOpenError).retryAfterMs).toBe(1_000);
|
||||
}
|
||||
|
||||
// Breaker has re-opened with a fresh openedAt.
|
||||
expect(breaker.getState()).toBe('open');
|
||||
|
||||
// r4: should see the fresh full cooldown, NOT the probe-in-flight 1000ms.
|
||||
let r4Caught: CircuitOpenError | null = null;
|
||||
try {
|
||||
await resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
} catch (err) {
|
||||
r4Caught = err as CircuitOpenError;
|
||||
}
|
||||
expect(r4Caught).toBeInstanceOf(CircuitOpenError);
|
||||
expect(r4Caught?.retryAfterMs).toBe(10_000);
|
||||
});
|
||||
|
||||
it('cancellation: probe AbortError releases permit; next caller becomes new probe', async () => {
|
||||
const { breaker } = preOpenedBreaker({ cooldownMs: 10_000 });
|
||||
const deferred1 = deferredFetch();
|
||||
const deferred2 = deferredFetch();
|
||||
let callIdx = 0;
|
||||
const fetchImpl = vi.fn(() => (callIdx++ === 0 ? deferred1.promise : deferred2.promise));
|
||||
|
||||
// r1 admitted as the probe; r2 blocked while r1 still in flight.
|
||||
const r1 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
const r2 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
await expect(r2).rejects.toBeInstanceOf(CircuitOpenError);
|
||||
|
||||
// Cancel the probe — `AbortError` routes through terminal-network →
|
||||
// `recordNeutral` → permit released, state stays half-open.
|
||||
deferred1.reject(new DOMException('aborted by caller', 'AbortError'));
|
||||
await expect(r1).rejects.toMatchObject({ name: 'AbortError' });
|
||||
|
||||
expect(breaker.isProbeInFlight()).toBe(false);
|
||||
expect(breaker.getState()).toBe('half-open');
|
||||
|
||||
// r3: now succeeds and becomes the new probe.
|
||||
const r3 = resilientFetch(URL_STR, undefined, {
|
||||
fetchImpl: fetchImpl as unknown as typeof fetch,
|
||||
breaker,
|
||||
retry: { sleep: async () => {}, maxAttempts: 1 },
|
||||
});
|
||||
deferred2.resolve(new Response(null, { status: 200 }));
|
||||
const r3Resp = await r3;
|
||||
expect(r3Resp.status).toBe(200);
|
||||
expect(breaker.getState()).toBe('closed');
|
||||
// Two fetches total: the cancelled probe + the recovery probe.
|
||||
expect(fetchImpl).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
});
|
||||
});
|
||||
128
gitnexus/test/unit/integrations/retry.test.ts
Normal file
128
gitnexus/test/unit/integrations/retry.test.ts
Normal file
|
|
@ -0,0 +1,128 @@
|
|||
import { describe, it, expect, vi } from 'vitest';
|
||||
import { computeBackoffMs, withRetry, type RetryOptions } from 'gitnexus-shared';
|
||||
|
||||
describe('computeBackoffMs', () => {
|
||||
it('returns afterMs (capped) when caller supplies it', () => {
|
||||
expect(computeBackoffMs(0, 500, 5000, 1500, () => 0.5)).toBe(1500);
|
||||
expect(computeBackoffMs(0, 500, 5000, 99_999, () => 0.5)).toBe(5000);
|
||||
expect(computeBackoffMs(0, 500, 5000, 0, () => 0.5)).toBe(0);
|
||||
expect(computeBackoffMs(0, 500, 5000, -1, () => 0.5)).toBe(0);
|
||||
});
|
||||
|
||||
it('full-jitter delay falls within [0, min(cap, base * 2^attempt)]', () => {
|
||||
// attempt 0: upper = min(5000, 500 * 1) = 500
|
||||
expect(computeBackoffMs(0, 500, 5000, undefined, () => 0)).toBe(0);
|
||||
expect(computeBackoffMs(0, 500, 5000, undefined, () => 0.999)).toBeLessThan(500);
|
||||
// attempt 1: upper = min(5000, 500 * 2) = 1000
|
||||
expect(computeBackoffMs(1, 500, 5000, undefined, () => 0.5)).toBe(500);
|
||||
// attempt 4: 500 * 16 = 8000, capped at 5000
|
||||
expect(computeBackoffMs(4, 500, 5000, undefined, () => 0.5)).toBe(2500);
|
||||
expect(computeBackoffMs(4, 500, 5000, undefined, () => 0.999)).toBeLessThan(5000);
|
||||
});
|
||||
});
|
||||
|
||||
describe('withRetry', () => {
|
||||
function makeOpts(overrides: Partial<RetryOptions> = {}): RetryOptions {
|
||||
return {
|
||||
maxAttempts: 3,
|
||||
baseDelayMs: 10,
|
||||
capDelayMs: 100,
|
||||
isRetryable: () => ({ retry: true }),
|
||||
sleep: vi.fn(async () => {}),
|
||||
random: () => 0.5,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
it('returns immediately when fn succeeds first try', async () => {
|
||||
const sleep = vi.fn(async () => {});
|
||||
const fn = vi.fn(async () => 'ok');
|
||||
const result = await withRetry(fn, makeOpts({ sleep }));
|
||||
expect(result).toBe('ok');
|
||||
expect(fn).toHaveBeenCalledTimes(1);
|
||||
expect(sleep).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('retries when isRetryable returns retry:true and second call succeeds', async () => {
|
||||
const sleep = vi.fn(async () => {});
|
||||
let calls = 0;
|
||||
const fn = async () => {
|
||||
calls += 1;
|
||||
if (calls === 1) throw new Error('boom');
|
||||
return 'ok';
|
||||
};
|
||||
const result = await withRetry(fn, makeOpts({ sleep }));
|
||||
expect(result).toBe('ok');
|
||||
expect(calls).toBe(2);
|
||||
expect(sleep).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('honors afterMs returned by isRetryable', async () => {
|
||||
const sleep = vi.fn(async () => {});
|
||||
let calls = 0;
|
||||
const fn = async () => {
|
||||
calls += 1;
|
||||
if (calls === 1) throw new Error('throttle');
|
||||
return 'ok';
|
||||
};
|
||||
await withRetry(
|
||||
fn,
|
||||
makeOpts({
|
||||
sleep,
|
||||
isRetryable: () => ({ retry: true, afterMs: 1500 }),
|
||||
capDelayMs: 5000,
|
||||
}),
|
||||
);
|
||||
expect(sleep).toHaveBeenCalledWith(1500);
|
||||
});
|
||||
|
||||
it('caps afterMs at capDelayMs', async () => {
|
||||
const sleep = vi.fn(async () => {});
|
||||
let calls = 0;
|
||||
const fn = async () => {
|
||||
calls += 1;
|
||||
if (calls === 1) throw new Error('throttle');
|
||||
return 'ok';
|
||||
};
|
||||
await withRetry(
|
||||
fn,
|
||||
makeOpts({
|
||||
sleep,
|
||||
isRetryable: () => ({ retry: true, afterMs: 10_000 }),
|
||||
capDelayMs: 3000,
|
||||
}),
|
||||
);
|
||||
expect(sleep).toHaveBeenCalledWith(3000);
|
||||
});
|
||||
|
||||
it('rethrows immediately when isRetryable returns retry:false', async () => {
|
||||
const sleep = vi.fn(async () => {});
|
||||
const fn = vi.fn(async () => {
|
||||
throw new Error('terminal');
|
||||
});
|
||||
await expect(
|
||||
withRetry(fn, makeOpts({ sleep, isRetryable: () => ({ retry: false }) })),
|
||||
).rejects.toThrow('terminal');
|
||||
expect(fn).toHaveBeenCalledTimes(1);
|
||||
expect(sleep).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('throws the last error when maxAttempts exhausted', async () => {
|
||||
const sleep = vi.fn(async () => {});
|
||||
let calls = 0;
|
||||
const fn = async () => {
|
||||
calls += 1;
|
||||
throw new Error(`boom-${calls}`);
|
||||
};
|
||||
await expect(withRetry(fn, makeOpts({ sleep, maxAttempts: 3 }))).rejects.toThrow('boom-3');
|
||||
expect(calls).toBe(3);
|
||||
// 3 attempts → 2 sleeps between them; final attempt does not sleep.
|
||||
expect(sleep).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('rejects maxAttempts < 1', async () => {
|
||||
await expect(withRetry(async () => 'ok', makeOpts({ maxAttempts: 0 }))).rejects.toThrow(
|
||||
/maxAttempts must be >= 1/,
|
||||
);
|
||||
});
|
||||
});
|
||||
|
|
@ -5,9 +5,11 @@ import os from 'os';
|
|||
import fs from 'fs';
|
||||
|
||||
describe('--skip-git CLI flag', () => {
|
||||
const cliPath = path.resolve(__dirname, '../../dist/cli/index.js');
|
||||
|
||||
it('Commander maps --skip-git to options.skipGit (not --no-git inversion)', () => {
|
||||
// Verify the CLI defines --skip-git and --skip-agents-md in analyze help.
|
||||
const helpOutput = execSync('node dist/cli/index.js analyze --help', {
|
||||
const helpOutput = execSync(`node "${cliPath}" analyze --help`, {
|
||||
cwd: path.resolve(__dirname, '../..'),
|
||||
encoding: 'utf8',
|
||||
timeout: 10000,
|
||||
|
|
@ -37,8 +39,59 @@ describe('--skip-git CLI flag', () => {
|
|||
}
|
||||
});
|
||||
|
||||
it('still respects .gitnexusignore when run with --skip-git', () => {
|
||||
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-skip-git-ignore-'));
|
||||
const gitnexusHome = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-skip-git-ignore-home-'));
|
||||
fs.mkdirSync(path.join(tmpDir, 'src'), { recursive: true });
|
||||
fs.mkdirSync(path.join(tmpDir, 'customskip'), { recursive: true });
|
||||
fs.writeFileSync(path.join(tmpDir, '.gitnexusignore'), 'customskip/\n');
|
||||
fs.writeFileSync(path.join(tmpDir, 'src', 'keep.ts'), 'export function keep() { return 1; }\n');
|
||||
fs.writeFileSync(
|
||||
path.join(tmpDir, 'customskip', 'leaked.ts'),
|
||||
'export function leaked() { return 42; }\n',
|
||||
);
|
||||
|
||||
const env = {
|
||||
...process.env,
|
||||
HOME: gitnexusHome,
|
||||
GITNEXUS_HOME: gitnexusHome,
|
||||
GITNEXUS_LBUG_EXTENSION_INSTALL: 'never',
|
||||
};
|
||||
|
||||
try {
|
||||
execSync(`node "${cliPath}" analyze "${tmpDir}" --skip-git --skip-agents-md`, {
|
||||
encoding: 'utf8',
|
||||
timeout: 60000,
|
||||
env,
|
||||
});
|
||||
|
||||
const keepContext = execSync(
|
||||
`node "${cliPath}" context keep --repo "${path.basename(tmpDir)}"`,
|
||||
{
|
||||
encoding: 'utf8',
|
||||
timeout: 60000,
|
||||
env,
|
||||
},
|
||||
);
|
||||
expect(keepContext).toContain('"status": "found"');
|
||||
expect(keepContext).toContain('"filePath": "src/keep.ts"');
|
||||
|
||||
const leakedContext = execSync(
|
||||
`node "${cliPath}" context leaked --repo "${path.basename(tmpDir)}"`,
|
||||
{
|
||||
encoding: 'utf8',
|
||||
timeout: 60000,
|
||||
env,
|
||||
},
|
||||
);
|
||||
expect(leakedContext).toContain(`"error": "Symbol 'leaked' not found"`);
|
||||
} finally {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
fs.rmSync(gitnexusHome, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
describe('--skip-git does not walk up to parent git repo (#1232)', () => {
|
||||
const cliPath = path.resolve(__dirname, '../../dist/cli/index.js');
|
||||
let parentDir: string;
|
||||
let gitnexusHome: string;
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue