mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
Adds review_gate(), a state machine that keeps a `ready for review` label in sync with whether an external PR clears BOTH gates — the LLM rubric and Greptile's most recent confidence score: - pass (untagged) -> add label + "ready for review" / "all clear" comment - pass (already tagged) -> no-op (idempotent across re-runs) - regress (Greptile < 4/5 or QA proof removed) -> remove label + "what's missing" comment, PR stays open - recover after a regression -> "all clear again" comment + re-add the label - fail & untagged, < 24h old -> one-time "what's missing" notice (grace window) - fail & untagged, > 24h old -> close + comment (reopen via @agent-shin reconsider) The label itself is the persisted state, so comments fire only on transitions (never on every scheduled run). All side effects are gated behind --close, so the dry-run contract matches the existing triage flow. Lifecycle comments use hidden HTML markers and deliberately avoid the auto-close marker so they never trip the reconsider provenance check. Relocates the shared Greptile helpers (extract_greptile_score, SCORE_PATTERN, GREPTILE_BOT_LOGINS, parse_iso8601) into triage_with_llm.py so the daily sweep and the review gate read the score through one implementation, and adds the review_gate.yml workflow (dry-run unless AGENT_SHIN_ENABLED=true) plus 18 unit tests covering every branch and a full pass->regress->recover cycle. https://claude.ai/code/session_01XyyWa8t2VYmoGd6mKMEqkZ
128 lines
5.6 KiB
YAML
128 lines
5.6 KiB
YAML
name: Agent Shin — review gate
|
|
|
|
# Keeps the `ready for review` label in sync with whether an external PR
|
|
# currently clears BOTH the LLM rubric AND Greptile's confidence score.
|
|
#
|
|
# pass -> add `ready for review` + a "passed / all clear" comment
|
|
# regress -> remove the label + a "what's missing" comment (PR stays open)
|
|
# fail, <24h old -> a one-time "what's missing" notice (grace window)
|
|
# fail, >24h old -> close + a comment (reopen via `@agent-shin reconsider`)
|
|
#
|
|
# DRY-RUN BY DEFAULT. Every side effect (label add/remove, comment, close) is
|
|
# gated behind `--close`, which is only added when the repo variable
|
|
# `AGENT_SHIN_ENABLED == "true"`. Until then runs only write the verdict to the
|
|
# workflow step summary.
|
|
#
|
|
# Manual single PR: gh workflow run "Agent Shin — review gate" -f pr_number=NNN
|
|
# Manual dry-run: gh workflow run "Agent Shin — review gate" -f close=false
|
|
#
|
|
# We use `pull_request_target` so the workflow can read repo secrets and run
|
|
# against fork PRs. Fork code is never checked out — only PR metadata is read
|
|
# via `gh api`.
|
|
|
|
on:
|
|
pull_request_target:
|
|
types: [opened, reopened, synchronize, ready_for_review]
|
|
schedule:
|
|
# Daily at 09:30 UTC — re-reconciles labels as Greptile re-reviews land.
|
|
- cron: "30 9 * * *"
|
|
workflow_dispatch:
|
|
inputs:
|
|
pr_number:
|
|
description: "Single PR to reconcile (omit to sweep all open PRs)."
|
|
required: false
|
|
close:
|
|
description: "If AGENT_SHIN_ENABLED=true, actually act (false = dry run)."
|
|
required: false
|
|
default: "false"
|
|
type: choice
|
|
options:
|
|
- "true"
|
|
- "false"
|
|
grace_days:
|
|
description: "Hours/24 a failing, un-tagged PR may stay open before close."
|
|
required: false
|
|
default: "1"
|
|
min_greptile_score:
|
|
description: "Greptile score below which a PR counts as not passing (1-5)."
|
|
required: false
|
|
default: "4"
|
|
|
|
permissions:
|
|
contents: read
|
|
issues: write
|
|
pull-requests: write
|
|
|
|
jobs:
|
|
review-gate:
|
|
if: github.repository == 'BerriAI/litellm'
|
|
runs-on: ubuntu-latest
|
|
steps:
|
|
- name: Checkout triage script
|
|
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
|
|
with:
|
|
sparse-checkout: .github/scripts
|
|
persist-credentials: false
|
|
|
|
- name: Set up Python
|
|
uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
|
|
with:
|
|
python-version: "3.12"
|
|
|
|
- name: Install LLM client
|
|
run: pip install --no-cache-dir "openai>=1.40.0"
|
|
|
|
- name: Run review gate
|
|
env:
|
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
|
# Mirror the triage workflow: only expose the LLM key when the bot is
|
|
# enabled or a collaborator triggers it manually, so an external user
|
|
# can't force paid LLM calls by churning a fork PR while the bot is
|
|
# still in dry-run.
|
|
OPENAI_API_KEY: ${{ (vars.AGENT_SHIN_ENABLED == 'true' || github.event_name == 'workflow_dispatch') && secrets.OPENAI_API_KEY || '' }}
|
|
OPENAI_BASE_URL: ${{ vars.OPENAI_BASE_URL }}
|
|
TRIAGE_MODEL: ${{ vars.TRIAGE_MODEL }}
|
|
AGENT_SHIN_ENABLED: ${{ vars.AGENT_SHIN_ENABLED }}
|
|
CLOSE_FLAG: ${{ github.event.inputs.close || 'false' }}
|
|
GRACE_DAYS: ${{ github.event.inputs.grace_days || '1' }}
|
|
MIN_GREPTILE_SCORE: ${{ github.event.inputs.min_greptile_score || '4' }}
|
|
EVENT_PR: ${{ github.event.pull_request.number }}
|
|
INPUT_PR: ${{ github.event.inputs.pr_number }}
|
|
run: |
|
|
set -euo pipefail
|
|
COMMON=(--review-gate --grace-days "${GRACE_DAYS}" --min-greptile-score "${MIN_GREPTILE_SCORE}")
|
|
|
|
# Fail-safe gating, identical philosophy to the Greptile closer:
|
|
# - AGENT_SHIN_ENABLED must be the EXACT string "true" to act at all.
|
|
# - A manual dispatch can still preview with close=false.
|
|
# - Automatic triggers (PR events, schedule) act once enabled — that
|
|
# is the whole point of the gate (re-tag / un-tag automatically).
|
|
DO_CLOSE="false"
|
|
if [ "${AGENT_SHIN_ENABLED:-false}" != "true" ]; then
|
|
echo "::notice::AGENT_SHIN_ENABLED is not 'true' -> dry-run (no labels/comments/closes)."
|
|
elif [ "${GITHUB_EVENT_NAME:-}" = "workflow_dispatch" ] && [ "${CLOSE_FLAG:-false}" = "true" ]; then
|
|
DO_CLOSE="true"
|
|
echo "::notice::Manual run -> acting for real."
|
|
elif [ "${GITHUB_EVENT_NAME:-}" != "workflow_dispatch" ]; then
|
|
DO_CLOSE="true"
|
|
echo "::notice::Enabled automatic trigger (${GITHUB_EVENT_NAME:-}) -> acting for real."
|
|
else
|
|
echo "::notice::Manual dispatch with close=false -> dry-run."
|
|
fi
|
|
if [ "${DO_CLOSE}" = "true" ]; then
|
|
COMMON+=(--close)
|
|
fi
|
|
|
|
# Single PR (PR event or explicit input) vs. sweep over all open PRs.
|
|
TARGET_PR="${EVENT_PR:-${INPUT_PR:-}}"
|
|
if [ -n "${TARGET_PR}" ]; then
|
|
python3 .github/scripts/triage_with_llm.py --repo "${{ github.repository }}" --pr "${TARGET_PR}" "${COMMON[@]}"
|
|
else
|
|
echo "::notice::Sweeping all open PRs."
|
|
mapfile -t NUMBERS < <(gh pr list --repo "${{ github.repository }}" --state open --limit 1000 --json number --jq '.[].number')
|
|
for n in "${NUMBERS[@]}"; do
|
|
echo "::group::PR #${n}"
|
|
python3 .github/scripts/triage_with_llm.py --repo "${{ github.repository }}" --pr "${n}" "${COMMON[@]}" || echo "::warning::review gate errored on #${n}"
|
|
echo "::endgroup::"
|
|
done
|
|
fi
|