ci(duplicate-check): move the flag step into a tested bun script

A verdict is now dropped when it names a pull request, the issue itself,
or a newer issue, and the label goes on before the comment so a failed
comment leaves no marker and the rerun finishes the job. The flag logic
lives in scripts/flag-duplicate-issue.ts next to the sweep it feeds,
sharing normalizeTitle and the marker format, with bun tests that run on
pull requests touching it
This commit is contained in:
ryan-crabbe-berri 2026-09-12 20:10:30 -07:00
parent 82f20793eb
commit 009d6f364b
4 changed files with 400 additions and 85 deletions

View file

@ -1,12 +1,5 @@
name: Duplicate issue check (Codex)
# Semantic duplicate detection for newly opened issues. This replaces the
# title-similarity bot in check_duplicate_issues.yml, which only matched
# wording and so missed the same bug reported in different words.
#
# DRY-RUN BY DEFAULT: set the repo variable DUPLICATE_CHECK_ENABLED=true to let
# it comment and label. Until then the verdict only appears in the job summary.
on:
issues:
types: [opened]
@ -15,12 +8,40 @@ on:
issue_number:
description: "Issue number to check manually."
required: true
pull_request:
paths:
- .github/workflows/duplicate_issue_check.yml
- .github/prompts/duplicate-issue-check.md
- .github/prompts/duplicate-issue-check.schema.json
- scripts/flag-duplicate-issue.ts
- scripts/flag-duplicate-issue.test.ts
- scripts/auto-close-duplicates.ts
permissions: {}
jobs:
flag-tests:
if: github.event_name == 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
steps:
- name: Checkout repository
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
persist-credentials: false
- name: Setup Bun
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: "1.4.0"
- name: Test the flag step
run: bun test scripts/flag-duplicate-issue.test.ts
classify:
if: github.repository == 'BerriAI/litellm'
if: github.event_name != 'pull_request' && github.repository == 'BerriAI/litellm'
runs-on: ubuntu-latest
timeout-minutes: 15
permissions:
@ -35,8 +56,7 @@ jobs:
sparse-checkout: .github/prompts
persist-credentials: false
# Fetched through the API rather than interpolated from github.event, so
# no issue text ever reaches a shell or an action input as template text.
# Read through the API so issue text never reaches a shell or an action input
- name: Fetch the issue under review
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
@ -63,22 +83,16 @@ jobs:
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
with:
# Routed through LiteLLM, so the credential is a virtual key and the
# spend lands in the proxy's own logs. The action hands this key to
# codex-responses-api-proxy, which forwards to the endpoint below.
openai-api-key: ${{ secrets.LITELLM_API_KEY }}
responses-api-endpoint: ${{ vars.LITELLM_API_BASE }}/v1/responses
prompt-file: .github/prompts/duplicate-issue-check.md
output-schema-file: .github/prompts/duplicate-issue-check.schema.json
sandbox: read-only
# read-only still denies network, and the whole method is Codex
# searching the issue tracker with `gh`, so it needs egress.
# read-only denies network, and the whole method is searching the tracker with gh
codex-args: '["-c", "sandbox_permissions=[\"network-full-access\"]"]'
model: ${{ vars.DUPLICATE_CHECK_MODEL || 'gpt-5.6' }}
# Issue authors are external users without write access, and the
# action's default is to refuse to run for them. Safe to open up
# here: the prompt is fixed, the sandbox is read-only, and the only
# credential Codex holds is a read-only token for a public repo.
# Issue authors have no write access and the action refuses them by default; the
# prompt is fixed, the sandbox read-only, and the only token is read-only on a public repo
allow-users: "*"
- name: Summary
@ -98,73 +112,24 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
issues: write
steps:
- name: Comment and label
uses: actions/github-script@f28e40c7f34bde8b3046d885e986cb6290c5673b # v7.1.0
env:
VERDICT: ${{ needs.classify.outputs.verdict }}
ENABLED: ${{ vars.DUPLICATE_CHECK_ENABLED }}
ISSUE_NUMBER: ${{ github.event.issue.number || github.event.inputs.issue_number }}
- name: Checkout scripts
uses: actions/checkout@08eba0b27e820071cde6df949e0beb9ba4906955 # v4.3.0
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
script: |
let verdict;
try {
verdict = JSON.parse(process.env.VERDICT);
} catch (e) {
core.warning(`Codex did not return JSON: ${e.message}`);
return;
}
const { duplicate_of: original, confidence, evidence } = verdict;
// 0.95, not 0.8: over a full week of issues the 0.80 gate posted 21
// comments of which 6 were wrong, while 0.95 posts 12 with 1 wrong
// and still catches 9 of the 11 real duplicates.
if (!Number.isInteger(original) || confidence < 0.95) {
core.notice(`No duplicate flagged (duplicate_of=${original}, confidence=${confidence}).`);
return;
}
const issue_number = Number(process.env.ISSUE_NUMBER);
const { owner, repo } = context.repo;
sparse-checkout: scripts
persist-credentials: false
const existing = await github.paginate(github.rest.issues.listComments, { owner, repo, issue_number });
if (existing.some((c) => c.body?.includes('litellm:potential-duplicate'))) {
core.notice(`#${issue_number} already carries a duplicate notice.`);
return;
}
- name: Setup Bun
uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: "1.4.0"
const { data: prior } = await github.rest.issues.get({ owner, repo, issue_number: original });
const { data: self } = await github.rest.issues.get({ owner, repo, issue_number });
const lead = prior.state === 'closed'
? `**Already reported in #${original}**, which is closed`
: `**Possible duplicate of #${original}**`;
const ask = prior.state === 'closed'
? `If that issue covers this one, follow up there. If this is a new case, say so here and the label comes off.`
: `If that is right, add a thumbs-up to #${original} and follow along there. If it is not, say so here and the label comes off.`;
// Mirrors normalizeTitle in scripts/auto-close-duplicates.ts. That
// sweep can close this issue on the marker below, but only when the
// titles match exactly, so only warn when they actually do.
const normalize = (t) => t.toLowerCase().replace(/^\s*\[[^\]]*\]\s*:?/, '').replace(/[^a-z0-9]+/g, ' ').trim();
const autoCloses = prior.state === 'open' && normalize(self.title) === normalize(prior.title);
const warning = autoCloses
? `\n\nYour title is identical to #${original}, so this issue closes automatically in 3 days unless someone responds here.`
: '';
// Same marker the title bot posts, so auto-close-duplicates.yml sees
// one pipeline. That sweep still needs an identical title to close,
// which a semantic-only match will almost never have.
const body = [
`<!-- litellm:potential-duplicate candidates=${original}, -->`,
lead,
'',
evidence,
'',
ask + warning,
].join('\n');
if (process.env.ENABLED !== 'true') {
core.notice(`DRY RUN. Would have commented on #${issue_number}:\n${body}`);
return;
}
await github.rest.issues.createComment({ owner, repo, issue_number, body });
await github.rest.issues.addLabels({ owner, repo, issue_number, labels: ['potential-duplicate'] });
- name: Comment and label
run: bun run scripts/flag-duplicate-issue.ts
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
VERDICT: ${{ needs.classify.outputs.verdict }}
ISSUE_NUMBER: ${{ github.event.issue.number || github.event.inputs.issue_number }}
DRY_RUN: ${{ vars.DUPLICATE_CHECK_ENABLED != 'true' }}

View file

@ -157,7 +157,7 @@ export function closingComment(duplicateOf: number, graceDays: number): string {
${CLOSED_MARKER}`;
}
async function listAll<T>(api: GitHubApi, path: string, page = 1): Promise<readonly T[]> {
export async function listAll<T>(api: GitHubApi, path: string, page = 1): Promise<readonly T[]> {
const separator = path.includes("?") ? "&" : "?";
const batch = await api.request<readonly T[]>("GET", `${path}${separator}per_page=${PAGE_SIZE}&page=${page}`);
return batch.length < PAGE_SIZE ? batch : [...batch, ...(await listAll<T>(api, path, page + 1))];

View file

@ -0,0 +1,200 @@
import { describe, expect, test } from "bun:test";
import { candidateNumbers, duplicateTarget, type Comment, type GitHubApi, type Issue } from "./auto-close-duplicates";
import {
MIN_CONFIDENCE,
flagIssue,
flagTarget,
noticeBody,
parseVerdict,
readConfig,
type FlagConfig,
type Verdict,
} from "./flag-duplicate-issue";
const issue = (number: number, title: string, overrides: Partial<Issue> = {}): Issue => ({
number,
title,
state: "open",
user: { login: "reporter" },
...overrides,
});
const verdict = (overrides: Partial<Verdict> = {}): Verdict => ({
duplicate_of: 10,
confidence: 0.99,
evidence: "Both report the same traceback from the same function.",
...overrides,
});
const config: FlagConfig = { repo: "BerriAI/litellm", issueNumber: 35, dryRun: false };
describe("parseVerdict", () => {
test("accepts the schema's shape, with a null duplicate_of", () => {
const parsed = parseVerdict('{"duplicate_of": null, "confidence": 0.9, "evidence": "Nothing matches.", "considered": [1]}');
expect(parsed).toEqual({ kind: "verdict", verdict: { duplicate_of: null, confidence: 0.9, evidence: "Nothing matches." } });
});
test("rejects non-JSON, a non-object, a non-integer target, a missing confidence and empty evidence", () => {
expect(parseVerdict("not json").kind).toBe("skip");
expect(parseVerdict('"just a string"').kind).toBe("skip");
expect(parseVerdict('{"duplicate_of": "10", "confidence": 0.99, "evidence": "x"}').kind).toBe("skip");
expect(parseVerdict('{"duplicate_of": 10.5, "confidence": 0.99, "evidence": "x"}').kind).toBe("skip");
expect(parseVerdict('{"duplicate_of": 10, "evidence": "x"}').kind).toBe("skip");
expect(parseVerdict('{"duplicate_of": 10, "confidence": 0.99, "evidence": " "}').kind).toBe("skip");
});
});
describe("flagTarget", () => {
test("flags at the gate and not one hundredth below it", () => {
expect(flagTarget(verdict({ confidence: MIN_CONFIDENCE }), 35)).toEqual({ kind: "target", original: 10 });
expect(flagTarget(verdict({ confidence: 0.94 }), 35).kind).toBe("skip");
});
test("never flags nothing, itself, or a newer issue", () => {
expect(flagTarget(verdict({ duplicate_of: null }), 35).kind).toBe("skip");
expect(flagTarget(verdict({ duplicate_of: 35 }), 35).kind).toBe("skip");
expect(flagTarget(verdict({ duplicate_of: 36 }), 35).kind).toBe("skip");
});
});
describe("noticeBody", () => {
const reporter = issue(35, "[Bug]: Gemma 4-e4b fails on Vertex");
test("an open original gets the thumbs-up ask, and the marker the sweep reads", () => {
const body = noticeBody(reporter, issue(10, "Vertex Gemma 4 crash"), "Same stack.");
expect(body).toContain("**Possible duplicate of #10**");
expect(body).toContain("add a thumbs-up to #10");
expect(body).toContain("Same stack.");
expect(body).not.toContain("closes automatically");
expect(candidateNumbers(body, 35)).toEqual([10]);
});
test("a closed original gets the follow-up-there ask", () => {
const body = noticeBody(reporter, issue(10, "Vertex Gemma 4 crash", { state: "closed" }), "Same stack.");
expect(body).toContain("**Already reported in #10**, which is closed");
expect(body).toContain("follow up there");
});
test("warns about the automatic close exactly when the sweep would close", () => {
const twin = issue(10, "[bug] gemma 4-e4b fails on vertex!");
const body = noticeBody(reporter, twin, "Same stack.");
expect(body).toContain("closes automatically in 3 days");
expect(duplicateTarget(reporter, [twin], []).kind).toBe("close");
const closedTwin = issue(10, "[bug] gemma 4-e4b fails on vertex!", { state: "closed" });
expect(noticeBody(reporter, closedTwin, "Same stack.")).not.toContain("closes automatically");
expect(duplicateTarget(reporter, [closedTwin], []).kind).toBe("skip");
});
});
describe("flagIssue", () => {
const reporter = issue(35, "[Bug]: Gemma 4-e4b fails on Vertex");
function fakeApi(
prior: Issue = issue(10, "Vertex Gemma 4 crash"),
comments: readonly Comment[] = [],
failing: readonly string[] = [],
): { readonly api: GitHubApi; readonly writes: string[] } {
const writes: string[] = [];
const api: GitHubApi = {
request: async <T>(method: string, path: string, body?: object): Promise<T> => {
if (method !== "GET") {
if (failing.includes(path)) {
throw new Error(`${method} ${path} failed: 502`);
}
writes.push(`${method} ${path} ${JSON.stringify(body)}`);
return {} as T;
}
if (path.startsWith("/repos/BerriAI/litellm/issues/35/comments")) {
return comments as T;
}
if (path === "/repos/BerriAI/litellm/issues/35") {
return reporter as T;
}
if (path === `/repos/BerriAI/litellm/issues/${prior.number}`) {
return prior as T;
}
throw new Error(`unexpected GET ${path}`);
},
};
return { api, writes };
}
test("a real run labels first, then comments with the marker", async () => {
const { api, writes } = fakeApi();
const result = await flagIssue(api, config, verdict());
expect(result.kind).toBe("flagged");
expect(writes.map((write) => write.split(" ").slice(0, 2).join(" "))).toEqual([
"POST /repos/BerriAI/litellm/issues/35/labels",
"POST /repos/BerriAI/litellm/issues/35/comments",
]);
expect(writes[0]).toContain('{"labels":["potential-duplicate"]}');
expect(writes[1]).toContain("<!-- litellm:potential-duplicate candidates=10, -->");
});
test("a dry run renders the comment and writes nothing", async () => {
const { api, writes } = fakeApi();
const result = await flagIssue(api, { ...config, dryRun: true }, verdict());
expect(result.kind).toBe("flagged");
expect(result.kind === "flagged" && result.body).toContain("**Possible duplicate of #10**");
expect(writes).toEqual([]);
});
test("a verdict naming a pull request is dropped without a write", async () => {
const { api, writes } = fakeApi(issue(10, "fix: Vertex Gemma 4 crash", { pull_request: {} }));
expect(await flagIssue(api, config, verdict())).toEqual({ kind: "skip", reason: "#10 is a pull request" });
expect(writes).toEqual([]);
});
test("a verdict below the gate never touches the API", async () => {
const { api, writes } = fakeApi();
expect((await flagIssue(api, config, verdict({ confidence: 0.9 }))).kind).toBe("skip");
expect(writes).toEqual([]);
});
test("an issue that already carries a notice is not flagged twice", async () => {
const existing: Comment = {
id: 1,
body: "<!-- litellm:potential-duplicate candidates=10, -->\n**Possible duplicate of #10**",
created_at: "2026-09-10T00:00:00Z",
user: { type: "Bot", login: "github-actions[bot]" },
};
const { api, writes } = fakeApi(undefined, [existing]);
expect(await flagIssue(api, config, verdict())).toEqual({ kind: "skip", reason: "already carries a duplicate notice" });
expect(writes).toEqual([]);
});
test("a failed comment leaves no marker, so the rerun finishes the job", async () => {
const commentsPath = "/repos/BerriAI/litellm/issues/35/comments";
const first = fakeApi(undefined, [], [commentsPath]);
await expect(flagIssue(first.api, config, verdict())).rejects.toThrow("failed: 502");
expect(first.writes).toEqual(['POST /repos/BerriAI/litellm/issues/35/labels {"labels":["potential-duplicate"]}']);
const rerun = fakeApi();
expect((await flagIssue(rerun.api, config, verdict())).kind).toBe("flagged");
expect(rerun.writes.map((write) => write.split(" ")[1])).toEqual([
"/repos/BerriAI/litellm/issues/35/labels",
commentsPath,
]);
});
});
describe("readConfig", () => {
const env = { GITHUB_TOKEN: "t", GITHUB_REPOSITORY: "BerriAI/litellm", ISSUE_NUMBER: "35" };
test("defaults to a real run", () => {
expect(readConfig(env)).toEqual({ token: "t", repo: "BerriAI/litellm", issueNumber: 35, dryRun: false });
});
test("honors DRY_RUN", () => {
expect(readConfig({ ...env, DRY_RUN: "true" }).dryRun).toBe(true);
});
test("refuses a missing token, a malformed repository, or a bad issue number", () => {
expect(() => readConfig({ ...env, GITHUB_TOKEN: undefined })).toThrow("GITHUB_TOKEN");
expect(() => readConfig({ ...env, GITHUB_REPOSITORY: "not a repo" })).toThrow("GITHUB_REPOSITORY");
expect(() => readConfig({ ...env, ISSUE_NUMBER: "" })).toThrow("ISSUE_NUMBER");
expect(() => readConfig({ ...env, ISSUE_NUMBER: "1.5" })).toThrow("ISSUE_NUMBER");
});
});

View file

@ -0,0 +1,150 @@
#!/usr/bin/env bun
import {
DEFAULT_GRACE_DAYS,
FLAG_LABEL,
githubApi,
listAll,
normalizeTitle,
type Comment,
type GitHubApi,
type Issue,
} from "./auto-close-duplicates";
declare const process: { readonly env: Readonly<Record<string, string | undefined>> };
export interface Verdict {
readonly duplicate_of: number | null;
readonly confidence: number;
readonly evidence: string;
}
export interface FlagConfig {
readonly repo: string;
readonly issueNumber: number;
readonly dryRun: boolean;
}
export type ParsedVerdict =
| { readonly kind: "verdict"; readonly verdict: Verdict }
| { readonly kind: "skip"; readonly reason: string };
export type FlagTarget =
| { readonly kind: "target"; readonly original: number }
| { readonly kind: "skip"; readonly reason: string };
export type FlagVerdict =
| { readonly kind: "flagged"; readonly original: number; readonly body: string }
| { readonly kind: "skip"; readonly reason: string };
export const MIN_CONFIDENCE = 0.95;
export const NOTICE_MARKER_PREFIX = "<!-- litellm:potential-duplicate candidates=";
const skip = (reason: string): { readonly kind: "skip"; readonly reason: string } => ({ kind: "skip", reason });
const parseJson = (raw: string): unknown => {
try {
return JSON.parse(raw);
} catch {
return undefined;
}
};
export function parseVerdict(raw: string): ParsedVerdict {
const parsed = parseJson(raw);
if (typeof parsed !== "object" || parsed === null) {
return skip("Codex did not return a JSON object");
}
const { duplicate_of, confidence, evidence } = parsed as Record<string, unknown>;
if (duplicate_of !== null && !Number.isInteger(duplicate_of)) {
return skip(`duplicate_of must be an integer or null, got ${JSON.stringify(duplicate_of)}`);
}
if (typeof confidence !== "number" || !Number.isFinite(confidence)) {
return skip(`confidence must be a number, got ${JSON.stringify(confidence)}`);
}
if (typeof evidence !== "string" || evidence.trim() === "") {
return skip("evidence must be a non-empty string");
}
return { kind: "verdict", verdict: { duplicate_of: duplicate_of as number | null, confidence, evidence } };
}
export function flagTarget(verdict: Verdict, issueNumber: number): FlagTarget {
if (verdict.duplicate_of === null) {
return skip("no duplicate named");
}
if (verdict.confidence < MIN_CONFIDENCE) {
return skip(`confidence ${verdict.confidence} is below ${MIN_CONFIDENCE}`);
}
if (verdict.duplicate_of >= issueNumber) {
return skip(`#${verdict.duplicate_of} is not older than #${issueNumber}`);
}
return { kind: "target", original: verdict.duplicate_of };
}
export function noticeBody(issue: Issue, prior: Issue, evidence: string): string {
const closed = prior.state === "closed";
const lead = closed
? `**Already reported in #${prior.number}**, which is closed`
: `**Possible duplicate of #${prior.number}**`;
const ask = closed
? "If that issue covers this one, follow up there. If this is a new case, say so here and the label comes off."
: `If that is right, add a thumbs-up to #${prior.number} and follow along there. If it is not, say so here and the label comes off.`;
const autoCloses = !closed && normalizeTitle(issue.title) === normalizeTitle(prior.title);
const warning = autoCloses
? `\n\nYour title is identical to #${prior.number}, so this issue closes automatically in ${DEFAULT_GRACE_DAYS} days unless someone responds here.`
: "";
return [`${NOTICE_MARKER_PREFIX}${prior.number}, -->`, lead, "", evidence, "", ask + warning].join("\n");
}
export async function flagIssue(api: GitHubApi, config: FlagConfig, verdict: Verdict): Promise<FlagVerdict> {
const target = flagTarget(verdict, config.issueNumber);
if (target.kind === "skip") {
return target;
}
const issuePath = `/repos/${config.repo}/issues/${config.issueNumber}`;
const comments = await listAll<Comment>(api, `${issuePath}/comments`);
if (comments.some((comment) => comment.body.includes(NOTICE_MARKER_PREFIX))) {
return skip("already carries a duplicate notice");
}
const prior = await api.request<Issue>("GET", `/repos/${config.repo}/issues/${target.original}`);
if (prior.pull_request !== undefined) {
return skip(`#${target.original} is a pull request`);
}
const issue = await api.request<Issue>("GET", issuePath);
const body = noticeBody(issue, prior, verdict.evidence);
if (!config.dryRun) {
await api.request("POST", `${issuePath}/labels`, { labels: [FLAG_LABEL] });
await api.request("POST", `${issuePath}/comments`, { body });
}
return { kind: "flagged", original: target.original, body };
}
export function readConfig(env: Readonly<Record<string, string | undefined>>): FlagConfig & { readonly token: string } {
const token = env.GITHUB_TOKEN;
const repo = env.GITHUB_REPOSITORY;
if (!token || !repo || !/^[\w.-]+\/[\w.-]+$/.test(repo)) {
throw new Error("GITHUB_TOKEN and GITHUB_REPOSITORY (owner/repo) are required");
}
const issueNumber = Number(env.ISSUE_NUMBER);
if (!Number.isInteger(issueNumber) || issueNumber <= 0) {
throw new Error(`ISSUE_NUMBER must be a positive integer, got "${env.ISSUE_NUMBER}"`);
}
return { token, repo, issueNumber, dryRun: env.DRY_RUN === "true" };
}
function describe(config: FlagConfig, verdict: FlagVerdict): string {
if (verdict.kind === "skip") {
return `#${config.issueNumber}: skipped, ${verdict.reason}`;
}
if (config.dryRun) {
return `#${config.issueNumber}: DRY RUN, set the DUPLICATE_CHECK_ENABLED repo variable to true to post this:\n\n${verdict.body}`;
}
return `#${config.issueNumber}: flagged as a possible duplicate of #${verdict.original}`;
}
if (import.meta.main) {
const { token, ...config } = readConfig(process.env);
const parsed = parseVerdict(process.env.VERDICT ?? "");
const verdict = parsed.kind === "skip" ? parsed : await flagIssue(githubApi(token), config, parsed.verdict);
console.log(describe(config, verdict));
}