From c9655a6d9fb5d03b9db4bcffe4eca943cd5f822a Mon Sep 17 00:00:00 2001 From: wowo-zZ Date: Thu, 9 Apr 2026 12:30:44 +0800 Subject: [PATCH] add issue triage automation mvp --- .github/scripts/github.ts | 230 +++++ .github/scripts/issue-backlog-rescore.ts | 128 +++ .github/scripts/issue-handoff-brief.ts | 257 ++++++ .github/scripts/issue-llm-config.ts | 139 +++ .github/scripts/issue-llm-evaluator.ts | 466 ++++++++++ .github/scripts/issue-llm-provider.ts | 206 +++++ .github/scripts/issue-llm-types.ts | 62 ++ .github/scripts/issue-triage-config.ts | 236 +++++ .github/scripts/issue-triage-lib.ts | 923 ++++++++++++++++++++ .github/scripts/issue-triage-merge.ts | 166 ++++ .github/scripts/issue-triage-types.ts | 93 ++ .github/scripts/issue-triage.ts | 129 +++ .github/workflows/issue-backlog-rescore.yml | 51 ++ .github/workflows/issue-triage.yml | 62 ++ docs/2026-04-08-issue-automation-design.md | 337 +++++++ 15 files changed, 3485 insertions(+) create mode 100644 .github/scripts/github.ts create mode 100644 .github/scripts/issue-backlog-rescore.ts create mode 100644 .github/scripts/issue-handoff-brief.ts create mode 100644 .github/scripts/issue-llm-config.ts create mode 100644 .github/scripts/issue-llm-evaluator.ts create mode 100644 .github/scripts/issue-llm-provider.ts create mode 100644 .github/scripts/issue-llm-types.ts create mode 100644 .github/scripts/issue-triage-config.ts create mode 100644 .github/scripts/issue-triage-lib.ts create mode 100644 .github/scripts/issue-triage-merge.ts create mode 100644 .github/scripts/issue-triage-types.ts create mode 100644 .github/scripts/issue-triage.ts create mode 100644 .github/workflows/issue-backlog-rescore.yml create mode 100644 .github/workflows/issue-triage.yml create mode 100644 docs/2026-04-08-issue-automation-design.md diff --git a/.github/scripts/github.ts b/.github/scripts/github.ts new file mode 100644 index 00000000..7b9778b5 --- /dev/null +++ b/.github/scripts/github.ts @@ -0,0 +1,230 @@ +interface GitHubUser { + login: string; +} + +interface GitHubLabelRef { + name?: string; +} + +export interface GitHubIssue { + number: number; + title: string; + body: string | null; + state: string; + labels: GitHubLabelRef[]; + comments: number; + created_at: string; + updated_at: string; + user: GitHubUser; + html_url: string; + pull_request?: Record; +} + +export interface GitHubIssueComment { + id: number; + body: string; + user: GitHubUser; + created_at: string; + updated_at: string; + html_url: string; +} + +export interface GitHubLabelDefinition { + name: string; + color: string; + description: string; +} + +function buildApiUrl(path: string) { + return `https://api.github.com${path}`; +} + +export class GitHubClient { + constructor( + private readonly token: string, + private readonly owner: string, + private readonly repo: string, + ) {} + + async getIssue(issueNumber: number): Promise { + return this.request( + "GET", + `/repos/${this.owner}/${this.repo}/issues/${issueNumber}`, + ); + } + + async listIssueComments(issueNumber: number): Promise { + return this.paginate( + `/repos/${this.owner}/${this.repo}/issues/${issueNumber}/comments?per_page=100`, + ); + } + + async listOpenIssuesByLabel( + label: string, + limit = 0, + ): Promise { + const collected: GitHubIssue[] = []; + const unlimited = limit === 0; + let page = 1; + + while (unlimited || collected.length < limit) { + const pageItems = await this.request( + "GET", + `/repos/${this.owner}/${this.repo}/issues?state=open&labels=${ + encodeURIComponent(label) + }&per_page=100&page=${page}`, + ); + + const nonPrIssues = pageItems.filter((item) => !item.pull_request); + collected.push(...nonPrIssues); + + if (pageItems.length < 100) { + break; + } + + page += 1; + } + + return unlimited ? collected : collected.slice(0, limit); + } + + async replaceIssueLabels(issueNumber: number, labels: string[]) { + await this.request( + "PUT", + `/repos/${this.owner}/${this.repo}/issues/${issueNumber}/labels`, + { labels }, + ); + } + + async upsertLabel(definition: GitHubLabelDefinition) { + const encodedName = encodeURIComponent(definition.name); + + try { + await this.request( + "PATCH", + `/repos/${this.owner}/${this.repo}/labels/${encodedName}`, + { + new_name: definition.name, + color: definition.color, + description: definition.description, + }, + ); + } catch (error) { + if (!(error instanceof GitHubApiError) || error.status !== 404) { + throw error; + } + + await this.request("POST", `/repos/${this.owner}/${this.repo}/labels`, { + name: definition.name, + color: definition.color, + description: definition.description, + }); + } + } + + async createIssueComment(issueNumber: number, body: string) { + return this.request( + "POST", + `/repos/${this.owner}/${this.repo}/issues/${issueNumber}/comments`, + { body }, + ); + } + + async updateIssueComment(commentId: number, body: string) { + return this.request( + "PATCH", + `/repos/${this.owner}/${this.repo}/issues/comments/${commentId}`, + { body }, + ); + } + + private async paginate(path: string): Promise { + const collected: T[] = []; + let nextPath: string | null = path; + + while (nextPath) { + const response = await fetch(buildApiUrl(nextPath), { + headers: this.headers(), + }); + + if (!response.ok) { + throw await GitHubApiError.fromResponse(response); + } + + const pageItems = (await response.json()) as T[]; + collected.push(...pageItems); + nextPath = parseNextLink(response.headers.get("link")); + } + + return collected; + } + + private async request( + method: string, + path: string, + body?: unknown, + ): Promise { + const response = await fetch(buildApiUrl(path), { + method, + headers: this.headers(), + body: body ? JSON.stringify(body) : undefined, + }); + + if (!response.ok) { + throw await GitHubApiError.fromResponse(response); + } + + if (response.status === 204) { + return undefined as T; + } + + return (await response.json()) as T; + } + + private headers() { + return { + Accept: "application/vnd.github+json", + Authorization: `Bearer ${this.token}`, + "Content-Type": "application/json", + "User-Agent": "skillhub-issue-triage", + "X-GitHub-Api-Version": "2022-11-28", + }; + } +} + +export class GitHubApiError extends Error { + constructor( + readonly status: number, + readonly responseBody: string, + ) { + super(`GitHub API request failed with status ${status}: ${responseBody}`); + } + + static async fromResponse(response: Response) { + return new GitHubApiError(response.status, await response.text()); + } +} + +function parseNextLink(linkHeader: string | null) { + if (!linkHeader) { + return null; + } + + const nextEntry = linkHeader + .split(",") + .map((item) => item.trim()) + .find((item) => item.endsWith('rel="next"')); + + if (!nextEntry) { + return null; + } + + const urlMatch = nextEntry.match(/<([^>]+)>/); + + if (!urlMatch) { + return null; + } + + const url = new URL(urlMatch[1]); + return `${url.pathname}${url.search}`; +} diff --git a/.github/scripts/issue-backlog-rescore.ts b/.github/scripts/issue-backlog-rescore.ts new file mode 100644 index 00000000..dd0cf9c0 --- /dev/null +++ b/.github/scripts/issue-backlog-rescore.ts @@ -0,0 +1,128 @@ +import { GitHubClient } from "./github.ts"; +import { readIssueLlmConfig, shouldUseLlm } from "./issue-llm-config.ts"; +import { evaluateIssueWithLlm } from "./issue-llm-evaluator.ts"; +import { TRIAGE_MANUAL_OVERRIDE_LABEL } from "./issue-triage-config.ts"; +import { + analyzeIssue, + buildManagedLabels, + ensureManagedLabels, + findTriageComment, + parseTriageMachineState, + previewTriageMutation, + syncManagedLabels, + upsertTriageComment, +} from "./issue-triage-lib.ts"; +import { mergeRuleAndLlm } from "./issue-triage-merge.ts"; + +function readFlag(name: string) { + const index = Deno.args.indexOf(`--${name}`); + return index >= 0 ? Deno.args[index + 1] : undefined; +} + +function hasFlag(name: string) { + return Deno.args.includes(`--${name}`); +} + +const owner = readFlag("owner"); +const repo = readFlag("repo"); +const limitValue = readFlag("limit") ?? "0"; +const dryRun = hasFlag("dry-run"); +const token = Deno.env.get("GH_TOKEN") ?? Deno.env.get("GITHUB_TOKEN"); + +if (!owner || !repo || !token) { + throw new Error( + "Usage: deno run issue-backlog-rescore.ts --owner --repo [--limit 0 for all] with GH_TOKEN set.", + ); +} + +const limit = Number.parseInt(limitValue, 10); + +if (Number.isNaN(limit) || limit < 0) { + throw new Error(`Invalid limit: ${limitValue}`); +} + +const client = new GitHubClient(token, owner, repo); +if (!dryRun) { + await ensureManagedLabels(client); +} +const llmConfig = readIssueLlmConfig(); + +const issues = await client.listOpenIssuesByLabel("triage/deferred", limit); +const dryRunResults: Array> = []; + +for (const issue of issues) { + if ( + issue.labels.some((label) => label.name === TRIAGE_MANUAL_OVERRIDE_LABEL) + ) { + console.log( + `Skipping #${issue.number} because ${TRIAGE_MANUAL_OVERRIDE_LABEL} is set.`, + ); + continue; + } + + const comments = await client.listIssueComments(issue.number); + const ruleResult = analyzeIssue(issue, comments); + const existingComment = findTriageComment(comments); + const previousState = existingComment + ? parseTriageMachineState(existingComment.body) + : null; + let result = ruleResult; + + if (llmConfig) { + const llmDecision = shouldUseLlm(issue, ruleResult); + + if (llmDecision.use) { + const { inputHash, assessment } = await evaluateIssueWithLlm( + llmConfig, + issue, + comments, + ruleResult, + previousState, + ); + + result = mergeRuleAndLlm({ + ...ruleResult, + inputHash, + llm: assessment, + mode: assessment.mode === "assist" ? "llm-assist" : "llm-shadow", + }); + } + } + + if (dryRun) { + const preview = previewTriageMutation(result, comments); + dryRunResults.push({ + issue: issue.number, + mode: result.mode, + route: result.route, + priority: result.priority, + effort: result.effort, + confidence: result.confidence, + riskLevel: result.riskLevel, + labels: preview.labels, + commentAction: preview.existingComment ? "update" : "create", + commentBody: preview.commentBody, + }); + continue; + } + + await syncManagedLabels(client, issue, result); + await upsertTriageComment(client, issue.number, result, comments); + + console.log( + JSON.stringify( + { + issue: issue.number, + route: result.route, + priority: result.priority, + labels: buildManagedLabels(issue, result), + }, + null, + 2, + ), + ); +} + +if (dryRun) { + console.log(JSON.stringify({ dryRun: true, issues: dryRunResults }, null, 2)); +} diff --git a/.github/scripts/issue-handoff-brief.ts b/.github/scripts/issue-handoff-brief.ts new file mode 100644 index 00000000..e19dacaa --- /dev/null +++ b/.github/scripts/issue-handoff-brief.ts @@ -0,0 +1,257 @@ +import { MaintainerHandoffBrief, TriageResult } from "./issue-triage-types.ts"; + +const AREA_RULES: Array<{ keywords: string[]; area: string }> = [ + { + keywords: ["clawhub publish", "publish skill", "publish", "namespace"], + area: + "CLI 发布命令参数解析与 namespace 感知发布流程 / CLI publish command option parsing and namespace-aware publish flow", + }, + { + keywords: ["clawhub install", "install skill", "install"], + area: + "技能安装流程与 registry/lockfile 集成 / Skill installation flow and registry/lockfile integration", + }, + { + keywords: ["clawhub update", "update skill", "update"], + area: + "已安装技能更新流程与版本解析 / Installed skill update flow and version resolution", + }, + { + keywords: ["clawhub sync", "sync skill", "sync"], + area: + "本地技能同步流程与发布 diff 检测 / Local skill sync flow and publish diff detection", + }, + { + keywords: ["inspect", "search", "explore"], + area: + "Registry 发现与 CLI 查询流程 / Registry discovery and CLI query workflow", + }, + { + keywords: ["auth", "login", "ldap", "sso", "token"], + area: + "认证、会话与身份集成 / Authentication, session, and identity integration", + }, + { + keywords: ["openapi", "sdk", "api contract", "contract"], + area: + "公开 API 契约、生成 SDK 与兼容性表面 / Public API contract, generated SDKs, and compatibility surface", + }, + { + keywords: ["docs", "documentation", "manual", "help", "--help"], + area: + "文档、操作指引与 CLI help 输出 / Documentation, operator guidance, and CLI help output", + }, + { + keywords: ["scanner", "security", "audit"], + area: + "安全扫描流程与审计/报告行为 / Security scanner pipeline and audit/reporting behavior", + }, +]; + +export function buildMaintainerHandoffBrief( + result: TriageResult, +): MaintainerHandoffBrief | undefined { + if (result.route !== "core") { + return undefined; + } + + const summary = buildSummary(result); + const whyCore = unique([ + result.requiresCoreMaintainer + ? "阻塞 OpenClaw/ClawHub 核心工作流,因此即便改动范围看起来可控,也需要 maintainer judgment / Blocks an OpenClaw/ClawHub core workflow, so maintainer judgment is required even if the code change looks bounded." + : "", + result.riskLevel === "high" + ? "触及高风险区域,未经 maintainer 审查不应直接信任自动修复 / Touches a higher-risk area where automated fixes should not be trusted without maintainer review." + : "", + result.effort >= 4 + ? "大概率跨多个模块或公共兼容面 / Likely spans multiple modules or a public compatibility surface." + : "", + result.confidence <= 3 + ? "问题本身重要,但仍需要 maintainer 先收敛范围再实施 / The issue is important, but a maintainer still needs to tighten scope before implementation." + : "", + ...result.highRiskReasons, + ]).slice(0, 4); + + const reproduction = buildReproduction(result); + const suspectedAreas = inferSuspectedAreas(result); + const risks = buildRisks(result, suspectedAreas); + const validation = buildValidation(result, suspectedAreas); + + return { + summary, + whyCore, + reproduction, + suspectedAreas, + risks, + validation, + }; +} + +function buildSummary(result: TriageResult) { + const llmSummary = result.llm?.summaryZh ?? result.llm?.summary ?? + result.llm?.summaryEn; + + if (llmSummary && llmSummary.trim().length > 0) { + return llmSummary.trim(); + } + + const preferred = [ + result.sections["summary"], + result.sections["problem"], + result.sections["expected behavior"], + ].find((value) => value && value.trim().length > 0); + + if (preferred) { + return compact(preferred); + } + + return result.issue.title.replace(/^\[[^\]]+\]\s*/, "").trim(); +} + +function buildReproduction(result: TriageResult) { + const commandFocusedSteps = extractCommandAndErrorLines( + result.sections["steps to reproduce"], + ); + + if (commandFocusedSteps.length > 0) { + return commandFocusedSteps.slice(0, 4); + } + + const steps = splitIntoBullets(result.sections["steps to reproduce"]); + + if (steps.length > 0) { + return steps.slice(0, 5); + } + + const problem = splitIntoBullets(result.sections["problem"]); + + if (problem.length > 0) { + return problem.slice(0, 4); + } + + return [ + "按 issue 中描述的操作路径复现,并确认当前失败模式 / Recreate the operator flow described in the issue and confirm the current failure mode.", + ]; +} + +function inferSuspectedAreas(result: TriageResult) { + const text = [ + result.issue.title, + result.sections["summary"] ?? "", + result.sections["problem"] ?? "", + result.sections["steps to reproduce"] ?? "", + result.sections["impact"] ?? "", + result.sections["api contract impact"] ?? "", + result.sections["contract or sdk impact"] ?? "", + ] + .join("\n") + .toLowerCase(); + + const areas = AREA_RULES.filter((rule) => + rule.keywords.some((keyword) => text.includes(keyword)) + ).map((rule) => rule.area); + + if (areas.length > 0) { + return unique(areas).slice(0, 5); + } + + return [ + "最接近该失败路径的 owner-facing 工作流模块 / The closest owner-facing workflow module for the issue's reported failure path", + "当前对外承诺该行为的文档或 help 文本 / Any docs or help text that currently promise the affected behavior", + ]; +} + +function buildRisks(result: TriageResult, suspectedAreas: string[]) { + const risks = unique([ + ...result.highRiskReasons, + result.llm?.riskFlags.includes("cli-protocol") + ? "CLI 行为、文档和操作预期可能发生漂移,需要同步更新命令 help 与兼容性说明 / CLI behavior, docs, and operator expectations may drift unless command help and compatibility notes are updated together." + : "", + suspectedAreas.some((area) => area.toLowerCase().includes("namespace")) + ? "namespace 范围行为如果没有保留 fallback routing,可能回归默认 publish/install 流程 / Namespace-scoped behavior can regress default publish/install flows if fallback routing is not preserved." + : "", + result.requiresCoreMaintainer + ? "该问题影响已定义主流程,回归会很快被终端用户感知 / This issue affects a documented primary workflow, so regressions would be visible to end users quickly." + : "", + ]); + + return risks.length > 0 ? risks.slice(0, 4) : [ + "合并前检查相邻用户路径是否出现回归 / Check for regressions in adjacent user-facing workflow paths before merging.", + ]; +} + +function buildValidation(result: TriageResult, suspectedAreas: string[]) { + const validation = unique([ + result.sections["steps to reproduce"] + ? "按 issue 中的复现步骤逐条回放,确认报告的问题已消失 / Replay the exact reproduction steps from the issue and confirm the reported failure disappears." + : "修复后端到端验证主报告流程 / Validate the primary reported workflow end-to-end after the fix.", + result.sections["expected behavior"] + ? `确认最终行为符合 issue 期望 / Confirm the final behavior matches the issue's expected outcome: ${ + compact(result.sections["expected behavior"]) + }` + : "", + suspectedAreas.some((area) => area.toLowerCase().includes("documentation")) + ? "更新或核对文档与 CLI help 输出,确保其与实现行为一致 / Update or verify documentation and CLI help output so they match the implemented behavior." + : "", + suspectedAreas.some((area) => area.toLowerCase().includes("api contract")) + ? "发布前检查下游 API/SDK/CLI 的兼容性预期 / Check for downstream API/SDK/CLI compatibility expectations before shipping." + : "", + suspectedAreas.some((area) => area.toLowerCase().includes("namespace")) + ? "同时验证 namespace 范围行为与默认非 namespace 流程 / Verify both namespace-scoped behavior and the default non-namespace flow." + : "", + result.requiresCoreMaintainer + ? "围绕受影响的 OpenClaw/ClawHub 用户路径执行最小必要回归测试 / Run the smallest relevant regression test around the affected OpenClaw/ClawHub user journey." + : "", + ]); + + return validation.slice(0, 5); +} + +function splitIntoBullets(value: string | undefined) { + if (!value) { + return []; + } + + return value + .split("\n") + .map((line) => line.trim()) + .filter((line) => + line.length > 0 && + line !== "```" && + !line.startsWith("PS ") && + !line.startsWith("Usage:") && + !line.startsWith("Options:") && + !line.startsWith("Arguments:") + ) + .map((line) => line.replace(/^[*-]\s*/, "")) + .slice(0, 6); +} + +function extractCommandAndErrorLines(value: string | undefined) { + if (!value) { + return []; + } + + return value + .split("\n") + .map((line) => line.trim()) + .filter((line) => + line.length > 0 && + ( + line.toLowerCase().includes("clawhub ") || + line.toLowerCase().startsWith("error:") || + line.toLowerCase().includes("unknown option") || + line.toLowerCase().includes("usage:") + ) + ) + .map((line) => line.replace(/^[>*-]\s*/, "")) + .slice(0, 4); +} + +function compact(value: string) { + return value.replace(/\s+/g, " ").trim(); +} + +function unique(values: string[]) { + return [...new Set(values.filter((value) => value.trim().length > 0))]; +} diff --git a/.github/scripts/issue-llm-config.ts b/.github/scripts/issue-llm-config.ts new file mode 100644 index 00000000..f204d89b --- /dev/null +++ b/.github/scripts/issue-llm-config.ts @@ -0,0 +1,139 @@ +import { GitHubIssue } from "./github.ts"; +import { IssueLlmConfig } from "./issue-llm-types.ts"; +import { TriageResult } from "./issue-triage-types.ts"; + +const DEFAULT_TIMEOUT_MS = 30000; +const DEFAULT_MAX_ATTEMPTS = 2; +const DEFAULT_RETRY_BACKOFF_MS = 1500; +const DEFAULT_TEMPERATURE = 0.1; +const DEFAULT_MAX_COMMENTS = 4; +const DEFAULT_MAX_COMMENT_CHARS = 900; +const DEFAULT_MAX_BODY_CHARS = 6000; + +export function readIssueLlmConfig(): IssueLlmConfig | null { + const mode = normalizeMode(Deno.env.get("ISSUE_TRIAGE_LLM_MODE")); + + if (mode === "off") { + return null; + } + + const baseUrl = normalizeUrl(Deno.env.get("ISSUE_TRIAGE_LLM_BASE_URL")); + const apiKey = Deno.env.get("ISSUE_TRIAGE_LLM_API_KEY")?.trim() ?? ""; + const model = Deno.env.get("ISSUE_TRIAGE_LLM_MODEL")?.trim() ?? ""; + + if (!baseUrl || !apiKey || !model) { + console.warn( + "LLM triage is configured in a non-off mode but base URL, model, or API key is missing. Falling back to rules-only.", + ); + return null; + } + + return { + mode, + provider: "openai-compatible", + baseUrl, + apiKey, + model, + timeoutMs: parseInteger( + Deno.env.get("ISSUE_TRIAGE_LLM_TIMEOUT_MS"), + DEFAULT_TIMEOUT_MS, + ), + maxAttempts: Math.max( + 1, + parseInteger( + Deno.env.get("ISSUE_TRIAGE_LLM_MAX_ATTEMPTS"), + DEFAULT_MAX_ATTEMPTS, + ), + ), + retryBackoffMs: Math.max( + 0, + parseInteger( + Deno.env.get("ISSUE_TRIAGE_LLM_RETRY_BACKOFF_MS"), + DEFAULT_RETRY_BACKOFF_MS, + ), + ), + temperature: parseFloatSetting( + Deno.env.get("ISSUE_TRIAGE_LLM_TEMPERATURE"), + DEFAULT_TEMPERATURE, + ), + maxComments: parseInteger( + Deno.env.get("ISSUE_TRIAGE_LLM_MAX_COMMENTS"), + DEFAULT_MAX_COMMENTS, + ), + maxCommentChars: parseInteger( + Deno.env.get("ISSUE_TRIAGE_LLM_MAX_COMMENT_CHARS"), + DEFAULT_MAX_COMMENT_CHARS, + ), + maxBodyChars: parseInteger( + Deno.env.get("ISSUE_TRIAGE_LLM_MAX_BODY_CHARS"), + DEFAULT_MAX_BODY_CHARS, + ), + }; +} + +export function shouldUseLlm(issue: GitHubIssue, result: TriageResult) { + const reasons: string[] = []; + + if (result.route === "needs-info") { + reasons.push("route-needs-info"); + } + + if (result.route === "core") { + reasons.push("route-core"); + } + + if (result.priority >= 3 && result.priority <= 4.2) { + reasons.push("priority-near-threshold"); + } + + if (result.confidence <= 3) { + reasons.push("confidence-low"); + } + + if (issue.comments >= 4) { + reasons.push("discussion-heavy"); + } + + if ((issue.body ?? "").length >= 1200) { + reasons.push("body-long"); + } + + if (result.issueKind === "feature" || result.issueKind === "reward") { + reasons.push("non-bug-judgment"); + } + + return { + use: reasons.length > 0, + reasons, + }; +} + +function normalizeMode(raw: string | undefined | null) { + const value = raw?.trim().toLowerCase(); + + if (value === "shadow" || value === "assist") { + return value; + } + + return "off"; +} + +function normalizeUrl(value: string | undefined | null) { + const trimmed = value?.trim(); + + if (!trimmed) { + return ""; + } + + return trimmed.endsWith("/") ? trimmed.slice(0, -1) : trimmed; +} + +function parseInteger(raw: string | undefined, fallback: number) { + const parsed = Number.parseInt(raw ?? "", 10); + return Number.isNaN(parsed) ? fallback : parsed; +} + +function parseFloatSetting(raw: string | undefined, fallback: number) { + const parsed = Number.parseFloat(raw ?? ""); + return Number.isNaN(parsed) ? fallback : parsed; +} diff --git a/.github/scripts/issue-llm-evaluator.ts b/.github/scripts/issue-llm-evaluator.ts new file mode 100644 index 00000000..c9ac93ef --- /dev/null +++ b/.github/scripts/issue-llm-evaluator.ts @@ -0,0 +1,466 @@ +import { GitHubIssue, GitHubIssueComment } from "./github.ts"; +import { + IssueLlmConfig, + IssueLlmPayload, + IssueLlmResponse, +} from "./issue-llm-types.ts"; +import { requestOpenAiCompatibleJson } from "./issue-llm-provider.ts"; +import { + IssueRoute, + LlmAssessment, + TriageMachineState, + TriageResult, +} from "./issue-triage-types.ts"; + +const ALLOWED_RISK_FLAGS = new Set([ + "auth", + "security", + "token", + "permission", + "migration", + "schema", + "api-contract", + "sdk", + "cli-protocol", + "data-loss", +]); +const PROMPT_VERSION = 3; + +export async function evaluateIssueWithLlm( + config: IssueLlmConfig, + issue: GitHubIssue, + comments: GitHubIssueComment[], + ruleResult: TriageResult, + previousState: TriageMachineState | null, +) { + const payload = buildPayload(config, issue, comments, ruleResult); + const inputHash = await buildIssueInputHash(payload); + const cached = previousState?.llm; + + if ( + cached && + cached.inputHash === inputHash && + cached.provider === config.provider && + cached.model === config.model && + cached.mode === config.mode && + !cached.failed + ) { + return { + inputHash, + assessment: { + ...cached, + reused: true, + } as LlmAssessment, + }; + } + + try { + const rawJson = await requestOpenAiCompatibleJson( + config, + buildSystemPrompt(), + JSON.stringify(payload, null, 2), + ); + const parsed = validateLlmResponse(JSON.parse(rawJson), ruleResult); + + return { + inputHash, + assessment: { + provider: config.provider, + model: config.model, + mode: config.mode, + inputHash, + summary: parsed.summary_zh || parsed.summary || parsed.summary_en || "", + summaryEn: parsed.summary_en || parsed.summary || parsed.summary_zh || + "", + summaryZh: parsed.summary_zh || parsed.summary || parsed.summary_en || + "", + impact: parsed.impact, + urgency: parsed.urgency, + effort: parsed.effort, + confidence: parsed.confidence, + riskFlags: parsed.risk_flags, + missingInfo: parsed.missing_info, + suggestedQuestions: parsed.suggested_questions, + recommendedRoute: parsed.recommended_route, + rationale: parsed.rationale, + reused: false, + failed: false, + } satisfies LlmAssessment, + }; + } catch (error) { + const failureReason = error instanceof Error + ? error.message + : String(error); + + return { + inputHash, + assessment: { + provider: config.provider, + model: config.model, + mode: config.mode, + inputHash, + summary: "", + summaryEn: "", + summaryZh: "", + impact: ruleResult.impact, + urgency: ruleResult.urgency, + effort: ruleResult.effort, + confidence: ruleResult.confidence, + riskFlags: [], + missingInfo: [], + suggestedQuestions: [], + recommendedRoute: ruleResult.route, + rationale: [], + reused: false, + failed: true, + failureReason, + } satisfies LlmAssessment, + }; + } +} + +function buildPayload( + config: IssueLlmConfig, + issue: GitHubIssue, + comments: GitHubIssueComment[], + ruleResult: TriageResult, +): IssueLlmPayload { + const latestComments = comments + .filter((comment) => + !comment.body.includes("`; +} + +function calculateConfidence( + issueKind: IssueKind, + rawBody: string, + sections: Record, + missingFields: string[], +) { + const required = requiredFields(issueKind); + const requiredFilled = + required.filter((field) => hasMeaningfulSection(sections[field])).length; + const supportFields = Object.entries(sections).filter( + ([key, value]) => !required.includes(key) && hasMeaningfulSection(value), + ).length; + + let score = 1; + score += requiredFilled; + score += supportFields >= 1 ? 0.5 : 0; + score += supportFields >= 3 ? 0.5 : 0; + score += rawBody.length >= 400 ? 0.5 : 0; + score -= missingFields.length > 0 ? 1 : 0; + + return clamp(Math.round(score), 1, 5); +} + +function calculateAgePolicy(createdAt: string, now: Date) { + const created = new Date(createdAt); + const openDays = Math.floor( + (now.getTime() - created.getTime()) / (24 * 60 * 60 * 1000), + ); + const safeOpenDays = Math.max(0, openDays); + + if (safeOpenDays >= 14) { + return { + openDays: safeOpenDays, + ageBoost: 1.5, + priorityFloor: 4.4, + reason: + `已打开 ${safeOpenDays} 天,超过 14 天闭环 SLA,优先级强制提升到 P0 / Open for ${safeOpenDays} days; the 14-day closure SLA is breached, so priority is forced to P0.`, + }; + } + + if (safeOpenDays >= 10) { + return { + openDays: safeOpenDays, + ageBoost: 1, + priorityFloor: 3.6, + reason: + `已打开 ${safeOpenDays} 天,为避免超过 14 天仍未闭环,强制进入 active lane / Open for ${safeOpenDays} days; forced into an active lane before the 14-day closure SLA is missed.`, + }; + } + + if (safeOpenDays >= 7) { + return { + openDays: safeOpenDays, + ageBoost: 0.6, + priorityFloor: 2.6, + reason: + `已打开 ${safeOpenDays} 天,开始进入 2 周闭环预热窗口 / Open for ${safeOpenDays} days; entering the 2-week closure warm-up window.`, + }; + } + + return { + openDays: safeOpenDays, + ageBoost: 0, + priorityFloor: 0, + reason: "", + }; +} + +function calculateEngagementBoost( + commentCount: number, + rewardAmountText?: string, +) { + let boost = Math.min(0.8, commentCount * 0.1); + const rewardAmount = Number.parseFloat( + (rewardAmountText ?? "").replaceAll(/[^0-9.]/g, ""), + ); + + if (!Number.isNaN(rewardAmount)) { + if (rewardAmount >= 500) { + boost += 0.6; + } else if (rewardAmount >= 100) { + boost += 0.3; + } else if (rewardAmount > 0) { + boost += 0.1; + } + } + + return Math.min(1, boost); +} + +function requiredFields(issueKind: IssueKind) { + return REQUIRED_SECTIONS[issueKind] ?? []; +} + +function buildSearchText(issue: GitHubIssue, sections: Record) { + return [issue.title, issue.body ?? "", ...Object.values(sections)].join("\n") + .toLowerCase(); +} + +function buildRiskText(issue: GitHubIssue, sections: Record) { + const preferredSections = [ + "summary", + "problem", + "proposed solution", + "expected behavior", + "steps to reproduce", + "impact", + "api contract impact", + "contract or sdk impact", + ]; + + return [ + issue.title, + ...preferredSections.map((section) => sections[section] ?? ""), + ] + .join("\n") + .toLowerCase(); +} + +function buildWorkflowText( + issue: GitHubIssue, + sections: Record, +) { + const preferredSections = [ + "summary", + "problem", + "steps to reproduce", + "expected behavior", + "impact", + ]; + + return [ + issue.title, + ...preferredSections.map((section) => sections[section] ?? ""), + ] + .join("\n") + .toLowerCase(); +} + +function normalizeHeading(value: string) { + return value.trim().toLowerCase(); +} + +function cleanupSectionContent(value: string) { + return value + .replaceAll(/^_No response_\s*$/gim, "") + .replaceAll(/^no response\s*$/gim, "") + .trim(); +} + +function hasMeaningfulSection(value: string | undefined) { + return Boolean(value && cleanupSectionContent(value).length >= 3); +} + +function uniqueNonEmpty(values: string[]) { + return [...new Set(values.filter((value) => value.trim().length > 0))]; +} + +function clamp(value: number, min: number, max: number) { + return Math.max(min, Math.min(max, value)); +} + +function roundToOneDecimal(value: number) { + return Math.round(value * 10) / 10; +} + +export function findTriageComment(comments: GitHubIssueComment[]) { + return comments.find((comment) => + comment.body.includes(TRIAGE_COMMENT_MARKER) + ); +} + +export function buildManagedLabels(issue: GitHubIssue, result: TriageResult) { + const existingLabels = issue.labels + .map((label) => label.name) + .filter((label): label is string => Boolean(label)); + + const unmanagedLabels = existingLabels.filter( + (label) => + !MANAGED_LABEL_PREFIXES.some((prefix) => label.startsWith(prefix)), + ); + + return [ + ...unmanagedLabels, + routeLabel(result.route), + priorityLabel(result.priority), + effortLabel(result.effort), + ...riskLabels(result.riskLevel), + ]; +} + +export function previewTriageMutation( + result: TriageResult, + comments: GitHubIssueComment[], +) { + return { + labels: uniqueNonEmpty(buildManagedLabels(result.issue, result)), + commentBody: renderTriageComment(result), + existingComment: findTriageComment(comments) ?? null, + }; +} + +export function parseTriageMachineState( + commentBody: string, +): TriageMachineState | null { + const start = commentBody.indexOf(TRIAGE_COMMENT_MARKER); + + if (start < 0) { + return null; + } + + const jsonStart = start + TRIAGE_COMMENT_MARKER.length; + const end = commentBody.indexOf("-->", jsonStart); + + if (end < 0) { + return null; + } + + const rawJson = commentBody.slice(jsonStart, end).trim(); + + try { + const parsed = JSON.parse(rawJson) as TriageMachineState; + + if ( + typeof parsed !== "object" || + parsed === null || + typeof parsed.issue !== "number" || + typeof parsed.route !== "string" + ) { + return null; + } + + return parsed; + } catch { + return null; + } +} diff --git a/.github/scripts/issue-triage-merge.ts b/.github/scripts/issue-triage-merge.ts new file mode 100644 index 00000000..63c162dc --- /dev/null +++ b/.github/scripts/issue-triage-merge.ts @@ -0,0 +1,166 @@ +import { TriageResult, TriageSnapshot } from "./issue-triage-types.ts"; +import { buildMaintainerHandoffBrief } from "./issue-handoff-brief.ts"; + +export function mergeRuleAndLlm(ruleResult: TriageResult): TriageResult { + const llm = ruleResult.llm; + + if (!llm || llm.failed || llm.mode !== "assist") { + return { + ...ruleResult, + handoffBrief: ruleResult.route === "core" + ? buildMaintainerHandoffBrief(ruleResult) + : undefined, + mode: llm && !llm.failed && llm.mode === "shadow" + ? "llm-shadow" + : "rules-only", + inputHash: llm?.inputHash ?? ruleResult.inputHash, + }; + } + + const impact = nudgeScore(ruleResult.impact, llm.impact); + const urgency = nudgeScore(ruleResult.urgency, llm.urgency); + const effort = nudgeScore(ruleResult.effort, llm.effort); + const confidence = nudgeScore(ruleResult.confidence, llm.confidence); + const missingFields = unique([ + ...ruleResult.missingFields, + ...llm.missingInfo, + ]); + const highRiskReasons = unique([ + ...ruleResult.highRiskReasons, + ...llm.riskFlags.map((flag) => + `LLM 标记了高风险区域:${flag} / LLM flagged high-risk area: ${flag}.` + ), + ]); + const requiresCoreMaintainer = ruleResult.requiresCoreMaintainer; + const riskLevel = highRiskReasons.length > 0 ? "high" : "low"; + const priority = clamp( + roundToOneDecimal( + impact * 0.45 + + urgency * 0.35 + + ruleResult.ageBoost + + ruleResult.engagementBoost, + ), + 1, + 5, + ); + const route = determineRoute( + priority, + effort, + confidence, + riskLevel, + missingFields, + requiresCoreMaintainer, + ); + const nextAction = describeNextAction(route, missingFields); + const reasons = unique([ + ...ruleResult.reasons, + ...llm.rationale, + llm.summary + ? `LLM 摘要:${llm.summaryZh || llm.summary} / LLM summary: ${ + llm.summaryEn || llm.summary + }` + : "", + ]).slice(0, 6); + + const mergedSnapshot: TriageSnapshot = { + route, + riskLevel, + requiresCoreMaintainer, + openDays: ruleResult.openDays, + impact, + urgency, + effort, + confidence, + priority, + ageBoost: ruleResult.ageBoost, + priorityFloor: ruleResult.priorityFloor, + engagementBoost: ruleResult.engagementBoost, + missingFields, + reasons, + highRiskReasons, + nextAction, + }; + + return { + ...ruleResult, + ...mergedSnapshot, + mode: "llm-assist", + inputHash: llm.inputHash, + handoffBrief: route === "core" + ? buildMaintainerHandoffBrief({ + ...ruleResult, + ...mergedSnapshot, + mode: "llm-assist", + inputHash: llm.inputHash, + }) + : undefined, + }; +} + +export function determineRoute( + priority: number, + effort: number, + confidence: number, + riskLevel: "low" | "high", + missingFields: string[], + requiresCoreMaintainer = false, +) { + if (requiresCoreMaintainer) { + return "core"; + } + + if (missingFields.length > 0 || confidence <= 2) { + return "needs-info"; + } + + if (priority < 3.6) { + return "deferred"; + } + + if (riskLevel === "high" || effort >= 4 || confidence <= 3) { + return "core"; + } + + return "agent-ready"; +} + +export function describeNextAction( + route: TriageResult["route"], + missingFields: string[], +) { + if (route === "needs-info") { + return `等待补充更多信息;作者更新 issue 或评论 \`/retriage\` 后重新分流 / Wait for more detail, then rerun triage after the author edits the issue or comments \`/retriage\`. Missing: ${ + missingFields.join(", ") + }.`; + } + + if (route === "deferred") { + return "将 issue 保留在 deferred 队列,并由 6 小时一次的 rescore 持续抬升;最晚在第 10 天强制进入 active lane。若第 14 天仍未闭环,应按 SLA 视为 P0 升级目标,并在下一次 triage 中重点处理 / Keep the issue in the deferred queue and let the 6-hour rescore keep lifting it; it is forced into an active lane by day 10. If it is still open on day 14, treat it as a P0 escalation target under the SLA and prioritize it in the next triage pass."; + } + + if (route === "core") { + return "交给 core maintainer,并结合本地编程Agent协助完成复现、收敛范围与验证闭环 / Hand the issue to a core maintainer and use a local programming agent for reproduction, scoping, and validation."; + } + + return "在 self-hosted issue-agent runner 启用后,将其标记为低风险 agent 可执行候选 / Mark as a candidate for low-risk agent execution once the self-hosted issue-agent runner is enabled."; +} + +function nudgeScore(ruleScore: number, llmScore: number) { + if (llmScore === ruleScore) { + return ruleScore; + } + + return clamp(ruleScore + Math.sign(llmScore - ruleScore), 1, 5); +} + +function unique(values: string[]) { + return [...new Set(values.filter((value) => value.trim().length > 0))]; +} + +function clamp(value: number, min: number, max: number) { + return Math.max(min, Math.min(max, value)); +} + +function roundToOneDecimal(value: number) { + return Math.round(value * 10) / 10; +} diff --git a/.github/scripts/issue-triage-types.ts b/.github/scripts/issue-triage-types.ts new file mode 100644 index 00000000..e2830a15 --- /dev/null +++ b/.github/scripts/issue-triage-types.ts @@ -0,0 +1,93 @@ +import { GitHubIssue } from "./github.ts"; + +export type IssueKind = "bug" | "feature" | "reward" | "other"; +export type IssueRoute = "needs-info" | "deferred" | "core" | "agent-ready"; +export type RiskLevel = "low" | "high"; +export type LlmMode = "off" | "shadow" | "assist"; +export type AnalysisMode = "rules-only" | "llm-shadow" | "llm-assist"; + +export interface ParsedIssueBody { + sections: Record; + missingFields: string[]; +} + +export interface TriageSnapshot { + route: IssueRoute; + riskLevel: RiskLevel; + requiresCoreMaintainer: boolean; + openDays: number; + impact: number; + urgency: number; + effort: number; + confidence: number; + priority: number; + ageBoost: number; + priorityFloor: number; + engagementBoost: number; + missingFields: string[]; + reasons: string[]; + highRiskReasons: string[]; + nextAction: string; +} + +export interface MaintainerHandoffBrief { + summary: string; + whyCore: string[]; + reproduction: string[]; + suspectedAreas: string[]; + risks: string[]; + validation: string[]; +} + +export interface LlmAssessment { + provider: string; + model: string; + mode: LlmMode; + inputHash: string; + summary: string; + summaryEn?: string; + summaryZh?: string; + impact: number; + urgency: number; + effort: number; + confidence: number; + riskFlags: string[]; + missingInfo: string[]; + suggestedQuestions: string[]; + recommendedRoute: IssueRoute; + rationale: string[]; + reused: boolean; + failed: boolean; + failureReason?: string; +} + +export interface TriageResult extends TriageSnapshot { + issue: GitHubIssue; + issueKind: IssueKind; + sections: Record; + mode: AnalysisMode; + inputHash: string; + rule: TriageSnapshot; + llm?: LlmAssessment; + handoffBrief?: MaintainerHandoffBrief; +} + +export interface TriageMachineState { + version: number; + issue: number; + inputHash?: string; + mode?: AnalysisMode; + route: IssueRoute; + priority: number; + requiresCoreMaintainer?: boolean; + impact: number; + urgency: number; + effort: number; + confidence: number; + riskLevel: RiskLevel; + ageBoost: number; + engagementBoost: number; + missingFields: string[]; + updatedAt: string; + llm?: LlmAssessment; +} diff --git a/.github/scripts/issue-triage.ts b/.github/scripts/issue-triage.ts new file mode 100644 index 00000000..c9269ac4 --- /dev/null +++ b/.github/scripts/issue-triage.ts @@ -0,0 +1,129 @@ +import { GitHubClient } from "./github.ts"; +import { readIssueLlmConfig, shouldUseLlm } from "./issue-llm-config.ts"; +import { evaluateIssueWithLlm } from "./issue-llm-evaluator.ts"; +import { TRIAGE_MANUAL_OVERRIDE_LABEL } from "./issue-triage-config.ts"; +import { + analyzeIssue, + buildManagedLabels, + ensureManagedLabels, + findTriageComment, + parseTriageMachineState, + previewTriageMutation, + syncManagedLabels, + upsertTriageComment, +} from "./issue-triage-lib.ts"; +import { mergeRuleAndLlm } from "./issue-triage-merge.ts"; + +function readFlag(name: string) { + const index = Deno.args.indexOf(`--${name}`); + return index >= 0 ? Deno.args[index + 1] : undefined; +} + +function hasFlag(name: string) { + return Deno.args.includes(`--${name}`); +} + +const owner = readFlag("owner"); +const repo = readFlag("repo"); +const issueNumberValue = readFlag("issue-number"); +const dryRun = hasFlag("dry-run"); +const token = Deno.env.get("GH_TOKEN") ?? Deno.env.get("GITHUB_TOKEN"); + +if (!owner || !repo || !issueNumberValue || !token) { + throw new Error( + "Usage: deno run issue-triage.ts --owner --repo --issue-number with GH_TOKEN set.", + ); +} + +const issueNumber = Number.parseInt(issueNumberValue, 10); + +if (Number.isNaN(issueNumber)) { + throw new Error(`Invalid issue number: ${issueNumberValue}`); +} + +const client = new GitHubClient(token, owner, repo); +const issue = await client.getIssue(issueNumber); + +if (issue.pull_request) { + console.log(`Skipping #${issue.number} because it is a pull request conversation.`); + Deno.exit(0); +} + +if (issue.labels.some((label) => label.name === TRIAGE_MANUAL_OVERRIDE_LABEL)) { + console.log(`Skipping #${issue.number} because ${TRIAGE_MANUAL_OVERRIDE_LABEL} is set.`); + Deno.exit(0); +} + +const comments = await client.listIssueComments(issueNumber); +const ruleResult = analyzeIssue(issue, comments); +const existingComment = findTriageComment(comments); +const previousState = existingComment + ? parseTriageMachineState(existingComment.body) + : null; +const llmConfig = readIssueLlmConfig(); +let result = ruleResult; + +if (llmConfig) { + const llmDecision = shouldUseLlm(issue, ruleResult); + + if (llmDecision.use) { + const { inputHash, assessment } = await evaluateIssueWithLlm( + llmConfig, + issue, + comments, + ruleResult, + previousState, + ); + + result = mergeRuleAndLlm({ + ...ruleResult, + inputHash, + llm: assessment, + mode: assessment.mode === "assist" ? "llm-assist" : "llm-shadow", + }); + } +} + +if (dryRun) { + const preview = previewTriageMutation(result, comments); + console.log( + JSON.stringify( + { + dryRun: true, + issue: issue.number, + mode: result.mode, + route: result.route, + priority: result.priority, + effort: result.effort, + confidence: result.confidence, + riskLevel: result.riskLevel, + labels: preview.labels, + commentAction: preview.existingComment ? "update" : "create", + commentBody: preview.commentBody, + }, + null, + 2, + ), + ); + Deno.exit(0); +} + +await ensureManagedLabels(client); +await syncManagedLabels(client, issue, result); +await upsertTriageComment(client, issueNumber, result, comments); + +console.log( + JSON.stringify( + { + issue: issue.number, + route: result.route, + priority: result.priority, + effort: result.effort, + confidence: result.confidence, + riskLevel: result.riskLevel, + labels: buildManagedLabels(issue, result), + }, + null, + 2, + ), +); diff --git a/.github/workflows/issue-backlog-rescore.yml b/.github/workflows/issue-backlog-rescore.yml new file mode 100644 index 00000000..99f11939 --- /dev/null +++ b/.github/workflows/issue-backlog-rescore.yml @@ -0,0 +1,51 @@ +name: Issue Backlog Rescore + +on: + schedule: + - cron: "0 */6 * * *" + workflow_dispatch: + inputs: + limit: + description: Maximum number of deferred issues to rescore + required: false + default: "0" + +concurrency: + group: issue-backlog-rescore + cancel-in-progress: false + +permissions: + contents: read + issues: write + +jobs: + rescore: + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - uses: denoland/setup-deno@667a34cdef165d8d2b2e98dde39547c9daac7282 # v2.0.4 + with: + deno-version: v2.x + + - name: Rescore deferred issues + env: + GH_TOKEN: ${{ github.token }} + ISSUE_TRIAGE_LLM_MODE: ${{ vars.ISSUE_TRIAGE_LLM_MODE }} + ISSUE_TRIAGE_LLM_BASE_URL: ${{ vars.ISSUE_TRIAGE_LLM_BASE_URL }} + ISSUE_TRIAGE_LLM_MODEL: ${{ vars.ISSUE_TRIAGE_LLM_MODEL }} + ISSUE_TRIAGE_LLM_TIMEOUT_MS: ${{ vars.ISSUE_TRIAGE_LLM_TIMEOUT_MS }} + ISSUE_TRIAGE_LLM_MAX_ATTEMPTS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_ATTEMPTS }} + ISSUE_TRIAGE_LLM_RETRY_BACKOFF_MS: ${{ vars.ISSUE_TRIAGE_LLM_RETRY_BACKOFF_MS }} + ISSUE_TRIAGE_LLM_TEMPERATURE: ${{ vars.ISSUE_TRIAGE_LLM_TEMPERATURE }} + ISSUE_TRIAGE_LLM_MAX_COMMENTS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_COMMENTS }} + ISSUE_TRIAGE_LLM_MAX_COMMENT_CHARS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_COMMENT_CHARS }} + ISSUE_TRIAGE_LLM_MAX_BODY_CHARS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_BODY_CHARS }} + ISSUE_TRIAGE_LLM_API_KEY: ${{ secrets.ISSUE_TRIAGE_LLM_API_KEY }} + run: | + deno run --allow-env --allow-net \ + .github/scripts/issue-backlog-rescore.ts \ + --owner "${{ github.repository_owner }}" \ + --repo "${{ github.event.repository.name }}" \ + --limit "${{ inputs.limit || '0' }}" diff --git a/.github/workflows/issue-triage.yml b/.github/workflows/issue-triage.yml new file mode 100644 index 00000000..c60d5d9e --- /dev/null +++ b/.github/workflows/issue-triage.yml @@ -0,0 +1,62 @@ +name: Issue Triage + +on: + issues: + types: + - opened + - edited + - reopened + issue_comment: + types: + - created + workflow_dispatch: + inputs: + issue_number: + description: Issue number to re-triage manually + required: true + +concurrency: + group: issue-triage-${{ github.event.issue.number || inputs.issue_number }} + cancel-in-progress: true + +permissions: + contents: read + issues: write + +jobs: + triage: + if: | + github.event_name != 'issue_comment' || + ( + github.event.issue.pull_request == null && + contains(github.event.comment.body, '/retriage') + ) + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - uses: denoland/setup-deno@667a34cdef165d8d2b2e98dde39547c9daac7282 # v2.0.4 + with: + deno-version: v2.x + + - name: Run triage + env: + GH_TOKEN: ${{ github.token }} + ISSUE_TRIAGE_LLM_MODE: ${{ vars.ISSUE_TRIAGE_LLM_MODE }} + ISSUE_TRIAGE_LLM_BASE_URL: ${{ vars.ISSUE_TRIAGE_LLM_BASE_URL }} + ISSUE_TRIAGE_LLM_MODEL: ${{ vars.ISSUE_TRIAGE_LLM_MODEL }} + ISSUE_TRIAGE_LLM_TIMEOUT_MS: ${{ vars.ISSUE_TRIAGE_LLM_TIMEOUT_MS }} + ISSUE_TRIAGE_LLM_MAX_ATTEMPTS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_ATTEMPTS }} + ISSUE_TRIAGE_LLM_RETRY_BACKOFF_MS: ${{ vars.ISSUE_TRIAGE_LLM_RETRY_BACKOFF_MS }} + ISSUE_TRIAGE_LLM_TEMPERATURE: ${{ vars.ISSUE_TRIAGE_LLM_TEMPERATURE }} + ISSUE_TRIAGE_LLM_MAX_COMMENTS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_COMMENTS }} + ISSUE_TRIAGE_LLM_MAX_COMMENT_CHARS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_COMMENT_CHARS }} + ISSUE_TRIAGE_LLM_MAX_BODY_CHARS: ${{ vars.ISSUE_TRIAGE_LLM_MAX_BODY_CHARS }} + ISSUE_TRIAGE_LLM_API_KEY: ${{ secrets.ISSUE_TRIAGE_LLM_API_KEY }} + run: | + deno run --allow-env --allow-net \ + .github/scripts/issue-triage.ts \ + --owner "${{ github.repository_owner }}" \ + --repo "${{ github.event.repository.name }}" \ + --issue-number "${{ github.event.issue.number || inputs.issue_number }}" diff --git a/docs/2026-04-08-issue-automation-design.md b/docs/2026-04-08-issue-automation-design.md new file mode 100644 index 00000000..cdcdcc3b --- /dev/null +++ b/docs/2026-04-08-issue-automation-design.md @@ -0,0 +1,337 @@ +# Issue Automation MVP Design + +## Goal + +Reduce maintainer load by automatically triaging GitHub issues into three +queues: + +- `triage/deferred`: low-priority issues that should age upward over time +- `triage/core`: high-priority or high-risk issues that need core maintainer + ownership +- `triage/agent-ready`: high-priority, low-risk issues that are candidates for + future agent execution + +The MVP does not auto-fix issues yet. It focuses on scoring, routing, labeling, +and keeping the backlog fresh. + +This version supports two execution modes: + +- rules-only triage +- rules + OpenAI-compatible LLM assistance + +## Why This Split + +The initial proposal mixed priority and execution difficulty into one decision. +In practice, the system is easier to tune if it separates: + +- Priority: should we spend time on this issue now? +- Route: who should handle the issue once it is worth doing? + +This lets a high-value but difficult issue stay high priority while still +routing to `triage/core`. + +## Inputs + +The automation reads the live issue title, body, labels, comments, and +timestamps. + +Structured issue form fields are parsed from: + +- [bug_report.yml](../.github/ISSUE_TEMPLATE/bug_report.yml) +- [feature_request.yml](../.github/ISSUE_TEMPLATE/feature_request.yml) +- [reward-task.yml](../.github/ISSUE_TEMPLATE/reward-task.yml) + +## Scoring Model + +Each issue is scored across four axes: + +- `impact` (1-5): user and workflow impact +- `urgency` (1-5): release pressure, breakage, or repeated discussion +- `effort` (1-5): estimated change size and coordination cost +- `confidence` (1-5): how complete and actionable the issue description is + +Priority is computed from: + +```text +priority = impact * 0.45 + urgency * 0.35 + age_boost + engagement_boost +``` + +Where: + +- `age_boost`: SLA-driven escalation + - day 7-9: warm-up, minimum `priority/p2` + - day 10-13: forced out of `triage/deferred`, minimum `priority/p1` + - day 14+: on the next triage/rescore pass, treat the issue as SLA-breached + and raise it to at least `priority/p0` +- `engagement_boost`: comment pressure plus reward amount, capped at +1.0 + +Effort does not directly lower priority in the MVP. It only affects routing. + +## LLM-Assisted Triage + +When configured, the workflow can call an OpenAI-compatible chat completions +API. + +The LLM does not replace the rule engine. It only helps with: + +- issue summaries +- soft score adjustments +- `needs-info` follow-up questions +- better rationale for maintainers +- `triage/core` maintainer handoff briefs + +Hard gates stay in rules: + +- missing required information +- high-risk areas like auth, schema, migration, SDK, or public contract changes +- final promotion into `triage/agent-ready` + +The issue body and comments are treated as untrusted input. The workflow: + +- truncates long bodies and comments before sending them to the model +- tells the model to treat issue text as data, not instructions +- validates the model output against a strict JSON contract +- falls back to rules-only if the provider call or JSON validation fails + +### Modes + +- `off`: rules-only +- `shadow`: call the LLM, show its recommendation, but keep the rule-only route + and labels +- `assist`: let the LLM nudge soft scores by at most `+/-1`, then re-apply hard + gates + +### When the LLM is used + +The workflow only calls the LLM for issues that look ambiguous or high-value, +such as: + +- `triage/needs-info` +- `triage/core` +- issues near the routing threshold +- low-confidence cases +- long issue descriptions or heavy discussion +- feature or reward issues that need more judgment + +## Routing Rules + +1. `triage/needs-info` Trigger when required fields are missing or + `confidence <= 2`. + +2. `triage/deferred` Trigger when `priority < 3.6`, the issue is not blocked on + missing information, and the issue age is still below the SLA escalation + floor. + +3. `triage/core` Trigger when `priority >= 3.6` and any of the following are + true: + - the issue blocks an OpenClaw/ClawHub core workflow such as install, + publish, update, sync, or namespace-based publishing + - `effort >= 4` + - `confidence <= 3` + - high-risk keywords or contract-impact fields are present + +4. `triage/agent-ready` Trigger when `priority >= 3.6`, `effort <= 3`, + `confidence >= 4`, and no high-risk signals are present. + +In `assist` mode, LLM suggestions can nudge `impact`, `urgency`, `effort`, and +`confidence` by at most one point. The rule engine then recomputes priority and +route. + +OpenClaw/ClawHub core workflow issues are a hard gate to `triage/core`; LLM +assistance does not relax that rule. + +## Managed Labels + +The automation owns these label prefixes: + +- `triage/` +- `priority/` +- `effort/` +- `risk/` + +Current concrete labels: + +- `triage/needs-info` +- `triage/deferred` +- `triage/core` +- `triage/agent-ready` +- `priority/p0` +- `priority/p1` +- `priority/p2` +- `priority/p3` +- `effort/s` +- `effort/m` +- `effort/l` +- `risk/high` + +All other labels remain untouched. + +Separately, the automation recognizes one non-managed operator label: + +- `triage-manual`: freeze automated triage updates for that issue + +## Workflows + +### 1. Issue Triage + +File: [issue-triage.yml](../.github/workflows/issue-triage.yml) + +Triggers: + +- `issues.opened` +- `issues.edited` +- `issues.reopened` +- `issue_comment.created` when the comment contains `/retriage` +- `workflow_dispatch` + +Actions: + +- fetch issue and comments +- compute scores and route +- upsert managed labels +- upsert a single triage comment containing both human-readable reasoning and + hidden machine state +- optionally call the OpenAI-compatible provider and merge the result + +### 2. Deferred Backlog Rescore + +File: +[issue-backlog-rescore.yml](../.github/workflows/issue-backlog-rescore.yml) + +Triggers: + +- every 6 hours +- `workflow_dispatch` + +Actions: + +- list all open issues labeled `triage/deferred` +- recompute priority with age and engagement boosts +- promote or keep each issue +- update the triage comment in place +- reuse cached LLM results when the issue content has not changed + +Trial-run note: + +- the scheduled rescore currently scans `triage/deferred` issues only +- this guarantees low-priority backlog will not sit idle in `deferred` past day + 10 +- once an issue has already been promoted out of `deferred`, any later day-14 + escalation depends on a new triage event or a manual `/retriage` +- during trial run, the 14-day rule should be treated as an operational SLA + target, not yet as a repo-wide hard timer + +## Scripts + +New GitHub automation scripts live under +[`.github/scripts`](/Users/wowo/workspace/skillhub/.github/scripts): + +- [github.ts](/Users/wowo/workspace/skillhub/.github/scripts/github.ts): minimal + GitHub REST client +- [issue-triage-config.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-triage-config.ts): + labels, thresholds, and keyword rules +- [issue-llm-config.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-llm-config.ts): + LLM mode, environment variables, and call heuristics +- [issue-llm-provider.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-llm-provider.ts): + OpenAI-compatible chat completions client +- [issue-llm-evaluator.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-llm-evaluator.ts): + prompt construction, JSON validation, and cache key generation +- [issue-triage-lib.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-triage-lib.ts): + parsing, scoring, routing, and comment rendering +- [issue-triage-merge.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-triage-merge.ts): + bounded merge and hard-gate re-application +- [issue-triage.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-triage.ts): + single-issue entrypoint +- [issue-backlog-rescore.ts](/Users/wowo/workspace/skillhub/.github/scripts/issue-backlog-rescore.ts): + deferred queue rescoring entrypoint + +## Configuration + +Set these GitHub repository variables and secrets to enable LLM-assisted triage: + +Repository variables: + +- `ISSUE_TRIAGE_LLM_MODE` +- `ISSUE_TRIAGE_LLM_BASE_URL` +- `ISSUE_TRIAGE_LLM_MODEL` +- `ISSUE_TRIAGE_LLM_TIMEOUT_MS` optional +- `ISSUE_TRIAGE_LLM_TEMPERATURE` optional +- `ISSUE_TRIAGE_LLM_MAX_COMMENTS` optional +- `ISSUE_TRIAGE_LLM_MAX_COMMENT_CHARS` optional +- `ISSUE_TRIAGE_LLM_MAX_BODY_CHARS` optional + +Repository secret: + +- `ISSUE_TRIAGE_LLM_API_KEY` + +Recommended first rollout: + +- `ISSUE_TRIAGE_LLM_MODE=shadow` +- watch the triage comments for a few days +- switch to `assist` once the LLM suggestions look stable + +Example OpenAI-compatible variable set: + +```text +ISSUE_TRIAGE_LLM_MODE=shadow +ISSUE_TRIAGE_LLM_BASE_URL=https://your-provider.example.com/v1 +ISSUE_TRIAGE_LLM_MODEL=gpt-4.1-mini +``` + +## Rollout Plan + +### Phase 1: Now + +- enable triage and backlog rescore +- tune thresholds by observing a few weeks of issue traffic +- let maintainers freeze automation on specific issues via `triage-manual` +- if using an LLM, start in `shadow` mode + +### Phase 2: Maintainer Handoff + +Add an issue-brief generator for `triage/core` issues that prepares: + +- reproduction hints +- likely modules +- risk notes +- validation checklist + +This output can feed local programming-agent sessions and the existing parallel +worktree flow. + +The current MVP now embeds a `Maintainer Brief` section directly into the triage +comment for `triage/core` issues. That brief includes: + +- a concise issue summary +- why the issue was escalated to core +- reproduction or operator path notes +- suspected modules or workflow owners +- risk callouts +- a validation checklist + +### Phase 3: Self-Hosted Issue Agent + +Add a self-hosted runner that listens for `triage/agent-ready` and: + +- creates an isolated branch and worktree +- runs the issue-solving agent +- executes the smallest relevant test set +- opens a draft PR + +This phase should keep hard blockers in place for: + +- auth and permission changes +- security-sensitive changes +- schema or migration work +- public API, SDK, or CLI contract changes + +## Open Tuning Questions + +- Whether comment count alone is enough for engagement boost, or if reactions + should also be fetched +- Whether reward issues should receive a stronger value boost than the current + MVP gives them +- Whether `agent-ready` should require `effort <= 2` instead of `<= 3` +- Whether certain areas like `scanner` should be considered high-risk by default +- Whether some teams should keep `shadow` mode permanently and reserve `assist` + for a narrower repository subset