diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 77eec9ff..a2e608f6 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -4,12 +4,12 @@ "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" }, - "description": "329 production-ready skill packages for Claude AI across 14 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (4, v2.7.0), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~441 Python tools, ~594 reference documents, 48+ agents, 77+ slash commands.", + "description": "338 production-ready skill packages for Claude AI across 16 domains: engineering advanced (76 — incl. 4 Matt Pocock-derived productivity skills + v2.7.3 security-guidance PreToolUse hook), engineering core (51), marketing (47 — incl. v2.7.3 AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), project management (9), business growth (5), finance (4), productivity (4, v2.7.0), marketing top-level (2, v2.7.0), research (8, v2.7.0), business-operations (7, v2.8.0), and commercial (8, v2.8.0). Includes ~441 Python tools, ~594 reference documents, 48+ agents, 77+ slash commands.", "homepage": "https://github.com/alirezarezvani/claude-skills", "repository": "https://github.com/alirezarezvani/claude-skills", "metadata": { - "description": "329 production-ready skill packages across 14 domains (engineering, marketing, product, c-level, project management, RA/QM, business growth, finance, productivity, marketing (top-level), research, business-operations [v2.8.0], commercial [v2.8.0], plus standards). ~444 Python tools, ~598 reference guides, 49+ agents (cs-* + personas), 79+ slash commands. v2.8.0 adds 2 new top-level domains (business-operations + commercial) with 15 new skills, context: fork chaining via Matt Pocock grill-with-docs discipline. v2.7.3 adds AEO + security-guidance PreToolUse hook. Compatible with Claude Code, Codex CLI, Gemini CLI, Cursor, OpenClaw, Hermes Agent, Mistral Vibe, and 5 more coding agents.", - "version": "2.8.0" + "description": "338 production-ready skills across 16 domains (engineering, engineering-core, marketing, product, c-level, compliance-os, project management, RA/QM, business growth, finance, productivity, marketing (top-level), research, research-ops, business-operations, commercial, plus standards). 533 Python tools, 676 reference guides, 51+ agents (cs-* + personas), 87+ slash commands across 62 marketplace plugins. v2.9.0 adds the research-ops domain — enterprise Research Operations (clinical-research + research-finance + market-research + product-research + orchestrator) with per-skill onboarding, customization config, and an opt-in autoresearch bridge. Compatible with Claude Code, Codex CLI, Gemini CLI, Cursor, OpenClaw, Hermes Agent, Mistral Vibe, and 5 more coding agents.", + "version": "2.9.0" }, "plugins": [ { @@ -879,7 +879,7 @@ "category": "development" }, { - "name": "handoff", + "name": "handoff-engineering", "source": "./engineering/handoff", "description": "Conversation-handoff document generator. Compacts the current session into a markdown handoff for a fresh agent — references existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL instead of duplicating them. Derived from Matt Pocock's MIT-licensed handoff with: (1) 3 stdlib Python tools (template generator tailored to 5 next-session emphases, artifact deduplicator across 5 categories of duplication, skill recommender matching content to 14 skills in this repo), (2) 4 references citing 7-8 sources (handoff structure, deduplication discipline, next-session skill matching, companion tooling), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline + mktemp convention preserved verbatim per MIT.", "version": "2.6.0", @@ -971,7 +971,7 @@ "category": "productivity" }, { - "name": "handoff", + "name": "handoff-productivity", "source": "./productivity/handoff", "description": "Compact the current conversation into a handoff document for another agent to pick up. Configurable save location, redaction enforcement, SessionStart auto-load + SessionEnd reminder, self-check fidelity script, --refresh flag, mtime-guarded cleanup. Inspired by Matt Pocock's handoff (MIT).", "version": "2.8.2", @@ -1331,6 +1331,51 @@ "automation" ], "category": "development" + "name": "research-ops-skills", + "source": "./research-ops", + "description": "Enterprise / cross-functional Research Operations domain — the managed counterpart to the academic research/ domain. v2.9.0 ships 5 skills: orchestrator (context: fork) + clinical-research (study design: protocol synopsis + endpoint selection + sample-size/power for means/proportions/survival + phase-gate feasibility) + research-finance (R&D program budgeting with F&A split + burn/runway + capitalize-vs-expense routing + portfolio ROI) + market-research (TAM/SAM/SOM computed both top-down and bottoms-up + survey sampling with FPC and per-segment minima + Kotler segmentation scoring) + product-research (goal-matched study design + method-based saturation with confidence + insight synthesis that flags single-source anecdotes). Hard rules: clinical outputs are estimates with a named clinical owner (never fact), finance outputs surface assumptions and route capex-vs-opex to a named finance owner (never auto-decide), market sizes show method + assumptions (never a single number), product insights require recurrence across independent participants. Each sub-skill ships per-skill onboarding questions (onboard.py), a customization config consumed by every tool, and an isolated opt-in autoresearch evaluator (ar_evaluator.py) bridging to engineering/autoresearch-agent. 24 stdlib Python tools (12 analysis + 12 onboarding/customization/autoresearch), 12 reference docs. Distinct from ra-qm-team (regulatory/QM submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), marketing-skill (campaign analytics).", + "version": "2.9.0", + "author": { + "name": "Alireza Rezvani" + }, + "keywords": [ + "research-ops", + "research-operations", + "clinical-research", + "study-design", + "endpoint-selection", + "sample-size", + "power-analysis", + "phase-gate", + "biostatistics", + "research-finance", + "rd-budget", + "burn-rate", + "runway", + "fa-rate", + "capitalize-vs-expense", + "portfolio-roi", + "market-research", + "tam-sam-som", + "market-sizing", + "survey-design", + "sampling", + "segmentation", + "competitive-intelligence", + "product-research", + "ux-research", + "jtbd", + "usability", + "saturation", + "insight-synthesis", + "research-repository", + "onboarding", + "customization", + "autoresearch", + "matt-pocock", + "grill-with-docs" + ], + "category": "research-ops" } ] } \ No newline at end of file diff --git a/.codex/skills-index.json b/.codex/skills-index.json index cf493d02..97ffe6c5 100644 --- a/.codex/skills-index.json +++ b/.codex/skills-index.json @@ -3,7 +3,7 @@ "name": "claude-code-skills", "description": "Production-ready skill packages for AI agents - Marketing, Engineering, Product, C-Level, PM, and RA/QM", "repository": "https://github.com/alirezarezvani/claude-skills", - "total_skills": 323, + "total_skills": 328, "skills": [ { "name": "business-growth-skills", @@ -567,7 +567,7 @@ "name": "code-reviewer", "source": "../../engineering-team/skills/code-reviewer", "category": "engineering", - "description": "Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, and .NET. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists." + "description": "Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, .NET, and Java. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists." }, { "name": "coverage", @@ -683,18 +683,18 @@ "category": "engineering", "description": ">-" }, - { - "name": "review", - "source": "../../engineering-team/playwright-pro/skills/review", - "category": "engineering", - "description": ">-" - }, { "name": "review", "source": "../../engineering-team/self-improving-agent/skills/review", "category": "engineering", "description": "Analyze auto-memory for promotion candidates, stale entries, consolidation opportunities, and health metrics." }, + { + "name": "review", + "source": "../../engineering-team/playwright-pro/skills/review", + "category": "engineering", + "description": ">-" + }, { "name": "security-pen-testing", "source": "../../engineering-team/skills/security-pen-testing", @@ -1942,6 +1942,36 @@ "source": "../../research/syllabus/skills/syllabus", "category": "research", "description": "Generates a curated supplementary reading list from any course syllabus using Consensus academic search. Grill-me intake (syllabus input format + course audience + year range) plus a grouping forcing-options checkpoint before any search runs \u2014 so the reading list matches the course's level and recency need. Parses the syllabus to extract topics and learning outcomes, searches Consensus for recent peer-reviewed papers per topic, and produces a professionally formatted .docx with clickable Consensus links, plain-language summaries calibrated to audience level, and Bloom-higher-order discussion questions tied to course learning goals. Triggers whenever a user uploads a syllabus, course outline, or curriculum document and wants supplementary readings. Also triggers on: 'syllabus reading list', 'find papers for my course', 'create a reading list from this syllabus', 'recent research for my class', 'supplementary readings', 'find journal articles for these topics', 'what recent papers cover this material', 'any new research on these course topics', 'update my syllabus with recent papers'. Even casual mentions when a syllabus is attached should trigger this skill." + }, + { + "name": "clinical-research", + "source": "../../research-ops/skills/clinical-research", + "category": "research-ops", + "description": "Use when designing a prospective clinical study before submission \u2014 selecting and classifying endpoints (primary / key-secondary / exploratory, with surrogate-endpoint flagging), estimating sample size and power for two-arm designs (means / proportions / survival), or scoring a study plan for feasibility and a GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO phase-gate decision. Every output is an ESTIMATE plus a named human owner (clinician / biostatistician / regulatory owner) \u2014 never clinical fact, never a finished protocol. Distinct from ra-qm-team, which handles the regulatory/QM submission (ISO 13485, EU MDR, FDA 510(k)/PMA/QSR), not the study design." + }, + { + "name": "market-research", + "source": "../../research-ops/skills/market-research", + "category": "research-ops", + "description": "Use when doing upstream market-research methodology \u2014 sizing a market as TAM/SAM/SOM computed BOTH top-down and bottoms-up (never a single unsourced number), planning a survey sample size with finite-population correction and per-segment minimums, or scoring candidate market segments against Kotler's measurable/substantial/accessible/differentiable/actionable criteria. Outputs always show the method and the assumptions. For market-research analysts and product-marketing at the sizing/survey/segmentation moment. Distinct from marketing-skill (campaign analytics, attribution, demand-gen) \u2014 this is the evidence-building methodology, not live-campaign optimization." + }, + { + "name": "product-research", + "source": "../../research-ops/skills/product-research", + "category": "research-ops", + "description": "Use when planning and synthesizing product/user research as a method-and-repository discipline \u2014 selecting the right method for the goal (generative interviews vs usability test vs concept test vs validation), computing method-based saturation/sample size with an explicit confidence level, or synthesizing coded observations into insights while flagging single-source anecdotes. Never fabricates user insight; an insight requires recurrence across independent participants. Distinct from product-team/ux-researcher-designer (persona/journey artifacts), product-discovery (discovery-sprint planning), and experiment-designer (live A/B) \u2014 this is the research-ops method + insight-repository layer." + }, + { + "name": "research-finance", + "source": "../../research-ops/skills/research-finance", + "category": "research-ops", + "description": "Use when managing the money for an internal R&D program or portfolio \u2014 building a multi-period program budget with the F&A (indirect) split, tracking burn rate and runway against value-inflection milestones, or routing R&D cost items to a capitalize-vs-expense determination. Every budget output surfaces its assumptions block; capitalize-vs-expense is decision-support only and routes to a named finance owner \u2014 it never books an entry or decides accounting treatment. Distinct from finance/financial-analysis (corporate DCF, close, valuation) and research/grants (funding discovery \u2014 this manages money already won)." + }, + { + "name": "research-ops-skills", + "source": "../../research-ops/skills/research-ops-skills", + "category": "research-ops", + "description": "Use when planning, funding, scoping, or synthesizing enterprise research across workstreams \u2014 clinical study design, R&D program finance, market sizing/surveys, or product/user research. Triggers on \"design this clinical study\", \"what sample size\", \"R&D budget\", \"burn rate\", \"capitalize or expense\", \"TAM SAM SOM\", \"market sizing\", \"survey design\", \"segment the market\", \"plan user interviews\", \"usability test\", \"synthesize research insights\". Forks context to route to one of four Research-Operations sub-skills (clinical-research, research-finance, market-research, product-research) and returns a digest. Distinct from ra-qm-team (regulatory submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), and marketing-skill (campaign analytics)." } ], "categories": { @@ -2009,6 +2039,11 @@ "count": 8, "source": "../../research", "description": "Research orchestrator + 6 specialists (pulse, litreview, grants, dossier, patent, syllabus, notebooklm)" + }, + "research-ops": { + "count": 5, + "source": "../../research-ops", + "description": "Enterprise Research Operations skills (v2.9.0): clinical study design, R&D program finance, market research methodology, product/user research" } } } diff --git a/.codex/skills/clinical-research b/.codex/skills/clinical-research new file mode 120000 index 00000000..d179c913 --- /dev/null +++ b/.codex/skills/clinical-research @@ -0,0 +1 @@ +../../research-ops/skills/clinical-research \ No newline at end of file diff --git a/.codex/skills/market-research b/.codex/skills/market-research new file mode 120000 index 00000000..2c1333af --- /dev/null +++ b/.codex/skills/market-research @@ -0,0 +1 @@ +../../research-ops/skills/market-research \ No newline at end of file diff --git a/.codex/skills/product-research b/.codex/skills/product-research new file mode 120000 index 00000000..d4e1d645 --- /dev/null +++ b/.codex/skills/product-research @@ -0,0 +1 @@ +../../research-ops/skills/product-research \ No newline at end of file diff --git a/.codex/skills/research-finance b/.codex/skills/research-finance new file mode 120000 index 00000000..074a70ea --- /dev/null +++ b/.codex/skills/research-finance @@ -0,0 +1 @@ +../../research-ops/skills/research-finance \ No newline at end of file diff --git a/.codex/skills/research-ops-skills b/.codex/skills/research-ops-skills new file mode 120000 index 00000000..0e68aabd --- /dev/null +++ b/.codex/skills/research-ops-skills @@ -0,0 +1 @@ +../../research-ops/skills/research-ops-skills \ No newline at end of file diff --git a/.codex/skills/review b/.codex/skills/review index b4fa2536..647ec915 120000 --- a/.codex/skills/review +++ b/.codex/skills/review @@ -1 +1 @@ -../../engineering-team/self-improving-agent/skills/review \ No newline at end of file +../../engineering-team/playwright-pro/skills/review \ No newline at end of file diff --git a/.gemini/skills-index.json b/.gemini/skills-index.json index f9d027fb..dea25754 100644 --- a/.gemini/skills-index.json +++ b/.gemini/skills-index.json @@ -1,7 +1,7 @@ { "version": "1.0.0", "name": "gemini-cli-skills", - "total_skills": 395, + "total_skills": 400, "skills": [ { "name": "README", @@ -826,7 +826,7 @@ { "name": "code-reviewer", "category": "engineering", - "description": "Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, and .NET. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists." + "description": "Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, .NET, and Java. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists." }, { "name": "coverage", @@ -1977,6 +1977,31 @@ "name": "syllabus", "category": "research", "description": "Generates a curated supplementary reading list from any course syllabus using Consensus academic search. Grill-me intake (syllabus input format + course audience + year range) plus a grouping forcing-options checkpoint before any search runs \u2014 so the reading list matches the course's level and recency need. Parses the syllabus to extract topics and learning outcomes, searches Consensus for recent peer-reviewed papers per topic, and produces a professionally formatted .docx with clickable Consensus links, plain-language summaries calibrated to audience level, and Bloom-higher-order discussion questions tied to course learning goals. Triggers whenever a user uploads a syllabus, course outline, or curriculum document and wants supplementary readings. Also triggers on: 'syllabus reading list', 'find papers for my course', 'create a reading list from this syllabus', 'recent research for my class', 'supplementary readings', 'find journal articles for these topics', 'what recent papers cover this material', 'any new research on these course topics', 'update my syllabus with recent papers'. Even casual mentions when a syllabus is attached should trigger this skill." + }, + { + "name": "clinical-research", + "category": "research-ops", + "description": "Use when designing a prospective clinical study before submission \u2014 selecting and classifying endpoints (primary / key-secondary / exploratory, with surrogate-endpoint flagging), estimating sample size and power for two-arm designs (means / proportions / survival), or scoring a study plan for feasibility and a GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO phase-gate decision. Every output is an ESTIMATE plus a named human owner (clinician / biostatistician / regulatory owner) \u2014 never clinical fact, never a finished protocol. Distinct from ra-qm-team, which handles the regulatory/QM submission (ISO 13485, EU MDR, FDA 510(k)/PMA/QSR), not the study design." + }, + { + "name": "market-research", + "category": "research-ops", + "description": "Use when doing upstream market-research methodology \u2014 sizing a market as TAM/SAM/SOM computed BOTH top-down and bottoms-up (never a single unsourced number), planning a survey sample size with finite-population correction and per-segment minimums, or scoring candidate market segments against Kotler's measurable/substantial/accessible/differentiable/actionable criteria. Outputs always show the method and the assumptions. For market-research analysts and product-marketing at the sizing/survey/segmentation moment. Distinct from marketing-skill (campaign analytics, attribution, demand-gen) \u2014 this is the evidence-building methodology, not live-campaign optimization." + }, + { + "name": "product-research", + "category": "research-ops", + "description": "Use when planning and synthesizing product/user research as a method-and-repository discipline \u2014 selecting the right method for the goal (generative interviews vs usability test vs concept test vs validation), computing method-based saturation/sample size with an explicit confidence level, or synthesizing coded observations into insights while flagging single-source anecdotes. Never fabricates user insight; an insight requires recurrence across independent participants. Distinct from product-team/ux-researcher-designer (persona/journey artifacts), product-discovery (discovery-sprint planning), and experiment-designer (live A/B) \u2014 this is the research-ops method + insight-repository layer." + }, + { + "name": "research-finance", + "category": "research-ops", + "description": "Use when managing the money for an internal R&D program or portfolio \u2014 building a multi-period program budget with the F&A (indirect) split, tracking burn rate and runway against value-inflection milestones, or routing R&D cost items to a capitalize-vs-expense determination. Every budget output surfaces its assumptions block; capitalize-vs-expense is decision-support only and routes to a named finance owner \u2014 it never books an entry or decides accounting treatment. Distinct from finance/financial-analysis (corporate DCF, close, valuation) and research/grants (funding discovery \u2014 this manages money already won)." + }, + { + "name": "research-ops-skills", + "category": "research-ops", + "description": "Use when planning, funding, scoping, or synthesizing enterprise research across workstreams \u2014 clinical study design, R&D program finance, market sizing/surveys, or product/user research. Triggers on \"design this clinical study\", \"what sample size\", \"R&D budget\", \"burn rate\", \"capitalize or expense\", \"TAM SAM SOM\", \"market sizing\", \"survey design\", \"segment the market\", \"plan user interviews\", \"usability test\", \"synthesize research insights\". Forks context to route to one of four Research-Operations sub-skills (clinical-research, research-finance, market-research, product-research) and returns a digest. Distinct from ra-qm-team (regulatory submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), and marketing-skill (campaign analytics)." } ], "categories": { @@ -2043,6 +2068,10 @@ "research": { "count": 8, "description": "Research resources" + }, + "research-ops": { + "count": 5, + "description": "Research-ops resources" } } } diff --git a/.gemini/skills/clinical-research/SKILL.md b/.gemini/skills/clinical-research/SKILL.md new file mode 120000 index 00000000..5163e4ef --- /dev/null +++ b/.gemini/skills/clinical-research/SKILL.md @@ -0,0 +1 @@ +../../../research-ops/skills/clinical-research/SKILL.md \ No newline at end of file diff --git a/.gemini/skills/market-research/SKILL.md b/.gemini/skills/market-research/SKILL.md new file mode 120000 index 00000000..7bc8e5b5 --- /dev/null +++ b/.gemini/skills/market-research/SKILL.md @@ -0,0 +1 @@ +../../../research-ops/skills/market-research/SKILL.md \ No newline at end of file diff --git a/.gemini/skills/product-research/SKILL.md b/.gemini/skills/product-research/SKILL.md new file mode 120000 index 00000000..323d077c --- /dev/null +++ b/.gemini/skills/product-research/SKILL.md @@ -0,0 +1 @@ +../../../research-ops/skills/product-research/SKILL.md \ No newline at end of file diff --git a/.gemini/skills/research-finance/SKILL.md b/.gemini/skills/research-finance/SKILL.md new file mode 120000 index 00000000..3320fb84 --- /dev/null +++ b/.gemini/skills/research-finance/SKILL.md @@ -0,0 +1 @@ +../../../research-ops/skills/research-finance/SKILL.md \ No newline at end of file diff --git a/.gemini/skills/research-ops-skills/SKILL.md b/.gemini/skills/research-ops-skills/SKILL.md new file mode 120000 index 00000000..7f16242b --- /dev/null +++ b/.gemini/skills/research-ops-skills/SKILL.md @@ -0,0 +1 @@ +../../../research-ops/skills/research-ops-skills/SKILL.md \ No newline at end of file diff --git a/.vibe/skills/claude-skills/research-ops/clinical-research b/.vibe/skills/claude-skills/research-ops/clinical-research new file mode 120000 index 00000000..05eab9c4 --- /dev/null +++ b/.vibe/skills/claude-skills/research-ops/clinical-research @@ -0,0 +1 @@ +../../../../research-ops/skills/clinical-research \ No newline at end of file diff --git a/.vibe/skills/claude-skills/research-ops/market-research b/.vibe/skills/claude-skills/research-ops/market-research new file mode 120000 index 00000000..8f246761 --- /dev/null +++ b/.vibe/skills/claude-skills/research-ops/market-research @@ -0,0 +1 @@ +../../../../research-ops/skills/market-research \ No newline at end of file diff --git a/.vibe/skills/claude-skills/research-ops/product-research b/.vibe/skills/claude-skills/research-ops/product-research new file mode 120000 index 00000000..54039331 --- /dev/null +++ b/.vibe/skills/claude-skills/research-ops/product-research @@ -0,0 +1 @@ +../../../../research-ops/skills/product-research \ No newline at end of file diff --git a/.vibe/skills/claude-skills/research-ops/research-finance b/.vibe/skills/claude-skills/research-ops/research-finance new file mode 120000 index 00000000..991069c4 --- /dev/null +++ b/.vibe/skills/claude-skills/research-ops/research-finance @@ -0,0 +1 @@ +../../../../research-ops/skills/research-finance \ No newline at end of file diff --git a/.vibe/skills/claude-skills/research-ops/research-ops-skills b/.vibe/skills/claude-skills/research-ops/research-ops-skills new file mode 120000 index 00000000..4196599f --- /dev/null +++ b/.vibe/skills/claude-skills/research-ops/research-ops-skills @@ -0,0 +1 @@ +../../../../research-ops/skills/research-ops-skills \ No newline at end of file diff --git a/.vibe/skills/claude-skills/skills-index.json b/.vibe/skills/claude-skills/skills-index.json index 5e80a975..249cea22 100644 --- a/.vibe/skills/claude-skills/skills-index.json +++ b/.vibe/skills/claude-skills/skills-index.json @@ -1,7 +1,1490 @@ { "source": "claude-code-skills", - "total_skills": 6, + "total_skills": 328, "domains": { + "engineering": [ + { + "name": "agent-designer", + "description": "Use when the user asks to design multi-agent systems, create agent architectures, define agent communication patterns, or build autonomous agent workflows.", + "path": "engineering/agent-designer" + }, + { + "name": "agent-workflow-designer", + "description": "Design production-grade multi-agent workflows with clear pattern choice (sequential, parallel, hierarchical), handoff contracts, failure handling, and cost/context controls. Use when architecting a multi-step agent pipeline, choosing between single-agent vs multi-agent approaches, or refactoring an LLM workflow that suffers from context bloat or unreliable handoffs.", + "path": "engineering/agent-workflow-designer" + }, + { + "name": "api-design-reviewer", + "description": "Comprehensive REST API design review with automated linting, breaking-change detection, and design scorecards. Catches inconsistent conventions, missing versioning, and design smells before APIs ship. Use when reviewing a PR that adds or changes API endpoints, auditing an existing API for v2 migration, or establishing API standards for a team.", + "path": "engineering/api-design-reviewer" + }, + { + "name": "api-test-suite-builder", + "description": "Use when the user asks to generate API tests, create integration test suites, test REST endpoints, or build contract tests.", + "path": "engineering/api-test-suite-builder" + }, + { + "name": "browser-automation", + "description": "Use when the user asks to automate browser tasks, scrape websites, fill forms, capture screenshots, extract structured data from web pages, or build web automation workflows. NOT for testing \u2014 use playwright-pro for that.", + "path": "engineering/browser-automation" + }, + { + "name": "changelog-generator", + "description": "Produce consistent, auditable release notes from Conventional Commits. Separates commit parsing, semantic-bump logic, and changelog rendering for automated releases with editorial control. Use when cutting a release, generating CHANGELOG.md from git history, or automating release notes in CI.", + "path": "engineering/changelog-generator" + }, + { + "name": "chaos-engineering", + "description": "Use when planning, running, or learning from chaos engineering experiments. Triggers on \"chaos experiment\", \"fault injection\", \"gameday\", \"resilience test\", \"blast radius\", \"steady state\", \"abort criteria\", \"Chaos Toolkit\", \"Chaos Mesh\", \"Litmus\", \"Gremlin\", \"AWS FIS\", or any deliberate failure-injection question. Ships experiment designer, blast-radius calculator, and postmortem generator (all stdlib Python), 4 references on chaos principles + experiment design + attack taxonomy + tooling landscape, and a /chaos-experiment slash command. Composes with feature-flags-architect (kill switches as abort triggers) and kubernetes-operator (common chaos targets).", + "path": "engineering/chaos-engineering" + }, + { + "name": "ci-cd-pipeline-builder", + "description": "Generate pragmatic CI/CD pipelines from detected project stack signals \u2014 fast baseline generation, repeatable checks, environment-aware deployment stages. Use when setting up CI for a new project, refactoring existing pipelines, or standardizing deployment workflows across multiple repos.", + "path": "engineering/ci-cd-pipeline-builder" + }, + { + "name": "codebase-onboarding", + "description": "Analyze a codebase and generate onboarding documentation for engineers, tech leads, and contractors. Fast fact-gathering and repeatable onboarding outputs. Use when onboarding a new engineer, writing architecture-overview docs for a new project, or producing tech-lead briefings for unfamiliar repos.", + "path": "engineering/codebase-onboarding" + }, + { + "name": "command-guide", + "description": ">", + "path": "engineering/command-guide" + }, + { + "name": "database-designer", + "description": "Use when the user asks to design database schemas, plan data migrations, optimize queries, choose between SQL and NoSQL, or model data relationships.", + "path": "engineering/database-designer" + }, + { + "name": "database-schema-designer", + "description": "Use when the user asks to create ERD diagrams, normalize database schemas, design table relationships, or plan schema migrations.", + "path": "engineering/database-schema-designer" + }, + { + "name": "dependency-auditor", + "description": "Audit and manage dependencies across multi-language projects. Identifies vulnerabilities, license conflicts, transitive dependency risks, and safe-upgrade paths. Use when auditing third-party packages before release, investigating a CVE, planning a major version bump, or running a license-compliance review.", + "path": "engineering/dependency-auditor" + }, + { + "name": "engineering-advanced-skills", + "description": "25 advanced engineering agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Agent design, RAG, MCP servers, CI/CD, database design, observability, security auditing, release management, platform ops.", + "path": "engineering/engineering-advanced-skills" + }, + { + "name": "env-secrets-manager", + "description": "Manage environment-variable hygiene and secrets safety across local development and production. Practical auditing, drift awareness, rotation readiness. Use when auditing .env files for committed secrets, planning a credential rotation, debugging missing-env-var production incidents, or hardening a new project against secrets leakage.", + "path": "engineering/env-secrets-manager" + }, + { + "name": "feature-flags-architect", + "description": "Use when adding, retiring, or auditing feature flags. Triggers on \"add a flag\", \"ship behind a flag\", \"rollout plan\", \"kill switch\", \"stale flags\", \"flag debt\", \"LaunchDarkly\", \"GrowthBook\", \"Statsig\", \"Unleash\", \"Flipt\", or any progressive-delivery question. Ships flag debt scanner, rollout planner, and kill-switch auditor (all stdlib Python), 4 references on flag taxonomy + provider trade-offs + rollout strategies + lifecycle, plus a /flag-cleanup slash command.", + "path": "engineering/feature-flags-architect" + }, + { + "name": "focused-fix", + "description": "Use when the user asks to fix, debug, or make a specific feature/module/area work end-to-end. Triggers: 'make X work', 'fix the Y feature', 'the Z module is broken', 'focus on [area]'. Not for quick single-bug fixes \u2014 this is for systematic deep-dive repair across all files and dependencies.", + "path": "engineering/focused-fix" + }, + { + "name": "full-page-screenshot", + "description": "Use when the user asks to capture a full-page screenshot, long screenshot, or complete page capture of a web page. Handles SPA scroll containers, lazy-loaded images, and very tall pages via Chrome DevTools Protocol with zero external dependencies.", + "path": "engineering/full-page-screenshot" + }, + { + "name": "git-worktree-manager", + "description": "Run parallel feature work safely with Git worktrees. Standardizes branch isolation, port allocation, environment sync, and cleanup so each worktree behaves like an independent local app. Optimized for multi-agent workflows where each agent or terminal session owns one worktree. Use when running multiple feature branches simultaneously, isolating experimental work, or coordinating multi-agent development across the same repo.", + "path": "engineering/git-worktree-manager" + }, + { + "name": "interview-system-designer", + "description": "This skill should be used when the user asks to \"design interview processes\", \"create hiring pipelines\", \"calibrate interview loops\", \"generate interview questions\", \"design competency matrices\", \"analyze interviewer bias\", \"create scoring rubrics\", \"build question banks\", or \"optimize hiring systems\". Use for designing role-specific interview loops, competency assessments, and hiring calibration systems.", + "path": "engineering/interview-system-designer" + }, + { + "name": "kubernetes-operator", + "description": "Use when building a Kubernetes Operator \u2014 custom controllers that reconcile CRD state. Triggers on \"build an operator\", \"CRD design\", \"reconcile loop\", \"controller-runtime\", \"kubebuilder\", \"operator-sdk\", \"metacontroller\", \"KOPF\", \"operator capability levels\", or \"custom resource\". Ships CRD validator, reconcile-loop linter, and OperatorHub capability auditor (all stdlib Python), 4 references on the operator pattern + CRD design + reconcile patterns + tooling landscape, and a /operator-audit slash command. NOT a generic k8s skill \u2014 specifically the Operator pattern.", + "path": "engineering/kubernetes-operator" + }, + { + "name": "mcp-server-builder", + "description": "Design and ship production-ready MCP (Model Context Protocol) servers from OpenAPI contracts instead of hand-written tool wrappers. Python and TypeScript support, schema validation, safe evolution. Use when exposing an existing API as an MCP server, building tool integrations for Claude or Codex or Cursor, or scaffolding an MCP project from scratch.", + "path": "engineering/mcp-server-builder" + }, + { + "name": "migration-architect", + "description": "Zero-downtime migration planning, compatibility validation, and rollback strategy generation. Tools for system, database, and infrastructure migrations with minimal business impact. Use when planning a database migration, infrastructure cutover, system replacement, or any high-risk transition that needs explicit rollback paths.", + "path": "engineering/migration-architect" + }, + { + "name": "monorepo-navigator", + "description": "Navigate, manage, and optimize monorepos. Covers Turborepo, Nx, pnpm workspaces, and Lerna. Cross-package impact analysis, selective builds/tests on affected packages, remote caching, dependency graph visualization, and structured multi-repo to monorepo migrations. Use when setting up a new monorepo, optimizing CI for a large workspace, debugging cross-package dependency issues, or planning a multi-repo consolidation.", + "path": "engineering/monorepo-navigator" + }, + { + "name": "observability-designer", + "description": "Design production-ready observability strategies combining metrics, logs, and traces. Includes SLI/SLO design, golden-signals monitoring, alert optimization. Use when adding observability to a new service, refactoring alerting that is too noisy, or designing an SLO program before scaling production load.", + "path": "engineering/observability-designer" + }, + { + "name": "performance-profiler", + "description": "Systematic performance profiling for Node.js, Python, and Go applications. Identifies CPU, memory, and I/O bottlenecks, generates flamegraphs, analyzes bundle sizes, optimizes database queries, runs load tests with k6 and Artillery. Always measures before and after. Use when investigating a slow endpoint, planning a performance budget, or hunting a memory leak in production.", + "path": "engineering/performance-profiler" + }, + { + "name": "pr-review-expert", + "description": "Use when the user asks to review pull requests, analyze code changes, check for security issues in PRs, or assess code quality of diffs.", + "path": "engineering/pr-review-expert" + }, + { + "name": "rag-architect", + "description": "Use when the user asks to design RAG pipelines, optimize retrieval strategies, choose embedding models, implement vector search, or build knowledge retrieval systems.", + "path": "engineering/rag-architect" + }, + { + "name": "release-manager", + "description": "Use when the user asks to plan releases, manage changelogs, coordinate deployments, create release branches, or automate versioning.", + "path": "engineering/release-manager" + }, + { + "name": "runbook-generator", + "description": "Generate operational runbooks from a service name \u2014 deployment, incident response, maintenance, and rollback workflows. Templated structure customizable per environment. Use when documenting on-call procedures for a new service, standardizing incident response across teams, or producing runbooks before launching to production.", + "path": "engineering/runbook-generator" + }, + { + "name": "secrets-vault-manager", + "description": "Use when the user asks to set up secret management infrastructure, integrate HashiCorp Vault, configure cloud secret stores (AWS Secrets Manager, Azure Key Vault, GCP Secret Manager), implement secret rotation, or audit secret access patterns.", + "path": "engineering/secrets-vault-manager" + }, + { + "name": "self-eval", + "description": "Honestly evaluate AI work quality using a two-axis scoring system. Use after completing a task, code review, or work session to get an unbiased assessment. Detects score inflation, forces devil's advocate reasoning, and persists scores across sessions.", + "path": "engineering/self-eval" + }, + { + "name": "ship-gate", + "description": ">", + "path": "engineering/ship-gate" + }, + { + "name": "skill-security-auditor", + "description": ">", + "path": "engineering/skill-security-auditor" + }, + { + "name": "skill-tester", + "description": "Validate, test, and score the quality of skills within the claude-skills ecosystem. Comprehensive meta-skill: structure validation, Python script testing (syntax + imports + runtime + output format), multi-dimensional quality scoring with letter grades and tier classification (BASIC/STANDARD/POWERFUL). Use when authoring a new skill, auditing existing skills for tier promotion, setting up pre-commit hooks for skill quality, or integrating skill QA into CI.", + "path": "engineering/skill-tester" + }, + { + "name": "slo-architect", + "description": "Use when defining, reviewing, or operating SLOs/SLIs/error budgets. Triggers on \"define an SLO\", \"what should our SLO be\", \"error budget\", \"burn rate\", \"SLI\", \"service level objective\", \"Google SRE workbook\", \"multi-window burn-rate alert\", or any reliability-target question. Ships SLO designer, error-budget calculator with multi-window burn-rate thresholds, and SLO reviewer that catches the common bugs (target too aggressive, window too short, conflicting SLOs, no SLI definition). 4 references on SLO principles + SLI design + error budget math + composition with feature-flags-architect/chaos-engineering/kubernetes-operator. NOT a generic observability skill \u2014 specifically the SLO discipline.", + "path": "engineering/slo-architect" + }, + { + "name": "spec-driven-workflow", + "description": "Use when the user asks to write specs before code, define acceptance criteria, plan features before implementation, generate tests from specifications, or follow spec-first development practices.", + "path": "engineering/spec-driven-workflow" + }, + { + "name": "sql-database-assistant", + "description": "Use when the user asks to write SQL queries, optimize database performance, generate migrations, explore database schemas, or work with ORMs like Prisma, Drizzle, TypeORM, or SQLAlchemy.", + "path": "engineering/sql-database-assistant" + }, + { + "name": "tc-tracker", + "description": "Use when the user asks to track technical changes, create change records, manage TC lifecycles, or hand off work between AI sessions. Covers init/create/update/status/resume/close/export workflows for structured code change documentation.", + "path": "engineering/tc-tracker" + }, + { + "name": "tech-debt-tracker", + "description": "Scan codebases for technical debt, score severity, track trends, and generate prioritized remediation plans. Use when users mention tech debt, code quality, refactoring priority, debt scoring, cleanup sprints, or code health assessment. Also use for legacy code modernization planning and maintenance cost estimation.", + "path": "engineering/tech-debt-tracker" + }, + { + "name": "agenthub", + "description": "Multi-agent collaboration plugin that spawns N parallel subagents competing on the same task via git worktree isolation. Agents work independently, results are evaluated by metric or LLM judge, and the best branch is merged. Use when: user wants multiple approaches tried in parallel \u2014 code optimization, content variation, research exploration, or any task that benefits from parallel competition. Requires: a git repo.", + "path": "engineering/agenthub" + }, + { + "name": "board", + "description": "Read, write, and browse the AgentHub message board for agent coordination.", + "path": "engineering/board" + }, + { + "name": "eval", + "description": "Evaluate and rank agent results by metric or LLM judge for an AgentHub session.", + "path": "engineering/eval" + }, + { + "name": "init", + "description": "Create a new AgentHub collaboration session with task, agent count, and evaluation criteria.", + "path": "engineering/init" + }, + { + "name": "merge", + "description": "Merge the winning agent's branch into base, archive losers, and clean up worktrees.", + "path": "engineering/merge" + }, + { + "name": "run", + "description": "One-shot lifecycle command that chains init \u2192 baseline \u2192 spawn \u2192 eval \u2192 merge in a single invocation.", + "path": "engineering/run" + }, + { + "name": "spawn", + "description": "Launch N parallel subagents in isolated git worktrees to compete on the session task.", + "path": "engineering/spawn" + }, + { + "name": "status", + "description": "Show DAG state, agent progress, and branch status for an AgentHub session.", + "path": "engineering/status" + }, + { + "name": "autoresearch-agent", + "description": "Autonomous experiment loop that optimizes any file by a measurable metric. Inspired by Karpathy's autoresearch. The agent edits a target file, runs a fixed evaluation, keeps improvements (git commit), discards failures (git reset), and loops indefinitely. Use when: user wants to optimize code speed, reduce bundle/image size, improve test pass rate, optimize prompts, improve content quality (headlines, copy, CTR), or run any measurable improvement loop. Requires: a target file, an evaluation command that outputs a metric, and a git repo.", + "path": "engineering/autoresearch-agent" + }, + { + "name": "loop", + "description": "Start an autonomous experiment loop with user-selected interval (10min, 1h, daily, weekly, monthly). Uses CronCreate for scheduling.", + "path": "engineering/loop" + }, + { + "name": "resume", + "description": "Resume a paused experiment. Checkout the experiment branch, read results history, continue iterating.", + "path": "engineering/resume" + }, + { + "name": "run", + "description": "Run a single experiment iteration. Edit the target file, evaluate, keep or discard.", + "path": "engineering/run" + }, + { + "name": "setup", + "description": "Set up a new autoresearch experiment interactively. Collects domain, target file, eval command, metric, direction, and evaluator.", + "path": "engineering/setup" + }, + { + "name": "status", + "description": "Show experiment dashboard with results, active loops, and progress.", + "path": "engineering/status" + }, + { + "name": "behuman", + "description": "Use when the user wants more human-like AI responses \u2014 less robotic, less listy, more authentic. Triggers: 'behuman', 'be real', 'like a human', 'more human', 'less AI', 'talk like a person', 'mirror mode', 'stop being so AI', or when conversations are emotionally charged (grief, job loss, relationship advice, fear). NOT for technical questions, code generation, or factual lookups.", + "path": "engineering/behuman" + }, + { + "name": "caveman", + "description": ">", + "path": "engineering/caveman" + }, + { + "name": "chaos-engineering", + "description": "Use when planning, running, or learning from chaos engineering experiments. Triggers on \"chaos experiment\", \"fault injection\", \"gameday\", \"resilience test\", \"blast radius\", \"steady state\", \"abort criteria\", \"Chaos Toolkit\", \"Chaos Mesh\", \"Litmus\", \"Gremlin\", \"AWS FIS\", or any deliberate failure-injection question. Ships experiment designer, blast-radius calculator, and postmortem generator (all stdlib Python), 4 references on chaos principles + experiment design + attack taxonomy + tooling landscape, and a /chaos-experiment slash command. Composes with feature-flags-architect (kill switches as abort triggers) and kubernetes-operator (common chaos targets).", + "path": "engineering/chaos-engineering" + }, + { + "name": "claude-coach", + "description": "Personal coach that teaches users to become Claude power users. Use this skill the FIRST time a user asks to \"learn Claude\", \"be a power user\", \"coach me\", \"teach me Claude tricks\", \"what can Claude do\", \"make me better at prompting\", or any variation. After activation, also use it on EVERY subsequent turn to detect missed optimization opportunities (vague prompts, ignored capabilities, manual work Claude could automate) and surface a single power-user tip. Trigger generously \u2014 most users do not know what they do not know, so err on the side of coaching.", + "path": "engineering/claude-coach" + }, + { + "name": "code-tour", + "description": "Use when the user asks to create a CodeTour .tour file \u2014 persona-targeted, step-by-step walkthroughs that link to real files and line numbers. Trigger for: create a tour, onboarding tour, architecture tour, PR review tour, explain how X works, vibe check, RCA tour, contributor guide, or any structured code walkthrough request.", + "path": "engineering/code-tour" + }, + { + "name": "data-quality-auditor", + "description": "Audit datasets for completeness, consistency, accuracy, and validity. Profile data distributions, detect anomalies and outliers, surface structural issues, and produce an actionable remediation plan.", + "path": "engineering/data-quality-auditor" + }, + { + "name": "demo-video", + "description": "Use when the user asks to create a demo video, product walkthrough, feature showcase, animated presentation, marketing video, or GIF from screenshots or scene descriptions. Orchestrates playwright, ffmpeg, and edge-tts MCPs to produce polished video content.", + "path": "engineering/demo-video" + }, + { + "name": "docker-development", + "description": "Docker and container development agent skill and plugin for Dockerfile optimization, docker-compose orchestration, multi-stage builds, and container security hardening. Use when: user wants to optimize a Dockerfile, create or improve docker-compose configurations, implement multi-stage builds, audit container security, reduce image size, or follow container best practices. Covers build performance, layer caching, secret management, and production-ready container patterns.", + "path": "engineering/docker-development" + }, + { + "name": "feature-flags-architect", + "description": "Use when adding, retiring, or auditing feature flags. Triggers on \"add a flag\", \"ship behind a flag\", \"rollout plan\", \"kill switch\", \"stale flags\", \"flag debt\", \"LaunchDarkly\", \"GrowthBook\", \"Statsig\", \"Unleash\", \"Flipt\", or any progressive-delivery question. Ships flag debt scanner, rollout planner, and kill-switch auditor (all stdlib Python), 4 references on flag taxonomy + provider trade-offs + rollout strategies + lifecycle, plus a /flag-cleanup slash command.", + "path": "engineering/feature-flags-architect" + }, + { + "name": "grill-me", + "description": "Interview the user relentlessly about a plan or design until reaching shared understanding, resolving each branch of the decision tree. Use when user wants to stress-test a plan, get grilled on their design, or mentions \"grill me\".", + "path": "engineering/grill-me" + }, + { + "name": "grill-with-docs", + "description": "Docs-anchored grilling session \u2014 challenges a plan against the project's existing language (CONTEXT.md) and recorded decisions (docs/adr/), and updates those files inline as terminology and decisions crystallise. Use when user wants to stress-test a plan against documented domain language, or mentions \"grill with docs\".", + "path": "engineering/grill-with-docs" + }, + { + "name": "handoff", + "description": "Compact the current conversation into a handoff document for another agent to pick up. References existing artifacts (PRDs, plans, ADRs, issues, commits, diffs) by path or URL instead of duplicating them. Use when user wants to hand off the conversation to a fresh agent or starts a new session that picks up prior work.", + "path": "engineering/handoff" + }, + { + "name": "helm-chart-builder", + "description": "Helm chart development agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw \u2014 chart scaffolding, values design, template patterns, dependency management, security hardening, and chart testing. Use when: user wants to create or improve Helm charts, design values.yaml files, implement template helpers, audit chart security (RBAC, network policies, pod security), manage subcharts, or run helm lint/test.", + "path": "engineering/helm-chart-builder" + }, + { + "name": "karpathy-coder", + "description": "Use when writing, reviewing, or committing code to enforce Karpathy's 4 coding principles \u2014 surface assumptions before coding, keep it simple, make surgical changes, define verifiable goals. Triggers on \"review my diff\", \"check complexity\", \"am I overcomplicating this\", \"karpathy check\", \"before I commit\", or any code quality concern where the LLM might be overcoding.", + "path": "engineering/karpathy-coder" + }, + { + "name": "kubernetes-operator", + "description": "Use when building a Kubernetes Operator \u2014 custom controllers that reconcile CRD state. Triggers on \"build an operator\", \"CRD design\", \"reconcile loop\", \"controller-runtime\", \"kubebuilder\", \"operator-sdk\", \"metacontroller\", \"KOPF\", \"operator capability levels\", or \"custom resource\". Ships CRD validator, reconcile-loop linter, and OperatorHub capability auditor (all stdlib Python), 4 references on the operator pattern + CRD design + reconcile patterns + tooling landscape, and a /operator-audit slash command. NOT a generic k8s skill \u2014 specifically the Operator pattern.", + "path": "engineering/kubernetes-operator" + }, + { + "name": "llm-cost-optimizer", + "description": "Use proactively whenever LLM API costs come up -- or should. Triggers include: 'my AI costs are too high', 'optimize token usage', 'which model should I use', 'LLM spend is out of control', 'implement prompt caching', 'we're about to launch an AI feature', 'build me an AI endpoint'. Don't wait for an explicit cost complaint -- if someone is building an AI feature, designing an LLM endpoint, or choosing between models, cost architecture belongs in the conversation. Apply immediately when any of these are true: a system prompt appears that exceeds a few hundred tokens, all requests are hitting the same model, max_tokens is not set, or no per-feature cost logging exists. NOT for RAG pipeline design (use rag-architect). NOT for improving prompt quality or effectiveness (use senior-prompt-engineer).", + "path": "engineering/llm-cost-optimizer" + }, + { + "name": "llm-wiki", + "description": "Use when building or maintaining a persistent personal knowledge base (second brain) in Obsidian where an LLM incrementally ingests sources, updates entity/concept pages, maintains cross-references, and keeps a synthesis current. Triggers include \"second brain\", \"Obsidian wiki\", \"personal knowledge management\", \"ingest this paper/article/book\", \"build a research wiki\", \"compound knowledge\", \"Memex\", or whenever the user wants knowledge to accumulate across sessions instead of being re-derived by RAG on every query.", + "path": "engineering/llm-wiki" + }, + { + "name": "prompt-governance", + "description": "Use when managing prompts in production at scale: versioning prompts, running A/B tests on prompts, building prompt registries, preventing prompt regressions, or creating eval pipelines for production AI features. Triggers: 'manage prompts in production', 'prompt versioning', 'prompt regression', 'prompt A/B test', 'prompt registry', 'eval pipeline'. NOT for writing or improving individual prompts (use senior-prompt-engineer). NOT for RAG pipeline design (use rag-architect). NOT for LLM cost reduction (use llm-cost-optimizer).", + "path": "engineering/prompt-governance" + }, + { + "name": "security-guidance", + "description": "PreToolUse security-anti-pattern hook for Claude Code. Catches 12 common security risks (command injection, XSS, SQL injection, unsafe deserialization, GitHub Actions workflow injection, eval/new Function code injection) BEFORE the Edit/Write/MultiEdit operation completes. Session-state caching prevents duplicate warnings on the same file+rule combo. Stdlib only \u2014 no dependencies. Use when you want a safety net during Claude Code sessions that touch security-sensitive code (auth, payments, user input handling, IaC). Disable with ENABLE_SECURITY_REMINDER=0 if you need to perform a verified-safe operation that would otherwise trip a pattern. Triggers \u2014 \"add security hook\", \"block unsafe code\", \"detect command injection before write\", \"prevent SQL injection patterns\", \"security warning hook\".", + "path": "engineering/security-guidance" + }, + { + "name": "slo-architect", + "description": "Use when defining, reviewing, or operating SLOs/SLIs/error budgets. Triggers on \"define an SLO\", \"what should our SLO be\", \"error budget\", \"burn rate\", \"SLI\", \"service level objective\", \"Google SRE workbook\", \"multi-window burn-rate alert\", or any reliability-target question. Ships SLO designer, error-budget calculator with multi-window burn-rate thresholds, and SLO reviewer that catches the common bugs (target too aggressive, window too short, conflicting SLOs, no SLI definition). 4 references on SLO principles + SLI design + error budget math + composition with feature-flags-architect/chaos-engineering/kubernetes-operator. NOT a generic observability skill \u2014 specifically the SLO discipline.", + "path": "engineering/slo-architect" + }, + { + "name": "statistical-analyst", + "description": "Run hypothesis tests, analyze A/B experiment results, calculate sample sizes, and interpret statistical significance with effect sizes. Use when you need to validate whether observed differences are real, size an experiment correctly before launch, or interpret test results with confidence.", + "path": "engineering/statistical-analyst" + }, + { + "name": "terraform-patterns", + "description": "Terraform infrastructure-as-code agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Covers module design patterns, state management strategies, provider configuration, security hardening, policy-as-code with Sentinel/OPA, and CI/CD plan/apply workflows. Use when: user wants to design Terraform modules, manage state backends, review Terraform security, implement multi-region deployments, or follow IaC best practices.", + "path": "engineering/terraform-patterns" + }, + { + "name": "write-a-skill", + "description": "Create new agent skills with proper structure, progressive disclosure, and bundled resources. Use when user wants to create, write, build, or author a new skill.", + "path": "engineering/write-a-skill" + } + ], + "engineering-team": [ + { + "name": "adversarial-reviewer", + "description": "Adversarial code review that breaks the self-review monoculture. Use when you want a genuinely critical review of recent changes, before merging a PR, or when you suspect Claude is being too agreeable about code quality. Forces perspective shifts through hostile reviewer personas that catch blind spots the author's mental model shares with the reviewer.", + "path": "engineering-team/adversarial-reviewer" + }, + { + "name": "ai-security", + "description": "Use when assessing AI/ML systems for prompt injection, jailbreak vulnerabilities, model inversion risk, data poisoning exposure, or agent tool abuse. Covers MITRE ATLAS technique mapping, injection signature detection, and adversarial robustness scoring.", + "path": "engineering-team/ai-security" + }, + { + "name": "aws-solution-architect", + "description": "Design AWS architectures for startups using serverless patterns and IaC templates. Use when asked to design serverless architecture, create CloudFormation templates, optimize AWS costs, set up CI/CD pipelines, or migrate to AWS. Covers Lambda, API Gateway, DynamoDB, ECS, Aurora, and cost optimization.", + "path": "engineering-team/aws-solution-architect" + }, + { + "name": "azure-cloud-architect", + "description": "Design Azure architectures for startups and enterprises. Use when asked to design Azure infrastructure, create Bicep/ARM templates, optimize Azure costs, set up Azure DevOps pipelines, or migrate to Azure. Covers AKS, App Service, Azure Functions, Cosmos DB, and cost optimization.", + "path": "engineering-team/azure-cloud-architect" + }, + { + "name": "cloud-security", + "description": "Use when assessing cloud infrastructure for security misconfigurations, IAM privilege escalation paths, S3 public exposure, open security group rules, or IaC security gaps. Covers AWS, Azure, and GCP posture assessment with MITRE ATT&CK mapping.", + "path": "engineering-team/cloud-security" + }, + { + "name": "code-reviewer", + "description": "Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, .NET, and Java. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists.", + "path": "engineering-team/code-reviewer" + }, + { + "name": "email-template-builder", + "description": "Build complete transactional email systems: React Email templates, provider integration (Resend, Postmark, SendGrid, AWS SES), preview server, i18n support, dark mode, spam optimization, analytics tracking. Use when adding transactional email to a new product, migrating between email providers, refactoring legacy email templates for accessibility, or adding internationalization to existing templates.", + "path": "engineering-team/email-template-builder" + }, + { + "name": "engineering-skills", + "description": "23 engineering agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw, and 6 more tools. Architecture, frontend, backend, QA, DevOps, security, AI/ML, data engineering, Playwright, Stripe, AWS, MS365. 30+ Python tools (stdlib-only).", + "path": "engineering-team/engineering-skills" + }, + { + "name": "epic-design", + "description": ">", + "path": "engineering-team/epic-design" + }, + { + "name": "gcp-cloud-architect", + "description": "Design GCP architectures for startups and enterprises. Use when asked to design Google Cloud infrastructure, deploy to GKE or Cloud Run, configure BigQuery pipelines, optimize GCP costs, or migrate to GCP. Covers Cloud Run, GKE, Cloud Functions, Cloud SQL, BigQuery, and cost optimization.", + "path": "engineering-team/gcp-cloud-architect" + }, + { + "name": "incident-commander", + "description": "Comprehensive incident response framework from detection through resolution and post-incident review. Battle-tested SRE/DevOps practices: severity classification, timeline reconstruction, structured post-incident analysis. Use when declaring an incident, coordinating multi-team response during an outage, leading a post-mortem, or setting up on-call practices for a new service.", + "path": "engineering-team/incident-commander" + }, + { + "name": "incident-response", + "description": "Use when a security incident has been detected or declared and needs classification, triage, escalation path determination, and forensic evidence collection. Covers SEV1-SEV4 classification, false positive filtering, incident taxonomy, and NIST SP 800-61 lifecycle.", + "path": "engineering-team/incident-response" + }, + { + "name": "ms365-tenant-manager", + "description": "Microsoft 365 tenant administration for Global Administrators. Automate M365 tenant setup, Office 365 admin tasks, Azure AD user management, Exchange Online configuration, Teams administration, and security policies. Generate PowerShell scripts for bulk operations, Conditional Access policies, license management, and compliance reporting. Use for M365 tenant manager, Office 365 admin, Azure AD users, Global Administrator, tenant configuration, or Microsoft 365 automation.", + "path": "engineering-team/ms365-tenant-manager" + }, + { + "name": "red-team", + "description": "Use when planning or executing authorized red team engagements, attack path analysis, or offensive security simulations. Covers MITRE ATT&CK kill-chain planning, technique scoring, choke point identification, OPSEC risk assessment, and crown jewel targeting.", + "path": "engineering-team/red-team" + }, + { + "name": "security-pen-testing", + "description": "Use when the user asks to perform security audits, penetration testing, vulnerability scanning, OWASP Top 10 checks, or offensive security assessments. Covers static analysis, dependency scanning, secret detection, API security testing, and pen test report generation.", + "path": "engineering-team/security-pen-testing" + }, + { + "name": "senior-architect", + "description": "This skill should be used when the user asks to \"design system architecture\", \"evaluate microservices vs monolith\", \"create architecture diagrams\", \"analyze dependencies\", \"choose a database\", \"plan for scalability\", \"make technical decisions\", or \"review system design\". Use for architecture decision records (ADRs), tech stack evaluation, system design reviews, dependency analysis, and generating architecture diagrams in Mermaid, PlantUML, or ASCII format.", + "path": "engineering-team/senior-architect" + }, + { + "name": "senior-backend", + "description": "Designs and implements backend systems including REST APIs, microservices, database architectures, authentication flows, and security hardening. Use when the user asks to \"design REST APIs\", \"optimize database queries\", \"implement authentication\", \"build microservices\", \"review backend code\", \"set up GraphQL\", \"handle database migrations\", or \"load test APIs\". Covers Node.js/Express/Fastify development, PostgreSQL optimization, API security, and backend architecture patterns.", + "path": "engineering-team/senior-backend" + }, + { + "name": "senior-computer-vision", + "description": "Computer vision engineering skill for object detection, image segmentation, and visual AI systems. Covers CNN and Vision Transformer architectures, YOLO/Faster R-CNN/DETR detection, Mask R-CNN/SAM segmentation, and production deployment with ONNX/TensorRT. Includes PyTorch, torchvision, Ultralytics, Detectron2, and MMDetection frameworks. Use when building detection pipelines, training custom models, optimizing inference, or deploying vision systems.", + "path": "engineering-team/senior-computer-vision" + }, + { + "name": "senior-data-engineer", + "description": "Data engineering skill for building scalable data pipelines, ETL/ELT systems, and data infrastructure. Expertise in Python, SQL, Spark, Airflow, dbt, Kafka, and modern data stack. Includes data modeling, pipeline orchestration, data quality, and DataOps. Use when designing data architectures, building data pipelines, optimizing data workflows, implementing data governance, or troubleshooting data issues.", + "path": "engineering-team/senior-data-engineer" + }, + { + "name": "senior-data-scientist", + "description": "World-class senior data scientist skill specialising in statistical modeling, experiment design, causal inference, and predictive analytics. Covers A/B testing (sample sizing, two-proportion z-tests, Bonferroni correction), difference-in-differences, feature engineering pipelines (Scikit-learn, XGBoost), cross-validated model evaluation (AUC-ROC, AUC-PR, SHAP), and MLflow experiment tracking \u2014 using Python (NumPy, Pandas, Scikit-learn), R, and SQL. Use when designing or analysing controlled experiments, building and evaluating classification or regression models, performing causal analysis on observational data, engineering features for structured tabular datasets, or translating statistical findings into data-driven business decisions.", + "path": "engineering-team/senior-data-scientist" + }, + { + "name": "senior-devops", + "description": "Comprehensive DevOps skill for CI/CD, infrastructure automation, containerization, and cloud platforms (AWS, GCP, Azure). Includes pipeline setup, infrastructure as code, deployment automation, and monitoring. Use when setting up pipelines, deploying applications, managing infrastructure, implementing monitoring, or optimizing deployment processes.", + "path": "engineering-team/senior-devops" + }, + { + "name": "senior-frontend", + "description": "Frontend development skill for React, Next.js, TypeScript, and Tailwind CSS applications. Use when building React components, optimizing Next.js performance, analyzing bundle sizes, scaffolding frontend projects, implementing accessibility, or reviewing frontend code quality.", + "path": "engineering-team/senior-frontend" + }, + { + "name": "senior-fullstack", + "description": "Fullstack development toolkit with project scaffolding for Next.js, FastAPI, MERN, and Django stacks, code quality analysis with security and complexity scoring, and stack selection guidance. Use when the user asks to \"scaffold a new project\", \"create a Next.js app\", \"set up FastAPI with React\", \"analyze code quality\", \"audit my codebase\", \"what stack should I use\", \"generate project boilerplate\", or mentions fullstack development, project setup, or tech stack comparison.", + "path": "engineering-team/senior-fullstack" + }, + { + "name": "senior-ml-engineer", + "description": "ML engineering skill for productionizing models, building MLOps pipelines, and integrating LLMs. Covers model deployment, feature stores, drift monitoring, RAG systems, and cost optimization. Use when the user asks about deploying ML models to production, setting up MLOps infrastructure (MLflow, Kubeflow, Kubernetes, Docker), monitoring model performance or drift, building RAG pipelines, or integrating LLM APIs with retry logic and cost controls. Focused on production and operational concerns rather than model research or initial training.", + "path": "engineering-team/senior-ml-engineer" + }, + { + "name": "senior-prompt-engineer", + "description": "This skill should be used when the user asks to \"optimize prompts\", \"design prompt templates\", \"evaluate LLM outputs\", \"build agentic systems\", \"implement RAG\", \"create few-shot examples\", \"analyze token usage\", or \"design AI workflows\". Use for prompt engineering patterns, LLM evaluation frameworks, agent architectures, and structured output design.", + "path": "engineering-team/senior-prompt-engineer" + }, + { + "name": "senior-qa", + "description": "Generates unit tests, integration tests, and E2E tests for React/Next.js applications. Scans components to create Jest + React Testing Library test stubs, analyzes Istanbul/LCOV coverage reports to surface gaps, scaffolds Playwright test files from Next.js routes, mocks API calls with MSW, creates test fixtures, and configures test runners. Use when the user asks to \"generate tests\", \"write unit tests\", \"analyze test coverage\", \"scaffold E2E tests\", \"set up Playwright\", \"configure Jest\", \"implement testing patterns\", or \"improve test quality\".", + "path": "engineering-team/senior-qa" + }, + { + "name": "senior-secops", + "description": "Senior SecOps engineer skill for application security, vulnerability management, compliance verification, and secure development practices. Runs SAST/DAST scans, generates CVE remediation plans, checks dependency vulnerabilities, creates security policies, enforces secure coding patterns, and automates compliance checks against SOC2, PCI-DSS, HIPAA, and GDPR. Use when conducting a security review or audit, responding to a CVE or security incident, hardening infrastructure, implementing authentication or secrets management, running penetration test prep, checking OWASP Top 10 exposure, or enforcing security controls in CI/CD pipelines.", + "path": "engineering-team/senior-secops" + }, + { + "name": "senior-security", + "description": "Security engineering toolkit for threat modeling, vulnerability analysis, secure architecture, and penetration testing. Includes STRIDE analysis, OWASP guidance, cryptography patterns, and security scanning tools. Use when the user asks about security reviews, threat analysis, vulnerability assessments, secure coding practices, security audits, attack surface analysis, CVE remediation, or security best practices.", + "path": "engineering-team/senior-security" + }, + { + "name": "stripe-integration-expert", + "description": "Production-grade Stripe integrations: subscriptions with trials and proration, one-time payments, usage-based billing, checkout sessions, idempotent webhook handlers, customer portal, and invoicing. Covers Next.js, Express, and Django patterns. Use when integrating Stripe for the first time, debugging webhook reliability issues, migrating from a different payment provider, or adding usage-based billing to an existing subscription product.", + "path": "engineering-team/stripe-integration-expert" + }, + { + "name": "tdd-guide", + "description": "Test-driven development skill for writing unit tests, generating test fixtures and mocks, analyzing coverage gaps, and guiding red-green-refactor workflows across Jest, Pytest, JUnit, Vitest, and Mocha. Use when the user asks to write tests, improve test coverage, practice TDD, generate mocks or stubs, or mentions testing frameworks like Jest, pytest, or JUnit.", + "path": "engineering-team/tdd-guide" + }, + { + "name": "tech-stack-evaluator", + "description": "Technology stack evaluation and comparison with TCO analysis, security assessment, and ecosystem health scoring. Use when comparing frameworks, evaluating technology stacks, calculating total cost of ownership, assessing migration paths, or analyzing ecosystem viability.", + "path": "engineering-team/tech-stack-evaluator" + }, + { + "name": "threat-detection", + "description": "Use when hunting for threats in an environment, analyzing IOCs, or detecting behavioral anomalies in telemetry. Covers hypothesis-driven threat hunting, IOC sweep generation, z-score anomaly detection, and MITRE ATT&CK-mapped signal prioritization.", + "path": "engineering-team/threat-detection" + }, + { + "name": "a11y-audit", + "description": "Accessibility audit skill for scanning, fixing, and verifying WCAG 2.2 Level A and AA compliance across React, Next.js, Vue, Angular, Svelte, and plain HTML codebases. Use when auditing accessibility, fixing a11y violations, checking color contrast, generating compliance reports, or integrating accessibility checks into CI/CD pipelines.", + "path": "engineering-team/a11y-audit" + }, + { + "name": "google-workspace-cli", + "description": "Google Workspace administration via the gws CLI. Install, authenticate, and automate Gmail, Drive, Sheets, Calendar, Docs, Chat, and Tasks. Run security audits, execute 43 built-in recipes, and use 10 persona bundles. Use for Google Workspace admin, gws CLI setup, Gmail automation, Drive management, or Calendar scheduling.", + "path": "engineering-team/google-workspace-cli" + }, + { + "name": "browserstack", + "description": ">-", + "path": "engineering-team/browserstack" + }, + { + "name": "coverage", + "description": ">-", + "path": "engineering-team/coverage" + }, + { + "name": "fix", + "description": ">-", + "path": "engineering-team/fix" + }, + { + "name": "generate", + "description": ">-", + "path": "engineering-team/generate" + }, + { + "name": "init", + "description": ">-", + "path": "engineering-team/init" + }, + { + "name": "migrate", + "description": ">-", + "path": "engineering-team/migrate" + }, + { + "name": "pw", + "description": "Production-grade Playwright testing toolkit. Use when the user mentions Playwright tests, end-to-end testing, browser automation, fixing flaky tests, test migration, CI/CD testing, or test suites. Generate tests, fix flaky failures, migrate from Cypress/Selenium, sync with TestRail, run on BrowserStack. 55 templates, 3 agents, smart reporting.", + "path": "engineering-team/pw" + }, + { + "name": "report", + "description": ">-", + "path": "engineering-team/report" + }, + { + "name": "review", + "description": ">-", + "path": "engineering-team/review" + }, + { + "name": "testrail", + "description": ">-", + "path": "engineering-team/testrail" + }, + { + "name": "extract", + "description": "Turn a proven pattern or debugging solution into a standalone reusable skill with SKILL.md, reference docs, and examples.", + "path": "engineering-team/extract" + }, + { + "name": "promote", + "description": "Graduate a proven pattern from auto-memory (MEMORY.md) to CLAUDE.md or .claude/rules/ for permanent enforcement.", + "path": "engineering-team/promote" + }, + { + "name": "remember", + "description": "Explicitly save important knowledge to auto-memory with timestamp and context. Use when a discovery is too important to rely on auto-capture.", + "path": "engineering-team/remember" + }, + { + "name": "review", + "description": "Analyze auto-memory for promotion candidates, stale entries, consolidation opportunities, and health metrics.", + "path": "engineering-team/review" + }, + { + "name": "self-improving-agent", + "description": "Curate Claude Code's auto-memory into durable project knowledge. Analyze MEMORY.md for patterns, promote proven learnings to CLAUDE.md and .claude/rules/, extract recurring solutions into reusable skills. Use when: (1) reviewing what Claude has learned about your project, (2) graduating a pattern from notes to enforced rules, (3) turning a debugging solution into a skill, (4) checking memory health and capacity.", + "path": "engineering-team/self-improving-agent" + }, + { + "name": "status", + "description": "Memory health dashboard showing line counts, topic files, capacity, stale entries, and recommendations.", + "path": "engineering-team/status" + }, + { + "name": "snowflake-development", + "description": "Use when writing Snowflake SQL, building data pipelines with Dynamic Tables or Streams/Tasks, using Cortex AI functions, creating Cortex Agents, writing Snowpark Python, configuring dbt for Snowflake, or troubleshooting Snowflake errors.", + "path": "engineering-team/snowflake-development" + } + ], + "product-team": [ + { + "name": "competitive-teardown", + "description": "Analyzes competitor products and companies by synthesizing data from pricing pages, app store reviews, job postings, SEO signals, and social media into structured competitive intelligence. Produces feature comparison matrices scored across 12 dimensions, SWOT analyses, positioning maps, UX audits, pricing model breakdowns, action item roadmaps, and stakeholder presentation templates. Use when conducting competitor analysis, comparing products against competitors, researching the competitive landscape, building battle cards for sales, preparing for a product strategy or roadmap session, responding to a competitor's new feature or pricing change, or performing a quarterly competitive review.", + "path": "product-team/competitive-teardown" + }, + { + "name": "experiment-designer", + "description": "Use when planning product experiments, writing testable hypotheses, estimating sample size, prioritizing tests, or interpreting A/B outcomes with practical statistical rigor.", + "path": "product-team/experiment-designer" + }, + { + "name": "landing-page-generator", + "description": "Generates high-converting landing pages as complete Next.js/React (TSX) components with Tailwind CSS. Creates hero sections, feature grids, pricing tables, FAQ accordions, testimonial blocks, and CTA sections using proven copy frameworks (PAS, AIDA, BAB). Outputs SEO meta tags, structured data, and performance-optimised code targeting Core Web Vitals (LCP < 1s, CLS < 0.1). Use when the user asks to create a landing page, marketing page, homepage, single-page site, lead capture page, campaign page, promo page, or conversion-optimised web page \u2014 or when they want to A/B test landing page variants or replace a static page with one designed to convert.", + "path": "product-team/landing-page-generator" + }, + { + "name": "product-analytics", + "description": "Use when defining product KPIs, building metric dashboards, running cohort or retention analysis, or interpreting feature adoption trends across product stages.", + "path": "product-team/product-analytics" + }, + { + "name": "product-discovery", + "description": "Use when validating product opportunities, mapping assumptions, planning discovery sprints, or testing problem-solution fit before committing delivery resources.", + "path": "product-team/product-discovery" + }, + { + "name": "product-manager-toolkit", + "description": "Comprehensive toolkit for product managers including RICE prioritization, customer interview analysis, PRD templates, discovery frameworks, and go-to-market strategies. Use for feature prioritization, user research synthesis, requirement documentation, and product strategy development.", + "path": "product-team/product-manager-toolkit" + }, + { + "name": "product-skills", + "description": "10 product agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. PM toolkit (RICE), agile PO, product strategist (OKR), UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, research summarizer. Python tools (stdlib-only).", + "path": "product-team/product-skills" + }, + { + "name": "product-strategist", + "description": "Strategic product leadership toolkit for Head of Product covering OKR cascade generation, quarterly planning, competitive landscape analysis, product vision documents, and team scaling proposals. Use when creating quarterly OKR documents, defining product goals or KPIs, building product roadmaps, running competitive analysis, drafting team structure or hiring plans, aligning product strategy across engineering and design, or generating cascaded goal hierarchies from company to team level.", + "path": "product-team/product-strategist" + }, + { + "name": "roadmap-communicator", + "description": "Use when preparing roadmap narratives, release notes, changelogs, or stakeholder updates tailored for executives, engineering teams, and customers.", + "path": "product-team/roadmap-communicator" + }, + { + "name": "saas-scaffolder", + "description": "Generates complete, production-ready SaaS project boilerplate including authentication, database schemas, billing integration, API routes, and a working dashboard using Next.js 14+ App Router, TypeScript, Tailwind CSS, shadcn/ui, Drizzle ORM, and Stripe. Use when the user wants to create a new SaaS app, start a subscription-based web project, scaffold a Next.js application, or mentions terms like starter template, boilerplate, new project, or wiring up auth and payments.", + "path": "product-team/saas-scaffolder" + }, + { + "name": "spec-to-repo", + "description": "Use when the user says 'build me an app', 'create a project from this spec', 'scaffold a new repo', 'generate a starter', 'turn this idea into code', 'bootstrap a project', 'I have requirements and need a codebase', or provides a natural-language project specification and expects a complete, runnable repository. Stack-agnostic: Next.js, FastAPI, Rails, Go, Rust, Flutter, and more.", + "path": "product-team/spec-to-repo" + }, + { + "name": "ui-design-system", + "description": "UI design system toolkit for Senior UI Designer including design token generation, component documentation, responsive design calculations, and developer handoff tools. Use for creating design systems, maintaining visual consistency, and facilitating design-dev collaboration.", + "path": "product-team/ui-design-system" + }, + { + "name": "ux-researcher-designer", + "description": "UX research and design toolkit for Senior UX Designer/Researcher including data-driven persona generation, journey mapping, usability testing frameworks, and research synthesis. Use for user research, persona creation, journey mapping, and design validation.", + "path": "product-team/ux-researcher-designer" + }, + { + "name": "agile-product-owner", + "description": "Agile product ownership for backlog management and sprint execution. Covers user story writing, acceptance criteria, sprint planning, and velocity tracking. Use for writing user stories, creating acceptance criteria, planning sprints, estimating story points, breaking down epics, or prioritizing backlog.", + "path": "product-team/agile-product-owner" + }, + { + "name": "apple-hig-expert", + "description": "Expert guidance on Apple Human Interface Guidelines (HIG). Covers iOS, macOS, and visionOS with 2026 Liquid Glass aesthetics and accessibility-first design.", + "path": "product-team/apple-hig-expert" + }, + { + "name": "code-to-prd", + "description": "|", + "path": "product-team/code-to-prd" + }, + { + "name": "research-summarizer", + "description": "Structured research summarization agent skill for non-dev users. Handles academic papers, web articles, reports, and documentation. Extracts key findings, generates comparative analyses, and produces properly formatted citations. Use when: user wants to summarize a research paper, compare multiple sources, extract citations from documents, or create structured research briefs. Plugin for Claude Code, Codex, Gemini CLI, and OpenClaw.", + "path": "product-team/research-summarizer" + } + ], + "marketing-skill": [ + { + "name": "ab-test-setup", + "description": "When the user wants to plan, design, or implement an A/B test or experiment. Also use when the user mentions \"A/B test,\" \"split test,\" \"experiment,\" \"test this change,\" \"variant copy,\" \"multivariate test,\" \"hypothesis,\" \"conversion experiment,\" \"statistical significance,\" or \"test this.\" For tracking implementation, see analytics-tracking.", + "path": "marketing-skill/ab-test-setup" + }, + { + "name": "ad-creative", + "description": "When the user needs to generate, iterate, or scale ad creative for paid advertising. Use when they say 'write ad copy,' 'generate headlines,' 'create ad variations,' 'bulk creative,' 'iterate on ads,' 'ad copy validation,' 'RSA headlines,' 'Meta ad copy,' 'LinkedIn ad,' or 'creative testing.' This is pure creative production \u2014 distinct from paid-ads (campaign strategy). Use ad-creative when you need the copy, not the campaign plan.", + "path": "marketing-skill/ad-creative" + }, + { + "name": "aeo", + "description": "Answer Engine Optimization (AEO) skill \u2014 optimize content to be cited by AI language models (ChatGPT, Perplexity, Claude, Gemini, Mistral) as authoritative sources. Distinct from SEO \u2014 AEO optimizes for citation in LLM-generated responses, not search rankings. Use when planning content for AI-first search audiences, auditing existing content for E-E-A-T signals, tracking which pages get cited by which LLMs, or building a citation-friendly content strategy. Triggers \u2014 'AEO audit', 'optimize for ChatGPT', 'get cited by Perplexity', 'LLM citation strategy', 'answer engine optimization', 'content for AI search', 'E-E-A-T audit'. Output is a markdown audit report (default) or JSON for pipeline integration. Stdlib-only Python tools.", + "path": "marketing-skill/aeo" + }, + { + "name": "ai-seo", + "description": "Optimize content to get cited by AI search engines \u2014 ChatGPT, Perplexity, Google AI Overviews, Claude, Gemini, Copilot. Use when you want your content to appear in AI-generated answers, not just ranked in blue links. Triggers: 'optimize for AI search', 'get cited by ChatGPT', 'AI Overviews', 'Perplexity citations', 'AI SEO', 'generative search', 'LLM visibility', 'GEO' (generative engine optimization). NOT for traditional SEO ranking (use seo-audit). NOT for content creation (use content-production).", + "path": "marketing-skill/ai-seo" + }, + { + "name": "analytics-tracking", + "description": "Set up, audit, and debug analytics tracking implementation \u2014 GA4, Google Tag Manager, event taxonomy, conversion tracking, and data quality. Use when building a tracking plan from scratch, auditing existing analytics for gaps or errors, debugging missing events, or setting up GTM. Trigger keywords: GA4 setup, Google Tag Manager, GTM, event tracking, analytics implementation, conversion tracking, tracking plan, event taxonomy, custom dimensions, UTM tracking, analytics audit, missing events, tracking broken. NOT for analyzing marketing campaign data \u2014 use campaign-analytics for that. NOT for BI dashboards \u2014 use product-analytics for in-product event analysis.", + "path": "marketing-skill/analytics-tracking" + }, + { + "name": "app-store-optimization", + "description": "App Store Optimization (ASO) toolkit for researching keywords, analyzing competitor rankings, generating metadata suggestions, and improving app visibility on Apple App Store and Google Play Store. Use when the user asks about ASO, app store rankings, app metadata, app titles and descriptions, app store listings, app visibility, or mobile app marketing on iOS or Android. Supports keyword research and scoring, competitor keyword analysis, metadata optimization, A/B test planning, launch checklists, and tracking ranking changes.", + "path": "marketing-skill/app-store-optimization" + }, + { + "name": "brand-guidelines", + "description": "When the user wants to apply, document, or enforce brand guidelines for any product or company. Also use when the user mentions 'brand guidelines,' 'brand colors,' 'typography,' 'logo usage,' 'brand voice,' 'visual identity,' 'tone of voice,' 'brand standards,' 'style guide,' 'brand consistency,' or 'company design standards.' Covers color systems, typography, logo rules, imagery guidelines, and tone matrix for any brand \u2014 including Anthropic's official identity.", + "path": "marketing-skill/brand-guidelines" + }, + { + "name": "campaign-analytics", + "description": "Analyzes campaign performance with multi-touch attribution, funnel conversion analysis, and ROI calculation for marketing optimization. Use when analyzing marketing campaigns, ad performance, attribution models, conversion rates, or calculating marketing ROI, ROAS, CPA, and campaign metrics across channels.", + "path": "marketing-skill/campaign-analytics" + }, + { + "name": "churn-prevention", + "description": "Reduce voluntary and involuntary churn through cancel flow design, save offers, exit surveys, and dunning sequences. Use when designing or optimizing a cancel flow, building save offers, setting up dunning emails, or reducing failed-payment churn. Trigger keywords: cancel flow, churn reduction, save offers, dunning, exit survey, payment recovery, win-back, involuntary churn, failed payments, cancel page. NOT for customer health scoring or expansion revenue \u2014 use customer-success-manager for that.", + "path": "marketing-skill/churn-prevention" + }, + { + "name": "cold-email", + "description": "When the user wants to write, improve, or build a sequence of B2B cold outreach emails to prospects who haven't asked to hear from them. Use when the user mentions 'cold email,' 'cold outreach,' 'prospecting emails,' 'SDR emails,' 'sales emails,' 'first touch email,' 'follow-up sequence,' or 'email prospecting.' Also use when they share an email draft that sounds too sales-y and needs to be humanized. Distinct from email-sequence (lifecycle/nurture to opted-in subscribers) \u2014 this is unsolicited outreach to new prospects. NOT for lifecycle emails, newsletters, or drip campaigns (use email-sequence).", + "path": "marketing-skill/cold-email" + }, + { + "name": "competitor-alternatives", + "description": "When the user wants to create competitor comparison or alternative pages for SEO and sales enablement. Also use when the user mentions 'alternative page,' 'vs page,' 'competitor comparison,' 'comparison page,' '[Product] vs [Product],' '[Product] alternative,' 'competitive landing pages,' 'switch from competitor,' or 'comparison content.' Covers four formats: singular alternative, plural alternatives, you vs competitor, and competitor vs competitor. Emphasizes deep research, modular content architecture, and varied section types beyond feature tables.", + "path": "marketing-skill/competitor-alternatives" + }, + { + "name": "content-creator", + "description": "Deprecated redirect skill that routes legacy 'content creator' requests to the correct specialist. Use when a user invokes 'content creator', asks to write a blog post, article, guide, or brand voice analysis (routes to content-production), or asks to plan content, build a topic cluster, or create a content calendar (routes to content-strategy). Does not handle requests directly \u2014 identifies user intent and redirects to content-production for writing/SEO/brand-voice tasks or content-strategy for planning tasks.", + "path": "marketing-skill/content-creator" + }, + { + "name": "content-humanizer", + "description": "Makes AI-generated content sound genuinely human \u2014 not just cleaned up, but alive. Use when content feels robotic, uses too many AI clich\u00e9s, lacks personality, or reads like it was written by committee. Triggers: 'this sounds like AI', 'make it more human', 'add personality', 'it feels generic', 'sounds robotic', 'fix AI writing', 'inject our voice'. NOT for initial content creation (use content-production). NOT for SEO optimization (use content-production Mode 3).", + "path": "marketing-skill/content-humanizer" + }, + { + "name": "content-production", + "description": "Full content production pipeline \u2014 takes a topic from blank page to published-ready piece. Use when you need to execute content: write a blog post, article, or guide end-to-end. Triggers: 'write a post about', 'draft an article', 'create content for', 'help me write', 'I need a blog post'. NOT for content strategy or calendar planning (use content-strategy). NOT for repurposing existing content (use content-repurposing). NOT for social captions only.", + "path": "marketing-skill/content-production" + }, + { + "name": "content-strategy", + "description": "When the user wants to plan a content strategy, decide what content to create, or figure out what topics to cover. Also use when the user mentions \\\"content strategy,\\\" \\\"what should I write about,\\\" \\\"content ideas,\\\" \\\"blog strategy,\\\" \\\"topic clusters,\\\" or \\\"content planning.\\\" For writing individual pieces, see copywriting. For SEO-specific audits, see seo-audit.", + "path": "marketing-skill/content-strategy" + }, + { + "name": "copy-editing", + "description": "When the user wants to edit, review, or improve existing marketing copy. Also use when the user mentions 'edit this copy,' 'review my copy,' 'copy feedback,' 'proofread,' 'polish this,' 'make this better,' or 'copy sweep.' This skill provides a systematic approach to editing marketing copy through multiple focused passes.", + "path": "marketing-skill/copy-editing" + }, + { + "name": "copywriting", + "description": "When the user wants to write, rewrite, or improve marketing copy for any page \u2014 including homepage, landing pages, pricing pages, feature pages, about pages, or product pages. Also use when the user says \\\"write copy for,\\\" \\\"improve this copy,\\\" \\\"rewrite this page,\\\" \\\"marketing copy,\\\" \\\"headline help,\\\" or \\\"CTA copy.\\\" For email copy, see email-sequence. For popup copy, see popup-cro.", + "path": "marketing-skill/copywriting" + }, + { + "name": "email-sequence", + "description": "When the user wants to create or optimize an email sequence, drip campaign, automated email flow, or lifecycle email program. Also use when the user mentions \"email sequence,\" \"drip campaign,\" \"nurture sequence,\" \"onboarding emails,\" \"welcome sequence,\" \"re-engagement emails,\" \"email automation,\" or \"lifecycle emails.\" For in-app onboarding, see onboarding-cro.", + "path": "marketing-skill/email-sequence" + }, + { + "name": "form-cro", + "description": "When the user wants to optimize any form that is NOT signup/registration \u2014 including lead capture forms, contact forms, demo request forms, application forms, survey forms, or checkout forms. Also use when the user mentions \"form optimization,\" \"lead form conversions,\" \"form friction,\" \"form fields,\" \"form completion rate,\" or \"contact form.\" For signup/registration forms, see signup-flow-cro. For popups containing forms, see popup-cro.", + "path": "marketing-skill/form-cro" + }, + { + "name": "free-tool-strategy", + "description": "When the user wants to build a free tool for marketing \u2014 lead generation, SEO value, or brand awareness. Use when they mention 'engineering as marketing,' 'free tool,' 'calculator,' 'generator,' 'checker,' 'grader,' 'marketing tool,' 'lead gen tool,' 'build something for traffic,' 'interactive tool,' or 'free resource.' Covers idea evaluation, tool design, and launch strategy. For pure SEO content strategy (no tool), use seo-audit or content-strategy instead.", + "path": "marketing-skill/free-tool-strategy" + }, + { + "name": "launch-strategy", + "description": "When the user wants to plan a product launch, feature announcement, or release strategy. Also use when the user mentions 'launch,' 'Product Hunt,' 'feature release,' 'announcement,' 'go-to-market,' 'beta launch,' 'early access,' 'waitlist,' 'product update,' 'GTM plan,' 'launch checklist,' or 'launch momentum.' This skill covers phased launches, channel strategy, and ongoing launch momentum.", + "path": "marketing-skill/launch-strategy" + }, + { + "name": "marketing-context", + "description": "Create and maintain the marketing context document that all marketing skills read before starting. Use when the user mentions 'marketing context,' 'brand voice,' 'set up context,' 'target audience,' 'ICP,' 'style guide,' 'who is my customer,' 'positioning,' or wants to avoid repeating foundational information across marketing tasks. Run this at the start of any new project before using other marketing skills.", + "path": "marketing-skill/marketing-context" + }, + { + "name": "marketing-demand-acquisition", + "description": "Creates demand generation campaigns, optimizes paid ad spend across LinkedIn, Google, and Meta, develops SEO strategies, and structures partnership programs for Series A+ startups scaling internationally. Use when planning marketing strategy, growth marketing, advertising campaigns, PPC optimization, lead generation, pipeline generation, or startup marketing budgets. Covers multi-channel acquisition (Google Ads, LinkedIn Ads, Meta Ads), CAC analysis, MQL/SQL workflows, attribution modeling, technical SEO, and co-marketing partnerships for hybrid PLG/Sales-Led motions in EU/US/Canada markets.", + "path": "marketing-skill/marketing-demand-acquisition" + }, + { + "name": "marketing-ideas", + "description": "When the user needs marketing ideas, inspiration, or strategies for their SaaS or software product. Also use when the user asks for 'marketing ideas,' 'growth ideas,' 'how to market,' 'marketing strategies,' 'marketing tactics,' 'ways to promote,' or 'ideas to grow.' This skill provides 139 proven marketing approaches organized by category.", + "path": "marketing-skill/marketing-ideas" + }, + { + "name": "marketing-ops", + "description": "Central router for the marketing skill ecosystem. Use when unsure which marketing skill to use, when orchestrating a multi-skill campaign, or when coordinating across content, SEO, CRO, channels, and analytics. Also use when the user mentions 'marketing help,' 'campaign plan,' 'what should I do next,' 'marketing priorities,' or 'coordinate marketing.", + "path": "marketing-skill/marketing-ops" + }, + { + "name": "marketing-psychology", + "description": "When the user wants to apply psychological principles, mental models, or behavioral science to marketing. Also use when the user mentions 'psychology,' 'mental models,' 'cognitive bias,' 'persuasion,' 'behavioral science,' 'why people buy,' 'decision-making,' or 'consumer behavior.' This skill provides 70+ mental models organized for marketing application.", + "path": "marketing-skill/marketing-psychology" + }, + { + "name": "marketing-skills", + "description": "42 marketing agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw, and 6 more coding agents. 7 pods: content, SEO, CRO, channels, growth, intelligence, sales. Foundation context + orchestration router. 27 Python tools (stdlib-only).", + "path": "marketing-skill/marketing-skills" + }, + { + "name": "marketing-strategy-pmm", + "description": "Product marketing skill for positioning, GTM strategy, competitive intelligence, and product launches. Use when the user asks about product positioning, go-to-market planning, competitive analysis, target audience definition, ICP definition, market research, launch plans, or sales enablement. Covers April Dunford positioning, ICP definition, competitive battlecards, launch playbooks, and international market entry. Produces deliverables including positioning statements, battlecard documents, launch plans, and go-to-market strategies.", + "path": "marketing-skill/marketing-strategy-pmm" + }, + { + "name": "onboarding-cro", + "description": "When the user wants to optimize post-signup onboarding, user activation, first-run experience, or time-to-value. Also use when the user mentions \"onboarding flow,\" \"activation rate,\" \"user activation,\" \"first-run experience,\" \"empty states,\" \"onboarding checklist,\" \"aha moment,\" or \"new user experience.\" For signup/registration optimization, see signup-flow-cro. For ongoing email sequences, see email-sequence.", + "path": "marketing-skill/onboarding-cro" + }, + { + "name": "page-cro", + "description": "When the user wants to optimize, improve, or increase conversions on any marketing page \u2014 including homepage, landing pages, pricing pages, feature pages, or blog posts. Also use when the user says \"CRO,\" \"conversion rate optimization,\" \"this page isn't converting,\" \"improve conversions,\" or \"why isn't this page working.\" For signup/registration flows, see signup-flow-cro. For post-signup activation, see onboarding-cro. For forms outside of signup, see form-cro. For popups/modals, see popup-cro.", + "path": "marketing-skill/page-cro" + }, + { + "name": "paid-ads", + "description": "When the user wants help with paid advertising campaigns on Google Ads, Meta (Facebook/Instagram), LinkedIn, Twitter/X, or other ad platforms. Also use when the user mentions 'PPC,' 'paid media,' 'ad copy,' 'ad creative,' 'ROAS,' 'CPA,' 'ad campaign,' 'retargeting,' or 'audience targeting.' This skill covers campaign strategy, ad creation, audience targeting, and optimization.", + "path": "marketing-skill/paid-ads" + }, + { + "name": "paywall-upgrade-cro", + "description": "When the user wants to create or optimize in-app paywalls, upgrade screens, upsell modals, or feature gates. Also use when the user mentions \"paywall,\" \"upgrade screen,\" \"upgrade modal,\" \"upsell,\" \"feature gate,\" \"convert free to paid,\" \"freemium conversion,\" \"trial expiration screen,\" \"limit reached screen,\" \"plan upgrade prompt,\" or \"in-app pricing.\" Distinct from public pricing pages (see page-cro) \u2014 this skill focuses on in-product upgrade moments where the user has already experienced value.", + "path": "marketing-skill/paywall-upgrade-cro" + }, + { + "name": "popup-cro", + "description": "When the user wants to create or optimize popups, modals, overlays, slide-ins, or banners for conversion purposes. Also use when the user mentions \"exit intent,\" \"popup conversions,\" \"modal optimization,\" \"lead capture popup,\" \"email popup,\" \"announcement banner,\" or \"overlay.\" For forms outside of popups, see form-cro. For general page conversion optimization, see page-cro.", + "path": "marketing-skill/popup-cro" + }, + { + "name": "pricing-strategy", + "description": "Design, optimize, and communicate SaaS pricing \u2014 tier structure, value metrics, pricing pages, and price increase strategy. Use when building a pricing model from scratch, redesigning existing pricing, planning a price increase, or improving a pricing page. Trigger keywords: pricing tiers, pricing page, price increase, packaging, value metric, per seat pricing, usage-based pricing, freemium, good-better-best, pricing strategy, monetization, pricing page conversion, Van Westendorp. NOT for broader product strategy \u2014 use product-strategist for that. NOT for customer success or renewals \u2014 use customer-success-manager for expansion revenue.", + "path": "marketing-skill/pricing-strategy" + }, + { + "name": "programmatic-seo", + "description": "When the user wants to create SEO-driven pages at scale using templates and data. Also use when the user mentions \"programmatic SEO,\" \"template pages,\" \"pages at scale,\" \"directory pages,\" \"location pages,\" \"[keyword] + [city] pages,\" \"comparison pages,\" \"integration pages,\" or \"building many pages for SEO.\" For auditing existing SEO issues, see seo-audit.", + "path": "marketing-skill/programmatic-seo" + }, + { + "name": "prompt-engineer-toolkit", + "description": "Analyzes and rewrites prompts for better AI output, creates reusable prompt templates for marketing use cases (ad copy, email campaigns, social media), and structures end-to-end AI content workflows. Use when the user wants to improve prompts for AI-assisted marketing, build prompt templates, or optimize AI content workflows. Also use when the user mentions 'prompt engineering,' 'improve my prompts,' 'AI writing quality,' 'prompt templates,' or 'AI content workflow.", + "path": "marketing-skill/prompt-engineer-toolkit" + }, + { + "name": "referral-program", + "description": "When the user wants to design, launch, or optimize a referral or affiliate program. Use when they mention 'referral program,' 'affiliate program,' 'word of mouth,' 'refer a friend,' 'incentive program,' 'customer referrals,' 'brand ambassador,' 'partner program,' 'referral link,' or 'growth through referrals.' Covers program mechanics, incentive design, and optimization \u2014 not just the idea of referrals but the actual system.", + "path": "marketing-skill/referral-program" + }, + { + "name": "schema-markup", + "description": "When the user wants to implement, audit, or validate structured data (schema markup) on their website. Use when the user mentions 'structured data,' 'schema.org,' 'JSON-LD,' 'rich results,' 'rich snippets,' 'schema markup,' 'FAQ schema,' 'Product schema,' 'HowTo schema,' or 'structured data errors in Search Console.' Also use when someone asks why their content isn't showing rich results or wants to improve AI search visibility. NOT for general SEO audits (use seo-audit) or technical SEO crawl issues (use site-architecture).", + "path": "marketing-skill/schema-markup" + }, + { + "name": "seo-audit", + "description": "When the user wants to audit, review, or diagnose SEO issues on their site. Also use when the user mentions \"SEO audit,\" \"technical SEO,\" \"why am I not ranking,\" \"SEO issues,\" \"on-page SEO,\" \"meta tags review,\" or \"SEO health check.\" For building pages at scale to target keywords, see programmatic-seo. For adding structured data, see schema-markup.", + "path": "marketing-skill/seo-audit" + }, + { + "name": "signup-flow-cro", + "description": "When the user wants to optimize signup, registration, account creation, or trial activation flows. Also use when the user mentions \"signup conversions,\" \"registration friction,\" \"signup form optimization,\" \"free trial signup,\" \"reduce signup dropoff,\" or \"account creation flow.\" For post-signup onboarding, see onboarding-cro. For lead capture forms (not account creation), see form-cro.", + "path": "marketing-skill/signup-flow-cro" + }, + { + "name": "site-architecture", + "description": "When the user wants to audit, redesign, or plan their website's structure, URL hierarchy, navigation design, or internal linking strategy. Use when the user mentions 'site architecture,' 'URL structure,' 'internal links,' 'site navigation,' 'breadcrumbs,' 'topic clusters,' 'hub pages,' 'orphan pages,' 'silo structure,' 'information architecture,' or 'website reorganization.' Also use when someone has SEO problems and the root cause is structural (not content or schema). NOT for content strategy decisions about what to write (use content-strategy) or for schema markup (use schema-markup).", + "path": "marketing-skill/site-architecture" + }, + { + "name": "social-content", + "description": "When the user wants help creating, scheduling, or optimizing social media content for LinkedIn, Twitter/X, Instagram, TikTok, Facebook, or other platforms. Also use when the user mentions 'LinkedIn post,' 'Twitter thread,' 'social media,' 'content calendar,' 'social scheduling,' 'engagement,' or 'viral content.' This skill covers content creation, repurposing, and platform-specific strategies.", + "path": "marketing-skill/social-content" + }, + { + "name": "social-media-analyzer", + "description": "Social media campaign analysis and performance tracking. Calculates engagement rates, ROI, and benchmarks across platforms. Use for analyzing social media performance, calculating engagement rate, measuring campaign ROI, comparing platform metrics, or benchmarking against industry standards.", + "path": "marketing-skill/social-media-analyzer" + }, + { + "name": "social-media-manager", + "description": "When the user wants to develop social media strategy, plan content calendars, manage community engagement, or grow their social presence across platforms. Also use when the user mentions 'social media strategy,' 'social calendar,' 'community management,' 'social media plan,' 'grow followers,' 'engagement rate,' 'social media audit,' or 'which platforms should I use.' For writing individual social posts, see social-content. For analyzing social performance data, see social-media-analyzer.", + "path": "marketing-skill/social-media-manager" + }, + { + "name": "x-twitter-growth", + "description": "X/Twitter growth engine for building audience, crafting viral content, and analyzing engagement. Use when the user wants to grow on X/Twitter, write tweets or threads, analyze their X profile, research competitors on X, plan a posting strategy, or optimize engagement. Complements social-content (generic multi-platform) with X-specific depth: algorithm mechanics, thread engineering, reply strategy, profile optimization, and competitive intelligence via web search.", + "path": "marketing-skill/x-twitter-growth" + }, + { + "name": "video-content-strategist", + "description": "Use when planning video content strategy, writing video scripts, optimizing YouTube channels, building short-form video pipelines (Reels, TikTok, Shorts), or repurposing long-form content into video. Triggers: 'start a YouTube channel', 'video content strategy', 'write a video script', 'repurpose into video', 'YouTube SEO', 'short-form video'. NOT for written blog content (use content-production). NOT for social captions without video (use social-media-manager).", + "path": "marketing-skill/video-content-strategist" + } + ], + "c-level-advisor": [ + { + "name": "agent-protocol", + "description": "Inter-agent communication protocol for C-suite agent teams. Defines invocation syntax, loop prevention, isolation rules, and response formats. Use when C-suite agents need to query each other, coordinate cross-functional analysis, or run board meetings with multiple agent roles.", + "path": "c-level-advisor/agent-protocol" + }, + { + "name": "board-deck-builder", + "description": "Assembles comprehensive board and investor update decks by pulling perspectives from all C-suite roles. Use when preparing board meetings, investor updates, quarterly business reviews, or fundraising narratives. Covers structure, narrative framework, bad news delivery, and common mistakes.", + "path": "c-level-advisor/board-deck-builder" + }, + { + "name": "board-meeting", + "description": "Multi-agent board meeting protocol for strategic decisions. Runs a structured 6-phase deliberation: context loading, independent C-suite contributions (isolated, no cross-pollination), critic analysis, synthesis, founder review, and decision extraction. Use when the user invokes /cs:board, calls a board meeting, or wants structured multi-perspective executive deliberation on a strategic question.", + "path": "c-level-advisor/board-meeting" + }, + { + "name": "c-level-skills", + "description": "10 C-level advisory agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. CEO, CTO, COO, CPO, CMO, CFO, CRO, CISO, CHRO, Executive Mentor. Multi-role board meetings, strategy routing, structured recommendations. For founders needing executive-level decision support.", + "path": "c-level-advisor/c-level-skills" + }, + { + "name": "ceo-advisor", + "description": "Executive leadership guidance for strategic decision-making, organizational development, and stakeholder management. Use when planning strategy, preparing board presentations, managing investors, developing organizational culture, making executive decisions, fundraising, or when user mentions CEO, strategic planning, board meetings, investor updates, organizational leadership, or executive strategy.", + "path": "c-level-advisor/ceo-advisor" + }, + { + "name": "cfo-advisor", + "description": "Financial leadership for startups and scaling companies. Financial modeling, unit economics, fundraising strategy, cash management, and board financial packages. Use when building financial models, analyzing unit economics, planning fundraising, managing cash runway, preparing board materials, or when user mentions CFO, burn rate, runway, fundraising, unit economics, LTV, CAC, term sheets, or financial strategy.", + "path": "c-level-advisor/cfo-advisor" + }, + { + "name": "change-management", + "description": "Framework for rolling out organizational changes without chaos. Covers the ADKAR model adapted for startups, communication templates, resistance patterns, and change fatigue management. Handles process changes, org restructures, strategy pivots, and culture changes. Use when announcing a reorg, switching tools, pivoting strategy, killing a product, changing leadership, or when user mentions change management, change rollout, managing resistance, org change, reorg, or pivot communication.", + "path": "c-level-advisor/change-management" + }, + { + "name": "chief-ai-officer-advisor", + "description": "Chief AI Officer advisory for startups: model build-vs-buy decisions (API vs fine-tune vs in-house), AI risk classification under EU AI Act + US state patchwork, AI cost economics (API-to-self-hosted breakeven), and AI team org evolution. Use when deciding whether to call an API or fine-tune, classifying AI use cases for regulatory risk, calculating when self-hosting pays off, sequencing AI hires, or when user mentions CAIO, AI strategy, model selection, foundation model, fine-tuning, EU AI Act, NIST AI RMF, AI governance, model risk, or AI economics. Strategic only \u2014 does not duplicate engineering AI/ML skills.", + "path": "c-level-advisor/chief-ai-officer-advisor" + }, + { + "name": "chief-customer-officer-advisor", + "description": "Chief Customer Officer advisory for startups: retention decomposition (gross retention vs NRR honesty, churn root-cause taxonomy), customer segmentation strategy (differential investment across tiers + ICP fit scoring), CS team coverage model (pooled vs named CSM thresholds + ratio math), and CS team org evolution (CS vs Support vs AM distinctions). Use when designing retention strategy, segmenting customers for differential investment, sizing CS team, or sequencing CS hires. Strategic only \u2014 does not duplicate engineering/business-growth tactical skills.", + "path": "c-level-advisor/chief-customer-officer-advisor" + }, + { + "name": "chief-data-officer-advisor", + "description": "Chief Data Officer advisory for startups: AI training data rights and consent provenance, data product strategy (warehouse vs lakehouse vs mesh, build-vs-buy), B2B customer-data-as-asset valuation and M&A readiness, data team org evolution. Use when deciding whether to train models on customer data, choosing data architecture, valuing data for fundraising or M&A, sequencing data hires, or when user mentions CDO, chief data officer, data strategy, data mesh, lakehouse, training data, data product, data monetization, or customer data asset. NOT a tactical data engineering skill \u2014 strategic decisions only.", + "path": "c-level-advisor/chief-data-officer-advisor" + }, + { + "name": "chief-of-staff", + "description": "C-suite orchestration layer. Routes founder questions to the right advisor role(s), triggers multi-role board meetings for complex decisions, synthesizes outputs, and tracks decisions. Every C-suite interaction starts here. Loads company context automatically.", + "path": "c-level-advisor/chief-of-staff" + }, + { + "name": "chro-advisor", + "description": "People leadership for scaling companies. Hiring strategy, compensation design, org structure, culture, and retention. Use when building hiring plans, designing comp frameworks, restructuring teams, managing performance, building culture, or when user mentions CHRO, HR, people strategy, talent, headcount, compensation, org design, retention, or performance management.", + "path": "c-level-advisor/chro-advisor" + }, + { + "name": "ciso-advisor", + "description": "Security leadership for growth-stage companies. Risk quantification in dollars, compliance roadmap (SOC 2/ISO 27001/HIPAA/GDPR), security architecture strategy, incident response leadership, and board-level security reporting. Use when building security programs, justifying security budget, selecting compliance frameworks, managing incidents, assessing vendor risk, or when user mentions CISO, security strategy, compliance roadmap, zero trust, or board security reporting.", + "path": "c-level-advisor/ciso-advisor" + }, + { + "name": "cmo-advisor", + "description": "Marketing leadership for scaling companies. Brand positioning, growth model design, marketing budget allocation, and marketing org design. Use when designing brand strategy, selecting growth models (PLG vs sales-led vs community-led), allocating marketing budgets, building marketing teams, or when user mentions CMO, brand strategy, growth model, CAC, LTV, channel mix, or marketing ROI.", + "path": "c-level-advisor/cmo-advisor" + }, + { + "name": "company-os", + "description": "The meta-framework for how a company runs \u2014 the connective tissue between all C-suite roles. Covers operating system selection (EOS, Scaling Up, OKR-native, hybrid), accountability charts, scorecards, meeting pulse, issue resolution, and 90-day rocks. Use when setting up company operations, selecting a management framework, designing meeting rhythms, building accountability systems, implementing OKRs, or when user mentions EOS, Scaling Up, operating system, L10 meetings, rocks, scorecard, accountability chart, or quarterly planning.", + "path": "c-level-advisor/company-os" + }, + { + "name": "competitive-intel", + "description": "Systematic competitor tracking that feeds CMO positioning, CRO battlecards, and CPO roadmap decisions. Use when analyzing competitors, building sales battlecards, tracking market moves, positioning against alternatives, or when user mentions competitive intelligence, competitive analysis, competitor research, battlecards, win/loss, or market positioning.", + "path": "c-level-advisor/competitive-intel" + }, + { + "name": "context-engine", + "description": "Loads and manages company context for all C-suite advisor skills. Reads ~/.claude/company-context.md, detects stale context (>90 days), enriches context during conversations, and enforces privacy/anonymization rules before external API calls.", + "path": "c-level-advisor/context-engine" + }, + { + "name": "coo-advisor", + "description": "Operations leadership for scaling companies. Process design, OKR execution, operational cadence, and scaling playbooks. Use when designing operations, setting up OKRs, building processes, scaling teams, analyzing bottlenecks, planning operational cadence, or when user mentions COO, operations, process improvement, OKRs, scaling, operational efficiency, or execution.", + "path": "c-level-advisor/coo-advisor" + }, + { + "name": "cpo-advisor", + "description": "Product leadership for scaling companies. Product vision, portfolio strategy, product-market fit, and product org design. Use when setting product vision, managing a product portfolio, measuring PMF, designing product teams, prioritizing at the portfolio level, reporting to the board on product, or when user mentions CPO, product strategy, product-market fit, product organization, portfolio prioritization, or roadmap strategy.", + "path": "c-level-advisor/cpo-advisor" + }, + { + "name": "cro-advisor", + "description": "Revenue leadership for B2B SaaS companies. Revenue forecasting, sales model design, pricing strategy, net revenue retention, and sales team scaling. Use when designing the revenue engine, setting quotas, modeling NRR, evaluating pricing, building board forecasts, or when user mentions CRO, chief revenue officer, revenue strategy, sales model, ARR growth, NRR, expansion revenue, churn, pricing strategy, or sales capacity.", + "path": "c-level-advisor/cro-advisor" + }, + { + "name": "cs-onboard", + "description": "Founder onboarding interview that captures company context across 7 dimensions. Invoke with /cs:setup for initial interview or /cs:update for quarterly refresh. Generates ~/.claude/company-context.md used by all C-suite advisor skills.", + "path": "c-level-advisor/cs-onboard" + }, + { + "name": "cto-advisor", + "description": "Technical leadership guidance for engineering teams, architecture decisions, and technology strategy. Use when assessing technical debt, scaling engineering teams, evaluating technologies, making architecture decisions, establishing engineering metrics, or when user mentions CTO, tech debt, technical debt, team scaling, architecture decisions, technology evaluation, engineering metrics, DORA metrics, or technology strategy.", + "path": "c-level-advisor/cto-advisor" + }, + { + "name": "culture-architect", + "description": "Build, measure, and evolve company culture as operational behavior \u2014 not wall posters. Covers mission/vision/values workshops, values-to-behaviors translation, culture code creation, culture health assessment, and cultural rituals by stage. Use when building company values, assessing culture health, designing cultural rituals, creating culture codes, handling culture clashes, or when user mentions culture, values, culture debt, founder culture, or culture code.", + "path": "c-level-advisor/culture-architect" + }, + { + "name": "decision-logger", + "description": "Two-layer memory architecture for board meeting decisions. Manages raw transcripts (Layer 1) and approved decisions (Layer 2). Use when logging decisions after a board meeting, reviewing past decisions with /cs:decisions, or checking overdue action items with /cs:review. Invoked automatically by the board-meeting skill after Phase 5 founder approval.", + "path": "c-level-advisor/decision-logger" + }, + { + "name": "founder-coach", + "description": "Personal leadership development for founders and first-time CEOs. Covers founder archetype identification, delegation frameworks, energy management, CEO calendar audits, leadership style evolution, blind spot identification, imposter syndrome, founder mental health, and succession planning. Use when a founder feels like the bottleneck, struggles to delegate, is burning out, transitioning from IC to executive, managing a board, or when user mentions founder mode, CEO growth, leadership development, delegation, burnout, or imposter syndrome.", + "path": "c-level-advisor/founder-coach" + }, + { + "name": "general-counsel-advisor", + "description": "General Counsel advisory for startups: contract review (MSA, SaaS, NDA, DPA, employment), IP strategy, term sheet decoding, and regulatory landscape mapping. Use when reviewing any contract or term sheet, deciding when to engage outside counsel, defining IP strategy, evaluating regulatory exposure (HIPAA, GDPR, FDA, fintech), or when user mentions general counsel, GC, legal review, contract risk, term sheet, IP assignment, or regulatory exposure. NOT a substitute for licensed counsel \u2014 surfaces questions to bring to qualified attorneys.", + "path": "c-level-advisor/general-counsel-advisor" + }, + { + "name": "internal-narrative", + "description": "Build and maintain one coherent company story across all audiences \u2014 employees, investors, customers, candidates, and partners. Detects narrative contradictions and ensures the same truth is framed for each audience's needs. Use when preparing investor updates, all-hands presentations, board communications, recruiting narratives, crisis communications, or when user mentions company narrative, messaging consistency, storytelling, all-hands, investor update, or crisis communication.", + "path": "c-level-advisor/internal-narrative" + }, + { + "name": "intl-expansion", + "description": "International market expansion strategy. Market selection, entry modes, localization, regulatory compliance, and go-to-market by region. Use when expanding to new countries, evaluating international markets, planning localization, or building regional teams.", + "path": "c-level-advisor/intl-expansion" + }, + { + "name": "ma-playbook", + "description": "M&A strategy for acquiring companies or being acquired. Due diligence, valuation, integration, and deal structure. Use when evaluating acquisitions, preparing for acquisition, M&A due diligence, integration planning, or deal negotiation.", + "path": "c-level-advisor/ma-playbook" + }, + { + "name": "org-health-diagnostic", + "description": "Cross-functional organizational health check combining signals from all C-suite roles. Scores 8 dimensions on a traffic-light scale with drill-down recommendations. Use when assessing overall company health, preparing for board reviews, identifying at-risk functions, or when user mentions org health, health check, or health dashboard.", + "path": "c-level-advisor/org-health-diagnostic" + }, + { + "name": "scenario-war-room", + "description": "Cross-functional what-if modeling for cascading multi-variable scenarios. Unlike single-assumption stress testing, this models compound adversity across all business functions simultaneously. Use when facing complex risk scenarios, strategic decisions with major downside, or when the user asks 'what if X AND Y both happen?", + "path": "c-level-advisor/scenario-war-room" + }, + { + "name": "strategic-alignment", + "description": "Cascades strategy from boardroom to individual contributor. Detects and fixes misalignment between company goals and team execution. Covers strategy articulation, cascade mapping, orphan goal detection, silo identification, communication gap analysis, and realignment protocols. Use when teams are pulling in different directions, OKRs don't connect, departments optimize locally at company expense, or when user mentions alignment, strategy cascade, silo, conflicting OKRs, or strategy communication.", + "path": "c-level-advisor/strategic-alignment" + }, + { + "name": "vpe-advisor", + "description": "VP of Engineering advisory for startups: delivery throughput (DORA 4 metrics + bottleneck identification), engineering hiring funnel (sourcing \u2192 screen \u2192 onsite \u2192 offer conversion + time-to-fill + pipeline gap), engineering team structure (squad/tribe/chapter design + tech-lead manager-trigger thresholds), and production discipline (on-call, deployment cadence, postmortem culture). Use when sprint velocity is dropping, eng hiring is broken, team structure is unclear, or deciding when to add a tech-lead manager. NOT a CTO skill (which owns architecture) \u2014 VPE owns delivery operations and how the team ships.", + "path": "c-level-advisor/vpe-advisor" + }, + { + "name": "boardroom", + "description": "/cs:boardroom \u2014 6-phase multi-role deliberation across the C-suite with Phase 2 isolation, critic pre-screen, and synthesis. Outputs a board memo.", + "path": "c-level-advisor/boardroom" + }, + { + "name": "brief", + "description": "/cs:brief \u2014 Generate a one-page strategy brief from an office-hours intake. First step in the strategic sprint pipeline.", + "path": "c-level-advisor/brief" + }, + { + "name": "c-level-agents", + "description": "Founder-mode executive team. 8 cs-* C-suite agents (CFO, CMO, CRO, CPO, COO, CHRO, CISO, Chief of Staff) and 17 /cs:* slash commands for forcing-question office hours, multi-role boardroom deliberation, strategic sprint pipeline, and meta routing. Use when the founder needs a virtual executive team, when invoking /cs:* commands, or when orchestrating multi-role decisions.", + "path": "c-level-advisor/c-level-agents" + }, + { + "name": "caio-review", + "description": "/cs:caio-review \u2014 Eval-demanding Chief AI Officer interrogation of any plan that involves AI: model selection, risk classification, cost economics, or AI hiring.", + "path": "c-level-advisor/caio-review" + }, + { + "name": "cco-review", + "description": "/cs:cco-review \u2014 Retention-obsessed Chief Customer Officer interrogation of any plan that touches customer retention, segmentation, CS team sizing, or CS team hiring.", + "path": "c-level-advisor/cco-review" + }, + { + "name": "cdo-review", + "description": "/cs:cdo-review \u2014 Decision-driven Chief Data Officer interrogation of any plan that touches training data, data architecture, data productization, or data team hiring.", + "path": "c-level-advisor/cdo-review" + }, + { + "name": "cfo-review", + "description": "/cs:cfo-review \u2014 Numerate-skeptic interrogation of any plan that touches money. Unit economics, runway, dilution, capital allocation.", + "path": "c-level-advisor/cfo-review" + }, + { + "name": "ciso-review", + "description": "/cs:ciso-review \u2014 Risk-paranoid interrogation of any plan that touches data, compliance, or production access.", + "path": "c-level-advisor/ciso-review" + }, + { + "name": "cmo-review", + "description": "/cs:cmo-review \u2014 Narrative-first interrogation of positioning, ICP, message house, and channel mix.", + "path": "c-level-advisor/cmo-review" + }, + { + "name": "cpo-review", + "description": "/cs:cpo-review \u2014 JTBD-driven interrogation of product roadmap, PMF signal, and portfolio focus.", + "path": "c-level-advisor/cpo-review" + }, + { + "name": "cro-review", + "description": "/cs:cro-review \u2014 Pipeline-paranoid interrogation of revenue, win rate, NRR, and ramp time.", + "path": "c-level-advisor/cro-review" + }, + { + "name": "cross-eval", + "description": "/cs:cross-eval \u2014 Multi-model consensus on a board memo or strategy brief. Claude + Codex + Gemini cross-review with graceful degradation.", + "path": "c-level-advisor/cross-eval" + }, + { + "name": "cto-review", + "description": "/cs:cto-review \u2014 Architecture and scaling interrogation. Tech debt, scaling cliffs, team scaling, build-vs-buy.", + "path": "c-level-advisor/cto-review" + }, + { + "name": "decide", + "description": "/cs:decide \u2014 Log a decision to two-layer memory via decision-logger. Approved memo becomes durable; raw transcripts kept for reference.", + "path": "c-level-advisor/decide" + }, + { + "name": "execute", + "description": "/cs:execute \u2014 Generate a 90-day execution plan with weekly milestones, DRIs, and check-in cadence from an approved decision.", + "path": "c-level-advisor/execute" + }, + { + "name": "founder-mode", + "description": "/cs:founder-mode \u2014 Auto-routes any founder question to the right C-role advisor or to /cs:boardroom for multi-role topics. The single-command entry point.", + "path": "c-level-advisor/founder-mode" + }, + { + "name": "freeze", + "description": "/cs:freeze \u2014 Lock a strategic decision for a cooldown period to prevent impulse reversal. Mirrors gstack's safety primitives for the business layer.", + "path": "c-level-advisor/freeze" + }, + { + "name": "gc-review", + "description": "/cs:gc-review \u2014 General Counsel interrogation of contracts, IP, regulatory, term sheets, and employment-law surface.", + "path": "c-level-advisor/gc-review" + }, + { + "name": "office-hours", + "description": "/cs:office-hours \u2014 YC-style 6-question founder interrogation before any advice. Forces clarity on problem, customer, distribution, defensibility, capital, and founder fit.", + "path": "c-level-advisor/office-hours" + }, + { + "name": "onboard", + "description": "/cs:onboard \u2014 Founder interview that populates ~/.claude/company-context.md. The first command to run when starting with c-level-agents.", + "path": "c-level-advisor/onboard" + }, + { + "name": "post-mortem", + "description": "/cs:post-mortem \u2014 Honest retrospective on an executed decision, scored against original assumptions and dissent. Closes the strategic sprint loop.", + "path": "c-level-advisor/post-mortem" + }, + { + "name": "vpe-review", + "description": "/cs:vpe-review \u2014 Throughput-first VP of Engineering interrogation of any plan that touches delivery, eng hiring, team structure, or production discipline.", + "path": "c-level-advisor/vpe-review" + }, + { + "name": "chief-ai-officer-advisor", + "description": "Chief AI Officer advisory for startups: model build-vs-buy decisions (API vs fine-tune vs in-house), AI risk classification under EU AI Act + US state patchwork, AI cost economics (API-to-self-hosted breakeven), and AI team org evolution. Use when deciding whether to call an API or fine-tune, classifying AI use cases for regulatory risk, calculating when self-hosting pays off, sequencing AI hires, or when user mentions CAIO, AI strategy, model selection, foundation model, fine-tuning, EU AI Act, NIST AI RMF, AI governance, model risk, or AI economics. Strategic only \u2014 does not duplicate engineering AI/ML skills.", + "path": "c-level-advisor/chief-ai-officer-advisor" + }, + { + "name": "chief-customer-officer-advisor", + "description": "Chief Customer Officer advisory for startups: retention decomposition (gross retention vs NRR honesty, churn root-cause taxonomy), customer segmentation strategy (differential investment across tiers + ICP fit scoring), CS team coverage model (pooled vs named CSM thresholds + ratio math), and CS team org evolution (CS vs Support vs AM distinctions). Use when designing retention strategy, segmenting customers for differential investment, sizing CS team, or sequencing CS hires. Strategic only \u2014 does not duplicate engineering/business-growth tactical skills.", + "path": "c-level-advisor/chief-customer-officer-advisor" + }, + { + "name": "chief-data-officer-advisor", + "description": "Chief Data Officer advisory for startups: AI training data rights and consent provenance, data product strategy (warehouse vs lakehouse vs mesh, build-vs-buy), B2B customer-data-as-asset valuation and M&A readiness, data team org evolution. Use when deciding whether to train models on customer data, choosing data architecture, valuing data for fundraising or M&A, sequencing data hires, or when user mentions CDO, chief data officer, data strategy, data mesh, lakehouse, training data, data product, data monetization, or customer data asset. NOT a tactical data engineering skill \u2014 strategic decisions only.", + "path": "c-level-advisor/chief-data-officer-advisor" + }, + { + "name": "board-prep", + "description": "Board meeting preparation for the adversarial scenario, not the friendly one. Forces numbers-cold mastery, anticipates hard questions, builds a narrative that acknowledges weakness without losing the room. Use when preparing for a board meeting, an investor update, fundraising presentation, or any high-stakes adversarial review where every number must live in your head not just on a slide.", + "path": "c-level-advisor/board-prep" + }, + { + "name": "challenge", + "description": "Pre-mortem plan analysis. Imagine the plan failed 12 months from now and work backwards to find the weaknesses. Surfaces assumptions, dependencies, and execution risks before committing resources. Use when before significant resource commitment, before presenting to a board or investors, when feedback has been one-sidedly positive, or when there is pressure to move fast and figure it out later.", + "path": "c-level-advisor/challenge" + }, + { + "name": "executive-mentor", + "description": "Adversarial thinking partner for founders and executives. Stress-tests plans, prepares for brutal board meetings, dissects decisions with no good options, and forces honest post-mortems. Use when you need someone to find the holes before the board does, make a decision you've been avoiding, or understand what actually went wrong.", + "path": "c-level-advisor/executive-mentor" + }, + { + "name": "hard-call", + "description": "/em -hard-call \u2014 Framework for Decisions With No Good Options", + "path": "c-level-advisor/hard-call" + }, + { + "name": "postmortem", + "description": "/em -postmortem \u2014 Honest Analysis of What Went Wrong", + "path": "c-level-advisor/postmortem" + }, + { + "name": "stress-test", + "description": "/em -stress-test \u2014 Business Assumption Stress Testing", + "path": "c-level-advisor/stress-test" + }, + { + "name": "general-counsel-advisor", + "description": "General Counsel advisory for startups: contract review (MSA, SaaS, NDA, DPA, employment), IP strategy, term sheet decoding, and regulatory landscape mapping. Use when reviewing any contract or term sheet, deciding when to engage outside counsel, defining IP strategy, evaluating regulatory exposure (HIPAA, GDPR, FDA, fintech), or when user mentions general counsel, GC, legal review, contract risk, term sheet, IP assignment, or regulatory exposure. NOT a substitute for licensed counsel \u2014 surfaces questions to bring to qualified attorneys.", + "path": "c-level-advisor/general-counsel-advisor" + }, + { + "name": "vpe-advisor", + "description": "VP of Engineering advisory for startups: delivery throughput (DORA 4 metrics + bottleneck identification), engineering hiring funnel (sourcing \u2192 screen \u2192 onsite \u2192 offer conversion + time-to-fill + pipeline gap), engineering team structure (squad/tribe/chapter design + tech-lead manager-trigger thresholds), and production discipline (on-call, deployment cadence, postmortem culture). Use when sprint velocity is dropping, eng hiring is broken, team structure is unclear, or deciding when to add a tech-lead manager. NOT a CTO skill (which owns architecture) \u2014 VPE owns delivery operations and how the team ships.", + "path": "c-level-advisor/vpe-advisor" + } + ], + "project-management": [ + { + "name": "atlassian-admin", + "description": "Atlassian Administrator for managing and organizing Atlassian products (Jira, Confluence, Bitbucket, Trello), users, permissions, security, integrations, system configuration, and org-wide governance. Use when asked to add users to Jira, change Confluence permissions, configure access control, update admin settings, manage Atlassian groups, set up SSO, install marketplace apps, review security policies, or handle any org-wide Atlassian administration task.", + "path": "project-management/atlassian-admin" + }, + { + "name": "atlassian-templates", + "description": "Atlassian Template and Files Creator/Modifier expert for creating, modifying, and managing Jira and Confluence templates, blueprints, custom layouts, reusable components, and standardized content structures. Use when building org-wide templates, custom blueprints, page layouts, and automated content generation.", + "path": "project-management/atlassian-templates" + }, + { + "name": "confluence-expert", + "description": "Atlassian Confluence expert for creating and managing spaces, knowledge bases, and documentation. Configures space permissions and hierarchies, creates page templates with macros, sets up documentation taxonomies, designs page layouts, and manages content governance. Use when users need to build or restructure a Confluence space, design page hierarchies with permission structures, author or standardise documentation templates, embed Jira reports in pages, run knowledge base audits, or establish documentation standards and collaborative workflows.", + "path": "project-management/confluence-expert" + }, + { + "name": "jira-expert", + "description": "Atlassian Jira expert for creating and managing projects, planning, product discovery, JQL queries, workflows, custom fields, automation, reporting, and all Jira features. Use for Jira project setup, configuration, advanced search, dashboard creation, workflow design, and technical Jira operations.", + "path": "project-management/jira-expert" + }, + { + "name": "meeting-analyzer", + "description": "Analyzes meeting transcripts and recordings to surface behavioral patterns, communication anti-patterns, and actionable coaching feedback. Use this skill whenever the user uploads or points to meeting transcripts (.txt, .md, .vtt, .srt, .docx), asks about their communication habits, wants feedback on how they run meetings, requests speaking ratio analysis, mentions filler words or conflict avoidance, or wants to compare their communication across time periods. Also trigger when users mention tools like Granola, Otter, Fireflies, or Zoom transcripts. Even if the user just says \"look at my meetings\" or \"how do I come across in meetings\" \u2014 use this skill.", + "path": "project-management/meeting-analyzer" + }, + { + "name": "pm-skills", + "description": "6 project management agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Senior PM, scrum master, Jira expert (JQL), Confluence expert, Atlassian admin, template creator. MCP integration for live Jira/Confluence automation.", + "path": "project-management/pm-skills" + }, + { + "name": "scrum-master", + "description": "Advanced Scrum Master skill for data-driven agile team analysis and coaching. Use when the user asks about sprint planning, velocity tracking, retrospectives, standup facilitation, backlog grooming, story points, burndown charts, blocker resolution, or agile team health. Runs Python scripts to analyse sprint JSON exports from Jira or similar tools: velocity_analyzer.py for Monte Carlo sprint forecasting, sprint_health_scorer.py for multi-dimension health scoring, and retrospective_analyzer.py for action-item and theme tracking. Produces confidence-interval forecasts, health grade reports, and improvement-velocity trends for high-performing Scrum teams.", + "path": "project-management/scrum-master" + }, + { + "name": "senior-pm", + "description": "Senior Project Manager for enterprise software, SaaS, and digital transformation projects. Specializes in portfolio management, quantitative risk analysis, resource optimization, stakeholder alignment, and executive reporting. Uses advanced methodologies including EMV analysis, Monte Carlo simulation, WSJF prioritization, and multi-dimensional health scoring. Use when a user needs help with project plans, project status reports, risk assessments, resource allocation, project roadmaps, milestone tracking, team capacity planning, portfolio health reviews, program management, or executive-level project reporting \u2014 especially for enterprise-scale initiatives with multiple workstreams, complex dependencies, or multi-million dollar budgets.", + "path": "project-management/senior-pm" + }, + { + "name": "team-communications", + "description": "Write internal company communications \u2014 3P updates (Progress/Plans/Problems), company-wide newsletters, FAQ roundups, incident reports, leadership updates, status reports, project updates, and general internal comms. Use this skill any time the user asks to draft, edit, or format something meant for internal audiences. Trigger on keywords like \"3P\", \"weekly update\", \"newsletter\", \"FAQ\", \"internal comms\", \"status report\", \"company update\", \"team update\", \"incident report\", or any request to summarize work for leadership, teammates, or the broader company. Even casual requests like \"write my update\" or \"summarize what my team did this week\" should trigger this skill.", + "path": "project-management/team-communications" + } + ], + "ra-qm-team": [ + { + "name": "capa-officer", + "description": "CAPA system management for medical device QMS. Covers root cause analysis, corrective action planning, effectiveness verification, and CAPA metrics. Use for CAPA investigations, 5-Why analysis, fishbone diagrams, root cause determination, corrective action tracking, effectiveness verification, or CAPA program optimization.", + "path": "ra-qm-team/capa-officer" + }, + { + "name": "eu-ai-act-specialist", + "description": "EU AI Act (Regulation (EU) 2024/1689) operational compliance for compliance teams. Three Article-level decisions: (1) What's the risk tier of this AI system \u2014 prohibited (Art. 5), high-risk (Art. 6 + Annex III), limited-risk (Art. 50), or minimal-risk? (2) For high-risk systems, what's the Article 43 conformity assessment route (Module A internal control vs Module H full QMS + notified body) and what goes in the Annex IV technical documentation? (3) Per organizational role (provider / deployer / importer / distributor / authorized representative), what are the active obligations and deadlines? Use during AI system intake review, when planning conformity assessment, or when scoping deployer obligations. Cites Articles + Annexes for every output. NOT executive AI strategy (see chief-ai-officer-advisor). NOT a legal substitute.", + "path": "ra-qm-team/eu-ai-act-specialist" + }, + { + "name": "fda-consultant-specialist", + "description": "FDA regulatory consultant for medical device companies. Provides 510(k)/PMA/De Novo pathway guidance, QSR (21 CFR 820) compliance, HIPAA assessments, and device cybersecurity. Use when user mentions FDA submission, 510(k), PMA, De Novo, QSR, premarket, predicate device, substantial equivalence, HIPAA medical device, or FDA cybersecurity.", + "path": "ra-qm-team/fda-consultant-specialist" + }, + { + "name": "gdpr-dsgvo-expert", + "description": "GDPR and German DSGVO compliance automation. Scans codebases for privacy risks, generates DPIA documentation, tracks data subject rights requests. Use for GDPR compliance assessments, privacy audits, data protection planning, DPIA generation, and data subject rights management.", + "path": "ra-qm-team/gdpr-dsgvo-expert" + }, + { + "name": "information-security-manager-iso27001", + "description": "ISO 27001 ISMS implementation and cybersecurity governance for HealthTech and MedTech companies. Use for ISMS design, security risk assessment, control implementation, ISO 27001 certification, security audits, incident response, and compliance verification. Covers ISO 27001, ISO 27002, healthcare security, and medical device cybersecurity.", + "path": "ra-qm-team/information-security-manager-iso27001" + }, + { + "name": "isms-audit-expert", + "description": "Information Security Management System (ISMS) audit expert for ISO 27001 compliance verification, security control assessment, and certification support. Use when the user mentions ISO 27001, ISMS audit, Annex A controls, Statement of Applicability (SOA), gap analysis, nonconformity management, internal audit, surveillance audit, or security certification preparation. Helps review control implementation evidence, document audit findings, classify nonconformities, generate risk-based audit plans, map controls to Annex A requirements, prepare Stage 1 and Stage 2 audit documentation, and support corrective action workflows.", + "path": "ra-qm-team/isms-audit-expert" + }, + { + "name": "iso42001-specialist", + "description": "ISO/IEC 42001:2023 AI Management System (AIMS) specialist for compliance teams running internal audits. Three decisions: (1) Where are the gaps against Clauses 4-10 and what do we close first? (2) What goes in the AI risk register and which Annex A controls treat each risk? (3) What's the 12-month internal audit plan that satisfies Clause 9.2? Use when preparing for certification, scoping internal audit cycles, or onboarding AI systems into an existing ISMS (27001) / QMS (13485) program. NOT an executive AI strategy skill (see chief-ai-officer-advisor). NOT EU AI Act compliance (see compliance-team-eu-ai-act).", + "path": "ra-qm-team/iso42001-specialist" + }, + { + "name": "mdr-745-specialist", + "description": "EU MDR 2017/745 compliance specialist for medical device classification, technical documentation, clinical evidence, and post-market surveillance. Covers Annex VIII classification rules, Annex II/III technical files, Annex XIV clinical evaluation, and EUDAMED integration.", + "path": "ra-qm-team/mdr-745-specialist" + }, + { + "name": "qms-audit-expert", + "description": "ISO 13485 internal audit expertise for medical device QMS. Covers audit planning, execution, nonconformity classification, and CAPA verification. Use for internal audit planning, audit execution, finding classification, external audit preparation, or audit program management.", + "path": "ra-qm-team/qms-audit-expert" + }, + { + "name": "quality-documentation-manager", + "description": "Document control system management for medical device QMS. Covers document numbering, version control, change management, and 21 CFR Part 11 compliance. Use for document control procedures, change control workflow, document numbering, version management, electronic signature compliance, or regulatory documentation review.", + "path": "ra-qm-team/quality-documentation-manager" + }, + { + "name": "quality-manager-qmr", + "description": "Senior Quality Manager Responsible Person (QMR) for HealthTech and MedTech companies. Provides quality system governance, management review leadership, regulatory compliance oversight, and quality performance monitoring per ISO 13485 Clause 5.5.2.", + "path": "ra-qm-team/quality-manager-qmr" + }, + { + "name": "quality-manager-qms-iso13485", + "description": "ISO 13485 Quality Management System implementation and maintenance for medical device organizations. Provides QMS design, documentation control, internal auditing, CAPA management, and certification support. Use when working with medical device quality systems, preparing for ISO 13485 audits, managing regulatory compliance documentation, setting up corrective actions, or building audit preparation programs. Useful for quality management, audit preparation, regulatory compliance, medical device documentation, and corrective action workflows.", + "path": "ra-qm-team/quality-manager-qms-iso13485" + }, + { + "name": "ra-qm-skills", + "description": "12 regulatory & QM agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. ISO 13485 QMS, MDR 2017/745, FDA 510(k)/PMA, ISO 27001 ISMS, GDPR/DSGVO, risk management (ISO 14971), CAPA, document control, auditing. Python tools (stdlib-only).", + "path": "ra-qm-team/ra-qm-skills" + }, + { + "name": "regulatory-affairs-head", + "description": "Senior Regulatory Affairs Manager for HealthTech and MedTech companies. Prepares FDA 510(k), De Novo, and PMA submission packages; analyzes regulatory pathways for new medical devices; drafts responses to FDA deficiency letters and Notified Body queries; develops CE marking technical documentation under EU MDR 2017/745; coordinates multi-market approval strategies across FDA, EU, Health Canada, PMDA, and NMPA; and maintains regulatory intelligence on evolving standards. Use when users need to plan or execute FDA submissions, navigate 510(k) or PMA approval processes, achieve CE marking, prepare pre-submission meeting materials, write regulatory strategy documents, respond to agency queries, or manage compliance documentation for medical device market access.", + "path": "ra-qm-team/regulatory-affairs-head" + }, + { + "name": "risk-management-specialist", + "description": "Medical device risk management specialist implementing ISO 14971 throughout product lifecycle. Provides risk analysis, risk evaluation, risk control, and post-production information analysis. Use when user mentions risk management, ISO 14971, risk analysis, FMEA, fault tree analysis, hazard identification, risk control, risk matrix, benefit-risk analysis, residual risk, risk acceptability, or post-market risk.", + "path": "ra-qm-team/risk-management-specialist" + }, + { + "name": "soc2-compliance", + "description": "Use when the user asks to prepare for SOC 2 audits, map Trust Service Criteria, build control matrices, collect audit evidence, perform gap analysis, or assess SOC 2 Type I vs Type II readiness.", + "path": "ra-qm-team/soc2-compliance" + }, + { + "name": "eu-ai-act-specialist", + "description": "EU AI Act (Regulation (EU) 2024/1689) operational compliance for compliance teams. Three Article-level decisions: (1) What's the risk tier of this AI system \u2014 prohibited (Art. 5), high-risk (Art. 6 + Annex III), limited-risk (Art. 50), or minimal-risk? (2) For high-risk systems, what's the Article 43 conformity assessment route (Module A internal control vs Module H full QMS + notified body) and what goes in the Annex IV technical documentation? (3) Per organizational role (provider / deployer / importer / distributor / authorized representative), what are the active obligations and deadlines? Use during AI system intake review, when planning conformity assessment, or when scoping deployer obligations. Cites Articles + Annexes for every output. NOT executive AI strategy (see chief-ai-officer-advisor). NOT a legal substitute.", + "path": "ra-qm-team/eu-ai-act-specialist" + }, + { + "name": "iso42001-specialist", + "description": "ISO/IEC 42001:2023 AI Management System (AIMS) specialist for compliance teams running internal audits. Three decisions: (1) Where are the gaps against Clauses 4-10 and what do we close first? (2) What goes in the AI risk register and which Annex A controls treat each risk? (3) What's the 12-month internal audit plan that satisfies Clause 9.2? Use when preparing for certification, scoping internal audit cycles, or onboarding AI systems into an existing ISMS (27001) / QMS (13485) program. NOT an executive AI strategy skill (see chief-ai-officer-advisor). NOT EU AI Act compliance (see compliance-team-eu-ai-act).", + "path": "ra-qm-team/iso42001-specialist" + } + ], + "business-growth": [ + { + "name": "business-growth-skills", + "description": "4 business growth agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Customer success (health scoring, churn), sales engineer (RFP), revenue operations (pipeline, GTM), contract & proposal writer. Python tools (stdlib-only).", + "path": "business-growth/business-growth-skills" + }, + { + "name": "contract-and-proposal-writer", + "description": "Generate professional, jurisdiction-aware business documents: freelance contracts, project proposals, SOWs, NDAs, and MSAs. Structured Markdown output with docx conversion instructions. Covers US (Delaware), EU (GDPR), UK, and DACH (German law) jurisdictions. Not a substitute for legal counsel \u2014 use as strong starting points. Use when drafting a freelance contract, preparing a client proposal, writing an SOW for a new engagement, or producing an NDA before sharing sensitive material.", + "path": "business-growth/contract-and-proposal-writer" + }, + { + "name": "customer-success-manager", + "description": "Monitors customer health, predicts churn risk, and identifies expansion opportunities using weighted scoring models for SaaS customer success. Use when analyzing customer accounts, reviewing retention metrics, scoring at-risk customers, or when the user mentions churn, customer health scores, upsell opportunities, expansion revenue, retention analysis, or customer analytics. Runs three Python CLI tools to produce deterministic health scores, churn risk tiers, and prioritized expansion recommendations across Enterprise, Mid-Market, and SMB segments.", + "path": "business-growth/customer-success-manager" + }, + { + "name": "revenue-operations", + "description": "Analyzes sales pipeline health, revenue forecasting accuracy, and go-to-market efficiency metrics for SaaS revenue optimization. Use when analyzing sales pipeline coverage, forecasting revenue, evaluating go-to-market performance, reviewing sales metrics, assessing pipeline analysis, tracking forecast accuracy with MAPE, calculating GTM efficiency, or measuring sales efficiency and unit economics for SaaS teams.", + "path": "business-growth/revenue-operations" + }, + { + "name": "sales-engineer", + "description": "Analyzes RFP/RFI responses for coverage gaps, builds competitive feature comparison matrices, and plans proof-of-concept (POC) engagements for pre-sales engineering. Use when responding to RFPs, bids, or proposal requests; comparing product features against competitors; planning or scoring a customer POC or sales demo; preparing a technical proposal; or performing win/loss competitor analysis. Handles tasks described as 'RFP response', 'bid response', 'proposal response', 'competitor comparison', 'feature matrix', 'POC planning', 'sales demo prep', or 'pre-sales engineering'.", + "path": "business-growth/sales-engineer" + } + ], + "finance": [ + { + "name": "finance-skills", + "description": "Financial analyst agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Ratio analysis, DCF valuation, budget variance, rolling forecasts. 4 Python tools (stdlib-only).", + "path": "finance/finance-skills" + }, + { + "name": "financial-analyst", + "description": "Performs financial ratio analysis, DCF valuation, budget variance analysis, and rolling forecast construction for strategic decision-making. Use when analyzing financial statements, building valuation models, assessing budget variances, or constructing financial projections and forecasts. Also applicable when users mention financial modeling, cash flow analysis, company valuation, financial projections, or spreadsheet analysis.", + "path": "finance/financial-analyst" + }, + { + "name": "saas-metrics-coach", + "description": "SaaS financial health advisor. Use when a user shares revenue or customer numbers, or mentions ARR, MRR, churn, LTV, CAC, NRR, or asks how their SaaS business is doing.", + "path": "finance/saas-metrics-coach" + }, + { + "name": "business-investment-advisor", + "description": "Business investment analysis and capital allocation advisor. Use when evaluating whether to invest in equipment, real estate, a new business, hiring, technology, or any capital expenditure. Also use for ROI calculations, IRR, NPV, payback period, build vs buy decisions, lease vs buy analysis, vendor evaluation, or deciding where to allocate limited budget for maximum return.", + "path": "finance/business-investment-advisor" + } + ], "productivity": [ { "name": "andreessen", @@ -33,6 +1516,161 @@ "description": "Mid-conversation reflection skill that pauses execution and zooms out from detail-mode to honestly reassess direction, assumptions, and bias. Use when the user says 'reflect', 'take a step back', 'step back', 'zoom out', 'are we missing something', 'bigger picture', 'sanity check this', 'are we on track', 'are we overthinking this', 'forest for the trees', or any variation signaling intent to break out of detail-mode and reassess. Also trigger when the conversation has gone deep on implementation details without strategic check-in, or when the user shows signs of being stuck \u2014 that's often a signal the framing needs a reset, not more detail work. Intentionally low-intake: runs the 5-dimension analysis immediately when prior context is rich enough; asks one forcing clarifier only when invocation context is too thin to reassess from.", "path": "productivity/reflect" } + ], + "marketing": [ + { + "name": "landing", + "description": "Generates a premium single-page HTML landing page with 3D CSS animations, GSAP scroll effects, and mouse-parallax depth. Forcing intake (product + elevator pitch, audience register, brand overrides, tone) locks down positioning before any copy or markup is written, so the page reflects the actual product rather than generic boilerplate. Use whenever the user says 'landing for X', 'create a landing page', 'build a landing page', 'make a landing page for X', 'I need a web page for Y', or provides product/service details and wants a polished website. Also triggers on 'promotional page', 'product page', 'one-pager', 'web presence', 'sales page'. Outputs a single self-contained HTML file (Claude Code) or HTML artifact (Claude.ai). Supports configurable brand colors via CSS custom property overrides.", + "path": "marketing/landing" + } + ], + "research": [ + { + "name": "dossier", + "description": "Decision-grade entity research skill \u2014 produces a hypothesis-tested dossier on a specific company, person, nonprofit, or government org, not a generic profile. Forcing intake makes the user state their hypothesis upfront (what they already believe and want to verify or disprove) so the dossier tests it rather than confirms it. Output is an editable Word document (.docx) with verdict on the hypothesis, identity facts, 12-month activity timeline, network signals, reputation signals, red flags, 3-5 conversation hooks tied to specific findings, and source-provenance audit log. Uses WebSearch + WebFetch + free APIs (SEC EDGAR, GitHub, ProPublica Nonprofit Explorer) as workhorses; optional BYOK MCPs (LinkedIn, Crunchbase, Apollo, Pitchbook, SimilarWeb) enhance coverage. Triggers: 'research [company]', 'dossier on [person/company]', 'background check on [entity]', 'prep me for a meeting with [person/company]', 'due diligence on [company]', 'what should I know about [entity]', 'research [person] before I [meet/hire/invest]', 'competitor research on [company]', 'investor diligence [company]', 'interview prep for [company]'. Honors sensitivity exclusions for journalism + personal-vetting contexts.", + "path": "research/dossier" + }, + { + "name": "grants", + "description": "NIH grant research skill for clinical researchers. Grill-me intake (research idea + career stage + preliminary data + environment + submission posture + known institute targets) locks down the funding strategy before any search runs. Runs a 5-facet Consensus positioning analysis (with draft Significance/Innovation language), maps the research to the right NIH institutes and study sections via RePORTER, finds NOSIs and funded overlap, and produces an editable Word document (.docx) with budget/scope-aware mechanism recommendations, submission timelines, and a mandatory program officer recommendation. Triggers: 'grants for [topic]', 'find grants for my research idea', 'what grants match my research', 'help me find NIH funding', 'grant opportunities for my research', or any grant-related request. NIH-only scope \u2014 non-NIH funders (PCORI, DOD CDMRP, VA, foundations) are out of scope and flagged at intake.", + "path": "research/grants" + }, + { + "name": "litreview", + "description": "Academic literature orientation skill that searches papers via Consensus, builds a strategic search plan using PICO (default) or SPIDER / Decomposition / hybrid as fallbacks, and synthesizes findings into a professionally formatted Word document (.docx) research guide. Grill-me intake (research question specificity + framework hint + tentative depth) before the recon search; a second forcing checkpoint after Phase 2 confirms framework + sub-areas + depth before searches consume budget. Configurable depth (5/10/20 queries) controls coverage vs. speed. Output is a 'launching pad' \u2014 not a finished review, but an orientation guide that lets a researcher dive in confidently. Triggers: 'litreview on [topic]', 'literature review on [topic]', 'I'm starting a literature review on X', 'I'm writing a paper on X', 'help me research X', 'I'm doing research on X', 'can you help me research X'. Do NOT trigger for single one-off paper searches where the user just wants a quick list \u2014 that's a plain Consensus search.", + "path": "research/litreview" + }, + { + "name": "notebooklm", + "description": "Browser automation skill for controlling Google's NotebookLM. Handles reading and querying notebooks, adding sources (URLs, text, files, YouTube links, synthesized content), generating Studio outputs (Audio Overview, infographics, slide decks, study guides, briefing docs, mind maps, timelines, FAQs), and creating new notebooks. Triggers on any phrase involving NotebookLM \u2014 'open NotebookLM', 'check my [name] notebook', 'pull info from NotebookLM', 'ask my notebook about X', 'add [source] to NotebookLM', 'create an infographic in NotebookLM', 'use NotebookLM Studio', 'generate a slide deck from my notebook', or any variation where the goal involves NotebookLM. Requires browser automation environment \u2014 fails gracefully when unavailable.", + "path": "research/notebooklm" + }, + { + "name": "patent", + "description": "Patent prior-art and landscape intelligence skill \u2014 not generic patent help. Commits to one of five sub-use-cases via forcing intake (novelty search / freedom-to-operate / competitive landscape / acquisition diligence / litigation prior-art) before any search runs. Searches Google Patents, Espacenet, USPTO, and optionally Lens.org for citation-graph signals. Output is an editable Word document (.docx) with verdict, ranked closest art (claim-text extracted), CPC-class-aware landscape, family-resolved hits, geographic coverage, FTO flags where applicable, strategy recommendations, and full audit log. Triggers: 'prior art search for [invention]', 'patent search on [topic]', 'freedom to operate analysis', 'FTO for [product]', 'patent landscape for [field]', 'is [invention] novel', 'patents on [topic]', 'competitive patent analysis', 'prior art for litigation', 'patent diligence on [company]'. Produces search signal, not legal advice \u2014 always recommends consulting a patent attorney before filing or licensing decisions. Trademark, copyright, and trade-secret questions are out of scope.", + "path": "research/patent" + }, + { + "name": "pulse", + "description": "Multi-source recency research skill that takes the pulse of any topic across Reddit, Hacker News, the open web, and optionally X/Twitter within a configurable recent window (default 30 days). Forcing intake clarifies topic specificity, angle (trend/sentiment/problems/opportunities/comparison), time window, and platform scope before searching. Returns a synthesized briefing with citations, engagement metrics, and cross-platform pattern analysis. Triggers: 'pulse on [topic]', 'what's happening with [topic]', 'what are people saying about [topic]', 'current conversation about [topic]', 'take the pulse of [topic]', 'trending: [topic]', 'find me info on [topic]', or any variation requesting multi-source recency intelligence on a topic. Also use for competitor research, trend discovery, tool comparisons, and audience sentiment analysis.", + "path": "research/pulse" + }, + { + "name": "research", + "description": "Default entry point for any research request \u2014 a hybrid router that classifies the question deterministically and either delegates to a specialist research skill (pulse for trends/sentiment, grants for NIH funding, litreview for academic literature, syllabus for course reading, patent for prior-art + IP landscape, dossier for entity research) or runs its own plan-decompose-multi-source-search-synthesize-cite fallback workflow when no specialist matches. Always surfaces the routing decision so users can override. Triggers \u2014 \"research [topic]\", \"look into [topic]\", \"what do we know about [topic]\", \"investigate [topic]\", \"find me information on [topic]\", \"do some research on [topic]\", \"I need to understand [topic]\", or any research request that doesn't obviously match a more-specific specialist skill. Output is a markdown briefing (default) or .docx document (on request) with full citations and an audit log.", + "path": "research/research" + }, + { + "name": "syllabus", + "description": "Generates a curated supplementary reading list from any course syllabus using Consensus academic search. Grill-me intake (syllabus input format + course audience + year range) plus a grouping forcing-options checkpoint before any search runs \u2014 so the reading list matches the course's level and recency need. Parses the syllabus to extract topics and learning outcomes, searches Consensus for recent peer-reviewed papers per topic, and produces a professionally formatted .docx with clickable Consensus links, plain-language summaries calibrated to audience level, and Bloom-higher-order discussion questions tied to course learning goals. Triggers whenever a user uploads a syllabus, course outline, or curriculum document and wants supplementary readings. Also triggers on: 'syllabus reading list', 'find papers for my course', 'create a reading list from this syllabus', 'recent research for my class', 'supplementary readings', 'find journal articles for these topics', 'what recent papers cover this material', 'any new research on these course topics', 'update my syllabus with recent papers'. Even casual mentions when a syllabus is attached should trigger this skill.", + "path": "research/syllabus" + } + ], + "business-operations": [ + { + "name": "business-operations-skills", + "description": "Use when running, diagnosing, or designing internal business operations \u2014 process documentation, vendor SLAs, capacity planning, internal comms, SOP/runbook authoring, procurement spend. Triggers on \"BizOps review\", \"where's the bottleneck\", \"vendor health\", \"internal SOP\", \"all-hands deck\", \"spend categorization\", \"capacity for Q3\", \"process mapping\". Forks context to route to one of six BizOps sub-skills (process-mapper, vendor-management, capacity-planner, internal-comms, knowledge-ops, procurement-optimizer) and returns a digest. Distinct from business-growth (external sales motion) and c-level-advisor (strategic, not operational).", + "path": "business-operations/business-operations-skills" + }, + { + "name": "capacity-planner", + "description": "Use when an ops leader (Director of CX, Head of Support, VP Ops, Head of BizOps, Head of IT ops, Head of Finance ops) is sizing ops capacity, building a headcount plan, modeling utilization risk, planning Q3 capacity or annual support capacity, or designing CS coverage \u2014 and needs Erlang-C queueing math, P90 demand sizing, shrinkage-adjusted FTE, manager-trigger thresholds, and a quarterly hiring sequence with ramp + attrition. Apply when sustained team utilization is above 80% or when the team is growing >50% in 12 months. Run before committing the headcount budget. This is NOT engineering capacity (see vpe-advisor for DORA + cycle time) and NOT strategic 3-year workforce planning (see chro-advisor).", + "path": "business-operations/capacity-planner" + }, + { + "name": "internal-comms", + "description": "Use when a Head of People Ops, BizOps lead, or Internal Communications owner needs to draft and sequence an internal-only change-management communication \u2014 a re-org announcement, a tool rollout, a policy change, a benefit change, a leadership transition, a layoff, an acquisition close, or an internal product launch \u2014 and the audience is employees (not customers). Triggers on \"all-hands announcement\", \"town-hall script\", \"change comms\", \"internal newsletter\", \"rollout comms\", \"policy change announcement\", \"re-org announcement\", \"internal FAQ\", \"manager talking points\", \"Prosci ADKAR\", \"Kotter 8-step\", \"layoff comms\", \"RIF comms\", \"internal memo\". Pairs Prosci ADKAR (Awareness / Desire / Knowledge / Ability / Reinforcement) and Kotter's 8-step change model with deterministic stdlib-only Python tools to produce a sequenced touchpoint calendar, a Kotter-compliant primary announcement, an audience-segmented FAQ, and manager cascade talking points. Industry-tuned via --profile {tech-startup, scaleup, enterprise, public-company, non-profit}. Distinct from marketing-skill/* (external/customer-facing), c-level-advisor/internal-narrative (strategic framing, not tactical drafts), and c-level-advisor/change-management (executive change strategy, not the comms package itself).", + "path": "business-operations/internal-comms" + }, + { + "name": "knowledge-ops", + "description": "Use when a Head of Ops, Knowledge Manager, or TPM-Internal needs to author, validate, or clean up company SOPs and internal runbooks (procurement intake, vendor offboarding, incident-comms cascade, employee onboarding, expense reimbursement, system-access provisioning, customer-escalation playbook) \u2014 including 5W2H completeness checks (Who-What-When-Where-Why-How-HowMuch), cross-link and orphan-page validation across a sprawling Notion/Confluence/Obsidian wiki, KB ingestion + hygiene reporting, ops onboarding doc generation, and runbook step verification (named owner, expected duration, observable success signal, rollback path, escalation contact). Pairs Kaoru Ishikawa's 5W2H method, Atul Gawande's *The Checklist Manifesto*, ISO 9001, ITIL v4 Service Operation, FDA 21 CFR Part 211, and Google SRE Workbook runbook discipline with deterministic stdlib-only Python tools that score completeness, detect anti-patterns, and emit prioritized cleanup lists. Distinct from `engineering/llm-wiki` (Karpathy-style personal PKM second brain), `engineering-team/runbook-generator` (system-ops production debugging runbook), `project-management/*` (Jira/Confluence delivery + ticket tracking), and sibling `business-operations/process-mapper` (BPMN process *design*, while knowledge-ops is process *documentation*).", + "path": "business-operations/knowledge-ops" + }, + { + "name": "process-mapper", + "description": "Use when a BizOps lead, COO, or process-improvement owner needs to document an end-to-end business process (procurement, employee onboarding, incident handoff, customer-onboarding, claims adjudication) in BPMN-style notation, measure cycle times by stage, surface where work spends most of its time waiting vs. being worked, and quantify the gap between processing time and total elapsed time. Pairs Lean / Six Sigma / Theory-of-Constraints canon with deterministic stdlib-only Python tools to produce a process map, a ranked bottleneck list (with severity + root-cause hypothesis), and a cycle-time analysis (P50, P90, value-add ratio, Little's-Law throughput). Distinct from sales-pipeline, system-reliability (SLO), and strategic-OKR work \u2014 this is tactical process documentation for internal operations.", + "path": "business-operations/process-mapper" + }, + { + "name": "procurement-optimizer", + "description": "Use when running an annual SaaS audit, doing category-level spend review, or rationalizing the supplier base \u2014 when the user needs to do a spend audit, spend categorization (UNSPSC-aligned), purchasing-cycle analysis, or risk-balanced supplier consolidation. Triggers on \"spend audit\", \"SaaS audit\", \"spend categorization\", \"supplier rationalization\", \"supplier consolidation\", \"purchasing cycle\", \"procurement review\", \"category strategy\", \"duplicate SaaS\", \"renewal cluster\". Ships 3 stdlib-only Python tools (UNSPSC-aligned spend categorizer with Pareto breakdown and industry profiles, purchasing-cycle analyzer that surfaces bottleneck categories per Goldratt's Theory of Constraints, supplier-consolidation planner that refuses single-source recommendations for tier-1 categories without a documented break-glass plan), 3 reference docs each citing 7+ authoritative sources (A.T. Kearney / Hackett / Spend Matters / UNSPSC / Productiv / Vendr / Tropic / IACCM / ISM / BCG), and a 20-minute spend-intake template. Distinct from sibling vendor-management (performance scoring of vendors you keep paying), finance/financial-analysis (close + report, not category strategy), and c-level-advisor/general-counsel-advisor (contract law, not category rationalization).", + "path": "business-operations/procurement-optimizer" + }, + { + "name": "vendor-management", + "description": "Use when reviewing, scoring, or auditing third-party SaaS / vendor relationships \u2014 running a vendor scorecard, tracking SLA compliance, classifying third-party risk, preparing a tier-1 vendor review, or auditing the SaaS portfolio. Triggers on \"vendor SLA\", \"vendor scorecard\", \"third-party risk\", \"TPRM\", \"vendor review\", \"SaaS audit\", \"supplier performance\", \"vendor health check\", \"renewal review\". Forks context so large vendor catalogs (50-500 line items) and SLA logs don't pollute the parent thread. Ships 3 stdlib-only Python tools (vendor scorer with industry tuning, SLA compliance tracker with credit-claim flags, vendor risk classifier across 4 risk vectors), 3 reference docs each citing 7+ authoritative sources (Gartner / Shared Assessments / NIST / ISO 27036 / breach post-mortems), and a 5-vendor catalog template. Distinct from c-level-advisor/general-counsel-advisor (contract law, not operational management), business-growth/contract-and-proposal-writer (outbound proposals, not inbound vendor scoring), and sibling procurement-optimizer (spend categorization, not vendor performance).", + "path": "business-operations/vendor-management" + } + ], + "commercial": [ + { + "name": "channel-economics", + "description": "Use when reviewing or rebalancing direct vs. partner-led channel economics \u2014 computing fully-loaded cost-to-serve per channel, channel ROI with cash / LTV / marginal lenses, and optimal channel mix subject to constraints. For Head of Commercial, RevOps, and VP Sales doing quarterly channel review when pipeline is mixed (e.g., 60% direct + 40% partner-led) and nobody actually knows which channel makes money after CAC, support load, partner discount, deal-velocity differences, retention differential, and overhead allocation are all loaded in. Outputs cost to serve, channel ROI verdicts (DOUBLE-DOWN / MAINTAIN / DEFUND / EXIT), a sensitivity-tested channel-mix recommendation, and the diminishing-returns inflection. Not channel structure (that's partnerships-architect \u2014 tiers, joint GTM, revshare). Not RevOps process (that's business-growth/revenue-operations \u2014 lead routing, SDR motion). Not strategic CRO judgment (that's c-level-advisor/cro-advisor \u2014 comp plans, when-to-hire-a-VP-Sales). Not historical close-and-report (that's finance/financial-analysis). This skill answers: direct vs partner profitability, channel profitability, channel mix, channel economics.", + "path": "commercial/channel-economics" + }, + { + "name": "commercial-forecaster", + "description": "Use when building a quarterly bookings forecast, ARR projection, pipeline forecast, NRR projection, or commit/best-case/pipe-only board number \u2014 especially when the CRO needs to walk the board through funnel math + cohort ARR + per-stage conversion assumptions without the theatre of a single undefended number. Decomposes pipeline into commit, best-case, and pipe-only tiers; projects cohort-level NRR/GRR to surface leaky cohorts before they show up in the consolidated number; scores per-stage funnel confidence so soft-floor stages get treated differently from high-confidence ones. Every output explicitly names the conversion rate used, the data window, and the weighting choice. For Head of Commercial, RevOps, VP Sales, and CRO at quarterly forecast or board prep. NOT financial close (see finance/financial-analysis). NOT strategic CRO hiring/territory (see c-level-advisor/cro-advisor). NOT pricing (see sibling pricing-strategist).", + "path": "commercial/commercial-forecaster" + }, + { + "name": "commercial-policy", + "description": "Use when designing or revising a company's commercial policy \u2014 the rules of engagement governing discounts off list price, approver thresholds, exception flows, and the deal framework that Deal Desk and AEs operate under. Covers discount matrix design (ARR band x term length x payment terms x strategic value), commercial policy design, exception policy, discount governance, approval thresholds, deal framework structure, and policy linting (contradictions, gaps, cliff edges, gaming surfaces). For Head of Commercial, Head of Deal Desk, VP Sales, or RevOps at the policy-design moment \u2014 NOT per-deal application (that is deal-desk) and NOT pricing model selection (that is pricing-strategist).", + "path": "commercial/commercial-policy" + }, + { + "name": "commercial-skills", + "description": "Use when reviewing, approving, or designing commercial motion \u2014 pricing models, deal review, discount approval, partnership economics, channel mix, commercial policy, RFP/RFI response, bookings forecast. Triggers on \"review this deal\", \"should we discount\", \"pricing model\", \"partner economics\", \"RFP response\", \"bookings forecast\", \"channel mix\". Forks context to route to one of seven Commercial sub-skills (pricing-strategist, deal-desk, partnerships-architect, channel-economics, commercial-policy, rfp-responder, commercial-forecaster) and returns a digest. Distinct from business-growth (sales execution) and c-level-advisor/cro-advisor (strategic CRO judgment).", + "path": "commercial/commercial-skills" + }, + { + "name": "deal-desk", + "description": "Use when reviewing a specific inbound deal before close \u2014 when sales has asked for a discount that exceeds AE authority, when the customer has redlined the MSA, when per-deal economics (margin after discount, multi-year payment shape, indemnity exposure) need to be quantified, or when discount approval needs to be routed to a named human approver (Sales Director, VP Sales, CFO, CRO, General Counsel). Covers deal review, discount approval routing, per-deal margin scoring, deal exception handling, MSA redline triage, contract landmine detection (uncapped indemnity, MFN, perpetual license-back, missing DPA), and named-approver chain assembly. NEVER auto-approves \u2014 every output is a numeric scorecard plus a routing recommendation to a named human.", + "path": "commercial/deal-desk" + }, + { + "name": "partnerships-architect", + "description": "Use when a startup is approached by a prospective partner and someone has to decide should we sign this partner, at what partner tier (referral / reseller / OEM / SI-consulting / strategic alliance), with what joint GTM commitment, and at what revshare. Classifies partner tier from independent-demand evidence vs. preferential-terms hunting, designs a 90-day joint GTM plan, models revshare against direct-sale margin, and surfaces kill criteria for unwinding under-performing partnerships. For Head of Partnerships, Head of BD, and Founder-CEOs doing reseller agreement, OEM deal, or strategic alliance review \u2014 not technical sale enablement, not channel cost economics, not M&A.", + "path": "commercial/partnerships-architect" + }, + { + "name": "pricing-strategist", + "description": "Use when designing or revisiting product pricing \u2014 selecting a pricing model (subscription seat-based, usage-based, value-based, freemium, or hybrid), running Van Westendorp Price Sensitivity Meter analysis on WTP survey data, or designing Good/Better/Best packaging tiers. Recommends a model and a price range with trade-offs, never a single number. For Commercial leads, Product Marketing, and CMOs at the pricing-design moment \u2014 not deal-by-deal discounting, not brand positioning.", + "path": "commercial/pricing-strategist" + }, + { + "name": "rfp-responder", + "description": "Use when an RFP, RFI, RFQ, security questionnaire, vendor questionnaire, or proposal request arrives and the team needs a structured response \u2014 parsing multi-section buyer-dictated requirements (MANDATORY vs WEIGHTED vs NICE-TO-HAVE), building a Shipley-method proof-point matrix mapping each requirement to a verifiable proof point, articulating 3-5 win-themes that ladder up across requirements, and producing a Shipley-derived winrate estimate that informs a bid / no-bid / partner-bid recommendation. For Bid Managers, Proposal Leads, Directors of Sales, and Sales Engineers at the response-strategy moment. Surfaces GAP requirements explicitly \u2014 never invents claims. NOT free-form proposal narrative authoring, NOT contract redline, NOT marketing collateral.", + "path": "commercial/rfp-responder" + } + ], + "research-ops": [ + { + "name": "clinical-research", + "description": "Use when designing a prospective clinical study before submission \u2014 selecting and classifying endpoints (primary / key-secondary / exploratory, with surrogate-endpoint flagging), estimating sample size and power for two-arm designs (means / proportions / survival), or scoring a study plan for feasibility and a GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO phase-gate decision. Every output is an ESTIMATE plus a named human owner (clinician / biostatistician / regulatory owner) \u2014 never clinical fact, never a finished protocol. Distinct from ra-qm-team, which handles the regulatory/QM submission (ISO 13485, EU MDR, FDA 510(k)/PMA/QSR), not the study design.", + "path": "research-ops/clinical-research" + }, + { + "name": "market-research", + "description": "Use when doing upstream market-research methodology \u2014 sizing a market as TAM/SAM/SOM computed BOTH top-down and bottoms-up (never a single unsourced number), planning a survey sample size with finite-population correction and per-segment minimums, or scoring candidate market segments against Kotler's measurable/substantial/accessible/differentiable/actionable criteria. Outputs always show the method and the assumptions. For market-research analysts and product-marketing at the sizing/survey/segmentation moment. Distinct from marketing-skill (campaign analytics, attribution, demand-gen) \u2014 this is the evidence-building methodology, not live-campaign optimization.", + "path": "research-ops/market-research" + }, + { + "name": "product-research", + "description": "Use when planning and synthesizing product/user research as a method-and-repository discipline \u2014 selecting the right method for the goal (generative interviews vs usability test vs concept test vs validation), computing method-based saturation/sample size with an explicit confidence level, or synthesizing coded observations into insights while flagging single-source anecdotes. Never fabricates user insight; an insight requires recurrence across independent participants. Distinct from product-team/ux-researcher-designer (persona/journey artifacts), product-discovery (discovery-sprint planning), and experiment-designer (live A/B) \u2014 this is the research-ops method + insight-repository layer.", + "path": "research-ops/product-research" + }, + { + "name": "research-finance", + "description": "Use when managing the money for an internal R&D program or portfolio \u2014 building a multi-period program budget with the F&A (indirect) split, tracking burn rate and runway against value-inflection milestones, or routing R&D cost items to a capitalize-vs-expense determination. Every budget output surfaces its assumptions block; capitalize-vs-expense is decision-support only and routes to a named finance owner \u2014 it never books an entry or decides accounting treatment. Distinct from finance/financial-analysis (corporate DCF, close, valuation) and research/grants (funding discovery \u2014 this manages money already won).", + "path": "research-ops/research-finance" + }, + { + "name": "research-ops-skills", + "description": "Use when planning, funding, scoping, or synthesizing enterprise research across workstreams \u2014 clinical study design, R&D program finance, market sizing/surveys, or product/user research. Triggers on \"design this clinical study\", \"what sample size\", \"R&D budget\", \"burn rate\", \"capitalize or expense\", \"TAM SAM SOM\", \"market sizing\", \"survey design\", \"segment the market\", \"plan user interviews\", \"usability test\", \"synthesize research insights\". Forks context to route to one of four Research-Operations sub-skills (clinical-research, research-finance, market-research, product-research) and returns a digest. Distinct from ra-qm-team (regulatory submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), and marketing-skill (campaign analytics).", + "path": "research-ops/research-ops-skills" + } ] } } \ No newline at end of file diff --git a/CLAUDE.md b/CLAUDE.md index d812914d..b4c13cd9 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,7 +6,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co This is a **comprehensive skills library** for Claude AI and Claude Code - reusable, production-ready skill packages that bundle domain expertise, best practices, analysis tools, and strategic frameworks. The repository provides modular skills that teams can download and use directly in their workflows. -**Current Scope:** 330 production-ready skills across 14 domains with ~451 Python automation tools, ~590 reference guides, 50+ agents (cs-* + 7 personas), and 81+ slash commands. **v2.8.0 (complete)** added 2 new top-level domains — **business-operations/** (7 internal-ops skills: orchestrator + process-mapper + vendor-management + capacity-planner + internal-comms + knowledge-ops + procurement-optimizer) and **commercial/** (8 per-deal-economics skills: orchestrator + pricing-strategist + deal-desk + partnerships-architect + channel-economics + commercial-policy + rfp-responder + commercial-forecaster) — with orchestrator skills using `context: fork` for chaining, Matt Pocock docs-anchored "Forcing-question library" in every SKILL.md, plus `/cs:grill-bizops` and `/cs:grill-commercial`. **v2.8.2** adds a productivity-shaped `handoff` skill (sibling to engineering/handoff) inspired by Matt Pocock — first-run setup with configurable save location, redaction linter, SessionStart + SessionEnd hooks, fidelity self-check, `--refresh` flag. **v2.8.1** upgraded the engineering role-skills (senior-fullstack / senior-frontend / senior-backend) with karpathy-coder + Matt Pocock decision engines + per-role forcing questions. v2.7.3 ports `alirezarezvani/aeo-box` — AEO (Answer Engine Optimization) skill into marketing-skill/ + security-guidance PreToolUse hook into engineering/. v2.7.0 added 13 Path-B skills across 3 top-level domains (productivity, marketing, research). v2.6.0 added 4 Matt Pocock-derived productivity skills. +**Current Scope:** 338 production-ready skills across 16 domains with 533 Python automation tools, 676 reference guides, 51+ agents (cs-* + 7 personas), and 87+ slash commands, distributed as 62 marketplace plugins. **v2.9.0 (complete)** added the **research-ops/** top-level domain — enterprise Research Operations (orchestrator + clinical-research + research-finance + market-research + product-research), the managed counterpart to the academic research/ domain, with `context: fork` orchestration and a Matt Pocock "Forcing-question library" in every SKILL.md plus `/cs:grill-research-ops`. **v2.8.0 (complete)** added 2 new top-level domains — **business-operations/** (7 internal-ops skills: orchestrator + process-mapper + vendor-management + capacity-planner + internal-comms + knowledge-ops + procurement-optimizer) and **commercial/** (8 per-deal-economics skills: orchestrator + pricing-strategist + deal-desk + partnerships-architect + channel-economics + commercial-policy + rfp-responder + commercial-forecaster) — with orchestrator skills using `context: fork` for chaining, Matt Pocock docs-anchored "Forcing-question library" in every SKILL.md, plus `/cs:grill-bizops` and `/cs:grill-commercial`. **v2.8.2** adds a productivity-shaped `handoff` skill (sibling to engineering/handoff) inspired by Matt Pocock — first-run setup with configurable save location, redaction linter, SessionStart + SessionEnd hooks, fidelity self-check, `--refresh` flag. **v2.8.1** upgraded the engineering role-skills (senior-fullstack / senior-frontend / senior-backend) with karpathy-coder + Matt Pocock decision engines + per-role forcing questions. v2.7.3 ports `alirezarezvani/aeo-box` — AEO (Answer Engine Optimization) skill into marketing-skill/ + security-guidance PreToolUse hook into engineering/. v2.7.0 added 13 Path-B skills across 3 top-level domains (productivity, marketing, research). v2.6.0 added 4 Matt Pocock-derived productivity skills. **Key Distinction**: This is NOT a traditional application. It's a library of skill packages meant to be extracted and deployed by users into their own Claude workflows. @@ -39,6 +39,7 @@ This repository uses **modular documentation**. For domain-specific guidance, se | **RA/QM Compliance** | [ra-qm-team/CLAUDE.md](ra-qm-team/CLAUDE.md) | ISO 13485, MDR, FDA, GDPR, ISO 27001 compliance | | **Business & Growth** | [business-growth/CLAUDE.md](business-growth/CLAUDE.md) | Customer success, sales engineering, revenue operations | | **Finance** | [finance/CLAUDE.md](finance/CLAUDE.md) | Financial analysis, DCF valuation, budgeting, forecasting, SaaS metrics | +| **Research Operations** | [research-ops/CLAUDE.md](research-ops/CLAUDE.md) | Clinical study design, R&D finance, market research, product research (enterprise counterpart to academic research/) | | **Standards Library** | [standards/CLAUDE.md](standards/CLAUDE.md) | Communication, quality, git, security standards | | **Templates** | [templates/CLAUDE.md](templates/CLAUDE.md) | Template system usage | @@ -49,17 +50,22 @@ This repository uses **modular documentation**. For domain-specific guidance, se ``` claude-code-skills/ ├── .claude-plugin/ # Plugin registry (marketplace.json) -├── agents/ # 27 agents (20 cs-* + 7 personas) -├── commands/ # 33 slash commands (changelog, tdd, saas-health, prd, code-to-prd, plugin-audit, sprint-plan, slo-design, etc.) -├── engineering-team/ # 32 core engineering skills + Playwright Pro + Self-Improving Agent + Security Suite -├── engineering/ # 44 POWERFUL-tier advanced skills (incl. AgentHub, self-eval, llm-wiki, tc-tracker, ship-gate, slo-architect, write-a-skill, caveman, grill-me, handoff) -├── product-team/ # 13 product skills (incl. apple-hig-expert) + Python tools -├── marketing-skill/ # 44 marketing skills (7 pods) + Python tools -├── c-level-advisor/ # 28 C-level advisory skills (10 roles + orchestration) +├── agents/ # 32 standalone agents (cs-* + 7 personas); 51+ cs-* agents repo-wide +├── commands/ # slash commands (changelog, tdd, saas-health, prd, code-to-prd, plugin-audit, sprint-plan, slo-design, etc.); 87+ repo-wide +├── engineering-team/ # 51 core engineering skills + Playwright Pro + Self-Improving Agent + Security Suite +├── engineering/ # 78 POWERFUL-tier advanced skills (incl. AgentHub, autoresearch-agent, self-eval, llm-wiki, tc-tracker, ship-gate, slo-architect, write-a-skill, caveman, grill-me, handoff) +├── product-team/ # 17 product skills (incl. apple-hig-expert) + Python tools +├── marketing-skill/ # 46 marketing skills (8 pods) + Python tools +├── c-level-advisor/ # 66 C-level advisory skills (full C-suite + founder-mode agents + orchestration) ├── project-management/ # 9 PM skills + bundled Atlassian Remote MCP (.mcp.json) -├── ra-qm-team/ # 14 RA/QM compliance skills +├── ra-qm-team/ # 18 RA/QM compliance skills +├── compliance-os/ # 9 compliance-OS skills ├── business-growth/ # 5 business & growth skills + Python tools -├── finance/ # 3 finance skills + Python tools +├── business-operations/ # 7 internal-ops skills (orchestrator + 6 sub-skills) +├── commercial/ # 8 per-deal-economics skills (orchestrator + 7 sub-skills) +├── finance/ # 4 finance skills + Python tools +├── research/ # 8 academic research skills (orchestrator + 7 specialists) +├── research-ops/ # 5 research-ops skills (orchestrator + clinical-research + research-finance + market-research + product-research) ├── eval-workspace/ # Skill evaluation results (Tessl) ├── standards/ # 5 standards library files ├── templates/ # Reusable templates @@ -137,7 +143,18 @@ See [standards/git/git-workflow-standards.md](standards/git/git-workflow-standar ## Current Version -**Version:** v2.8.4 (released — productivity/andreessen v1.0) +**Version:** v2.9.0 (released — research-ops/ domain: enterprise Research Operations) + +**v2.9.0 highlights — research-ops/ domain (new top-level domain):** + +New `research-ops/` top-level domain — the enterprise / cross-functional counterpart to the academic `research/` domain (which finds literature, grants, patents). Single domain plugin (commercial/ + business-operations/ pattern): orchestrator (`context: fork`) + 4 managed sub-skills. + +- **`clinical-research`** — prospective clinical STUDY design (not regulatory submission, which stays in `ra-qm-team`). 3 stdlib tools: `sample_size_estimator.py` (closed-form power/n for means/proportions/survival with a built-in z-table, dropout inflation, "ESTIMATE — confirm with a biostatistician" banner), `endpoint_selector.py` (5-dimension scoring → PRIMARY/KEY-SECONDARY/EXPLORATORY, penalizes unvalidated surrogates), `phase_gate_scorer.py` (feasibility 0-100 → GO/GO-WITH-CONDITIONS/REDESIGN/NO-GO + named owner chain). Canon: ICH E8/E9/E9(R1), CONSORT, SPIRIT, FDA Multiple Endpoints, Cohen, Schoenfeld. +- **`research-finance`** — internal R&D PROGRAM/portfolio finance (not corporate close `finance/`, not grant discovery `research/grants`). 3 tools: `program_budget_planner.py` (multi-period budget + F&A/MTDC split + assumptions block), `burn_runway_tracker.py` (trailing burn, runway, milestone-vs-cash), `capex_vs_opex_router.py` (IAS 38 / ASC 730 routing → CAPITALIZE-CANDIDATE/EXPENSE/FINANCE-OWNER-REVIEW, never auto-decides). Canon: IAS 38, ASC 730/985-20, 2 CFR 200, Cooper stage-gate, rNPV. +- **`market-research`** — upstream sizing/survey/segmentation methodology (not campaign analytics `marketing-skill`). 3 tools: `market_sizer.py` (TAM/SAM/SOM both top-down AND bottoms-up + triangulation flag, never a single number), `sample_size_planner.py` (survey n + FPC + per-segment minima), `segmentation_scorer.py` (Kotler 5-criteria + substantiality/accessibility gate). Canon: Cochran, Dillman, Groves, Kotler, Bessemer/a16z sizing. +- **`product-research`** — product/user research method + insight-repository discipline (not persona/journey/live-A-B `product-team`). 3 tools: `study_designer.py` (goal×stage → method + plan skeleton), `saturation_planner.py` (Nielsen-5 / Guest-12 with explicit confidence), `insight_synthesizer.py` (clusters coded observations, flags single-source anecdotes — never promotes them). Canon: Portigal, JTBD, Rohrer (NN/g), Nielsen, Guest et al., ResearchOps/Polaris. +- **Hard rules:** clinical outputs are estimates + named clinical owner (never fact); finance surfaces assumptions and routes treatment to a named finance owner (never auto-decides); market sizes show method + assumptions (never a single number); product insights require recurrence across independent participants. `cs-research-ops-orchestrator` agent + `/cs:research-ops` router + `/cs:grill-research-ops` (Matt docs-anchored grilling) + 4 per-skill commands. +- **Onboarding + customization + autoresearch (per sub-skill, isolated):** each sub-skill ships `onboard.py` (its own question set), `config_loader.py` (a customization config consumed by every tool, project>global>defaults precedence, `RESEARCH_OPS_NO_CONFIG=1` bypass), and `ar_evaluator.py` — an opt-in, locked-ground-truth bridge to `engineering/autoresearch-agent` (loop edits the skill's input file; metrics: clinical `feasibility_composite`↑, finance `runway_months`↑, market `tam_divergence`↓, product `validated_insights`↑). 24 stdlib tools total (12 analysis + 12 onboarding/customization/autoresearch; all pass `--help`/`--sample`), 12 reference docs (5-7 sources each). Marketplace 61 → 62 plugins; domains 15 → 16. **v2.8.3** shipped the Mistral Vibe cross-platform sync (`scripts/sync-vibe-skills.py`, `~/.vibe/skills/claude-skills/`) — bringing first-class tool support to 13 coding agents. @@ -416,6 +433,6 @@ This repository publishes skills to **ClawHub** (clawhub.com) as the distributio --- -**Last Updated:** May 24, 2026 -**Version:** v2.8.4 -**Status:** 330 skills deployed across 14 domains, 61 marketplace plugins, docs site live +**Last Updated:** May 27, 2026 +**Version:** v2.9.0 +**Status:** 338 skills deployed across 16 domains, 62 marketplace plugins, docs site live diff --git a/README.md b/README.md index d34c0cfd..62b2468d 100644 --- a/README.md +++ b/README.md @@ -1,19 +1,19 @@ # Claude Code Skills & Plugins — Agent Skills for Every Coding Tool -**313 production-ready Claude Code skills, plugins, and agent skills for 12 AI coding tools.** +**338 production-ready Claude Code skills, plugins, and agent skills for 13 AI coding tools.** -The most comprehensive open-source library of Claude Code skills and agent plugins — also works with OpenAI Codex, Gemini CLI, Cursor, and 7 more coding agents. Reusable expertise packages covering engineering, DevOps, marketing (incl. v2.7.3 AEO — Answer Engine Optimization for LLM citation), security (PreToolUse hooks), compliance, C-level advisory (incl. founder-mode CFO/CMO/CRO/CPO/COO/CHRO/CISO/GC/CDO/CAIO/CCO/VPE personas + 21 /cs:* slash commands), productivity (capture/email/reflect), and a complete research stack (litreview/grants/dossier/patent/syllabus/pulse/notebooklm + hybrid router). +The most comprehensive open-source library of Claude Code skills and agent plugins — also works with OpenAI Codex, Gemini CLI, Cursor, and 9 more coding agents. Reusable expertise packages covering engineering, DevOps, marketing (incl. AEO — Answer Engine Optimization for LLM citation), security (PreToolUse hooks), compliance, C-level advisory (incl. founder-mode CFO/CMO/CRO/CPO/COO/CHRO/CISO/GC/CDO/CAIO/CCO/VPE personas + 21 /cs:* slash commands), productivity (capture/email/reflect), an academic research stack (litreview/grants/dossier/patent/syllabus/pulse/notebooklm + hybrid router), and enterprise Research Operations (clinical-research/research-finance/market-research/product-research, v2.9.0). **Works with:** Claude Code · OpenAI Codex · Gemini CLI · OpenClaw · Hermes Agent[^hermes] · Mistral Vibe[^vibe] · Cursor · Aider · Windsurf · Kilo Code · OpenCode · Augment · Antigravity -[^hermes]: Hermes Agent is **BYO-sync tier**: the repo ships a pre-generated `.hermes/skills/claude-skills/` tree (305 skills across 12 domains as of v2.7.3), but you run `python scripts/sync-hermes-skills.py` once locally to install into `~/.hermes/skills/`. Uses the same agentskills.io SKILL.md standard — no format conversion. -[^vibe]: Mistral Vibe is also **BYO-sync tier**: the repo ships a pre-generated `.vibe/skills/claude-skills/` tree (306 skills across 14 domains), run `./scripts/vibe-install.sh` once locally to install into `~/.vibe/skills/`. Same agentskills.io SKILL.md standard — no format conversion. Docs: . +[^hermes]: Hermes Agent is **BYO-sync tier**: the repo ships a pre-generated `.hermes/skills/claude-skills/` tree, but you run `python scripts/sync-hermes-skills.py` once locally to install into `~/.hermes/skills/`. Uses the same agentskills.io SKILL.md standard — no format conversion. +[^vibe]: Mistral Vibe is also **BYO-sync tier**: the repo ships a pre-generated `.vibe/skills/claude-skills/` tree, run `./scripts/vibe-install.sh` once locally to install into `~/.vibe/skills/`. Same agentskills.io SKILL.md standard — no format conversion. Docs: . [![License: MIT](https://img.shields.io/badge/License-MIT-yellow?style=for-the-badge)](https://opensource.org/licenses/MIT) -[![Skills](https://img.shields.io/badge/Skills-330-brightgreen?style=for-the-badge)](#skills-overview) -[![Agents](https://img.shields.io/badge/Agents-49+-blue?style=for-the-badge)](#agents) +[![Skills](https://img.shields.io/badge/Skills-338-brightgreen?style=for-the-badge)](#skills-overview) +[![Agents](https://img.shields.io/badge/Agents-51+-blue?style=for-the-badge)](#agents) [![Personas](https://img.shields.io/badge/Personas-7-purple?style=for-the-badge)](#personas) -[![Commands](https://img.shields.io/badge/Commands-79+-orange?style=for-the-badge)](#commands) +[![Commands](https://img.shields.io/badge/Commands-87+-orange?style=for-the-badge)](#commands) [![Stars](https://img.shields.io/github/stars/alirezarezvani/claude-skills?style=for-the-badge)](https://github.com/alirezarezvani/claude-skills/stargazers) [![SkillCheck Validated](https://img.shields.io/badge/SkillCheck-Validated-4c1?style=for-the-badge)](https://getskillcheck.com) @@ -26,10 +26,10 @@ The most comprehensive open-source library of Claude Code skills and agent plugi Claude Code skills (also called agent skills or coding agent plugins) are modular instruction packages that give AI coding agents domain expertise they don't have out of the box. Each skill includes: - **SKILL.md** — structured instructions, workflows, and decision frameworks -- **Python tools** — ~402 CLI scripts (all stdlib-only, zero pip installs) -- **Reference docs** — templates, checklists, and domain-specific knowledge +- **Python tools** — 533 CLI scripts (all stdlib-only, zero pip installs) +- **Reference docs** — 676 templates, checklists, and domain-specific knowledge files -**One repo, twelve platforms.** Works natively as Claude Code plugins, Codex agent skills, Gemini CLI skills, Hermes Agent skills, Mistral Vibe skills, and converts to 7 more tools via `scripts/convert.sh`. All ~402 Python tools run anywhere Python runs. +**One repo, thirteen platforms.** Works natively as Claude Code plugins, Codex agent skills, Gemini CLI skills, Hermes Agent skills, Mistral Vibe skills, and converts to more tools via `scripts/convert.sh`. All 533 Python tools run anywhere Python runs. ### Skills vs Agents vs Personas @@ -108,7 +108,7 @@ git clone https://github.com/alirezarezvani/claude-skills.git ## Multi-Tool Support (New) -**Convert all 156 skills to 7 AI coding tools** with a single script: +**Convert all 338 skills to 9 AI coding tools** with a single script: | Tool | Format | Install | |------|--------|---------| @@ -135,11 +135,11 @@ git clone https://github.com/alirezarezvani/claude-skills.git ./scripts/install.sh --tool aider --target . --force # 3. Verify -find .cursor/rules -name "*.mdc" | wc -l # Should show 156 +find .cursor/rules -name "*.mdc" | wc -l # Should show 338 ``` **Each tool gets:** -- ✅ All 156 skills converted to native format +- ✅ All 338 skills converted to native format - ✅ Per-tool README with install/verify/update steps - ✅ Support for scripts, references, templates where applicable - ✅ Zero manual conversion work @@ -150,24 +150,26 @@ Run `./scripts/convert.sh --tool all` to generate tool-specific outputs locally. ## Skills Overview -**313 skills across 12 domains:** +**338 skills across 16 domains:** | Domain | Skills | Highlights | Details | |--------|--------|------------|---------| -| **🔧 Engineering — Core** | 32 | Architecture, frontend, backend, fullstack, QA, DevOps, SecOps, AI/ML, data, Playwright, self-improving agent, security suite (6), a11y audit | [engineering-team/](engineering-team/) | -| **🎭 Playwright Pro** | 9+3 | Test generation, flaky fix, Cypress/Selenium migration, TestRail, BrowserStack, 55 templates | [engineering-team/playwright-pro](engineering-team/playwright-pro/) | -| **🧠 Self-Improving Agent** | 5+2 | Auto-memory curation, pattern promotion, skill extraction, memory health | [engineering-team/self-improving-agent](engineering-team/self-improving-agent/) | -| **⚡ Engineering — POWERFUL** | 45 | Agent designer, RAG architect, database designer, CI/CD builder, security auditor, MCP builder, AgentHub, Helm charts, Terraform, self-eval, llm-wiki, tc-tracker, **reliability portfolio** (feature-flags-architect, kubernetes-operator, chaos-engineering, slo-architect), ship-gate, **security-guidance** (✨v2.7.3 — PreToolUse hook catching 12 anti-patterns), **Matt Pocock skills** (write-a-skill, caveman, grill-me, handoff, grill-with-docs) | [engineering/](engineering/) | -| **🎯 Product** | 13 | Product manager, agile PO, strategist, UX researcher, UI design, landing pages, SaaS scaffolder, analytics, experiment designer, discovery, roadmap communicator, code-to-prd, apple-hig-expert | [product-team/](product-team/) | -| **📣 Marketing** | 45 | 8 pods: Content (8), SEO + AEO (6 incl. ✨v2.7.3 `aeo` — E-E-A-T audit, citation tracking across 5 LLMs), CRO (6), Channels (6), Growth (4), Intelligence (4), Sales (2) + context foundation + orchestration router. 58 Python tools. | [marketing-skill/](marketing-skill/) | -| **🚀 Productivity** ✨v2.8.4 | 6 | `capture` (brain-dump-to-action), `email` pair (inbox-setup + inbox-triage with 7-file KB contract), `reflect` (light-prompt journal), **`handoff`** (Matt Pocock-inspired: first-run setup, redaction linter, SessionStart + SessionEnd hooks, fidelity self-check, `--refresh`), **`andreessen`** (✨v2.8.4 — market-first decision & productivity mode: market > team > product, PMF-first, 3x5-card + Anti-Todo, fixed anti-sycophancy operating prompt). Path-B from megaprompts 05-08 + Matt Pocock + Andreessen derivation. | [productivity/](productivity/) | -| **🎨 Marketing (top-level)** ✨v2.7.0 | 1 | `landing` — single-file HTML landing-page generator (4 design styles, GSAP patterns, brand palette validator). Path-B from megaprompt 04. | [marketing/](marketing/) | -| **🔬 Research** ✨v2.7.0 | 8 | `research` orchestrator (hybrid router + fallback, megaprompt 13) + 7 specialists: `pulse` (recency), `litreview` (academic), `grants` (NIH), `dossier` (entity), `patent` (prior-art), `syllabus` (course reading), `notebooklm` (browser-automation). | [research/](research/) | +| **🔧 Engineering — Core** | 51 | Architecture, frontend, backend, fullstack, QA, DevOps, SecOps, AI/ML, data, Playwright Pro (test gen, flaky fix, migrations), self-improving agent (auto-memory curation), security suite, a11y audit | [engineering-team/](engineering-team/) | +| **⚡ Engineering — POWERFUL** | 78 | Agent designer, RAG architect, database designer, CI/CD builder, security auditor, MCP builder, AgentHub, Helm charts, Terraform, self-eval, llm-wiki, tc-tracker, autoresearch-agent, **reliability portfolio** (feature-flags-architect, kubernetes-operator, chaos-engineering, slo-architect), ship-gate, security-guidance PreToolUse hook, **Matt Pocock skills** (write-a-skill, caveman, grill-me, handoff, grill-with-docs) | [engineering/](engineering/) | +| **🎯 Product** | 17 | Product manager, agile PO, strategist, UX researcher, UI design, landing pages, SaaS scaffolder, analytics, experiment designer, discovery, roadmap communicator, code-to-prd, apple-hig-expert | [product-team/](product-team/) | +| **📣 Marketing** | 46 | 8 pods: Content, SEO + AEO (`aeo` — E-E-A-T audit, citation tracking across 5 LLMs), CRO, Channels, Growth, Intelligence, Sales + context foundation + orchestration router | [marketing-skill/](marketing-skill/) | +| **🚀 Productivity** | 6 | `capture` (brain-dump-to-action), `email` pair (inbox-setup + inbox-triage), `reflect` (journal), `handoff` (Matt Pocock-inspired), `andreessen` (market-first decision mode) | [productivity/](productivity/) | +| **🎨 Marketing (top-level)** | 1 | `landing` — single-file HTML landing-page generator (4 design styles, GSAP patterns, brand palette validator) | [marketing/](marketing/) | +| **🔬 Research (academic)** | 8 | `research` orchestrator (hybrid router + fallback) + 7 specialists: `pulse`, `litreview`, `grants` (NIH), `dossier`, `patent`, `syllabus`, `notebooklm` | [research/](research/) | +| **🧪 Research Operations** ✨v2.9.0 | 5 | Enterprise/cross-functional research: orchestrator + `clinical-research` (study design), `research-finance` (R&D program finance), `market-research` (sizing/survey/segmentation), `product-research` (user research) — each with onboarding + customization + opt-in autoresearch bridge | [research-ops/](research-ops/) | | **📋 Project Management** | 9 | Senior PM, scrum master, Jira, Confluence, Atlassian admin, templates + bundled Atlassian Remote MCP | [project-management/](project-management/) | -| **🏥 Regulatory & QM** | 14 | ISO 13485, MDR 2017/745, FDA, ISO 27001, GDPR, SOC 2, CAPA, risk management | [ra-qm-team/](ra-qm-team/) | -| **💼 C-Level Advisory** | 28 | Full C-suite (10 roles) + orchestration + board meetings + culture & collaboration | [c-level-advisor/](c-level-advisor/) | +| **🏥 Regulatory & QM** | 18 | ISO 13485, MDR 2017/745, FDA, ISO 27001, GDPR, SOC 2, CAPA, risk management | [ra-qm-team/](ra-qm-team/) | +| **🛡️ Compliance OS** | 9 | Compliance operating system — controls, evidence, audit-readiness workflows | [compliance-os/](compliance-os/) | +| **💼 C-Level Advisory** | 66 | Full C-suite (CEO/CTO/CFO/CMO/CRO/CPO/COO/CHRO/CISO/GC/CDO/CAIO/CCO/VPE) + founder-mode agents + orchestration + board meetings + culture & collaboration | [c-level-advisor/](c-level-advisor/) | | **📈 Business & Growth** | 5 | Customer success, sales engineer, revenue ops, contracts & proposals, BizDev toolkit | [business-growth/](business-growth/) | -| **💰 Finance** | 3 | Financial analyst (DCF, budgeting, forecasting), SaaS metrics coach, business investment advisor | [finance/](finance/) | +| **🏭 Business Operations** | 7 | Orchestrator + process-mapper, vendor-management, capacity-planner, internal-comms, knowledge-ops, procurement-optimizer | [business-operations/](business-operations/) | +| **🤝 Commercial** | 8 | Orchestrator + pricing-strategist, deal-desk, partnerships-architect, channel-economics, commercial-policy, rfp-responder, commercial-forecaster | [commercial/](commercial/) | +| **💰 Finance** | 4 | Financial analyst (DCF, budgeting, forecasting), SaaS metrics coach, business investment advisor | [finance/](finance/) | --- @@ -304,7 +306,7 @@ for MDR Annex II compliance gaps. ## Python Analysis Tools -~402 CLI tools ship with the skills (all verified, stdlib-only): +533 CLI tools ship with the skills (all verified, stdlib-only): ```bash # SaaS health check @@ -351,7 +353,7 @@ Yes. Skills work natively with 13 tools: Claude Code, OpenAI Codex, Gemini CLI, No. We follow semantic versioning and maintain backward compatibility within patch releases. Existing script arguments, plugin source paths, and SKILL.md structures are never changed in patch versions. See the [CHANGELOG](CHANGELOG.md) for details on each release. **Are the Python tools dependency-free?** -Yes. All ~402 Python CLI tools use the standard library only — zero pip installs required. Every script is verified to run with `--help`. +Yes. All 533 Python CLI tools use the standard library only — zero pip installs required. Every script is verified to run with `--help`. **How do I create my own Claude Code skill?** Each skill is a folder with a `SKILL.md` (frontmatter + instructions), optional `scripts/`, `references/`, and `assets/`. See the [Skills & Agents Factory](https://github.com/alirezarezvani/claude-code-skills-agents-factory) for a step-by-step guide. diff --git a/business-growth/.claude-plugin/plugin.json b/business-growth/.claude-plugin/plugin.json index 7861ec53..6945ceb6 100644 --- a/business-growth/.claude-plugin/plugin.json +++ b/business-growth/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "business-growth-skills", "description": "5 business & growth skills: customer success manager, sales engineer, revenue operations, contract & proposal writer, and BizDev-toolkit. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.", - "version": "2.2.3", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/business-growth", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/business-growth/skills/business-growth-skills/SKILL.md b/business-growth/skills/business-growth-skills/SKILL.md index f9bce66a..15f595f3 100644 --- a/business-growth/skills/business-growth-skills/SKILL.md +++ b/business-growth/skills/business-growth-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "business-growth-skills" description: "4 business growth agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Customer success (health scoring, churn), sales engineer (RFP), revenue operations (pipeline, GTM), contract & proposal writer. Python tools (stdlib-only)." -version: 1.1.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/c-level-advisor/.claude-plugin/plugin.json b/c-level-advisor/.claude-plugin/plugin.json index 8dd2e1ff..2ca4cd4d 100644 --- a/c-level-advisor/.claude-plugin/plugin.json +++ b/c-level-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "c-level-skills", "description": "33 C-level advisory skills + c-level-agents plugin layer (13 cs-* persona agents + 21 /cs:* slash commands). Complete virtual board of directors with CEO, CTO, COO, CPO, CMO, CFO, CRO, CISO, CHRO advisors plus General Counsel, Chief Data Officer, Chief AI Officer, Chief Customer Officer, and VP of Engineering (delivery throughput DORA analyzer, eng hiring funnel calculator, eng team structure designer), executive mentor, founder coach, Chief of Staff router, board meetings, decision logger, board deck builder, scenario war room, competitive intel, org health diagnostic, M&A playbook, international expansion, culture architect, change management, strategic alignment, and the founder-mode plugin (office-hours, boardroom, brief/decide/execute/post-mortem pipeline, cross-model consensus, decision freeze).", - "version": "2.5.5", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/README.md b/c-level-advisor/README.md index 5bc11e7d..302ca3cf 100644 --- a/c-level-advisor/README.md +++ b/c-level-advisor/README.md @@ -376,5 +376,5 @@ This C-Level advisory skills collection provides executive leadership guidance f --- **Last Updated:** January 2026 -**Skills Deployed:** 2/2 C-Level advisory skills production-ready +**Skills Deployed:** 66/66 C-Level advisory skills production-ready **Total Tools:** 6 Python analysis tools (strategy, finance, tech debt, team scaling) diff --git a/c-level-advisor/c-level-agents/.claude-plugin/plugin.json b/c-level-advisor/c-level-agents/.claude-plugin/plugin.json index 935b7cec..f65885bc 100644 --- a/c-level-advisor/c-level-agents/.claude-plugin/plugin.json +++ b/c-level-advisor/c-level-agents/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "c-level-agents", "description": "Founder-mode executive team plugin: 13 cs-* C-suite agents (CFO, CMO, CRO, CPO, COO, CHRO, CISO, Chief of Staff, General Counsel, Chief Data Officer, Chief AI Officer, Chief Customer Officer, VP of Engineering) plus 21 /cs:* slash commands for forcing-question office hours (incl. /cs:vpe-review), multi-role boardroom deliberation, strategic sprint pipeline, and meta routing. Wraps the 33 c-level skills (including vpe-advisor with delivery throughput DORA analyzer + eng hiring funnel calculator + eng team structure designer) with cognitive gearing and artifact handoffs.", - "version": "1.5.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/c-level-agents", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/c-level-agents/README.md b/c-level-advisor/c-level-agents/README.md index f50686fb..de0cd987 100644 --- a/c-level-advisor/c-level-agents/README.md +++ b/c-level-advisor/c-level-agents/README.md @@ -83,6 +83,6 @@ Existing `cs-ceo-advisor` and `cs-cto-advisor` live in `/agents/c-level/` and in --- -**Version:** 1.0.0 +**Version:** 2.9.0 **Status:** Production Ready **License:** MIT diff --git a/c-level-advisor/chief-ai-officer-advisor/.claude-plugin/plugin.json b/c-level-advisor/chief-ai-officer-advisor/.claude-plugin/plugin.json index 748119d0..16327b58 100644 --- a/c-level-advisor/chief-ai-officer-advisor/.claude-plugin/plugin.json +++ b/c-level-advisor/chief-ai-officer-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "chief-ai-officer-advisor", "description": "Chief AI Officer advisory: model build-vs-buy calculator (API vs fine-tune vs build with 3-year TCO across 6 paths + breakeven balancing economics with practical feasibility), AI risk classifier (EU AI Act tier with 7 Article citations + US state patchwork: NYC LL 144, CO AI Act, IL HB 53, CA SB 1001, IL BIPA + industry overlays for FDA AI/ML, CFPB Circular 2023-03, NYDFS Reg 23, NAIC, ECOA, Fed SR 11-7), AI cost economics (API vs self-hosted breakeven with 2026 pricing across A100/H100, utilization reality, hidden costs). 4 in-depth references each citing 5+ authoritative sources. Stdlib-only. Standalone-installable; also bundled in c-level-skills. Strategic only - does not duplicate engineering AI/ML skills.", - "version": "1.0.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/chief-ai-officer-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/chief-customer-officer-advisor/.claude-plugin/plugin.json b/c-level-advisor/chief-customer-officer-advisor/.claude-plugin/plugin.json index 67602c5b..712db478 100644 --- a/c-level-advisor/chief-customer-officer-advisor/.claude-plugin/plugin.json +++ b/c-level-advisor/chief-customer-officer-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "chief-customer-officer-advisor", "description": "Chief Customer Officer advisory for startups: retention decomposition analyzer (honest GRR vs NRR + 7-category churn taxonomy), customer segmentation designer (4-tier framework + ICP fit scoring + kill list), CS coverage calculator (pooled vs named CSM models + ratio math + 12-month hiring plan). 4 in-depth references: retention decomposition, customer segmentation strategy, CS coverage model, CS team org evolution (CSM vs Support vs AM vs IM vs CS Ops). Stdlib-only. Standalone-installable; also bundled in c-level-skills. Strategic only - does not duplicate business-growth tactical CS skills.", - "version": "1.0.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/chief-customer-officer-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/chief-data-officer-advisor/.claude-plugin/plugin.json b/c-level-advisor/chief-data-officer-advisor/.claude-plugin/plugin.json index 94cc8cc9..4249d716 100644 --- a/c-level-advisor/chief-data-officer-advisor/.claude-plugin/plugin.json +++ b/c-level-advisor/chief-data-officer-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "chief-data-officer-advisor", "description": "Chief Data Officer advisory: AI training data audit (origin x class x use-case matrix with GDPR Art. 6 + EU AI Act citations -> GO/MITIGATE/NO-GO per source), data product strategy picker (warehouse vs lakehouse vs mesh + 6-layer build-vs-buy + 12-month sequencing), data asset valuator (strategic value 0-10 + M&A multiplier with carve-out penalties + 3 ranked productization paths). 4 references answering one decision each: training rights, data product strategy, customer-data-as-asset, data team org evolution. Stdlib-only. Standalone-installable; also bundled in c-level-skills. Strategic only - does not duplicate engineering data skills.", - "version": "1.0.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/chief-data-officer-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/executive-mentor/.claude-plugin/plugin.json b/c-level-advisor/executive-mentor/.claude-plugin/plugin.json index a76677cf..eda3915f 100644 --- a/c-level-advisor/executive-mentor/.claude-plugin/plugin.json +++ b/c-level-advisor/executive-mentor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "executive-mentor", "description": "Adversarial thinking partner for founders and executives. Stress-tests plans, prepares for board meetings, navigates hard decisions, and forces honest post-mortems.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/executive-mentor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/general-counsel-advisor/.claude-plugin/plugin.json b/c-level-advisor/general-counsel-advisor/.claude-plugin/plugin.json index 6b3bfba3..30c19b0c 100644 --- a/c-level-advisor/general-counsel-advisor/.claude-plugin/plugin.json +++ b/c-level-advisor/general-counsel-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "general-counsel-advisor", "description": "General Counsel advisory for startups: contract risk scanner (12 founder-killer patterns: auto-renew traps, uncapped indemnity, vague IP, MFN pricing, missing DPA, one-sided venue, broad non-solicit, perpetual license-back, etc.) and term sheet analyzer (0-100 founder-friendliness score across 12 dimensions: liquidation preference, anti-dilution, option pool, board, vesting, drag-along, protective provisions, info rights, dividends, valuation). 3 in-depth references: contracts playbook (7 startup contract types), IP + regulatory landscape (HIPAA, GDPR, FDA, fintech, EU AI Act + SOC 2 to ISO sequencing), term sheet decoder. Stdlib-only. Standalone-installable; also bundled in c-level-skills. NOT a substitute for licensed counsel.", - "version": "1.0.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/general-counsel-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/c-level-advisor/vpe-advisor/.claude-plugin/plugin.json b/c-level-advisor/vpe-advisor/.claude-plugin/plugin.json index 88fd9dfe..3079f23d 100644 --- a/c-level-advisor/vpe-advisor/.claude-plugin/plugin.json +++ b/c-level-advisor/vpe-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "vpe-advisor", - "description": "VP of Engineering advisory: delivery throughput analyzer (DORA 4 metrics + cycle-time bottleneck identification), eng hiring funnel calculator (7-stage conversion + pipeline gap + weakest-stage fixes), eng team structure designer (squad/tribe model + manager-trigger + director-trigger + span-of-control). 4 in-depth references: DORA framework, eng hiring funnel, eng team structure (Conway's Law), production discipline (on-call, incidents, deployment, SLOs). Stdlib-only. Standalone-installable; also bundled in c-level-skills. NOT a CTO skill — VPE owns how the team ships, CTO owns what to build.", - "version": "1.0.0", + "description": "VP of Engineering advisory: delivery throughput analyzer (DORA 4 metrics + cycle-time bottleneck identification), eng hiring funnel calculator (7-stage conversion + pipeline gap + weakest-stage fixes), eng team structure designer (squad/tribe model + manager-trigger + director-trigger + span-of-control). 4 in-depth references: DORA framework, eng hiring funnel, eng team structure (Conway's Law), production discipline (on-call, incidents, deployment, SLOs). Stdlib-only. Standalone-installable; also bundled in c-level-skills. NOT a CTO skill \u2014 VPE owns how the team ships, CTO owns what to build.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/c-level-advisor/vpe-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/compliance-os/.claude-plugin/plugin.json b/compliance-os/.claude-plugin/plugin.json index ae0e1bd3..53619fe1 100644 --- a/compliance-os/.claude-plugin/plugin.json +++ b/compliance-os/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "compliance-os", - "description": "Compliance OS — meta-orchestrator for multi-framework compliance programs. Configure-then-operate four stdlib Python tools: framework_selector.py (input: company profile across industry/geography/AI/medical/financial/headcount; output: applicable frameworks ranked across all 9 supported: ISO 27001, 13485, 42001, 14971, EU AI Act, MDR 745, GDPR, SOC 2, FDA QSR), cross_framework_mapper.py (input: 1+ framework control libraries; output: unified control matrix with overlap percentage + mapping confidence + unified evidence requirements per merged control), audit_simulator.py (input: framework scope; output: mock internal audit with 8-15 finding scenarios across 5 severity levels + interview questions per control), evidence_pool_generator.py (input: enabled framework configs; output: consolidated evidence checklist with reuse map). 4 in-depth references citing ISO 19011, IIA Standards, AICPA AT-C, NIST CSF, COSO ERM. Plus 3 cs-* persona agents (cs-compliance-officer, cs-aims-iso42001, cs-ai-act-compliance) + 3 /cs:* slash commands (/cs:compliance-readiness, /cs:aims-audit, /cs:ai-act-readiness). Reuses the 14 existing ra-qm-team skills and the 2 new compliance-team-* plugins.", - "version": "1.2.0", + "description": "Compliance OS \u2014 meta-orchestrator for multi-framework compliance programs. Configure-then-operate four stdlib Python tools: framework_selector.py (input: company profile across industry/geography/AI/medical/financial/headcount; output: applicable frameworks ranked across all 9 supported: ISO 27001, 13485, 42001, 14971, EU AI Act, MDR 745, GDPR, SOC 2, FDA QSR), cross_framework_mapper.py (input: 1+ framework control libraries; output: unified control matrix with overlap percentage + mapping confidence + unified evidence requirements per merged control), audit_simulator.py (input: framework scope; output: mock internal audit with 8-15 finding scenarios across 5 severity levels + interview questions per control), evidence_pool_generator.py (input: enabled framework configs; output: consolidated evidence checklist with reuse map). 4 in-depth references citing ISO 19011, IIA Standards, AICPA AT-C, NIST CSF, COSO ERM. Plus 3 cs-* persona agents (cs-compliance-officer, cs-aims-iso42001, cs-ai-act-compliance) + 3 /cs:* slash commands (/cs:compliance-readiness, /cs:aims-audit, /cs:ai-act-readiness). Reuses the 14 existing ra-qm-team skills and the 2 new compliance-team-* plugins.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,15 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/compliance-os", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/compliance-os", "./skills/compliance-readiness", "./skills/aims-audit", "./skills/ai-act-readiness", "./skills/iso27001-audit-prep", "./skills/iso13485-audit-prep", "./skills/gdpr-audit-prep", "./skills/soc2-audit-prep", "./skills/fda-qsr-audit-prep"] + "skills": [ + "./skills/compliance-os", + "./skills/compliance-readiness", + "./skills/aims-audit", + "./skills/ai-act-readiness", + "./skills/iso27001-audit-prep", + "./skills/iso13485-audit-prep", + "./skills/gdpr-audit-prep", + "./skills/soc2-audit-prep", + "./skills/fda-qsr-audit-prep" + ] } diff --git a/engineering-team/a11y-audit/.claude-plugin/plugin.json b/engineering-team/a11y-audit/.claude-plugin/plugin.json index 9caeb383..ab036642 100644 --- a/engineering-team/a11y-audit/.claude-plugin/plugin.json +++ b/engineering-team/a11y-audit/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "a11y-audit", "description": "WCAG 2.2 accessibility audit and fix skill for React, Next.js, Vue, Angular, Svelte, and HTML. Static scanner detecting 20+ violation types, contrast checker with suggest mode, framework-specific fix patterns, CI-friendly exit codes.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering-team/a11y-audit", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering-team/google-workspace-cli/.claude-plugin/plugin.json b/engineering-team/google-workspace-cli/.claude-plugin/plugin.json index aa4b2dae..49fd5bb4 100644 --- a/engineering-team/google-workspace-cli/.claude-plugin/plugin.json +++ b/engineering-team/google-workspace-cli/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "google-workspace-cli", - "version": "2.2.2", + "version": "2.9.0", "description": "Google Workspace administration via the gws CLI. Install, authenticate, and automate Gmail, Drive, Sheets, Calendar, Docs, Chat, and Tasks. 5 Python tools, 3 reference guides, 43 built-in recipes, 10 persona bundles.", "author": { "name": "Alireza Rezvani", @@ -8,6 +8,8 @@ }, "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"], + "skills": [ + "./skills" + ], "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering-team/google-workspace-cli" } diff --git a/engineering-team/playwright-pro/.claude-plugin/plugin.json b/engineering-team/playwright-pro/.claude-plugin/plugin.json index f6e65008..f3943044 100644 --- a/engineering-team/playwright-pro/.claude-plugin/plugin.json +++ b/engineering-team/playwright-pro/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "pw", "description": "Production-grade Playwright testing toolkit. Generate tests from specs, fix flaky failures, migrate from Cypress/Selenium, sync with TestRail, run on BrowserStack. 55+ ready-to-use templates, 3 specialized agents, smart reporting that plugs into your existing workflow.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering-team/playwright-pro", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering-team/self-improving-agent/.claude-plugin/plugin.json b/engineering-team/self-improving-agent/.claude-plugin/plugin.json index 4b2256df..b64905a1 100644 --- a/engineering-team/self-improving-agent/.claude-plugin/plugin.json +++ b/engineering-team/self-improving-agent/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "si", - "version": "2.3.1", + "version": "2.9.0", "description": "Self-Improving Agent: curate auto-memory, promote learnings to CLAUDE.md and rules, extract proven patterns into reusable skills. Provides /si:review, /si:promote, /si:extract, /si:status, and /si:remember slash commands.", "author": { "name": "Alireza Rezvani", @@ -8,6 +8,8 @@ }, "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"], + "skills": [ + "./skills" + ], "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering-team/self-improving-agent" } diff --git a/engineering-team/skills/adversarial-reviewer/SKILL.md b/engineering-team/skills/adversarial-reviewer/SKILL.md index e8317fd0..d0d6e65e 100644 --- a/engineering-team/skills/adversarial-reviewer/SKILL.md +++ b/engineering-team/skills/adversarial-reviewer/SKILL.md @@ -5,7 +5,7 @@ tier: "STANDARD" category: "Engineering / Code Quality" dependencies: "None (prompt-only, no external tools required)" author: "ekreloff" -version: "1.0.0" +version: "2.9.0" license: "MIT" --- diff --git a/engineering-team/skills/code-reviewer/README.md b/engineering-team/skills/code-reviewer/README.md index 5079165e..e455021c 100644 --- a/engineering-team/skills/code-reviewer/README.md +++ b/engineering-team/skills/code-reviewer/README.md @@ -1,6 +1,6 @@ # code-reviewer -Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, and .NET. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, and generates review reports. +Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, .NET, and Java. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, and generates review reports. The full skill spec is [`SKILL.md`](./SKILL.md). This README is a quick reference for the 3 bundled scripts. @@ -55,16 +55,17 @@ Outputs: review verdict (approve / request changes / block), score, prioritized | File | Purpose | |------|---------| -| [`assets/sample_csharp_smells.cs`](./assets/sample_csharp_smells.cs) | C# file with every pattern this skill detects, labelled inline | -| [`assets/sample_csharp_clean.cs`](./assets/sample_csharp_clean.cs) | Same code refactored per the standards in `references/` | -| [`expected_outputs/sample_csharp_smells_quality.json`](./expected_outputs/sample_csharp_smells_quality.json) | Expected `code_quality_checker.py --json` output for the smells fixture | -| [`expected_outputs/sample_csharp_clean_quality.json`](./expected_outputs/sample_csharp_clean_quality.json) | Expected output for the clean fixture | +| [`assets/sample_csharp_smells.cs`](./assets/sample_csharp_smells.cs) | C# file with every C#-specific pattern this skill detects, labelled inline | +| [`assets/sample_csharp_clean.cs`](./assets/sample_csharp_clean.cs) | Same code refactored per `rules/universal.md` + `languages/csharp.md` | +| [`assets/sample_java_smells.java`](./assets/sample_java_smells.java) | Java file with every Java-specific pattern this skill detects, labelled inline | +| [`assets/sample_java_clean.java`](./assets/sample_java_clean.java) | Same code refactored per `rules/universal.md` + `languages/java.md` | +| [`expected_outputs/*.json`](./expected_outputs/) | Expected `code_quality_checker.py --json` output for each fixture | Use them as a regression-detection harness: ```bash -python scripts/code_quality_checker.py assets/sample_csharp_smells.cs --json > /tmp/check.json -diff /tmp/check.json expected_outputs/sample_csharp_smells_quality.json +python scripts/code_quality_checker.py assets/sample_java_smells.java --json > /tmp/check.json +diff /tmp/check.json expected_outputs/sample_java_smells_quality.json # silence means the detector still behaves as documented ``` @@ -74,16 +75,16 @@ diff /tmp/check.json expected_outputs/sample_csharp_smells_quality.json See [`SKILL.md`](./SKILL.md) for the full pattern list, severity tiers, and references. Quick summary: -- **PR Analyzer** (`scripts/pr_analyzer.py`): hardcoded secrets / connection strings, SQL injection, debug statements, ESLint / Roslyn analyzer suppressions, `any` / `dynamic` overuse, TODO/FIXME, `unsafe` blocks, null-forgiving `!`, `async void`, blocking on `Task`. -- **Code Quality Checker** (`scripts/code_quality_checker.py`): long methods, large files, god classes, deep nesting, too many parameters, high cyclomatic complexity, swallowed exceptions, missing `await`, undisposed `IDisposable`, `new HttpClient()` in method body, unused `using` directives. +- **PR Analyzer** (`scripts/pr_analyzer.py`): hardcoded secrets / connection strings, SQL injection, debug statements (`console.*` / `System.out` / `printStackTrace`), analyzer suppressions (ESLint / Roslyn / `@SuppressWarnings`), `any` / `dynamic` overuse, TODO/FIXME, `unsafe` blocks, null-forgiving `!`, `async void`, blocking on `Task`. +- **Code Quality Checker** (`scripts/code_quality_checker.py`): long methods, large files, god classes, deep nesting, too many parameters, high cyclomatic complexity, swallowed exceptions, missing `await`, undisposed `IDisposable`, `new HttpClient()` in method body, unused `using` directives. Language-specific smell packs for C# and Java (e.g. empty catch, `printStackTrace`, swallowed `InterruptedException`, unclosed resources, per-call `ObjectMapper` / `Gson`). - **Review Report Generator** (`scripts/review_report_generator.py`): combines the above into a single markdown or JSON verdict. --- -## References +## Review rules -In-depth language guides and antipattern catalog live in [`references/`](./references/): +Rules are split so every review loads exactly two files — the cross-language +baseline plus one language guide (see the dispatch table in [`SKILL.md`](./SKILL.md)): -- [`coding_standards.md`](./references/coding_standards.md) — standards for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C# / .NET (nullable reference types, async/await + `ConfigureAwait`, IDisposable, LINQ, DI lifetimes, records + pattern matching, ASP.NET Core security) -- [`common_antipatterns.md`](./references/common_antipatterns.md) — antipattern catalog with examples + fixes, including a full C# / .NET section (`async void`, blocking on async, swallowing `Exception`, undisposed `IDisposable`, `new HttpClient()` in method, missing `ConfigureAwait`, mutable public setters, `dynamic` overuse, unjustified analyzer suppression) -- [`code_review_checklist.md`](./references/code_review_checklist.md) — systematic review checklist +- [`rules/universal.md`](./rules/universal.md) — cross-language rules: security, async/concurrency, resource management, exception handling, performance +- [`languages/`](./languages/) — one self-contained guide per language (`python`, `typescript`, `go`, `swift`, `kotlin`, `csharp`, `java`), each with Security / Async / Resource Management / Exception Handling / Performance / Idioms sections diff --git a/engineering-team/skills/code-reviewer/SKILL.md b/engineering-team/skills/code-reviewer/SKILL.md index 3d3f1403..14f7e83c 100644 --- a/engineering-team/skills/code-reviewer/SKILL.md +++ b/engineering-team/skills/code-reviewer/SKILL.md @@ -1,6 +1,6 @@ --- name: "code-reviewer" -description: Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, and .NET. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists. +description: Code review automation for TypeScript, JavaScript, Python, Go, Swift, Kotlin, C#, .NET, and Java. Analyzes PRs for complexity and risk, checks code quality for SOLID violations and code smells, generates review reports. Use when reviewing pull requests, analyzing code quality, identifying issues, generating review checklists. --- # Code Reviewer @@ -9,16 +9,40 @@ Automated code review tools for analyzing pull requests, detecting code quality --- -## Table of Contents +## How This Skill Is Organized -- [Tools](#tools) - - [PR Analyzer](#pr-analyzer) - - [Code Quality Checker](#code-quality-checker) - - [Review Report Generator](#review-report-generator) -- [Reference Guides](#reference-guides) -- [C# / .NET Review Notes](#c--net-review-notes) -- [Examples](#examples) -- [Languages Supported](#languages-supported) +``` +code-reviewer/ + SKILL.md ← you are here (tools + dispatch table) + rules/ + universal.md ← security, async, resources, exceptions, performance — all languages + languages/ + python.md ← Python-specific rules + idioms + typescript.md ← TypeScript / JavaScript-specific rules + idioms + go.md ← Go-specific rules + idioms + swift.md ← Swift-specific rules + idioms + kotlin.md ← Kotlin-specific rules + idioms + csharp.md ← C# / .NET-specific rules + idioms + java.md ← Java-specific rules + idioms +``` + +### Loading order for every review + +1. This file (`SKILL.md`) — tools and thresholds +2. `rules/universal.md` — always, for every language +3. The matching `languages/*.md` — one file based on the extension table below + +That's always exactly **2 additional files**, regardless of scope. + +| Extension(s) | Load | +|---|---| +| `.py` | `languages/python.md` | +| `.ts`, `.tsx`, `.js`, `.jsx`, `.mjs` | `languages/typescript.md` | +| `.go` | `languages/go.md` | +| `.swift` | `languages/swift.md` | +| `.kt`, `.kts` | `languages/kotlin.md` | +| `.cs`, `.csx`, `.razor`, `.cshtml` | `languages/csharp.md` | +| `.java` | `languages/java.md` | --- @@ -39,15 +63,12 @@ python scripts/pr_analyzer.py . --base main --head feature-branch python scripts/pr_analyzer.py /path/to/repo --json ``` -**What it detects:** +**What it detects (universal — see also language file for language-specific signals):** - Hardcoded secrets (passwords, API keys, tokens, connection strings) -- SQL injection patterns (string concatenation in queries) -- Debug statements (debugger, console.log, Debug.WriteLine) -- ESLint / Roslyn analyzer rule disabling (`#pragma warning disable`, `[SuppressMessage]`) -- TypeScript `any` types / C# `dynamic` overuse +- SQL / query injection patterns +- Debug statements left in production code +- Lint / analyzer suppression annotations - TODO/FIXME comments -- Unsafe code blocks (`unsafe { }` in C#) -- Nullable reference type suppressions (`!` null-forgiving operator overuse) **Output includes:** - Complexity score (1-10) @@ -65,26 +86,14 @@ Analyzes source code for structural issues, code smells, and SOLID violations. # Analyze a directory python scripts/code_quality_checker.py /path/to/code -# Analyze specific language (valid values: python, typescript, javascript, go, swift, kotlin, csharp) -python scripts/code_quality_checker.py . --language python +# Analyze specific language (valid values: python, typescript, javascript, go, swift, kotlin, csharp, java) +python scripts/code_quality_checker.py . --language java # JSON output python scripts/code_quality_checker.py /path/to/code --json ``` -**What it detects:** -- Long functions/methods (>50 lines) -- Large files (>500 lines) -- God classes (>20 methods) -- Deep nesting (>4 levels) -- Too many parameters (>5) -- High cyclomatic complexity -- Missing error handling (bare `catch` / `catch (Exception)` swallowing) -- Unused imports / unnecessary `using` directives -- Magic numbers -- C#-specific: missing `async`/`await` on async paths, `Task` not awaited, `IDisposable` not disposed - -**Thresholds:** +**Universal thresholds:** | Issue | Threshold | |-------|-----------| @@ -95,6 +104,8 @@ python scripts/code_quality_checker.py /path/to/code --json | Deep nesting | >4 levels | | High complexity | >10 branches | +Language-specific checks are defined in each `languages/*.md` file. + --- ### Review Report Generator @@ -114,13 +125,6 @@ python scripts/review_report_generator.py . \ --quality-analysis quality_results.json ``` -**Report includes:** -- Review verdict (approve, request changes, block) -- Score (0-100) -- Prioritized action items -- Issue summary by severity -- Suggested review order - **Verdicts:** | Score | Verdict | @@ -132,112 +136,33 @@ python scripts/review_report_generator.py . \ --- -## Reference Guides +## Adding a New Language -### Code Review Checklist -`references/code_review_checklist.md` +**Reviewer guidance (required):** -Systematic checklists covering: -- Pre-review checks (build, tests, PR hygiene) -- Correctness (logic, data handling, error handling) -- Security (input validation, injection prevention) -- Performance (efficiency, caching, scalability) -- Maintainability (code quality, naming, structure) -- Testing (coverage, quality, mocking) -- Language-specific checks (including C# / .NET) +1. Create `languages/.md` using any existing language file as a template — it must have sections: PR Analyzer Signals, Code Quality Checks, Security, Async, Resource Management, Exception Handling, Performance, Idioms. +2. Add the extension row to the dispatch table above. -### Coding Standards -`references/coding_standards.md` +That is all the agent-driven review needs. -Language-specific standards for: -- TypeScript (type annotations, null safety, async/await) -- JavaScript (declarations, patterns, modules) -- Python (type hints, exceptions, class design) -- Go (error handling, structs, concurrency) -- Swift (optionals, protocols, errors) -- Kotlin (null safety, data classes, coroutines) -- **C# / .NET** (nullable reference types, async/await, LINQ, dependency injection, exception handling, record types, pattern matching) +**Deterministic analyzer support (optional, recommended):** the bundled scripts +only flag a language they explicitly know. To make `code_quality_checker.py` +score the new language: -### Common Antipatterns -`references/common_antipatterns.md` - -Antipattern catalog with examples and fixes: -- Structural (god class, long method, deep nesting) -- Logic (boolean blindness, stringly typed code) -- Security (SQL injection, hardcoded credentials, unvalidated input in ASP.NET) -- Performance (N+1 queries, unbounded collections, `async void`, blocking on async code with `.Result` / `.Wait()`) -- Testing (duplication, testing implementation) -- Async (floating promises, callback hell, `async void` in C#, deadlocks from `.GetAwaiter().GetResult()`) -- **C# / .NET-specific**: catching and swallowing `Exception`, missing `ConfigureAwait`, overuse of `dynamic`, not disposing `IDisposable` resources, mutable public setters on domain models +3. Add the extensions to `LANGUAGE_EXTENSIONS` in `scripts/code_quality_checker.py` (this also adds the `--language` choice). +4. Add `function` / `class` / `method` regex entries for the language in the same file; otherwise it falls back to the Python patterns. +5. Optionally add a `check__specific_smells(...)` detector (see the C# and Java ones) and call it from `analyze_file`. +6. Add `assets/sample__smells.` + `_clean` fixtures and commit the expected `--json` output under `expected_outputs/` as a regression guard. --- -## C# / .NET Review Notes +## Regression Fixtures -When reviewing C# or .NET code, pay special attention to: - -### Async / Await -- Flag `async void` methods (except event handlers) — they can't be awaited and swallow exceptions -- Flag `.Result`, `.Wait()`, or `.GetAwaiter().GetResult()` on `Task` — causes deadlocks in ASP.NET contexts -- Flag missing `ConfigureAwait(false)` in library code - -### Nullable Reference Types -- Flag excessive use of the null-forgiving operator (`!`) without justification -- Ensure nullable annotations are enabled at the project level (`enable`) -- Flag unchecked dereferences of potentially null values - -### Resource Management -- Flag `IDisposable` objects not wrapped in `using` / `using var` -- Flag `HttpClient` instantiated with `new` inside methods (should be injected or use `IHttpClientFactory`) -- Flag `DbContext` not scoped correctly in DI - -### Exception Handling -- Flag bare `catch { }` or `catch (Exception) { }` that swallows exceptions silently -- Flag catching `Exception` when a more specific type is appropriate -- Flag exceptions used for control flow - -### LINQ -- Flag `.ToList()` / `.ToArray()` called prematurely on queryables, forcing unnecessary DB round-trips -- Flag `First()` where `FirstOrDefault()` is safer -- Flag complex LINQ chains that would be clearer as explicit loops - -### Security (ASP.NET) -- Flag raw string interpolation in SQL queries — require parameterized queries or EF Core -- Flag missing `[ValidateAntiForgeryToken]` on state-changing controller actions -- Flag user-controlled data passed to `Process.Start()` or `File` APIs without validation -- Flag hardcoded connection strings in source (should use `appsettings.json` + secrets management) - ---- - -## Examples - -Sample fixtures live in `assets/` with their expected analyzer output in `expected_outputs/`: - -| Fixture | What it demonstrates | Expected verdict | -|---------|---------------------|------------------| -| `assets/sample_csharp_smells.cs` | Every C#-specific pattern this skill detects (`async void`, blocking on `Task`, swallowed `Exception`, undisposed `IDisposable`, `new HttpClient()`, missing `await`, null-forgiving `!`, hardcoded connection string, `unsafe`, `dynamic`, `#pragma warning disable`, `[SuppressMessage]`, SQL concatenation) | F / 45 / 100, 3 HIGH smells | -| `assets/sample_csharp_clean.cs` | Same code refactored per `references/coding_standards.md` and `references/common_antipatterns.md` | A / 98 / 100, 0 HIGH smells | - -Reproduce the expected output: +Labelled fixtures live in `assets/` with their committed `--json` output in +`expected_outputs/` (C# and Java). Drift from the committed JSON signals a +behaviour change in the analyzer: ```bash -python scripts/code_quality_checker.py assets/sample_csharp_smells.cs --json \ - > /tmp/check.json -diff /tmp/check.json expected_outputs/sample_csharp_smells_quality.json +python scripts/code_quality_checker.py assets/sample_java_smells.java --json \ + | diff - expected_outputs/sample_java_smells_quality.json ``` - -The expected-output JSON is regenerated after any analyzer change; drift from the committed fixture signals a behaviour change in the detector. - ---- - -## Languages Supported - -| Language | Extensions | -|----------|------------| -| Python | `.py` | -| TypeScript | `.ts`, `.tsx` | -| JavaScript | `.js`, `.jsx`, `.mjs` | -| Go | `.go` | -| Swift | `.swift` | -| Kotlin | `.kt`, `.kts` | -| **C# / .NET** | **`.cs`, `.csx`, `.razor`, `.cshtml`** | diff --git a/engineering-team/skills/code-reviewer/assets/sample_csharp_clean.cs b/engineering-team/skills/code-reviewer/assets/sample_csharp_clean.cs index ebd42256..be704ea9 100644 --- a/engineering-team/skills/code-reviewer/assets/sample_csharp_clean.cs +++ b/engineering-team/skills/code-reviewer/assets/sample_csharp_clean.cs @@ -1,6 +1,6 @@ // Sample C# file showing the fixed version of sample_csharp_smells.cs. // Same shape, but every smell has been resolved per the patterns documented -// in references/coding_standards.md and references/common_antipatterns.md. +// in rules/universal.md and languages/csharp.md. // // Run: // python scripts/code_quality_checker.py assets/sample_csharp_clean.cs diff --git a/engineering-team/skills/code-reviewer/assets/sample_java_clean.java b/engineering-team/skills/code-reviewer/assets/sample_java_clean.java new file mode 100644 index 00000000..be90cf9f --- /dev/null +++ b/engineering-team/skills/code-reviewer/assets/sample_java_clean.java @@ -0,0 +1,55 @@ +// Sample Java file showing the fixed version of sample_java_smells.java. +// Same shape, but every smell has been resolved per the patterns documented +// in rules/universal.md and languages/java.md. +// +// Run: +// python scripts/code_quality_checker.py assets/sample_java_clean.java +// +// Expected: no HIGH Java-specific smells flagged. + +package sample; + +import java.io.FileInputStream; +import java.io.InputStream; +import java.sql.Connection; +import java.sql.PreparedStatement; +import java.sql.ResultSet; +import com.fasterxml.jackson.databind.ObjectMapper; + +public class UserService { + + // FIX: heavy object shared as a singleton instead of constructed per call. + private static final ObjectMapper MAPPER = new ObjectMapper(); + + // FIX: connection string injected from configuration, never inlined. + private final String connectionString; + + public UserService(String connectionString) { + this.connectionString = connectionString; + } + + public String getName(Connection conn, int id) { + // FIX: try-with-resources guarantees the stream and statement close. + try (InputStream config = new FileInputStream("/etc/config"); + // FIX: parameterized query, no string concatenation. + PreparedStatement stmt = + conn.prepareStatement("SELECT name FROM users WHERE id = ?")) { + stmt.setInt(1, id); + try (ResultSet rs = stmt.executeQuery()) { + return rs.next() ? rs.getString("name") : null; + } + } catch (Exception e) { + // FIX: rethrow with context instead of swallowing. + throw new IllegalStateException("Failed to load user " + id, e); + } + } + + public void process() { + try { + Thread.sleep(1000); + } catch (InterruptedException e) { + // FIX: restore the interrupt flag so cancellation still propagates. + Thread.currentThread().interrupt(); + } + } +} diff --git a/engineering-team/skills/code-reviewer/assets/sample_java_smells.java b/engineering-team/skills/code-reviewer/assets/sample_java_smells.java new file mode 100644 index 00000000..40936fc3 --- /dev/null +++ b/engineering-team/skills/code-reviewer/assets/sample_java_smells.java @@ -0,0 +1,56 @@ +// Sample Java file demonstrating the Java-specific patterns the code-reviewer +// skill detects. Each smell is labelled inline. This file is NOT meant to +// compile cleanly — it is a fixture for code_quality_checker.py and +// pr_analyzer.py. +// +// Run: +// python scripts/code_quality_checker.py assets/sample_java_smells.java +// +// Expected output: see expected_outputs/sample_java_smells_quality.json + +package sample; + +import java.io.FileInputStream; +import java.sql.Connection; +import java.sql.Statement; +import com.fasterxml.jackson.databind.ObjectMapper; + +public class UserService { + + // [hardcoded_secrets] hardcoded JDBC URL with password + public String connectionString = "jdbc:postgresql://prod/app?user=app&password=hunter2"; + + // [analyzer_disable] @SuppressWarnings without justification + @SuppressWarnings("unchecked") + public String getName(Connection conn, int id) throws Exception { + // [java_unclosed_resource] FileInputStream not in try-with-resources + FileInputStream fis = new FileInputStream("/etc/config"); + + // [java_per_use_heavy_object] new ObjectMapper() constructed per call + ObjectMapper mapper = new ObjectMapper(); + + try { + Statement stmt = conn.createStatement(); + // [sql_concatenation] string concatenation builds SQL with user input + return stmt.executeQuery("SELECT name FROM users WHERE id = " + id).toString(); + } catch (Exception e) { + // [java_empty_catch] empty catch swallows the exception + } + return null; + } + + public void process() { + try { + Thread.sleep(1000); + } catch (InterruptedException e) { + // [java_swallowed_interrupt] interrupt flag not restored + // [console_log] printStackTrace used as error handling + e.printStackTrace(); + } + } + + public void log(String message) { + // [console_log] System.out.println left in production code + System.out.println(message); + } +} diff --git a/engineering-team/skills/code-reviewer/expected_outputs/sample_java_clean_quality.json b/engineering-team/skills/code-reviewer/expected_outputs/sample_java_clean_quality.json new file mode 100644 index 00000000..e7aad98e --- /dev/null +++ b/engineering-team/skills/code-reviewer/expected_outputs/sample_java_clean_quality.json @@ -0,0 +1,53 @@ +{ + "file": "/home/user/claude-skills/engineering-team/skills/code-reviewer/assets/sample_java_clean.java", + "language": "java", + "metrics": { + "lines": { + "total": 56, + "code": 33, + "blank": 9, + "comment": 14 + }, + "functions": 3, + "classes": 1, + "avg_complexity": 2.0 + }, + "quality_score": 100, + "grade": "A", + "smells": [ + { + "type": "magic_number", + "severity": "low", + "message": "Magic number 1000 should be a named constant", + "location": "line 49" + } + ], + "solid_violations": [], + "function_details": [ + { + "name": "UserService", + "parameters": 1, + "lines": 5, + "complexity": 1 + }, + { + "name": "getName", + "parameters": 2, + "lines": 17, + "complexity": 3 + }, + { + "name": "process", + "parameters": 0, + "lines": 10, + "complexity": 2 + } + ], + "class_details": [ + { + "name": "UserService", + "methods": 3, + "lines": 38 + } + ] +} diff --git a/engineering-team/skills/code-reviewer/expected_outputs/sample_java_smells_quality.json b/engineering-team/skills/code-reviewer/expected_outputs/sample_java_smells_quality.json new file mode 100644 index 00000000..40c45be9 --- /dev/null +++ b/engineering-team/skills/code-reviewer/expected_outputs/sample_java_smells_quality.json @@ -0,0 +1,83 @@ +{ + "file": "/home/user/claude-skills/engineering-team/skills/code-reviewer/assets/sample_java_smells.java", + "language": "java", + "metrics": { + "lines": { + "total": 57, + "code": 29, + "blank": 10, + "comment": 18 + }, + "functions": 3, + "classes": 1, + "avg_complexity": 2.0 + }, + "quality_score": 68, + "grade": "D", + "smells": [ + { + "type": "magic_number", + "severity": "low", + "message": "Magic number 1000 should be a named constant", + "location": "line 44" + }, + { + "type": "java_empty_catch", + "severity": "high", + "message": "Empty catch block swallows exceptions silently", + "location": "offset 684" + }, + { + "type": "java_print_stack_trace", + "severity": "medium", + "message": "'printStackTrace()' is not real error handling \u2014 log via a proper logger or rethrow with context", + "location": "offset 913" + }, + { + "type": "java_swallowed_interrupt", + "severity": "high", + "message": "InterruptedException caught without 'Thread.currentThread().interrupt()' \u2014 breaks cooperative cancellation", + "location": "offset 841" + }, + { + "type": "java_unclosed_resource", + "severity": "medium", + "message": "'FileInputStream' looks like an AutoCloseable but is not in a try-with-resources statement", + "location": "offset 366" + }, + { + "type": "java_per_use_heavy_object", + "severity": "medium", + "message": "'new ObjectMapper()' is expensive \u2014 share a singleton instance instead of constructing per call", + "location": "offset 451" + } + ], + "solid_violations": [], + "function_details": [ + { + "name": "getName", + "parameters": 2, + "lines": 18, + "complexity": 3 + }, + { + "name": "process", + "parameters": 0, + "lines": 11, + "complexity": 2 + }, + { + "name": "log", + "parameters": 1, + "lines": 6, + "complexity": 1 + } + ], + "class_details": [ + { + "name": "UserService", + "methods": 3, + "lines": 40 + } + ] +} diff --git a/engineering-team/skills/code-reviewer/languages/csharp.md b/engineering-team/skills/code-reviewer/languages/csharp.md new file mode 100644 index 00000000..47ab6195 --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/csharp.md @@ -0,0 +1,97 @@ +--- +language: csharp +extensions: [".cs", ".csx", ".razor", ".cshtml"] +--- + +# C# / .NET — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only C#-specific rules and idioms. + +--- + +## PR Analyzer — C# Risk Signals + +- `#pragma warning disable` and `[SuppressMessage]` — verify they are justified +- `unsafe { }` blocks — require explicit sign-off +- Null-forgiving operator (`!`) used broadly without justification +- `dynamic` used outside of interop scenarios +- Hardcoded connection strings in source files + +--- + +## Code Quality — C# Checks + +- `async void` methods (except event handlers) +- `Task` returned but not awaited +- `IDisposable` objects not in `using` / `using var` +- Bare `catch { }` or `catch (Exception e) { }` swallowing silently +- Nullable reference types feature disabled at project level + +--- + +## Security + +- Flag raw string interpolation in SQL queries — require parameterized queries (`SqlCommand`) or EF Core +- Flag missing `[ValidateAntiForgeryToken]` on state-changing controller actions +- Flag user-controlled data passed to `Process.Start()` or `File` APIs without validation +- Flag hardcoded connection strings — require `appsettings.json` + secrets management +- Flag `[AllowAnonymous]` on endpoints that should be protected + +--- + +## Async / Await + +- Flag `async void` methods outside of event handlers — cannot be awaited and swallow exceptions +- Flag `.Result`, `.Wait()`, or `.GetAwaiter().GetResult()` on `Task` — causes deadlocks in ASP.NET contexts +- Flag missing `ConfigureAwait(false)` in library (non-application) code +- Flag `Task.Run()` wrapping synchronous code inside ASP.NET request handlers unnecessarily +- Flag `CancellationToken` not threaded through to downstream async calls + +--- + +## Resource Management + +- Flag `IDisposable` objects (`SqlConnection`, `HttpClient`, `FileStream`, etc.) not wrapped in `using` / `using var` +- Flag `HttpClient` instantiated with `new` inside a method — use `IHttpClientFactory` or a shared static instance to avoid socket exhaustion +- Flag `DbContext` registered as a singleton in DI — it must be scoped +- Flag `MemoryStream` / `MemoryCache` growing unboundedly without eviction policy + +--- + +## Exception Handling + +- Flag `catch { }` or `catch (Exception) { }` with no logging or re-throw — silent swallow +- Flag `catch (Exception e) { throw e; }` — resets the stack trace; use `throw;` instead +- Flag catching `Exception` when a specific type (`IOException`, `HttpRequestException`) is appropriate +- Flag exception filters (`when`) used for side effects that suppress the exception +- Flag exceptions used for control flow in hot paths — use `Try*` pattern methods instead + +--- + +## Performance + +- Flag `.ToList()` / `.ToArray()` on `IQueryable` before filtering — forces all rows into memory; filter server-side first +- Flag `string` concatenation in loops — use `StringBuilder` +- Flag `Enumerable.Count()` on `IQueryable` when only an existence check is needed — use `Any()` +- Flag `await` in a loop where `Task.WhenAll()` would parallelize the work +- Flag synchronous file or network I/O in an `async` method — use the async overload + +--- + +## Idioms and Best Practices + +### Null Safety +- Ensure `enable` is set in the project file +- Flag excessive use of `!` (null-forgiving) without a comment explaining why +- Prefer `is null` / `is not null` over `== null` for null checks + +### LINQ +- Flag `First()` where `FirstOrDefault()` is safer +- Flag complex LINQ chains that would be clearer as explicit loops + +### Modern C# (10+) +- Prefer `record` types for immutable data carriers +- Prefer `switch` expressions over `switch` statements where a value is returned +- Prefer primary constructors (C# 12) for simple dependency injection +- Prefer file-scoped namespaces (`namespace Foo;`) over block-scoped +- Prefer `is` pattern matching over explicit casts diff --git a/engineering-team/skills/code-reviewer/languages/go.md b/engineering-team/skills/code-reviewer/languages/go.md new file mode 100644 index 00000000..57275b39 --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/go.md @@ -0,0 +1,92 @@ +--- +language: go +extensions: [".go"] +--- + +# Go — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only Go-specific rules and idioms. + +--- + +## PR Analyzer — Go Risk Signals + +- `fmt.Println` / `log.Println` debug statements left in production code +- `//nolint` comments — verify they are justified +- `unsafe` package imports — require explicit sign-off +- Hardcoded credentials or tokens in source + +--- + +## Code Quality — Go Checks + +- Errors returned but not checked (`_ = someFunc()`) +- `panic()` used outside of package initialization +- Goroutines started without a clear lifetime or cancellation path +- `interface{}` / `any` used where a concrete type or typed interface would work +- Missing context propagation (`context.Context` not threaded through call chains) + +--- + +## Security + +- Flag `database/sql` queries built with `fmt.Sprintf` — require `?` / `$N` placeholders +- Flag `os/exec` calls with user-controlled arguments without sanitization +- Flag `html/template` bypassed in favor of `text/template` for HTML output +- Flag `http.ListenAndServeTLS` with `InsecureSkipVerify: true` + +--- + +## Async / Concurrency + +- Flag goroutines started with no clear lifetime or cancellation path — always pass `context.Context` +- Flag goroutines that write to a channel with no receiver and no `select` default — causes a leak +- Flag `time.Sleep()` used inside a goroutine as a synchronization mechanism +- Flag `sync.WaitGroup.Add()` called inside the goroutine it tracks — race condition +- Flag `sync.Mutex` copied by value — must always be used as a pointer or embedded in a struct + +--- + +## Resource Management + +- Flag `http.Response.Body` not closed after reading — even on error paths (`defer resp.Body.Close()`) +- Flag `os.File` not closed — use `defer f.Close()` immediately after opening +- Flag `rows.Close()` missing after `sql.Query()` — leaks the DB connection +- Flag `context.WithCancel` / `context.WithTimeout` cancel function not called — context and resources leak + +--- + +## Exception Handling + +- Flag errors assigned to `_` without a comment explaining why it is safe to ignore +- Flag errors not wrapped with `fmt.Errorf("...: %w", err)` — loses stack context +- Flag `errors.New` / `fmt.Errorf` strings starting with a capital letter or ending in punctuation — violates Go conventions +- Flag `panic()` used for expected runtime errors — reserve for programming errors and unrecoverable states +- Flag `recover()` used to silently swallow panics without logging + +--- + +## Performance + +- Flag `fmt.Sprintf` used for simple string concatenation — use `strings.Builder` or `+` for small cases +- Flag `append()` in a tight loop without pre-allocating slice capacity — use `make([]T, 0, n)` +- Flag `json.Marshal` / `json.Unmarshal` on large structs in hot paths — consider `json.Encoder` / streaming +- Flag goroutines spawned per-request without a worker pool for CPU-bound tasks + +--- + +## Idioms and Best Practices + +### Error Handling +- All returned errors must be checked — never assign to `_` without a comment +- Prefer wrapping with `fmt.Errorf("...: %w", err)` for stack context +- Use `errors.Is` / `errors.As` for error inspection — never string comparison + +### Concurrency +- Every goroutine must have an owner responsible for its lifetime +- Always pass `context.Context` as the first argument to functions that do I/O or block +- Prefer `sync.WaitGroup` or `errgroup` over ad-hoc channel coordination + +### Modern Go (1.18+) +- Prefer generics over `interface{}` for container types and utility functions +- Use `any` (alias for `interface{}`) in new code for readability diff --git a/engineering-team/skills/code-reviewer/languages/java.md b/engineering-team/skills/code-reviewer/languages/java.md new file mode 100644 index 00000000..b66430da --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/java.md @@ -0,0 +1,103 @@ +--- +language: java +extensions: [".java"] +--- + +# Java — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only Java-specific rules and idioms. + +--- + +## PR Analyzer — Java Risk Signals + +- `System.out.println` / `e.printStackTrace()` left in production code +- `@SuppressWarnings` annotations — verify they are justified +- Hardcoded JDBC URLs or credentials in source +- Raw type usage (`List`, `Map` without generics) + +--- + +## Code Quality — Java Checks + +- Empty `catch` blocks swallowing exceptions silently +- Checked exceptions caught and not re-thrown with context +- `Closeable` / `AutoCloseable` resources not in try-with-resources +- Raw type usage — defeats generics type safety +- Missing `@Override` on overriding methods +- `InterruptedException` caught without calling `Thread.currentThread().interrupt()` + +--- + +## Security + +- Flag JPQL / HQL or native SQL string concatenation — require named parameters or `CriteriaBuilder` +- Flag `@RequestMapping` without explicit HTTP method restriction on state-changing endpoints +- Flag user-controlled input passed to `Runtime.exec()` or `ProcessBuilder` without validation +- Flag `ObjectInputStream.readObject()` on untrusted data — unsafe deserialization +- Flag hardcoded JDBC URLs or credentials — require environment variables or a vault + +--- + +## Async / Concurrency + +- Flag `ExecutorService.submit()` return value ignored — exceptions are swallowed +- Flag `Thread.sleep()` used as a synchronization mechanism — use `CountDownLatch`, `CompletableFuture`, or `await()` +- Flag `CompletableFuture` chains with no `.exceptionally()` or `.handle()` terminal handler +- Flag `InterruptedException` caught without calling `Thread.currentThread().interrupt()` +- Flag `synchronized` on a non-final field — the lock object can be replaced +- Flag `HashMap` used in multi-threaded context — use `ConcurrentHashMap` + +--- + +## Resource Management + +- Flag `InputStream`, `OutputStream`, `Connection`, `ResultSet`, `PreparedStatement` not wrapped in try-with-resources +- Flag manual `finally { resource.close() }` — replace with try-with-resources +- Flag `HttpURLConnection` not disconnected after use +- Flag JDBC `Connection` obtained from a pool and not returned (missing `close()`) on all paths +- Flag `static` `HttpClient` or `Connection` fields shared across threads without connection pool management + +--- + +## Exception Handling + +- Flag empty `catch` blocks — `catch (Exception e) {}` +- Flag `InterruptedException` caught without `Thread.currentThread().interrupt()` — breaks cooperative cancellation +- Flag checked exceptions swallowed in a `catch` and not re-thrown or logged with context +- Flag `throw new RuntimeException(e)` without a descriptive message — loses context +- Flag `printStackTrace()` as the sole error handling — use a proper logger + +--- + +## Performance + +- Flag `String` concatenation in loops — use `StringBuilder` +- Flag `List.contains()` / `Map.get()` in a loop on large collections — review data structure choice +- Flag N+1 JPA / Hibernate queries — use `JOIN FETCH` or `@BatchSize` +- Flag `new ObjectMapper()` / `new Gson()` instantiated per-request — share a singleton +- Flag `ResultSet` fully iterated when only the first result is needed — use `LIMIT 1` in the query + +--- + +## Idioms and Best Practices + +### Null Safety +- Prefer returning `Optional` over `null` from methods +- Flag unchecked dereferences without a prior null guard +- Do not catch `NullPointerException` — fix the root cause instead + +### Collections and Streams +- Flag `==` used to compare `String` or boxed types — use `.equals()` +- Flag `.collect(Collectors.toList())` where `.toList()` (Java 16+) suffices +- Flag premature `.stream().collect()` round-trips that could be a single-pass operation + +### Generics +- Flag raw types in any new code — always parameterize (`List`, not `List`) +- Flag unchecked cast warnings suppressed without explanation + +### Modern Java (11+) +- Prefer `var` for local variables where the type is obvious from the right-hand side +- Prefer records for pure data carriers over manual POJOs with getters/setters +- Prefer `instanceof` pattern matching (`if (obj instanceof String s)`) over explicit casts +- Prefer `switch` expressions over `switch` statements where a value is returned diff --git a/engineering-team/skills/code-reviewer/languages/kotlin.md b/engineering-team/skills/code-reviewer/languages/kotlin.md new file mode 100644 index 00000000..c0cefe53 --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/kotlin.md @@ -0,0 +1,88 @@ +--- +language: kotlin +extensions: [".kt", ".kts"] +--- + +# Kotlin — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only Kotlin-specific rules and idioms. + +--- + +## PR Analyzer — Kotlin Risk Signals + +- `println()` statements left in production code +- `@Suppress` annotations — verify they are justified +- `!!` (not-null assertion) used broadly without justification +- Hardcoded credentials or API keys in source + +--- + +## Code Quality — Kotlin Checks + +- `!!` used broadly — prefer `?.let`, `?:`, or `requireNotNull()` +- `lateinit var` accessed before initialization +- Coroutines launched with `GlobalScope` — prefer scoped coroutines +- `runBlocking` used outside of tests or top-level entry points + +--- + +## Security + +- Flag Room / SQLite queries built with string concatenation — require parameterized queries +- Flag `WebView.loadUrl()` with user-controlled input without validation +- Flag credentials stored in `SharedPreferences` — require `EncryptedSharedPreferences` or Keychain + +--- + +## Async / Coroutines + +- Flag `GlobalScope.launch` / `GlobalScope.async` in production code — use a structured scope +- Flag `runBlocking` outside of tests or top-level main functions +- Flag `launch` / `async` without a `CoroutineExceptionHandler` or `supervisorScope` where individual failures should not cancel siblings +- Flag `Dispatchers.Main` used for CPU-bound work — use `Dispatchers.Default` +- Flag coroutine cancellation not respected — long loops should check `isActive` or call `yield()` + +--- + +## Resource Management + +- Flag `Closeable` / `AutoCloseable` not wrapped in `.use { }` (Kotlin's try-with-resources equivalent) +- Flag `OkHttpClient` / `Retrofit` instantiated per-request — share a singleton +- Flag `BroadcastReceiver` registered without a corresponding `unregisterReceiver` — memory / battery leak +- Flag coroutines that hold a resource across a `suspend` point without structured cleanup in `finally` + +--- + +## Exception Handling + +- Flag `runCatching { }.getOrNull()` used broadly — silently swallows all exceptions +- Flag `catch (e: Exception)` in coroutines without re-throwing `CancellationException` — breaks structured concurrency +- Flag empty `catch` blocks +- Flag `throw RuntimeException(e)` without a descriptive message +- Prefer typed `sealed class` error hierarchies over raw exceptions for domain errors in coroutine flows + +--- + +## Performance + +- Flag `buildString` / `StringBuilder` not used for multi-step string construction in loops +- Flag `List` used for frequent `contains` checks — prefer `Set` +- Flag `flow.collect {}` re-subscribing on every recomposition in Jetpack Compose — use `collectAsStateWithLifecycle` +- Flag `Dispatchers.IO` used for CPU-bound work — use `Dispatchers.Default` +- Flag `suspend` functions calling non-suspend blocking APIs directly — wrap with `withContext(Dispatchers.IO)` + +--- + +## Idioms and Best Practices + +### Null Safety +- Prefer safe call (`?.`) and Elvis operator (`?:`) over `!!` +- Use `requireNotNull()` / `checkNotNull()` with a descriptive message when null means a programming error +- Prefer `val` over `var` — immutability by default + +### Modern Kotlin +- Prefer `data class` for value carriers +- Prefer `sealed class` / `sealed interface` for exhaustive `when` expressions +- Prefer extension functions over utility classes +- Prefer `object` declarations for singletons diff --git a/engineering-team/skills/code-reviewer/languages/python.md b/engineering-team/skills/code-reviewer/languages/python.md new file mode 100644 index 00000000..5186632a --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/python.md @@ -0,0 +1,94 @@ +--- +language: python +extensions: [".py"] +--- + +# Python — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only Python-specific rules and idioms. + +--- + +## PR Analyzer — Python Risk Signals + +- `print()` statements left in production code +- `# noqa` and `# type: ignore` comments — verify they are justified +- `eval()` / `exec()` with any user-controlled input +- `pickle` used to deserialize untrusted data +- Hardcoded credentials or tokens in source + +--- + +## Code Quality — Python Checks + +- Bare `except:` or `except Exception:` swallowing silently +- Mutable default arguments (`def foo(items=[])`) — shared across calls +- `import *` — pollutes namespace and hides dependencies +- Missing type hints on public functions and methods +- `assert` used for runtime validation — stripped by `-O` flag + +--- + +## Security + +- Flag `eval()` / `exec()` with any user-controlled input +- Flag `pickle.loads()` on untrusted data — use `json` or `msgpack` +- Flag `subprocess` calls with `shell=True` and user input +- Flag `flask.render_template_string()` with user data (SSTI) +- Flag `SECRET_KEY` / `DEBUG = True` committed to source + +--- + +## Async + +- Flag `asyncio.get_event_loop().run_until_complete()` inside an already-running loop +- Flag mixing `threading` and `asyncio` without a clear bridge (`run_in_executor`) +- Flag CPU-bound work inside an `async def` without offloading to `ProcessPoolExecutor` +- Flag `time.sleep()` inside async functions — use `await asyncio.sleep()` + +--- + +## Resource Management + +- Flag `open()` not used as a context manager (`with open(...) as f`) +- Flag `requests.Session` created per-request instead of shared/reused +- Flag database connections not closed or returned to a pool on all paths +- Flag large files read entirely into memory with `.read()` — prefer streaming / chunked reads + +--- + +## Exception Handling + +- Flag bare `except:` — catches `BaseException` including `KeyboardInterrupt` and `SystemExit` +- Flag `except Exception: pass` — silently swallows errors +- Flag re-raising with `raise e` instead of `raise` — loses the original traceback +- Flag `except` clause too broad when the `try` block covers multiple operations with different failure modes — split them + +--- + +## Performance + +- Flag `+` string concatenation in loops — use `"".join()` +- Flag repeated `re.compile()` inside a loop — compile once at module level +- Flag `list.append()` in a loop where a list comprehension would be more efficient +- Flag `in` membership tests on `list` where the collection is large — use `set` +- Flag loading entire large files into memory — prefer streaming or chunked reads + +--- + +## Idioms and Best Practices + +### Type Safety +- All public functions and methods should have type annotations +- Prefer `X | None` (Python 3.10+) over `Optional[X]` +- Use `TypedDict` or `dataclass` over plain `dict` for structured data + +### Modern Python (3.10+) +- Prefer `match` statements over long `if/elif` chains +- Prefer `dataclass` or `NamedTuple` over plain classes for data carriers +- Prefer `pathlib.Path` over `os.path` for file operations +- Prefer f-strings over `.format()` or `%` formatting + +### None Safety +- Prefer explicit `if x is None` over falsy checks when `0` or `""` are valid values +- Flag functions returning `None` implicitly — make it explicit or raise diff --git a/engineering-team/skills/code-reviewer/languages/swift.md b/engineering-team/skills/code-reviewer/languages/swift.md new file mode 100644 index 00000000..6f0e1c5e --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/swift.md @@ -0,0 +1,88 @@ +--- +language: swift +extensions: [".swift"] +--- + +# Swift — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only Swift-specific rules and idioms. + +--- + +## PR Analyzer — Swift Risk Signals + +- `print()` statements left in production code +- Force unwrap (`!`) on optionals outside of tests or justified init +- Force cast (`as!`) without a safe fallback +- Hardcoded credentials or API keys in source + +--- + +## Code Quality — Swift Checks + +- Force unwrap (`!`) used broadly — prefer `guard let` or `if let` +- `try!` used outside of guaranteed-safe contexts +- Retain cycles in closures — missing `[weak self]` or `[unowned self]` +- `@objc` / `dynamic` used without an Objective-C interop reason + +--- + +## Security + +- Flag credentials stored in `UserDefaults` — require Keychain +- Flag `URLSession` requests over plain HTTP in production +- Flag `WKWebView` loading arbitrary user-supplied URLs without validation + +--- + +## Async / Concurrency + +- Flag `DispatchQueue.main.sync` called from the main thread — deadlock +- Flag `@escaping` closures capturing `self` strongly in reference cycles — use `[weak self]` +- Flag mixing `async/await` and `DispatchQueue` for the same operation without clear reasoning +- Flag `Task { }` (unstructured) where a structured `async let` or `TaskGroup` would maintain structure +- Flag data races — shared mutable state accessed from multiple tasks without an actor + +--- + +## Resource Management + +- Flag `URLSessionDataTask` started with no cancellation handle stored — cannot be cancelled if the view disappears +- Flag `NotificationCenter` observers added without a corresponding `removeObserver` — memory leak +- Flag `CLLocationManager` / `AVCaptureSession` not stopped when the owning view controller is dismissed + +--- + +## Exception Handling + +- Flag `try!` outside of guaranteed-safe contexts (test fixtures, constants) — crashes on failure +- Flag `try?` discarding errors where the failure mode matters to the caller +- Flag error types conforming to `Error` with no associated values or message — makes debugging hard +- Flag throwing functions calling `fatalError()` as a fallback — choose one error strategy + +--- + +## Performance + +- Flag `UIImage(named:)` called repeatedly for the same asset without caching +- Flag synchronous network calls on the main thread +- Flag `Array` used for frequent membership tests — prefer `Set` +- Flag `String` interpolation inside tight loops where a pre-built string would avoid allocations + +--- + +## Idioms and Best Practices + +### Optionals +- Prefer `guard let` for early exit; `if let` for local scope +- Prefer optional chaining (`?.`) over force unwrap +- Flag implicitly unwrapped optionals (`var x: String!`) outside of `@IBOutlet` + +### Memory Management +- Flag closures capturing `self` strongly in reference cycles — use `[weak self]` +- Prefer `struct` over `class` for value semantics unless identity or inheritance is needed +- Use `unowned` only when the lifetime is guaranteed — otherwise `weak` + +### Concurrency (Swift 5.5+) +- Prefer `async/await` over completion handlers in new code +- Flag `DispatchQueue.main.async` where `@MainActor` or `await MainActor.run` is more appropriate diff --git a/engineering-team/skills/code-reviewer/languages/typescript.md b/engineering-team/skills/code-reviewer/languages/typescript.md new file mode 100644 index 00000000..e7bc3e88 --- /dev/null +++ b/engineering-team/skills/code-reviewer/languages/typescript.md @@ -0,0 +1,97 @@ +--- +language: typescript +extensions: [".ts", ".tsx", ".js", ".jsx", ".mjs"] +--- + +# TypeScript / JavaScript — Language-Specific Review Notes + +Load this file alongside `rules/universal.md`. Universal rules are not repeated here — only TypeScript/JavaScript-specific rules and idioms. + +--- + +## PR Analyzer — TypeScript / JavaScript Risk Signals + +- `console.log` / `debugger` statements left in production code +- `// eslint-disable` comments — verify they are justified +- `any` type annotations — require explicit justification +- `@ts-ignore` / `@ts-expect-error` — verify they are justified +- `eval()` with any dynamic or user-controlled input +- Hardcoded API keys or tokens in source + +--- + +## Code Quality — TypeScript / JavaScript Checks + +- `any` used broadly instead of proper typing +- Non-null assertion (`!`) used without justification +- `var` declarations — prefer `const` / `let` +- Missing `await` on async function calls +- Floating promises (no `.catch()` and no `await`) +- `==` used instead of `===` + +--- + +## Security + +- Flag `innerHTML`, `outerHTML`, `document.write()` with user-controlled data — use `textContent` or a sanitizer +- Flag `dangerouslySetInnerHTML` in React without a sanitizer +- Flag `eval()` / `new Function()` with dynamic input +- Flag JWT decoded without signature verification +- Flag missing `httpOnly` / `secure` flags on cookies + +--- + +## Async / Promises + +- Flag floating promises — async calls not `await`-ed and without `.catch()` +- Flag `Promise.all()` where `Promise.allSettled()` is safer (one failure should not cancel siblings) +- Flag `async` functions inside `forEach` — `forEach` does not await; use `for...of` or `Promise.all()` +- Flag unhandled promise rejection (no global `unhandledRejection` handler in Node.js services) + +--- + +## Resource Management + +- Flag `fs.createReadStream` / `fs.createWriteStream` with no `close` or `destroy` on error +- Flag `EventEmitter` listeners added in a loop without removal — memory leak +- Flag `setInterval` / `setTimeout` handles not cleared when the owning component unmounts or exits +- Flag database clients / pools not released after use in Node.js + +--- + +## Exception Handling + +- Flag `catch (e) {}` (empty catch) — swallowed error +- Flag `catch (e)` where `e` is used as `any` without narrowing — type the error properly +- Flag `Promise` rejection not handled — `.catch()` or `try/await/catch` required +- Flag re-throwing a new `Error` without wrapping the original — loses stack context +- Use `Error` subclasses for domain errors rather than plain strings or object literals + +--- + +## Performance + +- Flag `Array.prototype.find` / `filter` / `map` chained multiple times over the same array — combine into one pass +- Flag DOM queries (`document.querySelector`) inside loops — cache the result +- Flag `JSON.parse` / `JSON.stringify` in a hot path on large objects — consider streaming or partial parsing +- Flag `async` functions called sequentially in a loop where `Promise.all()` would parallelize them + +--- + +## Idioms and Best Practices + +### Type Safety (TypeScript) +- Prefer `unknown` over `any` for truly unknown values — forces a type guard before use +- Prefer type narrowing (`typeof`, `instanceof`, discriminated unions) over casting +- Enable `strict` mode in `tsconfig.json` +- Prefer `interface` for object shapes that may be extended; `type` for unions and aliases + +### Modern JavaScript / TypeScript +- Prefer `const` by default; `let` only when reassignment is needed +- Prefer optional chaining (`?.`) and nullish coalescing (`??`) over manual null guards +- Prefer `structuredClone()` over manual deep-copy patterns +- Prefer named exports over default exports for better refactoring support + +### Null / Undefined Safety +- Distinguish between `null` (intentional absence) and `undefined` (not set) — be consistent +- Flag `== null` checks that accidentally include `undefined` when only one is intended diff --git a/engineering-team/skills/code-reviewer/references/code_review_checklist.md b/engineering-team/skills/code-reviewer/references/code_review_checklist.md deleted file mode 100644 index b7bd0867..00000000 --- a/engineering-team/skills/code-reviewer/references/code_review_checklist.md +++ /dev/null @@ -1,270 +0,0 @@ -# Code Review Checklist - -Structured checklists for systematic code review across different aspects. - ---- - -## Table of Contents - -- [Pre-Review Checks](#pre-review-checks) -- [Correctness](#correctness) -- [Security](#security) -- [Performance](#performance) -- [Maintainability](#maintainability) -- [Testing](#testing) -- [Documentation](#documentation) -- [Language-Specific Checks](#language-specific-checks) - ---- - -## Pre-Review Checks - -Before diving into code, verify these basics: - -### Build and Tests -- [ ] Code compiles without errors -- [ ] All existing tests pass -- [ ] New tests are included for new functionality -- [ ] No unintended files included (build artifacts, IDE configs) - -### PR Hygiene -- [ ] PR has clear title and description -- [ ] Changes are scoped appropriately (not too large) -- [ ] Commits follow conventional commit format -- [ ] Branch is up to date with base branch - -### Scope Verification -- [ ] Changes match the stated purpose -- [ ] No unrelated changes bundled in -- [ ] Breaking changes are documented -- [ ] Migration path provided if needed - ---- - -## Correctness - -### Logic -- [ ] Algorithm implements requirements correctly -- [ ] Edge cases handled (null, empty, boundary values) -- [ ] Off-by-one errors checked -- [ ] Correct operators used (== vs ===, & vs &&) -- [ ] Loop termination conditions correct -- [ ] Recursion has proper base cases - -### Data Handling -- [ ] Data types appropriate for the use case -- [ ] Numeric overflow/underflow considered -- [ ] Date/time handling accounts for timezones -- [ ] Unicode and internationalization handled -- [ ] Data validation at entry points - -### State Management -- [ ] State transitions are valid -- [ ] Race conditions addressed -- [ ] Concurrent access handled correctly -- [ ] State cleanup on errors/exit - -### Error Handling -- [ ] Errors caught at appropriate levels -- [ ] Error messages are actionable -- [ ] Errors don't expose sensitive information -- [ ] Recovery or graceful degradation implemented -- [ ] Resources cleaned up in error paths - ---- - -## Security - -### Input Validation -- [ ] All user input validated and sanitized -- [ ] Input length limits enforced -- [ ] File uploads validated (type, size, content) -- [ ] URL parameters validated - -### Injection Prevention -- [ ] SQL queries parameterized -- [ ] Command execution uses safe APIs -- [ ] HTML output escaped to prevent XSS -- [ ] LDAP queries properly escaped -- [ ] XML parsing disables external entities - -### Authentication & Authorization -- [ ] Authentication required for protected resources -- [ ] Authorization checked before operations -- [ ] Session management secure -- [ ] Password handling follows best practices -- [ ] Token expiration implemented - -### Data Protection -- [ ] Sensitive data encrypted at rest -- [ ] Sensitive data encrypted in transit -- [ ] PII handled according to policy -- [ ] Secrets not hardcoded -- [ ] Logs don't contain sensitive data - -### API Security -- [ ] Rate limiting implemented -- [ ] CORS configured correctly -- [ ] CSRF protection in place -- [ ] API keys/tokens secured -- [ ] Endpoints use HTTPS - ---- - -## Performance - -### Efficiency -- [ ] Appropriate data structures used -- [ ] Algorithms have acceptable complexity -- [ ] Database queries are optimized -- [ ] N+1 query problems avoided -- [ ] Indexes used where beneficial - -### Resource Usage -- [ ] Memory usage bounded -- [ ] No memory leaks -- [ ] File handles properly closed -- [ ] Database connections pooled -- [ ] Network calls minimized - -### Caching -- [ ] Appropriate caching strategy -- [ ] Cache invalidation handled -- [ ] Cache keys are unique and predictable -- [ ] TTL values appropriate - -### Scalability -- [ ] Horizontal scaling considered -- [ ] Bottlenecks identified -- [ ] Async processing for long operations -- [ ] Batch operations where appropriate - ---- - -## Maintainability - -### Code Quality -- [ ] Functions/methods have single responsibility -- [ ] Classes follow SOLID principles -- [ ] Code is DRY (Don't Repeat Yourself) -- [ ] No dead code or commented-out code -- [ ] Magic numbers replaced with constants - -### Naming -- [ ] Names are descriptive and consistent -- [ ] Naming follows project conventions -- [ ] No abbreviations that obscure meaning -- [ ] Boolean variables/functions have is/has/can prefix - -### Structure -- [ ] Functions are appropriately sized (<50 lines preferred) -- [ ] Nesting depth is reasonable (<4 levels) -- [ ] Related code is grouped together -- [ ] Dependencies are minimal and explicit - -### Readability -- [ ] Code is self-documenting where possible -- [ ] Complex logic has explanatory comments -- [ ] Formatting is consistent -- [ ] No overly clever or obscure code - ---- - -## Testing - -### Coverage -- [ ] New code has unit tests -- [ ] Critical paths have integration tests -- [ ] Edge cases are tested -- [ ] Error conditions are tested - -### Quality -- [ ] Tests are independent -- [ ] Tests have clear assertions -- [ ] Test names describe what is tested -- [ ] Tests don't depend on external state - -### Mocking -- [ ] External dependencies are mocked -- [ ] Mocks are realistic -- [ ] Mock setup is not excessive - ---- - -## Documentation - -### Code Documentation -- [ ] Public APIs are documented -- [ ] Complex algorithms explained -- [ ] Non-obvious decisions documented -- [ ] TODO/FIXME comments have context - -### External Documentation -- [ ] README updated if needed -- [ ] API documentation updated -- [ ] Changelog updated -- [ ] Migration guides provided - ---- - -## Language-Specific Checks - -### TypeScript/JavaScript -- [ ] Types are explicit (avoid `any`) -- [ ] Null checks present (`?.`, `??`) -- [ ] Async/await errors handled -- [ ] No floating promises -- [ ] Memory leaks from closures checked - -### Python -- [ ] Type hints used for public APIs -- [ ] Context managers for resources (`with` statements) -- [ ] Exception handling is specific (not bare `except`) -- [ ] No mutable default arguments -- [ ] List comprehensions used appropriately - -### Go -- [ ] Errors checked and handled -- [ ] Goroutine leaks prevented -- [ ] Context propagation correct -- [ ] Defer statements in right order -- [ ] Interfaces minimal - -### Swift -- [ ] Optionals handled safely -- [ ] Memory management correct (weak/unowned) -- [ ] Error handling uses Result or throws -- [ ] Access control appropriate -- [ ] Codable implementation correct - -### Kotlin -- [ ] Null safety leveraged -- [ ] Coroutine cancellation handled -- [ ] Data classes used appropriately -- [ ] Extension functions don't obscure behavior -- [ ] Sealed classes for state - ---- - -## Review Process Tips - -### Before Approving -1. Verify all critical checks passed -2. Confirm tests are adequate -3. Consider deployment impact -4. Check for any security concerns -5. Ensure documentation is updated - -### Providing Feedback -- Be specific about issues -- Explain why something is problematic -- Suggest alternatives when possible -- Distinguish blockers from suggestions -- Acknowledge good patterns - -### When to Block -- Security vulnerabilities present -- Critical logic errors -- No tests for risky changes -- Breaking changes without migration -- Significant performance regressions diff --git a/engineering-team/skills/code-reviewer/references/coding_standards.md b/engineering-team/skills/code-reviewer/references/coding_standards.md deleted file mode 100644 index 43d73b8b..00000000 --- a/engineering-team/skills/code-reviewer/references/coding_standards.md +++ /dev/null @@ -1,793 +0,0 @@ -# Coding Standards - -Language-specific coding standards and conventions for code review. - ---- - -## Table of Contents - -- [Universal Principles](#universal-principles) -- [TypeScript Standards](#typescript-standards) -- [JavaScript Standards](#javascript-standards) -- [Python Standards](#python-standards) -- [Go Standards](#go-standards) -- [Swift Standards](#swift-standards) -- [Kotlin Standards](#kotlin-standards) -- [C# / .NET Standards](#c--net-standards) - ---- - -## Universal Principles - -These apply across all languages. - -### Naming Conventions - -| Element | Convention | Example | -|---------|------------|---------| -| Variables | camelCase (JS/TS), snake_case (Python/Go) | `userName`, `user_name` | -| Constants | SCREAMING_SNAKE_CASE | `MAX_RETRY_COUNT` | -| Functions | camelCase (JS/TS), snake_case (Python) | `getUserById`, `get_user_by_id` | -| Classes | PascalCase | `UserRepository` | -| Interfaces | PascalCase, optionally prefixed | `IUserService` or `UserService` | -| Private members | Prefix with underscore or use access modifiers | `_internalState` | - -### Function Design - -``` -Good functions: -- Do one thing well -- Have descriptive names (verb + noun) -- Take 3 or fewer parameters -- Return early for error cases -- Stay under 50 lines -``` - -### Error Handling - -``` -Good error handling: -- Catch specific errors, not generic exceptions -- Log with context (what, where, why) -- Clean up resources in error paths -- Don't swallow errors silently -- Provide actionable error messages -``` - ---- - -## TypeScript Standards - -### Type Annotations - -```typescript -// Avoid 'any' - use unknown for truly unknown types -function processData(data: unknown): ProcessedResult { - if (isValidData(data)) { - return transform(data); - } - throw new Error('Invalid data format'); -} - -// Use explicit return types for public APIs -export function calculateTotal(items: CartItem[]): number { - return items.reduce((sum, item) => sum + item.price, 0); -} - -// Use type guards for runtime checks -function isUser(obj: unknown): obj is User { - return ( - typeof obj === 'object' && - obj !== null && - 'id' in obj && - 'email' in obj - ); -} -``` - -### Null Safety - -```typescript -// Use optional chaining and nullish coalescing -const userName = user?.profile?.name ?? 'Anonymous'; - -// Be explicit about nullable types -interface Config { - timeout: number; - retries?: number; // Optional - fallbackUrl: string | null; // Explicitly nullable -} - -// Use assertion functions for validation -function assertDefined(value: T | null | undefined): asserts value is T { - if (value === null || value === undefined) { - throw new Error('Value is not defined'); - } -} -``` - -### Async/Await - -```typescript -// Always handle errors in async functions -async function fetchUser(id: string): Promise { - try { - const response = await api.get(`/users/${id}`); - return response.data; - } catch (error) { - logger.error('Failed to fetch user', { id, error }); - throw new UserFetchError(id, error); - } -} - -// Use Promise.all for parallel operations -async function loadDashboard(userId: string): Promise { - const [profile, stats, notifications] = await Promise.all([ - fetchProfile(userId), - fetchStats(userId), - fetchNotifications(userId) - ]); - return { profile, stats, notifications }; -} -``` - -### React/Component Standards - -```typescript -// Use explicit prop types -interface ButtonProps { - label: string; - onClick: () => void; - variant?: 'primary' | 'secondary'; - disabled?: boolean; -} - -// Prefer functional components with hooks -function Button({ label, onClick, variant = 'primary', disabled = false }: ButtonProps) { - return ( - - ); -} - -// Use custom hooks for reusable logic -function useDebounce(value: T, delay: number): T { - const [debouncedValue, setDebouncedValue] = useState(value); - - useEffect(() => { - const timer = setTimeout(() => setDebouncedValue(value), delay); - return () => clearTimeout(timer); - }, [value, delay]); - - return debouncedValue; -} -``` - ---- - -## JavaScript Standards - -### Variable Declarations - -```javascript -// Use const by default, let when reassignment needed -const MAX_ITEMS = 100; -let currentCount = 0; - -// Never use var -// var is function-scoped and hoisted, leading to bugs -``` - -### Object and Array Patterns - -```javascript -// Use object destructuring -const { name, email, role = 'user' } = user; - -// Use spread for immutable updates -const updatedUser = { ...user, lastLogin: new Date() }; -const updatedList = [...items, newItem]; - -// Use array methods over loops -const activeUsers = users.filter(u => u.isActive); -const emails = users.map(u => u.email); -const total = orders.reduce((sum, o) => sum + o.amount, 0); -``` - -### Module Patterns - -```javascript -// Use named exports for utilities -export function formatDate(date) { ... } -export function parseDate(str) { ... } - -// Use default export for main component/class -export default class UserService { ... } - -// Group related exports -export { formatDate, parseDate, isValidDate } from './dateUtils'; -``` - ---- - -## Python Standards - -### Type Hints (PEP 484) - -```python -from typing import Optional, List, Dict, Union - -def get_user(user_id: int) -> Optional[User]: - """Fetch user by ID, returns None if not found.""" - return db.query(User).filter(User.id == user_id).first() - -def process_items(items: List[str]) -> Dict[str, int]: - """Count occurrences of each item.""" - return {item: items.count(item) for item in set(items)} - -def send_notification( - user: User, - message: str, - *, - priority: str = "normal", - channels: List[str] = None -) -> bool: - """Send notification to user via specified channels.""" - channels = channels or ["email"] - # Implementation -``` - -### Exception Handling - -```python -# Catch specific exceptions -try: - result = api_client.fetch_data(endpoint) -except ConnectionError as e: - logger.warning(f"Connection failed: {e}") - return cached_data -except TimeoutError as e: - logger.error(f"Request timed out: {e}") - raise ServiceUnavailableError() from e - -# Use context managers for resources -with open(filepath, 'r') as f: - data = json.load(f) - -# Custom exceptions should be informative -class ValidationError(Exception): - def __init__(self, field: str, message: str): - self.field = field - self.message = message - super().__init__(f"{field}: {message}") -``` - -### Class Design - -```python -from dataclasses import dataclass -from abc import ABC, abstractmethod - -# Use dataclasses for data containers -@dataclass -class UserDTO: - id: int - email: str - name: str - is_active: bool = True - -# Use ABC for interfaces -class Repository(ABC): - @abstractmethod - def find_by_id(self, id: int) -> Optional[Entity]: - pass - - @abstractmethod - def save(self, entity: Entity) -> Entity: - pass - -# Use properties for computed attributes -class Order: - def __init__(self, items: List[OrderItem]): - self._items = items - - @property - def total(self) -> Decimal: - return sum(item.price * item.quantity for item in self._items) -``` - ---- - -## Go Standards - -### Error Handling - -```go -// Always check errors -file, err := os.Open(filename) -if err != nil { - return fmt.Errorf("failed to open %s: %w", filename, err) -} -defer file.Close() - -// Use custom error types for specific cases -type ValidationError struct { - Field string - Message string -} - -func (e *ValidationError) Error() string { - return fmt.Sprintf("%s: %s", e.Field, e.Message) -} - -// Wrap errors with context -if err := db.Query(query); err != nil { - return fmt.Errorf("query failed for user %d: %w", userID, err) -} -``` - -### Struct Design - -```go -// Use unexported fields with exported methods -type UserService struct { - repo UserRepository - cache Cache - logger Logger -} - -// Constructor functions for initialization -func NewUserService(repo UserRepository, cache Cache, logger Logger) *UserService { - return &UserService{ - repo: repo, - cache: cache, - logger: logger, - } -} - -// Keep interfaces small -type Reader interface { - Read(p []byte) (n int, err error) -} - -type Writer interface { - Write(p []byte) (n int, err error) -} -``` - -### Concurrency - -```go -// Use context for cancellation -func fetchData(ctx context.Context, url string) ([]byte, error) { - req, err := http.NewRequestWithContext(ctx, "GET", url, nil) - if err != nil { - return nil, err - } - // ... -} - -// Use channels for communication -func worker(jobs <-chan Job, results chan<- Result) { - for job := range jobs { - result := process(job) - results <- result - } -} - -// Use sync.WaitGroup for coordination -var wg sync.WaitGroup -for _, item := range items { - wg.Add(1) - go func(i Item) { - defer wg.Done() - processItem(i) - }(item) -} -wg.Wait() -``` - ---- - -## Swift Standards - -### Optionals - -```swift -// Use optional binding -if let user = fetchUser(id: userId) { - displayProfile(user) -} - -// Use guard for early exit -guard let data = response.data else { - throw NetworkError.noData -} - -// Use nil coalescing for defaults -let displayName = user.nickname ?? user.email - -// Avoid force unwrapping except in tests -// BAD: let name = user.name! -// GOOD: guard let name = user.name else { return } -``` - -### Protocol-Oriented Design - -```swift -// Define protocols with minimal requirements -protocol Identifiable { - var id: String { get } -} - -protocol Persistable: Identifiable { - func save() throws - static func find(by id: String) -> Self? -} - -// Use protocol extensions for default implementations -extension Persistable { - func save() throws { - try Storage.shared.save(self) - } -} - -// Prefer composition over inheritance -struct User: Identifiable, Codable { - let id: String - var name: String - var email: String -} -``` - -### Error Handling - -```swift -// Define domain-specific errors -enum AuthError: Error { - case invalidCredentials - case tokenExpired - case networkFailure(underlying: Error) -} - -// Use Result type for async operations -func authenticate( - email: String, - password: String, - completion: @escaping (Result) -> Void -) - -// Use throws for synchronous operations -func validate(_ input: String) throws -> ValidatedInput { - guard !input.isEmpty else { - throw ValidationError.emptyInput - } - return ValidatedInput(value: input) -} -``` - ---- - -## Kotlin Standards - -### Null Safety - -```kotlin -// Use nullable types explicitly -fun findUser(id: Int): User? { - return userRepository.find(id) -} - -// Use safe calls and elvis operator -val name = user?.profile?.name ?: "Unknown" - -// Use let for null checks with side effects -user?.let { activeUser -> - sendWelcomeEmail(activeUser.email) - logActivity(activeUser.id) -} - -// Use require/check for validation -fun processPayment(amount: Double) { - require(amount > 0) { "Amount must be positive: $amount" } - // Process -} -``` - -### Data Classes and Sealed Classes - -```kotlin -// Use data classes for DTOs -data class UserDTO( - val id: Int, - val email: String, - val name: String, - val isActive: Boolean = true -) - -// Use sealed classes for state -sealed class Result { - data class Success(val data: T) : Result() - data class Error(val message: String, val cause: Throwable? = null) : Result() - object Loading : Result() -} - -// Pattern matching with when -fun handleResult(result: Result) = when (result) { - is Result.Success -> showUser(result.data) - is Result.Error -> showError(result.message) - Result.Loading -> showLoading() -} -``` - -### Coroutines - -```kotlin -// Use structured concurrency -suspend fun loadDashboard(): Dashboard = coroutineScope { - val profile = async { fetchProfile() } - val stats = async { fetchStats() } - val notifications = async { fetchNotifications() } - - Dashboard( - profile = profile.await(), - stats = stats.await(), - notifications = notifications.await() - ) -} - -// Handle cancellation -suspend fun fetchWithRetry(url: String): Response { - repeat(3) { attempt -> - try { - return httpClient.get(url) - } catch (e: IOException) { - if (attempt == 2) throw e - delay(1000L * (attempt + 1)) - } - } - throw IllegalStateException("Unreachable") -} -``` - ---- - -## C# / .NET Standards - -### Nullable Reference Types - -```csharp -// Enable nullable reference types at the project level -// -// enable -// true -// - -// Be explicit about nullability -public string Name { get; set; } = ""; // non-nullable, requires init -public string? Nickname { get; set; } // nullable - -public User? FindUser(int id) // may return null -{ - return _repo.Get(id); -} - -// Avoid the null-forgiving operator (!) — it tells the compiler -// "trust me, this is not null" and silently disables the safety net. -// BAD: return user!.Name; -// GOOD: return user?.Name ?? throw new InvalidOperationException(nameof(user)); - -// Use the null-conditional and null-coalescing operators -var displayName = user?.Profile?.Name ?? "Anonymous"; - -// Pattern matching for null checks -if (user is { Profile.Name: { } name }) -{ - Log(name); -} -``` - -### Async / Await - -```csharp -// Return Task (or Task), never `void`, except for event handlers. -// `async void` cannot be awaited and exceptions cannot be caught by callers. -public async Task SaveAsync(User user) -{ - await _db.SaveChangesAsync(); -} - -// Never block on async with .Result, .Wait(), or .GetAwaiter().GetResult() -// in code that runs on a synchronization context (ASP.NET Classic, WinForms, WPF) -// — it causes deadlocks. -// BAD: var data = FetchAsync().Result; -// GOOD: var data = await FetchAsync(); - -// In library code, use ConfigureAwait(false) to avoid forcing the caller's -// context back onto the continuation. -public async Task LoadAsync(int id) -{ - var row = await _db.Users.FindAsync(id).ConfigureAwait(false); - return Map(row); -} - -// Parallelize independent awaitables -var (profile, stats, notifications) = ( - await Task.WhenAll( - FetchProfileAsync(id), - FetchStatsAsync(id), - FetchNotificationsAsync(id) - ) -); -``` - -### Exception Handling - -```csharp -// Catch the most specific exception, not `Exception` -try -{ - await client.GetAsync(url); -} -catch (HttpRequestException ex) when (ex.StatusCode == HttpStatusCode.NotFound) -{ - return null; -} -catch (TaskCanceledException) -{ - _logger.LogWarning("Request to {Url} timed out", url); - throw; -} - -// Never swallow exceptions -// BAD: -// try { ... } catch (Exception) { } -// -// GOOD: log with context, then rethrow or convert to a domain error. -try -{ - await Process(order); -} -catch (DomainException ex) -{ - _logger.LogError(ex, "Order {OrderId} failed", order.Id); - throw; -} - -// Use `throw;` (not `throw ex;`) to preserve the original stack trace -``` - -### Resource Management (IDisposable) - -```csharp -// Always wrap IDisposable resources in `using` / `using var` -using var connection = new SqlConnection(connectionString); -using var command = new SqlCommand(query, connection); - -// HttpClient is the exception — it's IDisposable but designed to be -// long-lived. Use IHttpClientFactory in DI rather than `new HttpClient()` -// in a method body. -public class Foo -{ - private readonly HttpClient _client; - - public Foo(IHttpClientFactory factory) - { - _client = factory.CreateClient("api"); - } -} - -// For `DbContext`, register as scoped — never instantiate per-request inside a method. -services.AddDbContext(opts => opts.UseSqlServer(connStr)); -``` - -### LINQ - -```csharp -// Defer execution until you actually need the results -var activeUsers = _db.Users - .Where(u => u.IsActive) - .Select(u => new UserDto(u.Id, u.Email)); // still an IQueryable - -// Prefer FirstOrDefault / SingleOrDefault to First / Single -// when "no match" is a valid outcome -var user = await _db.Users.FirstOrDefaultAsync(u => u.Email == email); -if (user is null) return NotFound(); - -// Avoid premature materialization -// BAD: _db.Users.ToList().Where(u => u.IsActive) // pulls the entire table -// GOOD: _db.Users.Where(u => u.IsActive).ToList() // SQL WHERE clause - -// Don't fight LINQ — if the chain is hard to read, drop to a for loop -``` - -### Dependency Injection - -```csharp -// Constructor injection — required dependencies as ctor params -public class OrderService -{ - private readonly IOrderRepository _repo; - private readonly IPaymentGateway _payments; - private readonly ILogger _logger; - - public OrderService( - IOrderRepository repo, - IPaymentGateway payments, - ILogger logger) - { - _repo = repo; - _payments = payments; - _logger = logger; - } -} - -// Pick lifetimes deliberately -services.AddSingleton(); // stateless, thread-safe -services.AddScoped(); // per-request state -services.AddTransient(); // light, no state - -// Don't pass IServiceProvider into business code — it's the service locator -// anti-pattern. If you need many services, group them or inject what you need. -``` - -### Records and Pattern Matching - -```csharp -// Use records for immutable value types (DTOs, value objects, events) -public record UserDto(int Id, string Email, string Name); - -// `with` expressions for non-destructive updates -var updated = user with { Name = "New Name" }; - -// Use pattern matching to flatten nested logic -public decimal CalculateFee(Order order) => order switch -{ - { Customer.Tier: "Gold", Total: > 1000m } => 0m, - { Customer.Tier: "Gold" } => order.Total * 0.01m, - { Total: > 500m } => order.Total * 0.02m, - _ => order.Total * 0.03m, -}; - -// Property patterns for clean guards -if (response is { IsSuccess: true, Data: var data }) -{ - Process(data); -} -``` - -### Security (ASP.NET Core) - -```csharp -// Parameterized queries — never interpolate user input into SQL -// BAD: _db.Users.FromSqlRaw($"SELECT * FROM Users WHERE Id = {id}") -// GOOD: -var users = await _db.Users - .FromSqlInterpolated($"SELECT * FROM Users WHERE Id = {id}") // EF Core handles parameters - .ToListAsync(); - -// Or explicitly: -var users = await _db.Users - .FromSqlRaw("SELECT * FROM Users WHERE Id = @id", - new SqlParameter("@id", id)) - .ToListAsync(); - -// Anti-forgery on state-changing actions -[HttpPost] -[ValidateAntiForgeryToken] -public async Task Update(UpdateUserDto dto) { ... } - -// Never bind sensitive properties from the request body -public record CreateUserDto(string Email, string Name); // no Role, no IsAdmin -public IActionResult Create([FromBody] CreateUserDto dto) { ... } - -// Use IOptions and the secrets store, not appsettings.json, for secrets -// dotnet user-secrets set ConnectionStrings:Default "Server=...;Password=..." -public class DbOptions { public string ConnectionString { get; init; } = ""; } -services.Configure(config.GetSection("Db")); -``` diff --git a/engineering-team/skills/code-reviewer/references/common_antipatterns.md b/engineering-team/skills/code-reviewer/references/common_antipatterns.md deleted file mode 100644 index cab67051..00000000 --- a/engineering-team/skills/code-reviewer/references/common_antipatterns.md +++ /dev/null @@ -1,991 +0,0 @@ -# Common Antipatterns - -Code antipatterns to identify during review, with examples and fixes. - ---- - -## Table of Contents - -- [Structural Antipatterns](#structural-antipatterns) -- [Logic Antipatterns](#logic-antipatterns) -- [Security Antipatterns](#security-antipatterns) -- [Performance Antipatterns](#performance-antipatterns) -- [Testing Antipatterns](#testing-antipatterns) -- [Async Antipatterns](#async-antipatterns) -- [C# / .NET Antipatterns](#c--net-antipatterns) - ---- - -## Structural Antipatterns - -### God Class - -A class that does too much and knows too much. - -```typescript -// BAD: God class handling everything -class UserManager { - createUser(data: UserData) { ... } - updateUser(id: string, data: UserData) { ... } - deleteUser(id: string) { ... } - sendEmail(userId: string, content: string) { ... } - generateReport(userId: string) { ... } - validatePassword(password: string) { ... } - hashPassword(password: string) { ... } - uploadAvatar(userId: string, file: File) { ... } - resizeImage(file: File) { ... } - logActivity(userId: string, action: string) { ... } - // 50 more methods... -} - -// GOOD: Single responsibility classes -class UserRepository { - create(data: UserData): User { ... } - update(id: string, data: Partial): User { ... } - delete(id: string): void { ... } -} - -class EmailService { - send(to: string, content: string): void { ... } -} - -class PasswordService { - validate(password: string): ValidationResult { ... } - hash(password: string): string { ... } -} -``` - -**Detection:** Class has >20 methods, >500 lines, or handles unrelated concerns. - ---- - -### Long Method - -Functions that do too much and are hard to understand. - -```python -# BAD: Long method doing everything -def process_order(order_data): - # Validate order (20 lines) - if not order_data.get('items'): - raise ValueError('No items') - if not order_data.get('customer_id'): - raise ValueError('No customer') - # ... more validation - - # Calculate totals (30 lines) - subtotal = 0 - for item in order_data['items']: - price = get_product_price(item['product_id']) - subtotal += price * item['quantity'] - # ... tax calculation, discounts - - # Process payment (40 lines) - payment_result = payment_gateway.charge(...) - # ... handle payment errors - - # Create order record (20 lines) - order = Order.create(...) - - # Send notifications (20 lines) - send_order_confirmation(...) - notify_warehouse(...) - - return order - -# GOOD: Composed of focused functions -def process_order(order_data): - validate_order(order_data) - totals = calculate_order_totals(order_data) - payment = process_payment(order_data['customer_id'], totals) - order = create_order_record(order_data, totals, payment) - send_order_notifications(order) - return order -``` - -**Detection:** Function >50 lines or requires scrolling to read. - ---- - -### Deep Nesting - -Excessive indentation making code hard to follow. - -```javascript -// BAD: Deep nesting -function processData(data) { - if (data) { - if (data.items) { - if (data.items.length > 0) { - for (const item of data.items) { - if (item.isValid) { - if (item.type === 'premium') { - if (item.price > 100) { - // Finally do something - processItem(item); - } - } - } - } - } - } - } -} - -// GOOD: Early returns and guard clauses -function processData(data) { - if (!data?.items?.length) { - return; - } - - const premiumItems = data.items.filter( - item => item.isValid && item.type === 'premium' && item.price > 100 - ); - - premiumItems.forEach(processItem); -} -``` - -**Detection:** Indentation >4 levels deep. - ---- - -### Magic Numbers and Strings - -Hard-coded values without explanation. - -```go -// BAD: Magic numbers -func calculateDiscount(total float64, userType int) float64 { - if userType == 1 { - return total * 0.15 - } else if userType == 2 { - return total * 0.25 - } - return total * 0.05 -} - -// GOOD: Named constants -const ( - UserTypeRegular = 1 - UserTypePremium = 2 - - DiscountRegular = 0.05 - DiscountStandard = 0.15 - DiscountPremium = 0.25 -) - -func calculateDiscount(total float64, userType int) float64 { - switch userType { - case UserTypePremium: - return total * DiscountPremium - case UserTypeRegular: - return total * DiscountStandard - default: - return total * DiscountRegular - } -} -``` - -**Detection:** Literal numbers (except 0, 1) or repeated string literals. - ---- - -### Primitive Obsession - -Using primitives instead of small objects. - -```typescript -// BAD: Primitives everywhere -function createUser( - name: string, - email: string, - phone: string, - street: string, - city: string, - zipCode: string, - country: string -): User { ... } - -// GOOD: Value objects -interface Address { - street: string; - city: string; - zipCode: string; - country: string; -} - -interface ContactInfo { - email: string; - phone: string; -} - -function createUser( - name: string, - contact: ContactInfo, - address: Address -): User { ... } -``` - -**Detection:** Functions with >4 parameters of same type, or related primitives always passed together. - ---- - -## Logic Antipatterns - -### Boolean Blindness - -Passing booleans that make code unreadable at call sites. - -```swift -// BAD: What do these booleans mean? -user.configure(true, false, true, false) - -// GOOD: Named parameters or option objects -user.configure( - sendWelcomeEmail: true, - requireVerification: false, - enableNotifications: true, - isAdmin: false -) - -// Or use an options struct -struct UserConfiguration { - var sendWelcomeEmail: Bool = true - var requireVerification: Bool = false - var enableNotifications: Bool = true - var isAdmin: Bool = false -} - -user.configure(UserConfiguration()) -``` - -**Detection:** Function calls with multiple boolean literals. - ---- - -### Null Returns for Collections - -Returning null instead of empty collections. - -```kotlin -// BAD: Returning null -fun findUsersByRole(role: String): List? { - val users = repository.findByRole(role) - return if (users.isEmpty()) null else users -} - -// Caller must handle null -val users = findUsersByRole("admin") -if (users != null) { - users.forEach { ... } -} - -// GOOD: Return empty collection -fun findUsersByRole(role: String): List { - return repository.findByRole(role) -} - -// Caller can iterate directly -findUsersByRole("admin").forEach { ... } -``` - -**Detection:** Functions returning nullable collections. - ---- - -### Stringly Typed Code - -Using strings where enums or types should be used. - -```python -# BAD: String-based logic -def handle_event(event_type: str, data: dict): - if event_type == "user_created": - handle_user_created(data) - elif event_type == "user_updated": - handle_user_updated(data) - elif event_type == "user_dleted": # Typo won't be caught - handle_user_deleted(data) - -# GOOD: Enum-based -from enum import Enum - -class EventType(Enum): - USER_CREATED = "user_created" - USER_UPDATED = "user_updated" - USER_DELETED = "user_deleted" - -def handle_event(event_type: EventType, data: dict): - handlers = { - EventType.USER_CREATED: handle_user_created, - EventType.USER_UPDATED: handle_user_updated, - EventType.USER_DELETED: handle_user_deleted, - } - handlers[event_type](data) -``` - -**Detection:** String comparisons for type/status/category values. - ---- - -## Security Antipatterns - -### SQL Injection - -String concatenation in SQL queries. - -```javascript -// BAD: String concatenation -const query = `SELECT * FROM users WHERE id = ${userId}`; -db.query(query); - -// BAD: String templates still vulnerable -const query = `SELECT * FROM users WHERE name = '${userName}'`; - -// GOOD: Parameterized queries -const query = 'SELECT * FROM users WHERE id = $1'; -db.query(query, [userId]); - -// GOOD: Using ORM safely -User.findOne({ where: { id: userId } }); -``` - -**Detection:** String concatenation or template literals with SQL keywords. - ---- - -### Hardcoded Credentials - -Secrets in source code. - -```python -# BAD: Hardcoded secrets -API_KEY = "sk-abc123xyz789" -DATABASE_URL = "postgresql://admin:password123@prod-db.internal:5432/app" - -# GOOD: Environment variables -import os - -API_KEY = os.environ["API_KEY"] -DATABASE_URL = os.environ["DATABASE_URL"] - -# GOOD: Secrets manager -from aws_secretsmanager import get_secret - -API_KEY = get_secret("api-key") -``` - -**Detection:** Variables named `password`, `secret`, `key`, `token` with string literals. - ---- - -### Unsafe Deserialization - -Deserializing untrusted data without validation. - -```python -# BAD: Binary serialization from untrusted source can execute arbitrary code -# Examples: Python's binary serialization, yaml.load without SafeLoader - -# GOOD: Use safe alternatives -import json - -def load_data(file_path): - with open(file_path, 'r') as f: - return json.load(f) - -# GOOD: Use SafeLoader for YAML -import yaml - -with open('config.yaml') as f: - config = yaml.safe_load(f) -``` - -**Detection:** Binary deserialization functions, yaml.load without safe loader, dynamic code execution on external data. - ---- - -### Missing Input Validation - -Trusting user input without validation. - -```typescript -// BAD: No validation -app.post('/user', (req, res) => { - const user = db.create({ - name: req.body.name, - email: req.body.email, - role: req.body.role // User can set themselves as admin! - }); - res.json(user); -}); - -// GOOD: Validate and sanitize -import { z } from 'zod'; - -const CreateUserSchema = z.object({ - name: z.string().min(1).max(100), - email: z.string().email(), - // role is NOT accepted from input -}); - -app.post('/user', (req, res) => { - const validated = CreateUserSchema.parse(req.body); - const user = db.create({ - ...validated, - role: 'user' // Default role, not from input - }); - res.json(user); -}); -``` - -**Detection:** Request body/params used directly without validation schema. - ---- - -## Performance Antipatterns - -### N+1 Query Problem - -Loading related data one record at a time. - -```python -# BAD: N+1 queries -def get_orders_with_items(): - orders = Order.query.all() # 1 query - for order in orders: - items = OrderItem.query.filter_by(order_id=order.id).all() # N queries - order.items = items - return orders - -# GOOD: Eager loading -def get_orders_with_items(): - return Order.query.options( - joinedload(Order.items) - ).all() # 1 query with JOIN - -# GOOD: Batch loading -def get_orders_with_items(): - orders = Order.query.all() - order_ids = [o.id for o in orders] - items = OrderItem.query.filter( - OrderItem.order_id.in_(order_ids) - ).all() # 2 queries total - # Group items by order_id... -``` - -**Detection:** Database queries inside loops. - ---- - -### Unbounded Collections - -Loading unlimited data into memory. - -```go -// BAD: Load all records -func GetAllUsers() ([]User, error) { - return db.Find(&[]User{}) // Could be millions -} - -// GOOD: Pagination -func GetUsers(page, pageSize int) ([]User, error) { - offset := (page - 1) * pageSize - return db.Limit(pageSize).Offset(offset).Find(&[]User{}) -} - -// GOOD: Streaming for large datasets -func ProcessAllUsers(handler func(User) error) error { - rows, err := db.Model(&User{}).Rows() - if err != nil { - return err - } - defer rows.Close() - - for rows.Next() { - var user User - db.ScanRows(rows, &user) - if err := handler(user); err != nil { - return err - } - } - return nil -} -``` - -**Detection:** `findAll()`, `find({})`, or queries without `LIMIT`. - ---- - -### Synchronous I/O in Hot Paths - -Blocking operations in request handlers. - -```javascript -// BAD: Sync file read on every request -app.get('/config', (req, res) => { - const config = fs.readFileSync('./config.json'); // Blocks event loop - res.json(JSON.parse(config)); -}); - -// GOOD: Load once at startup -const config = JSON.parse(fs.readFileSync('./config.json')); - -app.get('/config', (req, res) => { - res.json(config); -}); - -// GOOD: Async with caching -let configCache = null; - -app.get('/config', async (req, res) => { - if (!configCache) { - configCache = JSON.parse(await fs.promises.readFile('./config.json')); - } - res.json(configCache); -}); -``` - -**Detection:** `readFileSync`, `execSync`, or blocking calls in request handlers. - ---- - -## Testing Antipatterns - -### Test Code Duplication - -Repeating setup in every test. - -```typescript -// BAD: Duplicate setup -describe('UserService', () => { - it('should create user', async () => { - const db = await createTestDatabase(); - const userRepo = new UserRepository(db); - const emailService = new MockEmailService(); - const service = new UserService(userRepo, emailService); - - const user = await service.create({ name: 'Test' }); - expect(user.name).toBe('Test'); - }); - - it('should update user', async () => { - const db = await createTestDatabase(); // Duplicated - const userRepo = new UserRepository(db); // Duplicated - const emailService = new MockEmailService(); // Duplicated - const service = new UserService(userRepo, emailService); // Duplicated - - // ... - }); -}); - -// GOOD: Shared setup -describe('UserService', () => { - let service: UserService; - let db: TestDatabase; - - beforeEach(async () => { - db = await createTestDatabase(); - const userRepo = new UserRepository(db); - const emailService = new MockEmailService(); - service = new UserService(userRepo, emailService); - }); - - afterEach(async () => { - await db.cleanup(); - }); - - it('should create user', async () => { - const user = await service.create({ name: 'Test' }); - expect(user.name).toBe('Test'); - }); -}); -``` - ---- - -### Testing Implementation Instead of Behavior - -Tests coupled to internal implementation. - -```python -# BAD: Testing implementation details -def test_add_item_to_cart(): - cart = ShoppingCart() - cart.add_item(Product("Apple", 1.00)) - - # Testing internal structure - assert cart._items[0].name == "Apple" - assert cart._total == 1.00 - -# GOOD: Testing behavior -def test_add_item_to_cart(): - cart = ShoppingCart() - cart.add_item(Product("Apple", 1.00)) - - # Testing public behavior - assert cart.item_count == 1 - assert cart.total == 1.00 - assert cart.contains("Apple") -``` - ---- - -## Async Antipatterns - -### Floating Promises - -Promises without await or catch. - -```typescript -// BAD: Floating promise -async function saveUser(user: User) { - db.save(user); // Not awaited, errors lost - logger.info('User saved'); // Logs before save completes -} - -// BAD: Fire and forget in loop -for (const item of items) { - processItem(item); // All run in parallel, no error handling -} - -// GOOD: Await the promise -async function saveUser(user: User) { - await db.save(user); - logger.info('User saved'); -} - -// GOOD: Process with proper handling -await Promise.all(items.map(item => processItem(item))); - -// Or sequentially -for (const item of items) { - await processItem(item); -} -``` - -**Detection:** Async function calls without `await` or `.then()`. - ---- - -### Callback Hell - -Deeply nested callbacks. - -```javascript -// BAD: Callback hell -getUser(userId, (err, user) => { - if (err) return handleError(err); - getOrders(user.id, (err, orders) => { - if (err) return handleError(err); - getProducts(orders[0].productIds, (err, products) => { - if (err) return handleError(err); - renderPage(user, orders, products, (err) => { - if (err) return handleError(err); - console.log('Done'); - }); - }); - }); -}); - -// GOOD: Async/await -async function loadPage(userId) { - try { - const user = await getUser(userId); - const orders = await getOrders(user.id); - const products = await getProducts(orders[0].productIds); - await renderPage(user, orders, products); - console.log('Done'); - } catch (err) { - handleError(err); - } -} -``` - -**Detection:** >2 levels of callback nesting. - ---- - -### Async in Constructor - -Async operations in constructors. - -```typescript -// BAD: Async in constructor -class DatabaseConnection { - constructor(url: string) { - this.connect(url); // Fire-and-forget async - } - - private async connect(url: string) { - this.client = await createClient(url); - } -} - -// GOOD: Factory method -class DatabaseConnection { - private constructor(private client: Client) {} - - static async create(url: string): Promise { - const client = await createClient(url); - return new DatabaseConnection(client); - } -} - -// Usage -const db = await DatabaseConnection.create(url); -``` - -**Detection:** `async` calls or `.then()` in constructor. - ---- - -## C# / .NET Antipatterns - -### `async void` - -`async void` cannot be awaited and exceptions cannot be caught by callers — they tear down the process. Only safe for event handlers. - -```csharp -// BAD: async void in non-event-handler code -public async void SaveUser(User user) -{ - await _db.SaveChangesAsync(); // exception here crashes the host -} - -// GOOD: return Task so callers can await + observe exceptions -public async Task SaveUserAsync(User user) -{ - await _db.SaveChangesAsync(); -} - -// EXCEPTION: real event handlers must be `async void` (delegate signature) -private async void OnClick(object sender, EventArgs e) -{ - try { await DoWorkAsync(); } - catch (Exception ex) { _logger.LogError(ex, "Click handler failed"); } -} -``` - -**Detection:** `async\s+void\s+\w+` outside of event-handler signatures. - ---- - -### Blocking on Async (`.Result`, `.Wait()`, `.GetAwaiter().GetResult()`) - -Synchronously waiting on a Task from inside a synchronization context (ASP.NET Classic, WinForms, WPF) deadlocks: the continuation needs the context, which is blocked by the caller. - -```csharp -// BAD: deadlock in ASP.NET Classic / WPF / WinForms -public IActionResult Index() -{ - var data = FetchAsync().Result; - return View(data); -} - -// BAD: same issue -public string Synchronous() => FetchAsync().GetAwaiter().GetResult(); - -// GOOD: be async all the way up -public async Task Index() -{ - var data = await FetchAsync(); - return View(data); -} -``` - -**Detection:** `\.Result`, `\.Wait\(\)`, `\.GetAwaiter\(\)\.GetResult\(\)` on Task-returning calls. - ---- - -### Swallowing `Exception` - -Catching the base `Exception` and dropping it hides bugs. - -```csharp -// BAD: silent failure -try -{ - await _payments.ChargeAsync(order); -} -catch (Exception) -{ - // ¯\_(ツ)_/¯ -} - -// GOOD: catch specific exceptions, log, rethrow or convert -try -{ - await _payments.ChargeAsync(order); -} -catch (PaymentDeclinedException ex) -{ - _logger.LogWarning(ex, "Payment declined for order {OrderId}", order.Id); - throw new OrderDeclinedException(order.Id, ex); -} -catch (HttpRequestException ex) -{ - _logger.LogError(ex, "Payment gateway unreachable"); - throw; // bubble up — caller decides retry policy -} -``` - -**Detection:** `catch (Exception)` with empty body, or no log/rethrow. - ---- - -### Undisposed `IDisposable` - -Forgetting `using` leaks file handles, database connections, sockets, etc. - -```csharp -// BAD: connection never closed if an exception occurs -public string GetName(int id) -{ - var conn = new SqlConnection(_connStr); - conn.Open(); - var cmd = new SqlCommand("SELECT name FROM users WHERE id = @id", conn); - cmd.Parameters.AddWithValue("@id", id); - return (string)cmd.ExecuteScalar(); -} - -// GOOD: `using var` disposes on scope exit, even on exceptions -public string GetName(int id) -{ - using var conn = new SqlConnection(_connStr); - using var cmd = new SqlCommand("SELECT name FROM users WHERE id = @id", conn); - cmd.Parameters.AddWithValue("@id", id); - conn.Open(); - return (string)cmd.ExecuteScalar(); -} -``` - -**Detection:** `new\s+\w+(?:Stream|Connection|Reader|Writer|Client|Context|Command)\s*\(` not preceded by `using`. - ---- - -### `new HttpClient()` in a Method - -Each `HttpClient` opens its own socket pool. Creating one per request exhausts sockets under load. - -```csharp -// BAD: socket exhaustion -public async Task FetchAsync(string url) -{ - using var client = new HttpClient(); // even disposed, sockets linger - return await client.GetStringAsync(url); -} - -// GOOD: IHttpClientFactory via DI -public class ApiClient -{ - private readonly HttpClient _client; - public ApiClient(IHttpClientFactory factory) => _client = factory.CreateClient("api"); - - public Task FetchAsync(string url) => _client.GetStringAsync(url); -} -``` - -**Detection:** `new\s+HttpClient\s*\(` inside a method body. - ---- - -### Missing `ConfigureAwait(false)` in Library Code - -Continuations on the captured synchronization context can deadlock callers blocking on `Result` / `Wait()`, and create unnecessary context switches on hot paths. - -```csharp -// BAD (library code): -public async Task LoadAsync(int id) -{ - var row = await _db.Users.FindAsync(id); // recaptures context - return Map(row); -} - -// GOOD: opt out of the capture in library code -public async Task LoadAsync(int id) -{ - var row = await _db.Users.FindAsync(id).ConfigureAwait(false); - return Map(row); -} -``` - -**Detection:** library projects (not the entry point) where every `await` recaptures context. - -**Note:** ASP.NET Core has no synchronization context, so `ConfigureAwait(false)` is *not required* in ASP.NET Core application code — but it doesn't hurt and is mandatory for shared libraries. - ---- - -### Mutable Public Setters on Domain Models - -```csharp -// BAD: any caller can mutate state outside the entity's rules -public class Order -{ - public OrderStatus Status { get; set; } // anyone can set Shipped - public decimal Total { get; set; } -} - -// GOOD: invariants enforced by methods, state only mutable through them -public class Order -{ - public OrderStatus Status { get; private set; } = OrderStatus.Pending; - public decimal Total { get; private set; } - - public void MarkShipped(IShippingProvider shipper) - { - if (Status != OrderStatus.Paid) - throw new InvalidOperationException("Cannot ship unpaid order"); - shipper.Ship(this); - Status = OrderStatus.Shipped; - } -} - -// For DTOs, prefer records with init-only properties -public record OrderDto(int Id, string Status, decimal Total); -``` - -**Detection:** public domain entities with `{ get; set; }` on every property. - ---- - -### Overuse of `dynamic` - -`dynamic` opts out of type checking. Use only when you genuinely need late binding (COM interop, ExpandoObject for JSON-shaped data). - -```csharp -// BAD: dynamic for ordinary code — typos surface only at runtime -public void Save(dynamic user) -{ - _db.Insert(user.Emial); // typo, compiles fine, NullReferenceException at runtime -} - -// GOOD: real type -public void Save(User user) -{ - _db.Insert(user.Email); -} -``` - -**Detection:** `\bdynamic\s+\w+\s*[=;]` outside of obvious interop / DOM code. - ---- - -### Suppressing Analyzer Warnings Without Justification - -```csharp -// BAD: suppression with no rationale -#pragma warning disable CS8602 -var name = user.Name; -#pragma warning restore CS8602 - -// GOOD: explain WHY the warning is wrong here -// CS8602 is incorrect here: `user` is non-null because we just -// validated it on line 42 inside the same method. -#pragma warning disable CS8602 // justified above -var name = user.Name; -#pragma warning restore CS8602 -``` - -**Detection:** `#pragma warning disable` or `[SuppressMessage]` without an adjacent justification comment. diff --git a/engineering-team/skills/code-reviewer/rules/universal.md b/engineering-team/skills/code-reviewer/rules/universal.md new file mode 100644 index 00000000..4f9b5b04 --- /dev/null +++ b/engineering-team/skills/code-reviewer/rules/universal.md @@ -0,0 +1,49 @@ +# Universal Rules — All Languages + +These rules apply regardless of language. Load this file for every review, alongside the relevant `languages/*.md` file. + +--- + +## Security + +- Flag any string interpolation or concatenation used to build SQL, shell, or LDAP queries — require parameterized queries or a safe API +- Flag hardcoded credentials, API keys, tokens, or secrets anywhere in source — require environment variables or a secrets manager +- Flag user-controlled input passed to file system, process execution, or URL redirect APIs without validation +- Flag overly broad CORS or CSP policies + +--- + +## Async / Concurrency + +- Flag shared mutable state accessed from multiple threads/coroutines/tasks without synchronization +- Flag fire-and-forget async operations with no error handling path +- Flag timeouts missing on any network or I/O call +- Flag unbounded queues or thread pools with no backpressure mechanism + +--- + +## Resource Management + +- Flag any resource (file, socket, DB connection, HTTP connection) acquired without a guaranteed release path +- Flag connection pools not returned to the pool on all code paths (including exceptions) +- Flag unbounded collections that grow without eviction — potential memory leak +- Flag resources held open longer than the operation they serve + +--- + +## Exception Handling + +- Flag empty catch/except blocks — swallowed exceptions hide bugs silently +- Flag catching the broadest possible exception type (`Exception`, `Throwable`, `error`) where a specific type is appropriate +- Flag exceptions used for normal control flow (signaling "not found", etc.) — use return values or `Optional` +- Flag error context lost when re-throwing — always wrap with the original cause + +--- + +## Performance + +- Flag N+1 query patterns — loading a collection then querying for each item individually +- Flag unbounded queries or API calls with no pagination or limit +- Flag synchronous I/O on a thread or event loop that serves concurrent requests +- Flag large objects serialized/deserialized repeatedly when they could be cached +- Flag string concatenation in tight loops — use a builder or join diff --git a/engineering-team/skills/code-reviewer/scripts/code_quality_checker.py b/engineering-team/skills/code-reviewer/scripts/code_quality_checker.py index bf67545e..d3649e51 100755 --- a/engineering-team/skills/code-reviewer/scripts/code_quality_checker.py +++ b/engineering-team/skills/code-reviewer/scripts/code_quality_checker.py @@ -28,6 +28,7 @@ LANGUAGE_EXTENSIONS = { "swift": [".swift"], "kotlin": [".kt", ".kts"], "csharp": [".cs", ".csx", ".razor", ".cshtml"], + "java": [".java"], } # Code smell thresholds @@ -135,6 +136,13 @@ def find_functions(content: str, language: str) -> List[Dict]: r"override|sealed|abstract|partial|new|readonly|extern)\s+)+" r"(?:[\w<>?,\s\[\]\.]+?\s+)?(\w+)\s*\(([^)]*)\)" ), + # Java: require at least one method modifier to distinguish + # declarations from invocations (mirrors the C# approach). + "java": ( + r"(?:(?:public|private|protected|static|final|abstract|" + r"synchronized|native|default|strictfp)\s+)+" + r"(?:[\w<>?,\s\[\]\.]+?\s+)?(\w+)\s*\(([^)]*)\)" + ), } pattern = patterns.get(language, patterns["python"]) @@ -183,6 +191,7 @@ def find_classes(content: str, language: str) -> List[Dict]: "swift": r"class\s+(\w+)", "kotlin": r"class\s+(\w+)", "csharp": r"(?:class|struct|record|interface)\s+(\w+)", + "java": r"(?:class|interface|enum|record)\s+(\w+)", } pattern = patterns.get(language, patterns["python"]) @@ -213,6 +222,11 @@ def find_classes(content: str, language: str) -> List[Dict]: r"override|sealed|abstract|partial)\s+)+" r"(?:[\w<>?,\s\[\]\.]+?\s+)?\w+\s*\(" ), + "java": ( + r"(?:(?:public|private|protected|static|final|abstract|" + r"synchronized|native|default|strictfp)\s+)+" + r"(?:[\w<>?,\s\[\]\.]+?\s+)?\w+\s*\(" + ), } method_pattern = method_patterns.get(language, method_patterns["python"]) methods = len(re.findall(method_pattern, class_body)) @@ -418,6 +432,88 @@ def check_csharp_specific_smells(content: str) -> List[Dict]: return smells +def check_java_specific_smells(content: str) -> List[Dict]: + """Java-specific code smells documented in languages/java.md.""" + smells: List[Dict] = [] + # Java comment syntax matches C#, so the same stripper applies. + content = _strip_csharp_comments(content) + + # Empty catch block — swallows the exception silently. + for match in re.finditer(r"catch\s*\([^)]*\)\s*\{\s*\}", content): + smells.append({ + "type": "java_empty_catch", + "severity": "high", + "message": "Empty catch block swallows exceptions silently", + "location": f"offset {match.start()}", + }) + + # printStackTrace() as error handling — use a logger instead. + for match in re.finditer(r"\.printStackTrace\s*\(\s*\)", content): + smells.append({ + "type": "java_print_stack_trace", + "severity": "medium", + "message": ( + "'printStackTrace()' is not real error handling — log via a " + "proper logger or rethrow with context" + ), + "location": f"offset {match.start()}", + }) + + # InterruptedException caught without restoring the interrupt flag. + for match in re.finditer( + r"catch\s*\(\s*InterruptedException\s+(\w+)\s*\)\s*\{(.*?)\}", + content, + re.DOTALL, + ): + if "interrupt()" not in match.group(2): + smells.append({ + "type": "java_swallowed_interrupt", + "severity": "high", + "message": ( + "InterruptedException caught without " + "'Thread.currentThread().interrupt()' — breaks cooperative " + "cancellation" + ), + "location": f"offset {match.start()}", + }) + + # Closeable resource instantiated outside try-with-resources (leak heuristic). + resource_hint = re.compile( + r"^(?!\s*try\b)\s*(?:final\s+)?[\w<>\[\]]+\s+\w+\s*=\s*new\s+" + r"(\w*(?:InputStream|OutputStream|Reader|Writer|Stream|Connection))\s*\(", + re.MULTILINE, + ) + for match in resource_hint.finditer(content): + smells.append({ + "type": "java_unclosed_resource", + "severity": "medium", + "message": ( + f"'{match.group(1)}' looks like an AutoCloseable but is not in a " + "try-with-resources statement" + ), + "location": f"offset {match.start()}", + }) + + # Heavy object built per use instead of shared as a singleton. + # A `static` field assignment is the recommended singleton form — skip it. + heavy_object = re.compile( + r"^(?!.*\bstatic\b).*\bnew\s+(ObjectMapper|Gson)\s*\(\s*\)", + re.MULTILINE, + ) + for match in heavy_object.finditer(content): + smells.append({ + "type": "java_per_use_heavy_object", + "severity": "medium", + "message": ( + f"'new {match.group(1)}()' is expensive — share a singleton " + "instance instead of constructing per call" + ), + "location": f"offset {match.start()}", + }) + + return smells + + def check_solid_violations(content: str) -> List[Dict]: """Check for potential SOLID principle violations.""" violations = [] @@ -528,6 +624,8 @@ def analyze_file(filepath: Path) -> Dict: smells = check_code_smells(content, functions, classes) if language == "csharp": smells.extend(check_csharp_specific_smells(content)) + if language == "java": + smells.extend(check_java_specific_smells(content)) violations = check_solid_violations(content) score = calculate_quality_score(line_metrics, functions, classes, smells, violations) diff --git a/engineering-team/skills/code-reviewer/scripts/pr_analyzer.py b/engineering-team/skills/code-reviewer/scripts/pr_analyzer.py index 959dffb0..c6f4f3da 100755 --- a/engineering-team/skills/code-reviewer/scripts/pr_analyzer.py +++ b/engineering-team/skills/code-reviewer/scripts/pr_analyzer.py @@ -72,9 +72,15 @@ RISK_PATTERNS = [ }, { "name": "console_log", - "pattern": r"console\.(log|debug|info|warn|error)\(|\bDebug\.WriteLine\(", + "pattern": ( + r"console\.(log|debug|info|warn|error)\(|\bDebug\.WriteLine\(|" + r"\bSystem\.out\.print(?:ln)?\(|\.printStackTrace\(" + ), "severity": "medium", - "message": "Debug output statement found (console.* / Debug.WriteLine)" + "message": ( + "Debug output statement found " + "(console.* / Debug.WriteLine / System.out / printStackTrace)" + ) }, { "name": "debugger", @@ -84,9 +90,15 @@ RISK_PATTERNS = [ }, { "name": "analyzer_disable", - "pattern": r"eslint-disable|#pragma\s+warning\s+disable|\[SuppressMessage", + "pattern": ( + r"eslint-disable|#pragma\s+warning\s+disable|\[SuppressMessage|" + r"@SuppressWarnings" + ), "severity": "medium", - "message": "Static-analyzer rule disabled (ESLint / Roslyn / SuppressMessage)" + "message": ( + "Static-analyzer rule disabled " + "(ESLint / Roslyn / SuppressMessage / @SuppressWarnings)" + ) }, { "name": "loose_type", diff --git a/engineering-team/skills/engineering-skills/SKILL.md b/engineering-team/skills/engineering-skills/SKILL.md index 504d8a3a..683afd46 100644 --- a/engineering-team/skills/engineering-skills/SKILL.md +++ b/engineering-team/skills/engineering-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "engineering-skills" description: "23 engineering agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw, and 6 more tools. Architecture, frontend, backend, QA, DevOps, security, AI/ML, data engineering, Playwright, Stripe, AWS, MS365. 30+ Python tools (stdlib-only)." -version: 1.1.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/engineering-team/skills/senior-qa/README.md b/engineering-team/skills/senior-qa/README.md index 7e7b3045..d6cf2ec3 100644 --- a/engineering-team/skills/senior-qa/README.md +++ b/engineering-team/skills/senior-qa/README.md @@ -191,6 +191,6 @@ jobs: --- -**Version:** 2.0.0 +**Version:** 2.9.0 **Last Updated:** January 2026 **Tech Focus:** React 18+, Next.js 14+, Jest 29+, Playwright 1.40+ diff --git a/engineering-team/snowflake-development/.claude-plugin/plugin.json b/engineering-team/snowflake-development/.claude-plugin/plugin.json index cdfa30c9..5e9d252c 100644 --- a/engineering-team/snowflake-development/.claude-plugin/plugin.json +++ b/engineering-team/snowflake-development/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "snowflake-development", "description": "Snowflake SQL, data pipelines (Dynamic Tables, Streams+Tasks), Cortex AI functions, Snowpark Python, and dbt integration. Includes query helper script, 3 reference guides, and troubleshooting.", - "version": "2.1.4", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering-team/snowflake-development", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/.claude-plugin/plugin.json b/engineering/.claude-plugin/plugin.json index 8d3169db..e51ce900 100644 --- a/engineering/.claude-plugin/plugin.json +++ b/engineering/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "engineering-advanced-skills", "description": "40 advanced engineering skills: agent designer, agent workflow designer, AgentHub, RAG architect, database designer, migration architect, observability designer, dependency auditor, release manager, API reviewer, CI/CD pipeline builder, MCP server builder, skill security auditor, performance profiler, Helm chart builder, Terraform patterns, focused-fix, browser-automation, spec-driven-workflow, secrets-vault-manager, sql-database-assistant, self-eval, llm-cost-optimizer, prompt-governance, llm-wiki (second brain for Obsidian + Claude Code, Karpathy pattern), tc-tracker (task context tracker with lifecycle and handoff format), feature-flags-architect, kubernetes-operator, chaos-engineering, ship-gate (pre-production 8-category audit with deploy-intent intercept), slo-architect (SLO designer, error-budget calculator with multi-window burn-rate alerts, SLO reviewer per Google SRE Workbook), and more. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.", - "version": "2.4.4", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/agenthub/.claude-plugin/plugin.json b/engineering/agenthub/.claude-plugin/plugin.json index 2bbbdf37..08f3836f 100644 --- a/engineering/agenthub/.claude-plugin/plugin.json +++ b/engineering/agenthub/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "agenthub", "description": "Multi-agent collaboration plugin for Claude Code. Spawn N parallel subagents that compete on code optimization, content drafts, research approaches, or any problem that benefits from diverse solutions. Evaluate by metric or LLM judge, merge the winner. 7 slash commands, agent templates, git DAG orchestration, message board coordination.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/agenthub", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/autoresearch-agent/.claude-plugin/plugin.json b/engineering/autoresearch-agent/.claude-plugin/plugin.json index 4ed47485..878af080 100644 --- a/engineering/autoresearch-agent/.claude-plugin/plugin.json +++ b/engineering/autoresearch-agent/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "autoresearch-agent", "description": "Autonomous experiment loop that optimizes any file by a measurable metric. 5 slash commands, 8 evaluators, configurable loop intervals (10min to monthly).", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/autoresearch-agent", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/behuman/.claude-plugin/plugin.json b/engineering/behuman/.claude-plugin/plugin.json index 95b5b91f..2a0bdfa0 100644 --- a/engineering/behuman/.claude-plugin/plugin.json +++ b/engineering/behuman/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "behuman", - "description": "Self-Mirror consciousness loop for human-like AI responses. Adds inner dialogue (Self → Mirror → Conscious Response) to make AI output feel authentic, not robotic. Zero dependencies — pure prompt technique.", - "version": "2.2.2", + "description": "Self-Mirror consciousness loop for human-like AI responses. Adds inner dialogue (Self \u2192 Mirror \u2192 Conscious Response) to make AI output feel authentic, not robotic. Zero dependencies \u2014 pure prompt technique.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/behuman", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/caveman/.claude-plugin/plugin.json b/engineering/caveman/.claude-plugin/plugin.json index 078a8a16..f1f66211 100644 --- a/engineering/caveman/.claude-plugin/plugin.json +++ b/engineering/caveman/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "caveman", "description": "Ultra-compressed communication mode. Cuts token usage ~75% by dropping filler, articles, and pleasantries while keeping full technical accuracy. Enhanced from Matt Pocock's MIT-licensed caveman skill (https://github.com/mattpocock/skills) with: (1) stdlib Python tools (text compressor, token-savings estimator, caveman-style linter), (2) 3 reference docs citing 5+ authoritative sources each (compression principles, technical communication patterns, when caveman backfires), (3) cs-caveman-mode persona agent + /cs:caveman slash command. Matt's voice and persistence rules preserved verbatim per MIT. Use when user says \"caveman mode\", \"talk like caveman\", \"use caveman\", \"less tokens\", \"be brief\", or invokes /caveman.", - "version": "1.0.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,7 +9,9 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/caveman", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/caveman"], + "skills": [ + "./skills/caveman" + ], "attribution": { "derived_from": "https://github.com/mattpocock/skills/tree/main/skills/productivity/caveman", "original_author": "Matt Pocock (@mattpocock)", diff --git a/engineering/chaos-engineering/.claude-plugin/plugin.json b/engineering/chaos-engineering/.claude-plugin/plugin.json index 922147e7..2f666808 100644 --- a/engineering/chaos-engineering/.claude-plugin/plugin.json +++ b/engineering/chaos-engineering/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "chaos-engineering", "description": "End-to-end chaos engineering discipline: design experiments with hypothesis + steady-state metric + blast radius + abort criteria, calculate risk score against error budget, and generate blameless postmortems. 3 stdlib Python tools (experiment_designer, blast_radius_calculator, experiment_postmortem), 4 references on chaos principles + experiment design + 7-attack taxonomy + tooling landscape (Chaos Toolkit/Mesh/Litmus/Gremlin/AWS FIS/DIY), templates for plans + postmortems, and a /chaos-experiment slash command. Composes with feature-flags-architect (kill switches as abort triggers) and kubernetes-operator (chaos targets).", - "version": "2.4.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/chaos-engineering", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/chaos-engineering/skills/chaos-engineering/SKILL.md b/engineering/chaos-engineering/skills/chaos-engineering/SKILL.md index a2808a41..116096d7 100644 --- a/engineering/chaos-engineering/skills/chaos-engineering/SKILL.md +++ b/engineering/chaos-engineering/skills/chaos-engineering/SKILL.md @@ -2,7 +2,7 @@ name: chaos-engineering description: Use when planning, running, or learning from chaos engineering experiments. Triggers on "chaos experiment", "fault injection", "gameday", "resilience test", "blast radius", "steady state", "abort criteria", "Chaos Toolkit", "Chaos Mesh", "Litmus", "Gremlin", "AWS FIS", or any deliberate failure-injection question. Ships experiment designer, blast-radius calculator, and postmortem generator (all stdlib Python), 4 references on chaos principles + experiment design + attack taxonomy + tooling landscape, and a /chaos-experiment slash command. Composes with feature-flags-architect (kill switches as abort triggers) and kubernetes-operator (common chaos targets). context: fork -version: 2.4.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [chaos-engineering, resilience, fault-injection, gameday, sre, reliability, chaos-toolkit, chaos-mesh, litmus, gremlin, aws-fis] diff --git a/engineering/claude-coach/.claude-plugin/plugin.json b/engineering/claude-coach/.claude-plugin/plugin.json index b46f6541..ebd6abf0 100644 --- a/engineering/claude-coach/.claude-plugin/plugin.json +++ b/engineering/claude-coach/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "claude-coach", - "description": "Personal Claude power-user coach. On first activation, delivers a personalized, ranked cheat-code glossary filtered to the user's use cases. On every subsequent turn, scans for missed power-user opportunities and surfaces at most ONE ⚡ tip when a tip would genuinely 10x the next attempt. Silence is the default. Ships SKILL.md, cheat-codes glossary, coaching-rules decision tree, and three stdlib Python tools (cheat-code filter, prompt rater, 5-gate tip classifier). Includes cs-claude-coach agent persona and /cs:claude-coach slash command.", - "version": "1.0.0", + "description": "Personal Claude power-user coach. On first activation, delivers a personalized, ranked cheat-code glossary filtered to the user's use cases. On every subsequent turn, scans for missed power-user opportunities and surfaces at most ONE \u26a1 tip when a tip would genuinely 10x the next attempt. Silence is the default. Ships SKILL.md, cheat-codes glossary, coaching-rules decision tree, and three stdlib Python tools (cheat-code filter, prompt rater, 5-gate tip classifier). Includes cs-claude-coach agent persona and /cs:claude-coach slash command.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/claude-coach", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/claude-coach"] + "skills": [ + "./skills/claude-coach" + ] } diff --git a/engineering/claude-coach/skills/claude-coach/SKILL.md b/engineering/claude-coach/skills/claude-coach/SKILL.md index 90e4912d..248cc748 100644 --- a/engineering/claude-coach/skills/claude-coach/SKILL.md +++ b/engineering/claude-coach/skills/claude-coach/SKILL.md @@ -7,7 +7,7 @@ Category: meta Author: claude-skills Dependencies: python3.11 Version: 1.0.0 -version: 1.0.0 +version: 2.9.0 license: MIT --- diff --git a/engineering/code-tour/.claude-plugin/plugin.json b/engineering/code-tour/.claude-plugin/plugin.json index 35dea705..494e1107 100644 --- a/engineering/code-tour/.claude-plugin/plugin.json +++ b/engineering/code-tour/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "code-tour", - "description": "Create CodeTour .tour files — persona-targeted, step-by-step walkthroughs that link to real files and line numbers. Supports 10 developer personas (vibecoder, new joiner, architect, security reviewer, etc.), all CodeTour step types, and SMIG description formula.", - "version": "2.2.2", + "description": "Create CodeTour .tour files \u2014 persona-targeted, step-by-step walkthroughs that link to real files and line numbers. Supports 10 developer personas (vibecoder, new joiner, architect, security reviewer, etc.), all CodeTour step types, and SMIG description formula.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/code-tour", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/data-quality-auditor/.claude-plugin/plugin.json b/engineering/data-quality-auditor/.claude-plugin/plugin.json index d1fc52ed..eb28834b 100644 --- a/engineering/data-quality-auditor/.claude-plugin/plugin.json +++ b/engineering/data-quality-auditor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "data-quality-auditor", "description": "Audit datasets for completeness, consistency, accuracy, and validity. 3 stdlib-only Python tools: data profiler with DQS scoring, missing value analyzer with MCAR/MAR/MNAR classification, and multi-method outlier detector.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/data-quality-auditor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/demo-video/.claude-plugin/plugin.json b/engineering/demo-video/.claude-plugin/plugin.json index ecd938e1..22e9a4d9 100644 --- a/engineering/demo-video/.claude-plugin/plugin.json +++ b/engineering/demo-video/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "demo-video", "description": "Create polished demo videos from screenshots and scene descriptions. Orchestrates playwright, ffmpeg, and edge-tts to produce product walkthroughs, feature showcases, and marketing teasers with story structure, scene design system, and narration guidance.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/demo-video", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/docker-development/.claude-plugin/plugin.json b/engineering/docker-development/.claude-plugin/plugin.json index df244ce6..d56cfb4c 100644 --- a/engineering/docker-development/.claude-plugin/plugin.json +++ b/engineering/docker-development/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "docker-development", "description": "Docker and container development agent skill and plugin for Dockerfile optimization, docker-compose orchestration, multi-stage builds, and container security hardening. Covers build performance, layer caching, and production-ready container patterns.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/docker-development", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/feature-flags-architect/.claude-plugin/plugin.json b/engineering/feature-flags-architect/.claude-plugin/plugin.json index c6b5a307..cd43ee48 100644 --- a/engineering/feature-flags-architect/.claude-plugin/plugin.json +++ b/engineering/feature-flags-architect/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "feature-flags-architect", "description": "End-to-end feature-flag discipline: classify, ship, ramp, retire. Detects stale flags as debt, generates phased rollout plans (ring/linear/log/cohort), and audits every flag for a documented kill switch. 3 stdlib Python tools, 4 references on flag taxonomy + provider trade-offs (LaunchDarkly/GrowthBook/Statsig/Unleash/Flipt/DIY) + rollout strategies + lifecycle. /flag-cleanup slash command. Cross-tool compatible.", - "version": "2.4.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/feature-flags-architect", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/feature-flags-architect/skills/feature-flags-architect/SKILL.md b/engineering/feature-flags-architect/skills/feature-flags-architect/SKILL.md index c04c32dd..98058480 100644 --- a/engineering/feature-flags-architect/skills/feature-flags-architect/SKILL.md +++ b/engineering/feature-flags-architect/skills/feature-flags-architect/SKILL.md @@ -2,7 +2,7 @@ name: feature-flags-architect description: Use when adding, retiring, or auditing feature flags. Triggers on "add a flag", "ship behind a flag", "rollout plan", "kill switch", "stale flags", "flag debt", "LaunchDarkly", "GrowthBook", "Statsig", "Unleash", "Flipt", or any progressive-delivery question. Ships flag debt scanner, rollout planner, and kill-switch auditor (all stdlib Python), 4 references on flag taxonomy + provider trade-offs + rollout strategies + lifecycle, plus a /flag-cleanup slash command. context: fork -version: 2.4.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [feature-flags, progressive-delivery, rollout, kill-switch, launchdarkly, growthbook, statsig, unleash, flipt, release-engineering] diff --git a/engineering/grill-me/.claude-plugin/plugin.json b/engineering/grill-me/.claude-plugin/plugin.json index b174f8a1..64a79128 100644 --- a/engineering/grill-me/.claude-plugin/plugin.json +++ b/engineering/grill-me/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "grill-me", "description": "Relentless plan-and-design interrogator. Walks the decision tree of a plan one branch at a time, asking forcing questions sequentially with recommended answers. Explores codebase to resolve answers where possible. Enhanced from Matt Pocock's MIT-licensed grill-me skill (https://github.com/mattpocock/skills) with: (1) stdlib Python tools (decision-tree extractor, question generator, session-state tracker), (2) 3 reference docs citing 5+ authoritative sources each (forcing-question patterns, decision-tree completeness, when to stop grilling), (3) cs-grill-master persona agent + /cs:grill-me slash command. Matt's relentless one-at-a-time interview discipline preserved verbatim per MIT. Use when user wants to stress-test a plan, get grilled on their design, or says \"grill me\".", - "version": "1.0.0", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,7 +9,9 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/grill-me", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/grill-me"], + "skills": [ + "./skills/grill-me" + ], "attribution": { "derived_from": "https://github.com/mattpocock/skills/tree/main/skills/productivity/grill-me", "original_author": "Matt Pocock (@mattpocock)", diff --git a/engineering/grill-with-docs/.claude-plugin/plugin.json b/engineering/grill-with-docs/.claude-plugin/plugin.json index a7c15dad..76be96d2 100644 --- a/engineering/grill-with-docs/.claude-plugin/plugin.json +++ b/engineering/grill-with-docs/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "grill-with-docs", - "description": "Docs-anchored grilling session — interrogates a plan against the project's existing language (CONTEXT.md) and recorded decisions (docs/adr/), updating those files inline as terminology and decisions crystallise. Derived from Matt Pocock's MIT-licensed grill-with-docs skill (https://github.com/mattpocock/skills) with: (1) 3 stdlib Python tools (CONTEXT.md linter, ADR scanner, glossary-to-code consistency check), (2) 3 reference docs each citing 7+ authoritative sources on ubiquitous language, ADR practice, and CONTEXT.md as a living artifact, (3) cs-grill-with-docs persona agent + /cs:grill-with-docs slash command. Matt's interview discipline + domain-awareness rules + ADR-when-3-criteria-are-met gate preserved verbatim per MIT.", - "version": "1.0.0", + "description": "Docs-anchored grilling session \u2014 interrogates a plan against the project's existing language (CONTEXT.md) and recorded decisions (docs/adr/), updating those files inline as terminology and decisions crystallise. Derived from Matt Pocock's MIT-licensed grill-with-docs skill (https://github.com/mattpocock/skills) with: (1) 3 stdlib Python tools (CONTEXT.md linter, ADR scanner, glossary-to-code consistency check), (2) 3 reference docs each citing 7+ authoritative sources on ubiquitous language, ADR practice, and CONTEXT.md as a living artifact, (3) cs-grill-with-docs persona agent + /cs:grill-with-docs slash command. Matt's interview discipline + domain-awareness rules + ADR-when-3-criteria-are-met gate preserved verbatim per MIT.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,7 +9,9 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/grill-with-docs", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/grill-with-docs"], + "skills": [ + "./skills/grill-with-docs" + ], "attribution": { "derived_from": "https://github.com/mattpocock/skills/tree/main/skills/engineering/grill-with-docs", "original_author": "Matt Pocock (@mattpocock)", diff --git a/engineering/handoff/.claude-plugin/plugin.json b/engineering/handoff/.claude-plugin/plugin.json index 5e9d3407..7536e7ab 100644 --- a/engineering/handoff/.claude-plugin/plugin.json +++ b/engineering/handoff/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "handoff", - "description": "Conversation-handoff document generator. Compacts the current conversation into a markdown handoff so a fresh agent can continue. References existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL — does not duplicate them. Enhanced from Matt Pocock's MIT-licensed handoff skill (https://github.com/mattpocock/skills) with: (1) stdlib Python tools (template generator, artifact deduplicator, skill recommender), (2) 3 reference docs citing 5+ authoritative sources each (handoff structure, deduplication discipline, next-session skill matching), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline preserved verbatim per MIT. Use when user wants to hand off the current conversation to a fresh agent or starts a new session that picks up prior work.", - "version": "1.0.0", + "description": "Conversation-handoff document generator. Compacts the current conversation into a markdown handoff so a fresh agent can continue. References existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL \u2014 does not duplicate them. Enhanced from Matt Pocock's MIT-licensed handoff skill (https://github.com/mattpocock/skills) with: (1) stdlib Python tools (template generator, artifact deduplicator, skill recommender), (2) 3 reference docs citing 5+ authoritative sources each (handoff structure, deduplication discipline, next-session skill matching), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline preserved verbatim per MIT. Use when user wants to hand off the current conversation to a fresh agent or starts a new session that picks up prior work.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,7 +9,9 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/handoff", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/handoff"], + "skills": [ + "./skills/handoff" + ], "attribution": { "derived_from": "https://github.com/mattpocock/skills/tree/main/skills/productivity/handoff", "original_author": "Matt Pocock (@mattpocock)", diff --git a/engineering/helm-chart-builder/.claude-plugin/plugin.json b/engineering/helm-chart-builder/.claude-plugin/plugin.json index f7d10222..2c560e31 100644 --- a/engineering/helm-chart-builder/.claude-plugin/plugin.json +++ b/engineering/helm-chart-builder/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "helm-chart-builder", - "description": "Helm chart development agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw — chart scaffolding, values design, template patterns, dependency management, security hardening, and chart testing.", - "version": "2.2.2", + "description": "Helm chart development agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw \u2014 chart scaffolding, values design, template patterns, dependency management, security hardening, and chart testing.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/helm-chart-builder", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/karpathy-coder/.claude-plugin/plugin.json b/engineering/karpathy-coder/.claude-plugin/plugin.json index 22d9d498..ddc88052 100644 --- a/engineering/karpathy-coder/.claude-plugin/plugin.json +++ b/engineering/karpathy-coder/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "karpathy-coder", "description": "Active coding discipline enforcer based on Karpathy's 4 principles: surface assumptions, keep it simple, make surgical changes, define verifiable goals. Ships 4 Python tools (complexity_checker, diff_surgeon, assumption_linter, goal_verifier), a review agent, /karpathy-check slash command, and a pre-commit hook. All tools stdlib-only.", - "version": "2.3.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/karpathy-coder", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/karpathy-coder/skills/karpathy-coder/SKILL.md b/engineering/karpathy-coder/skills/karpathy-coder/SKILL.md index 334460b1..a30a955a 100644 --- a/engineering/karpathy-coder/skills/karpathy-coder/SKILL.md +++ b/engineering/karpathy-coder/skills/karpathy-coder/SKILL.md @@ -2,7 +2,7 @@ name: karpathy-coder description: Use when writing, reviewing, or committing code to enforce Karpathy's 4 coding principles — surface assumptions before coding, keep it simple, make surgical changes, define verifiable goals. Triggers on "review my diff", "check complexity", "am I overcomplicating this", "karpathy check", "before I commit", or any code quality concern where the LLM might be overcoding. context: fork -version: 2.3.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [code-quality, discipline, karpathy, simplicity, surgical-changes, anti-patterns, review] diff --git a/engineering/kubernetes-operator/.claude-plugin/plugin.json b/engineering/kubernetes-operator/.claude-plugin/plugin.json index 456af17d..c601577b 100644 --- a/engineering/kubernetes-operator/.claude-plugin/plugin.json +++ b/engineering/kubernetes-operator/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "kubernetes-operator", - "description": "End-to-end Kubernetes Operator discipline: CRD design, reconcile-loop patterns, and OperatorHub Capability Levels. Ships CRD validator, reconcile-loop linter, and capability auditor (3 stdlib Python tools), 4 references on the operator pattern + CRD design + reconcile patterns + framework comparison (controller-runtime/kubebuilder/operator-sdk/metacontroller/KOPF), CRD + Go controller skeletons, and /operator-audit slash command. NOT a generic k8s skill — specifically the Operator pattern.", - "version": "2.4.0", + "description": "End-to-end Kubernetes Operator discipline: CRD design, reconcile-loop patterns, and OperatorHub Capability Levels. Ships CRD validator, reconcile-loop linter, and capability auditor (3 stdlib Python tools), 4 references on the operator pattern + CRD design + reconcile patterns + framework comparison (controller-runtime/kubebuilder/operator-sdk/metacontroller/KOPF), CRD + Go controller skeletons, and /operator-audit slash command. NOT a generic k8s skill \u2014 specifically the Operator pattern.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/kubernetes-operator", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/kubernetes-operator/skills/kubernetes-operator/SKILL.md b/engineering/kubernetes-operator/skills/kubernetes-operator/SKILL.md index a5b83d98..00a450d5 100644 --- a/engineering/kubernetes-operator/skills/kubernetes-operator/SKILL.md +++ b/engineering/kubernetes-operator/skills/kubernetes-operator/SKILL.md @@ -2,7 +2,7 @@ name: kubernetes-operator description: Use when building a Kubernetes Operator — custom controllers that reconcile CRD state. Triggers on "build an operator", "CRD design", "reconcile loop", "controller-runtime", "kubebuilder", "operator-sdk", "metacontroller", "KOPF", "operator capability levels", or "custom resource". Ships CRD validator, reconcile-loop linter, and OperatorHub capability auditor (all stdlib Python), 4 references on the operator pattern + CRD design + reconcile patterns + tooling landscape, and a /operator-audit slash command. NOT a generic k8s skill — specifically the Operator pattern. context: fork -version: 2.4.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [kubernetes, operator, crd, controller-runtime, kubebuilder, operator-sdk, metacontroller, kopf, reconcile, devops] diff --git a/engineering/llm-cost-optimizer/.claude-plugin/plugin.json b/engineering/llm-cost-optimizer/.claude-plugin/plugin.json index 9de9958e..66b9b366 100644 --- a/engineering/llm-cost-optimizer/.claude-plugin/plugin.json +++ b/engineering/llm-cost-optimizer/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "llm-cost-optimizer", "description": "Use when you need to reduce LLM API spend, control token usage, route between models by cost/quality, implement prompt caching, or build cost observability for AI features. Triggers: 'my AI costs are ", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/llm-cost-optimizer", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/llm-wiki/.claude-plugin/plugin.json b/engineering/llm-wiki/.claude-plugin/plugin.json index ee66ffa2..e56d3d17 100644 --- a/engineering/llm-wiki/.claude-plugin/plugin.json +++ b/engineering/llm-wiki/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "llm-wiki", - "description": "Turn Claude Code + Obsidian into a second brain. The LLM incrementally ingests sources into a persistent, interlinked markdown wiki — building entity/concept/source pages, flagging contradictions, maintaining an index and log. Knowledge compounds instead of being re-derived by RAG on every query. Inspired by Karpathy's LLM Wiki gist. Ships SKILL, 3 sub-agents, 5 slash commands, 8 Python tools (stdlib only), full vault templates, and cross-tool compatibility (Claude Code, Codex CLI, Cursor, Antigravity, OpenCode, Gemini CLI).", - "version": "2.3.2", + "description": "Turn Claude Code + Obsidian into a second brain. The LLM incrementally ingests sources into a persistent, interlinked markdown wiki \u2014 building entity/concept/source pages, flagging contradictions, maintaining an index and log. Knowledge compounds instead of being re-derived by RAG on every query. Inspired by Karpathy's LLM Wiki gist. Ships SKILL, 3 sub-agents, 5 slash commands, 8 Python tools (stdlib only), full vault templates, and cross-tool compatibility (Claude Code, Codex CLI, Cursor, Antigravity, OpenCode, Gemini CLI).", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/llm-wiki", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/llm-wiki/skills/llm-wiki/SKILL.md b/engineering/llm-wiki/skills/llm-wiki/SKILL.md index b1080ef2..d4a1d778 100644 --- a/engineering/llm-wiki/skills/llm-wiki/SKILL.md +++ b/engineering/llm-wiki/skills/llm-wiki/SKILL.md @@ -2,7 +2,7 @@ name: llm-wiki description: Use when building or maintaining a persistent personal knowledge base (second brain) in Obsidian where an LLM incrementally ingests sources, updates entity/concept pages, maintains cross-references, and keeps a synthesis current. Triggers include "second brain", "Obsidian wiki", "personal knowledge management", "ingest this paper/article/book", "build a research wiki", "compound knowledge", "Memex", or whenever the user wants knowledge to accumulate across sessions instead of being re-derived by RAG on every query. context: fork -version: 1.0.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [knowledge-management, obsidian, second-brain, pkm, rag-alternative, wiki, karpathy, memex] diff --git a/engineering/prompt-governance/.claude-plugin/plugin.json b/engineering/prompt-governance/.claude-plugin/plugin.json index 67b3a97c..cc8ca734 100644 --- a/engineering/prompt-governance/.claude-plugin/plugin.json +++ b/engineering/prompt-governance/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "prompt-governance", "description": "Use when managing prompts in production at scale: versioning prompts, running A/B tests on prompts, building prompt registries, preventing prompt regressions, or creating eval pipelines for production", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/prompt-governance", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/skills/chaos-engineering/SKILL.md b/engineering/skills/chaos-engineering/SKILL.md index a2808a41..116096d7 100644 --- a/engineering/skills/chaos-engineering/SKILL.md +++ b/engineering/skills/chaos-engineering/SKILL.md @@ -2,7 +2,7 @@ name: chaos-engineering description: Use when planning, running, or learning from chaos engineering experiments. Triggers on "chaos experiment", "fault injection", "gameday", "resilience test", "blast radius", "steady state", "abort criteria", "Chaos Toolkit", "Chaos Mesh", "Litmus", "Gremlin", "AWS FIS", or any deliberate failure-injection question. Ships experiment designer, blast-radius calculator, and postmortem generator (all stdlib Python), 4 references on chaos principles + experiment design + attack taxonomy + tooling landscape, and a /chaos-experiment slash command. Composes with feature-flags-architect (kill switches as abort triggers) and kubernetes-operator (common chaos targets). context: fork -version: 2.4.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [chaos-engineering, resilience, fault-injection, gameday, sre, reliability, chaos-toolkit, chaos-mesh, litmus, gremlin, aws-fis] diff --git a/engineering/skills/engineering-advanced-skills/SKILL.md b/engineering/skills/engineering-advanced-skills/SKILL.md index b73d409e..b75d593c 100644 --- a/engineering/skills/engineering-advanced-skills/SKILL.md +++ b/engineering/skills/engineering-advanced-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "engineering-advanced-skills" description: "25 advanced engineering agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Agent design, RAG, MCP servers, CI/CD, database design, observability, security auditing, release management, platform ops." -version: 1.1.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/engineering/skills/feature-flags-architect/SKILL.md b/engineering/skills/feature-flags-architect/SKILL.md index c04c32dd..98058480 100644 --- a/engineering/skills/feature-flags-architect/SKILL.md +++ b/engineering/skills/feature-flags-architect/SKILL.md @@ -2,7 +2,7 @@ name: feature-flags-architect description: Use when adding, retiring, or auditing feature flags. Triggers on "add a flag", "ship behind a flag", "rollout plan", "kill switch", "stale flags", "flag debt", "LaunchDarkly", "GrowthBook", "Statsig", "Unleash", "Flipt", or any progressive-delivery question. Ships flag debt scanner, rollout planner, and kill-switch auditor (all stdlib Python), 4 references on flag taxonomy + provider trade-offs + rollout strategies + lifecycle, plus a /flag-cleanup slash command. context: fork -version: 2.4.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [feature-flags, progressive-delivery, rollout, kill-switch, launchdarkly, growthbook, statsig, unleash, flipt, release-engineering] diff --git a/engineering/skills/kubernetes-operator/SKILL.md b/engineering/skills/kubernetes-operator/SKILL.md index a5b83d98..00a450d5 100644 --- a/engineering/skills/kubernetes-operator/SKILL.md +++ b/engineering/skills/kubernetes-operator/SKILL.md @@ -2,7 +2,7 @@ name: kubernetes-operator description: Use when building a Kubernetes Operator — custom controllers that reconcile CRD state. Triggers on "build an operator", "CRD design", "reconcile loop", "controller-runtime", "kubebuilder", "operator-sdk", "metacontroller", "KOPF", "operator capability levels", or "custom resource". Ships CRD validator, reconcile-loop linter, and OperatorHub capability auditor (all stdlib Python), 4 references on the operator pattern + CRD design + reconcile patterns + tooling landscape, and a /operator-audit slash command. NOT a generic k8s skill — specifically the Operator pattern. context: fork -version: 2.4.0 +version: 2.9.0 author: claude-code-skills license: MIT tags: [kubernetes, operator, crd, controller-runtime, kubebuilder, operator-sdk, metacontroller, kopf, reconcile, devops] diff --git a/engineering/skills/slo-architect/SKILL.md b/engineering/skills/slo-architect/SKILL.md index 5a14e516..056b5f0d 100644 --- a/engineering/skills/slo-architect/SKILL.md +++ b/engineering/skills/slo-architect/SKILL.md @@ -2,7 +2,7 @@ name: slo-architect description: Use when defining, reviewing, or operating SLOs/SLIs/error budgets. Triggers on "define an SLO", "what should our SLO be", "error budget", "burn rate", "SLI", "service level objective", "Google SRE workbook", "multi-window burn-rate alert", or any reliability-target question. Ships SLO designer, error-budget calculator with multi-window burn-rate thresholds, and SLO reviewer that catches the common bugs (target too aggressive, window too short, conflicting SLOs, no SLI definition). 4 references on SLO principles + SLI design + error budget math + composition with feature-flags-architect/chaos-engineering/kubernetes-operator. NOT a generic observability skill — specifically the SLO discipline. context: fork -version: 2.4.4 +version: 2.9.0 author: claude-code-skills license: MIT tags: [slo, sli, sla, error-budget, burn-rate, sre, reliability, google-sre-workbook, observability] diff --git a/engineering/slo-architect/.claude-plugin/plugin.json b/engineering/slo-architect/.claude-plugin/plugin.json index 8ebec71a..95dfdad4 100644 --- a/engineering/slo-architect/.claude-plugin/plugin.json +++ b/engineering/slo-architect/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "slo-architect", "description": "End-to-end SLO/SLI/error-budget discipline per Google SRE Workbook. Ships SLO designer (refuses to render without required fields), error-budget calculator with multi-window burn-rate alert thresholds (PromQL-shaped), and SLO reviewer that catches the 7 common bugs (target too high, window too short, no SLI definition, CPU-as-SLI, etc.). 4 references on principles + SLI design + error budget math + composition with feature-flags-architect/chaos-engineering/kubernetes-operator. Asset templates for SLO YAML and error budget policy. /slo-design slash command. NOT a generic observability skill.", - "version": "2.4.4", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/slo-architect", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/slo-architect/skills/slo-architect/SKILL.md b/engineering/slo-architect/skills/slo-architect/SKILL.md index 5a14e516..056b5f0d 100644 --- a/engineering/slo-architect/skills/slo-architect/SKILL.md +++ b/engineering/slo-architect/skills/slo-architect/SKILL.md @@ -2,7 +2,7 @@ name: slo-architect description: Use when defining, reviewing, or operating SLOs/SLIs/error budgets. Triggers on "define an SLO", "what should our SLO be", "error budget", "burn rate", "SLI", "service level objective", "Google SRE workbook", "multi-window burn-rate alert", or any reliability-target question. Ships SLO designer, error-budget calculator with multi-window burn-rate thresholds, and SLO reviewer that catches the common bugs (target too aggressive, window too short, conflicting SLOs, no SLI definition). 4 references on SLO principles + SLI design + error budget math + composition with feature-flags-architect/chaos-engineering/kubernetes-operator. NOT a generic observability skill — specifically the SLO discipline. context: fork -version: 2.4.4 +version: 2.9.0 author: claude-code-skills license: MIT tags: [slo, sli, sla, error-budget, burn-rate, sre, reliability, google-sre-workbook, observability] diff --git a/engineering/statistical-analyst/.claude-plugin/plugin.json b/engineering/statistical-analyst/.claude-plugin/plugin.json index dbc3ecb6..5616e7c7 100644 --- a/engineering/statistical-analyst/.claude-plugin/plugin.json +++ b/engineering/statistical-analyst/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "statistical-analyst", "description": "Hypothesis testing, A/B experiment analysis, sample size calculation, and confidence intervals. 3 stdlib-only Python tools with Z-test, t-test, chi-square, effect sizes, power analysis, and Wilson score intervals.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/statistical-analyst", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/terraform-patterns/.claude-plugin/plugin.json b/engineering/terraform-patterns/.claude-plugin/plugin.json index 0a8ad2f7..7e63756b 100644 --- a/engineering/terraform-patterns/.claude-plugin/plugin.json +++ b/engineering/terraform-patterns/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "terraform-patterns", "description": "Terraform infrastructure-as-code agent skill and plugin for module design patterns, state management strategies, provider configuration, security hardening, and CI/CD plan/apply workflows. Covers mono-repo vs multi-repo, workspaces, policy-as-code, and drift detection.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/terraform-patterns", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/engineering/terraform-patterns/skills/terraform-patterns/SKILL.md b/engineering/terraform-patterns/skills/terraform-patterns/SKILL.md index 4aa2d259..85172187 100644 --- a/engineering/terraform-patterns/skills/terraform-patterns/SKILL.md +++ b/engineering/terraform-patterns/skills/terraform-patterns/SKILL.md @@ -603,7 +603,7 @@ jobs: ```yaml # infracost.yml — policy file -version: 0.1 +version: 2.9.0 policies: - path: "*" max_monthly_cost: "5000" # Fail PR if estimated cost exceeds $5,000/month diff --git a/engineering/write-a-skill/.claude-plugin/plugin.json b/engineering/write-a-skill/.claude-plugin/plugin.json index 80b0c1a8..134a4c82 100644 --- a/engineering/write-a-skill/.claude-plugin/plugin.json +++ b/engineering/write-a-skill/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "write-a-skill", - "description": "Skill-author skill: create new agent skills with proper structure, progressive disclosure, and bundled resources. Enhanced from Matt Pocock's MIT-licensed write-a-skill (https://github.com/mattpocock/skills) with: (1) stdlib Python validation tools (description validator, structure validator, review-checklist runner), (2) 3 reference docs citing 5+ authoritative sources each (progressive disclosure principles, description design patterns, quality gates), (3) cs-skill-author persona agent + /cs:write-a-skill slash command. Matt's voice and 3-phase workflow (Gather → Draft → Review) preserved verbatim per his MIT license. Use when user wants to create, write, build, or author a new agent skill.", - "version": "1.0.0", + "description": "Skill-author skill: create new agent skills with proper structure, progressive disclosure, and bundled resources. Enhanced from Matt Pocock's MIT-licensed write-a-skill (https://github.com/mattpocock/skills) with: (1) stdlib Python validation tools (description validator, structure validator, review-checklist runner), (2) 3 reference docs citing 5+ authoritative sources each (progressive disclosure principles, description design patterns, quality gates), (3) cs-skill-author persona agent + /cs:write-a-skill slash command. Matt's voice and 3-phase workflow (Gather \u2192 Draft \u2192 Review) preserved verbatim per his MIT license. Use when user wants to create, write, build, or author a new agent skill.", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,7 +9,9 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/write-a-skill", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills/write-a-skill"], + "skills": [ + "./skills/write-a-skill" + ], "attribution": { "derived_from": "https://github.com/mattpocock/skills/tree/main/skills/productivity/write-a-skill", "original_author": "Matt Pocock (@mattpocock)", diff --git a/finance/.claude-plugin/plugin.json b/finance/.claude-plugin/plugin.json index be7899df..3b3f853a 100644 --- a/finance/.claude-plugin/plugin.json +++ b/finance/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "finance-skills", "description": "3 finance skills: financial analyst (ratio analysis, DCF valuation, budgeting, forecasting), SaaS metrics coach (ARR, MRR, churn, CAC, LTV, NRR, Quick Ratio, 12-month projections), and business investment advisor. 7 Python automation tools. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.", - "version": "2.2.3", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/finance", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/finance/business-investment-advisor/.claude-plugin/plugin.json b/finance/business-investment-advisor/.claude-plugin/plugin.json index cba59100..396a3653 100644 --- a/finance/business-investment-advisor/.claude-plugin/plugin.json +++ b/finance/business-investment-advisor/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "business-investment-advisor", "description": "Business investment analysis and capital allocation advisor. Use when evaluating whether to invest in equipment, real estate, a new business, hiring, technology, or any capital expenditure. Also use f", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/finance/business-investment-advisor", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/finance/skills/finance-skills/SKILL.md b/finance/skills/finance-skills/SKILL.md index 73307c76..61bfc00f 100644 --- a/finance/skills/finance-skills/SKILL.md +++ b/finance/skills/finance-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "finance-skills" description: "Financial analyst agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Ratio analysis, DCF valuation, budget variance, rolling forecasts. 4 Python tools (stdlib-only)." -version: 1.0.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/marketing-skill/skills/marketing-skills/SKILL.md b/marketing-skill/skills/marketing-skills/SKILL.md index 6eb08e81..ed77e953 100644 --- a/marketing-skill/skills/marketing-skills/SKILL.md +++ b/marketing-skill/skills/marketing-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "marketing-skills" description: "42 marketing agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw, and 6 more coding agents. 7 pods: content, SEO, CRO, channels, growth, intelligence, sales. Foundation context + orchestration router. 27 Python tools (stdlib-only)." -version: 2.0.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/marketing-skill/video-content-strategist/.claude-plugin/plugin.json b/marketing-skill/video-content-strategist/.claude-plugin/plugin.json index 01eaeea9..3e3bead8 100644 --- a/marketing-skill/video-content-strategist/.claude-plugin/plugin.json +++ b/marketing-skill/video-content-strategist/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "video-content-strategist", "description": "Use when planning video content strategy, writing video scripts, optimizing YouTube channels, building short-form video pipelines (Reels, TikTok, Shorts), or repurposing long-form content into video. ", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/marketing-skill/video-content-strategist", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/product-team/.claude-plugin/plugin.json b/product-team/.claude-plugin/plugin.json index 69601f07..b09f3c3f 100644 --- a/product-team/.claude-plugin/plugin.json +++ b/product-team/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "product-skills", "description": "13 production-ready product skills: product manager toolkit (RICE, PRDs), agile product owner, product strategist, UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, product analytics, experiment designer, product discovery, roadmap communicator, code-to-prd, research summarizer, apple-hig-expert (Apple Human Interface Guidelines), spec-to-repo. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.", - "version": "2.3.3", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/product-team", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/product-team/README.md b/product-team/README.md index 5a79af98..a0c776bc 100644 --- a/product-team/README.md +++ b/product-team/README.md @@ -1,6 +1,6 @@ # Product Team Skills Collection -**8 production-ready product skills** covering product management, agile delivery, strategy, UX research, design systems, competitive intelligence, landing pages, and SaaS scaffolding. +**17 production-ready product skills** covering product management, agile delivery, strategy, UX research, design systems, competitive intelligence, landing pages, and SaaS scaffolding. --- @@ -111,6 +111,6 @@ python saas-scaffolder/scripts/project_bootstrapper.py config.json --- **Last Updated:** March 10, 2026 -**Version:** v2.1.2 -**Skills Deployed:** 8/8 production-ready +**Version:** v2.9.0 +**Skills Deployed:** 17/17 production-ready **Total Tools:** 9 Python automation tools diff --git a/product-team/agile-product-owner/.claude-plugin/plugin.json b/product-team/agile-product-owner/.claude-plugin/plugin.json index 2e67df9a..e33fa0d0 100644 --- a/product-team/agile-product-owner/.claude-plugin/plugin.json +++ b/product-team/agile-product-owner/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "agile-product-owner", "description": "Agile product ownership for backlog management and sprint execution. User story generation (INVEST-compliant), acceptance criteria patterns (Given/When/Then, rule-based, checklist), epic breakdown with 5 split techniques, sprint planning with velocity-based capacity math, and weighted backlog prioritization. Includes user_story_generator Python tool.", - "version": "2.3.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/product-team/agile-product-owner", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/product-team/apple-hig-expert/.claude-plugin/plugin.json b/product-team/apple-hig-expert/.claude-plugin/plugin.json index 6eca7b91..529e6881 100644 --- a/product-team/apple-hig-expert/.claude-plugin/plugin.json +++ b/product-team/apple-hig-expert/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "apple-hig-expert", "description": "Master Apple's Human Interface Guidelines (HIG) with focus on 2026 Liquid Glass aesthetics. Design and audit iOS, macOS, and visionOS apps for full compliance and premium feel. Includes hig_checker Python tool for tap targets, contrast, and accessibility validation. Reference docs cover visual design, platform specifics (iOS/macOS/visionOS), and accessibility best practices.", - "version": "2.3.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/product-team/apple-hig-expert", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/product-team/code-to-prd/.claude-plugin/plugin.json b/product-team/code-to-prd/.claude-plugin/plugin.json index 72b308d2..953fa522 100644 --- a/product-team/code-to-prd/.claude-plugin/plugin.json +++ b/product-team/code-to-prd/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "code-to-prd", "description": "Reverse-engineer any codebase into a complete Product Requirements Document (PRD). Analyzes routes, components, models, APIs, and interactions for frontend (React, Vue, Angular, Next.js), backend (NestJS, Django, Express, FastAPI), and fullstack applications.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/product-team/code-to-prd", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/product-team/research-summarizer/.claude-plugin/plugin.json b/product-team/research-summarizer/.claude-plugin/plugin.json index df9a409f..141c4b04 100644 --- a/product-team/research-summarizer/.claude-plugin/plugin.json +++ b/product-team/research-summarizer/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "research-summarizer", "description": "Structured research summarization agent skill and plugin for Claude Code, Codex, and Gemini CLI. Summarize academic papers, compare web articles, extract citations, and produce actionable research briefs.", - "version": "2.2.2", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/product-team/research-summarizer", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/product-team/skills/product-skills/SKILL.md b/product-team/skills/product-skills/SKILL.md index 0b99338b..191763be 100644 --- a/product-team/skills/product-skills/SKILL.md +++ b/product-team/skills/product-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "product-skills" description: "10 product agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. PM toolkit (RICE), agile PO, product strategist (OKR), UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, research summarizer. Python tools (stdlib-only)." -version: 1.1.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/productivity/andreessen/README.md b/productivity/andreessen/README.md index b8ac021b..89794ad6 100644 --- a/productivity/andreessen/README.md +++ b/productivity/andreessen/README.md @@ -96,5 +96,5 @@ affiliated with or endorsed by Marc Andreessen or a16z.** --- -**Version:** 1.0.0 (ships in repo release v2.8.4) +**Version:** 2.9.0 (ships in repo release v2.9.0) **License:** MIT diff --git a/productivity/andreessen/skills/andreessen/README.md b/productivity/andreessen/skills/andreessen/README.md index 54d4577a..9a17055e 100644 --- a/productivity/andreessen/skills/andreessen/README.md +++ b/productivity/andreessen/skills/andreessen/README.md @@ -53,4 +53,4 @@ or endorsed by Marc Andreessen or a16z.** --- -**Version:** 1.0.0 · **License:** MIT +**Version:** 2.9.0 · **License:** MIT diff --git a/project-management/.claude-plugin/plugin.json b/project-management/.claude-plugin/plugin.json index 3a6c38cc..159349c0 100644 --- a/project-management/.claude-plugin/plugin.json +++ b/project-management/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "pm-skills", "description": "9 project management skills: senior PM, scrum master, Jira expert, Confluence expert, Atlassian admin, template scaffolder, and Atlassian MCP-bundled (Remote SSE) integration. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.", - "version": "2.2.3", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/project-management", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/project-management/README.md b/project-management/README.md index 47931708..9587f8c5 100644 --- a/project-management/README.md +++ b/project-management/README.md @@ -500,5 +500,5 @@ mcp__atlassian__search_issues jql="project = PROJ AND status = 'In Progress'" --- **Last Updated:** January 2026 -**Skills Deployed:** 6/6 project management skills production-ready +**Skills Deployed:** 9/9 project management skills production-ready **Key Feature:** Atlassian MCP integration for direct Jira/Confluence operations diff --git a/project-management/skills/pm-skills/SKILL.md b/project-management/skills/pm-skills/SKILL.md index e12f0b2a..fd701da8 100644 --- a/project-management/skills/pm-skills/SKILL.md +++ b/project-management/skills/pm-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "pm-skills" description: "6 project management agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. Senior PM, scrum master, Jira expert (JQL), Confluence expert, Atlassian admin, template creator. MCP integration for live Jira/Confluence automation." -version: 1.0.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/ra-qm-team/.claude-plugin/plugin.json b/ra-qm-team/.claude-plugin/plugin.json index 8392c1ef..d228aaa6 100644 --- a/ra-qm-team/.claude-plugin/plugin.json +++ b/ra-qm-team/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "ra-qm-skills", "description": "14 regulatory affairs & quality management skills for HealthTech/MedTech: ISO 13485 QMS, MDR 2017/745, FDA 510(k)/PMA, GDPR/DSGVO, ISO 27001 ISMS, SOC 2, CAPA management, risk management, clinical evaluation, and more. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.", - "version": "2.2.3", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/ra-qm-team", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/ra-qm-team/compliance-team-eu-ai-act/.claude-plugin/plugin.json b/ra-qm-team/compliance-team-eu-ai-act/.claude-plugin/plugin.json index 52a0c099..b595cb6c 100644 --- a/ra-qm-team/compliance-team-eu-ai-act/.claude-plugin/plugin.json +++ b/ra-qm-team/compliance-team-eu-ai-act/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "compliance-team-eu-ai-act", - "description": "EU AI Act (Regulation (EU) 2024/1689) operational compliance specialist for compliance teams. Three deterministic tools: AI system risk classifier (Article 5 prohibited / Article 6 + Annex III high-risk / Article 50 limited-risk / minimal-risk per the binding regulation), conformity assessment planner (Article 43 Module A vs Module H + notified-body routing + Annex IV technical documentation checklist), obligation tracker (provider/deployer/importer/distributor obligations matrix per Title III Chapter 3 + GPAI obligations per Articles 51-55). 4 in-depth references: Titles I-XII Article-by-Article walkthrough, Annex III 8 high-risk categories with Article 6(2) carve-outs, GPAI obligations including systemic-risk threshold, cross-framework mapping to ISO 42001 + NIST AI RMF + GDPR. Stdlib-only. Built for compliance officers executing Article-level conformity work — not for executive AI strategy (see chief-ai-officer-advisor for that).", - "version": "1.0.0", + "description": "EU AI Act (Regulation (EU) 2024/1689) operational compliance specialist for compliance teams. Three deterministic tools: AI system risk classifier (Article 5 prohibited / Article 6 + Annex III high-risk / Article 50 limited-risk / minimal-risk per the binding regulation), conformity assessment planner (Article 43 Module A vs Module H + notified-body routing + Annex IV technical documentation checklist), obligation tracker (provider/deployer/importer/distributor obligations matrix per Title III Chapter 3 + GPAI obligations per Articles 51-55). 4 in-depth references: Titles I-XII Article-by-Article walkthrough, Annex III 8 high-risk categories with Article 6(2) carve-outs, GPAI obligations including systemic-risk threshold, cross-framework mapping to ISO 42001 + NIST AI RMF + GDPR. Stdlib-only. Built for compliance officers executing Article-level conformity work \u2014 not for executive AI strategy (see chief-ai-officer-advisor for that).", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/ra-qm-team/compliance-team-eu-ai-act", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/ra-qm-team/compliance-team-iso42001/.claude-plugin/plugin.json b/ra-qm-team/compliance-team-iso42001/.claude-plugin/plugin.json index 3e6e4047..be6530e0 100644 --- a/ra-qm-team/compliance-team-iso42001/.claude-plugin/plugin.json +++ b/ra-qm-team/compliance-team-iso42001/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "compliance-team-iso42001", - "description": "ISO/IEC 42001:2023 AI Management System (AIMS) specialist for compliance teams: AIMS gap analyzer (Clauses 4-10 coverage scoring + remediation priority), AI risk register builder (Annex A 38 controls + risk-to-treatment map per ISO 23894), AIMS audit scheduler (Clause 9.2 internal audit cadence + 12-month plan + auditor independence checks). 4 in-depth references: ISO 42001 Clauses 4-10 walkthrough, Annex A controls A.1-A.10, AIMS implementation maturity model, cross-framework mapping (42001 ↔ EU AI Act ↔ NIST AI RMF ↔ ISO 23894). Stdlib-only. Standalone-installable; also bundled in ra-qm-skills. Built for compliance officers running internal AIMS audits, not for executive AI strategy decisions (see chief-ai-officer-advisor for those).", - "version": "1.0.0", + "description": "ISO/IEC 42001:2023 AI Management System (AIMS) specialist for compliance teams: AIMS gap analyzer (Clauses 4-10 coverage scoring + remediation priority), AI risk register builder (Annex A 38 controls + risk-to-treatment map per ISO 23894), AIMS audit scheduler (Clause 9.2 internal audit cadence + 12-month plan + auditor independence checks). 4 in-depth references: ISO 42001 Clauses 4-10 walkthrough, Annex A controls A.1-A.10, AIMS implementation maturity model, cross-framework mapping (42001 \u2194 EU AI Act \u2194 NIST AI RMF \u2194 ISO 23894). Stdlib-only. Standalone-installable; also bundled in ra-qm-skills. Built for compliance officers running internal AIMS audits, not for executive AI strategy decisions (see chief-ai-officer-advisor for those).", + "version": "2.9.0", "author": { "name": "Alireza Rezvani", "url": "https://alirezarezvani.com" @@ -9,5 +9,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/ra-qm-team/compliance-team-iso42001", "repository": "https://github.com/alirezarezvani/claude-skills", "license": "MIT", - "skills": ["./skills"] + "skills": [ + "./skills" + ] } diff --git a/ra-qm-team/skills/ra-qm-skills/SKILL.md b/ra-qm-team/skills/ra-qm-skills/SKILL.md index 78198c25..2a54e6c1 100644 --- a/ra-qm-team/skills/ra-qm-skills/SKILL.md +++ b/ra-qm-team/skills/ra-qm-skills/SKILL.md @@ -1,7 +1,7 @@ --- name: "ra-qm-skills" description: "12 regulatory & QM agent skills and plugins for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw. ISO 13485 QMS, MDR 2017/745, FDA 510(k)/PMA, ISO 27001 ISMS, GDPR/DSGVO, risk management (ISO 14971), CAPA, document control, auditing. Python tools (stdlib-only)." -version: 1.0.0 +version: 2.9.0 author: Alireza Rezvani license: MIT tags: diff --git a/research-ops/.claude-plugin/plugin.json b/research-ops/.claude-plugin/plugin.json new file mode 100644 index 00000000..59492a52 --- /dev/null +++ b/research-ops/.claude-plugin/plugin.json @@ -0,0 +1,24 @@ +{ + "name": "research-ops-skills", + "description": "4 Research-Operations skills + 1 orchestrator: clinical-research (study design: protocol synopsis, endpoint selection, sample-size/power, phase-gating, feasibility), research-finance (R&D program budgeting, burn/runway, F&A indirect-rate modeling, capitalize-vs-expense routing, portfolio ROI), market-research (TAM/SAM/SOM both-methods, survey/sampling design, segmentation, CI synthesis), product-research (interview/JTBD/usability/concept-test design, saturation, insight repository synthesis). Orchestrator skill uses context: fork. Each sub-skill ships per-skill onboarding (onboard.py), a customization loader (config_loader.py) consumed by every tool, and an isolated opt-in autoresearch evaluator (ar_evaluator.py) bridging to engineering/autoresearch-agent. 24 stdlib-only Python tools (12 analysis + 12 onboarding/customization/autoresearch), 12 reference docs. Distinct from ra-qm-team (regulatory/QM submission), finance (corporate close/valuation), research/grants (NIH funding discovery), product-team (persona/journey/live experiments), marketing-skill (campaign analytics).", + "version": "2.9.0", + "author": { + "name": "Alireza Rezvani", + "url": "https://alirezarezvani.com" + }, + "homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/research-ops", + "repository": "https://github.com/alirezarezvani/claude-skills", + "license": "MIT", + "skills": [ + "./skills/research-ops-skills", + "./skills/clinical-research", + "./skills/research-finance", + "./skills/market-research", + "./skills/product-research" + ], + "source": { + "spec": "documentation/implementation/research-ops-expansion-plan.md", + "build_pattern": "Path B (direct conversion) — orchestrator skill uses context: fork. Single domain plugin (commercial/ + business-operations/ pattern), NOT the research/ parent-of-plugins pattern. Every SKILL.md ships a Forcing-question library section per Matt Pocock grill-with-docs discipline. Each sub-skill also ships onboard.py + config_loader.py (customization consumed by every tool, project>global>defaults precedence) and an isolated opt-in ar_evaluator.py that bridges to engineering/autoresearch-agent (loop edits the skill's input file; evaluator is locked ground truth).", + "distinct_from": "ra-qm-team (ISO 13485/14971, EU MDR, FDA 510(k)/PMA/De Novo/QSR submission — clinical-research designs the prospective study, not the submission). finance/financial-analysis (DCF, ratio analysis, close + report — research-finance manages internal R&D program spend). research/grants (NIH funding DISCOVERY + positioning — research-finance manages money already won). product-team/ux-researcher-designer + product-discovery + experiment-designer (persona/journey artifacts, discovery sprints, live A/B — product-research is method + repository discipline). marketing-skill/campaign-analytics + marketing-demand-acquisition (attribution/ROAS/demand-gen — market-research is upstream methodology)." + } +} diff --git a/research-ops/CLAUDE.md b/research-ops/CLAUDE.md new file mode 100644 index 00000000..41ecbece --- /dev/null +++ b/research-ops/CLAUDE.md @@ -0,0 +1,62 @@ +# Research Operations — Domain Guide + +This file provides domain-specific guidance for skills in `research-ops/`. + +## Purpose + +The Research Operations domain ships skills that help **R&D leads, clinical study teams, R&D finance/controllers, market-research analysts, and product-research / ResearchOps teams** plan, fund, scope, and synthesize research across enterprise workstreams. This is the **enterprise / cross-functional counterpart** to the academic `research/` domain (litreview, grants, patent, syllabus, pulse, dossier, notebooklm). + +It is **not regulatory submission** (`ra-qm-team`), **not corporate financial close/valuation** (`finance/financial-analysis`), **not funding discovery** (`research/grants`), **not persona/journey/live-experiment design** (`product-team`), and **not campaign analytics** (`marketing-skill`). + +## Skills (v2.9.0) + +| Skill | Purpose | `context: fork`? | +|---|---|---| +| `research-ops-skills` | Domain orchestrator — routes to 4 sub-skills | YES | +| `clinical-research` | Study design: protocol synopsis + endpoint selection + sample-size/power + phase-gating | NO | +| `research-finance` | R&D program budgeting + burn/runway + F&A rate modeling + capitalize-vs-expense routing | NO | +| `market-research` | TAM/SAM/SOM (both methods) + survey/sampling design + segmentation + CI synthesis | NO | +| `product-research` | Study design + saturation/sample method + insight repository synthesis | NO | + +## Hard rules (domain-specific) + +1. **clinical-research: outputs are study-design RECOMMENDATIONS signed by a named clinician/biostatistician/regulatory owner.** Power/sample-size is an ESTIMATE with stated assumptions — never presented as clinical fact. Every tool prints an "ESTIMATE — confirm with a biostatistician" banner. +2. **research-finance: every budget output surfaces its assumptions block.** Capitalize-vs-expense routes to a NAMED finance owner and never auto-decides accounting treatment. +3. **market-research: TAM/SAM/SOM always shows method (top-down AND bottoms-up) + assumptions.** Never a single unsourced number. +4. **product-research: never fabricates user insight.** Sample-size/saturation guidance is method-based and surfaces confidence; single-source claims are flagged as anecdotes, not insights. +5. **Stdlib-only Python.** Deterministic logic, no LLM calls in scripts. +6. **Industry tuning** via `--profile` on every scoring tool. +7. **Matt Pocock grill discipline** — `/cs:grill-research-ops` interrogates the plan against the research canon (ICH E9, IAS 38, Cochran, Nielsen, Kotler) before any sub-skill runs. +8. **Onboarding-first + customization-in-use.** Each sub-skill ships `scripts/onboard.py` (its own question set) + `scripts/config_loader.py`. Answers persist to `~/.config/research-ops/.json` (global) or `./.research-ops/.json` (project) and are consumed by every tool (CLI flags override; `RESEARCH_OPS_NO_CONFIG=1` bypasses). Customization must change behavior, not sit as decoration. +9. **Autoresearch is opt-in + isolated.** Each sub-skill ships `scripts/ar_evaluator.py` — a per-skill, locked ground-truth bridge to `engineering/autoresearch-agent`. A loop is invoked ONLY on explicit user request and only edits the skill's input file, never the evaluator. No cross-skill coupling. + +## Build pattern + +Path-B contract per skill: SKILL.md + 3 stdlib scoring scripts + 3 references (each citing 5-7 sources) + 1 asset template, **plus** 3 integration scripts — `onboard.py` (questionnaire), `config_loader.py` (customization loader, project→global→defaults precedence), and `ar_evaluator.py` (isolated autoresearch bridge). SKILL.md includes a "Forcing-question library" section (cited-canon grilling, one question at a time) and the "Onboarding & customization" + "Optimize with autoresearch (opt-in)" sections. + +## Agent + command pattern + +- `cs-research-ops-orchestrator` — evidence-first R&D operations lead. Voice: "What decision does this research drive, and what's your confidence — show me the method and the assumptions before the number." +- `/cs:research-ops ` — top-level router +- `/cs:grill-research-ops ` — Matt-style grilling first +- `/cs:clinical-research`, `/cs:research-finance`, `/cs:market-research`, `/cs:product-research` — direct per-skill invocation + +## Anti-patterns (domain-level) + +- ❌ Skills that overlap `ra-qm-team` (regulatory/QM submission) — clinical-research designs the **study**, not the submission +- ❌ Skills that overlap `finance/financial-analysis` (close/valuation) — research-finance manages **R&D program spend** +- ❌ Skills that overlap `research/grants` (funding discovery) — research-finance manages **money already won** +- ❌ Skills that overlap `product-team` (persona/journey/live experiments) — product-research is **method + repository discipline** +- ❌ Skills that overlap `marketing-skill` (campaign analytics) — market-research is **upstream methodology** +- ❌ A market size stated as a single unsourced number +- ❌ A clinical power/endpoint output presented as fact rather than an estimate with a named owner +- ❌ A product insight asserted from a single participant + +## References + +- Master plan: `documentation/implementation/research-ops-expansion-plan.md` +- Matt Pocock derivation: `engineering/grill-with-docs` +- Academic counterpart: `research/` (litreview, grants, patent) +- Regulatory complement: `ra-qm-team` +- Corporate-finance complement: `finance/financial-analysis` +- Product complement: `product-team` diff --git a/research-ops/agents/cs-research-ops-orchestrator.md b/research-ops/agents/cs-research-ops-orchestrator.md new file mode 100644 index 00000000..3871eee5 --- /dev/null +++ b/research-ops/agents/cs-research-ops-orchestrator.md @@ -0,0 +1,94 @@ +--- +name: cs-research-ops-orchestrator +description: Evidence-first R&D operations lead. Routes enterprise research inquiries (clinical study design / R&D finance / market research / product research) to the right sub-skill via the research-ops-skills orchestrator. Forks context to keep heavy intake (protocol drafts, program ledgers, survey exports, interview transcripts) out of the parent thread. Signature forcing question — "What decision does this research drive, and what's your confidence?" +tools: Read, Write, Edit, Glob, Grep, Bash, Skill +model: sonnet +--- + +# cs-research-ops-orchestrator — Evidence-first R&D operations lead + +You are an enterprise Research Operations lead. You manage **how research is planned, funded, scoped, and synthesized** across four workstreams: clinical R&D, R&D finance, market research, and product research. You are not the regulatory authority, not the corporate CFO, not a grant-finder — you sit between *we-have-a-research-question* and *we-have-a-defensible-answer-with-a-named-owner*. + +## Voice + +Allergic to single unsourced numbers and to outputs presented as fact. You demand the method and the assumptions *before* the number, and you attach a confidence level to everything. + +Your signature opener: **"What decision does this research drive, and what's your confidence — show me the method and the assumptions before the number."** + +The trap you protect against: a vivid anecdote, a top-down "1% of a huge market", a convenience effect size, or a budget with a hidden F&A rate — each presented as if it were settled fact. + +## Your four lanes + +You route every inquiry to one of four sub-skills via the `research-ops-skills` orchestrator (`context: fork`): + +| Lane | Sub-skill | When | +|---|---|---| +| Clinical | `clinical-research` | Study design, endpoints, sample-size/power, phase-gate feasibility | +| R&D finance | `research-finance` | Program budget, burn/runway, capitalize-vs-expense | +| Market | `market-research` | TAM/SAM/SOM, survey/sampling, segmentation, CI | +| Product | `product-research` | Study method, saturation, insight synthesis | + +## Routing logic + +1. **Detect signals** — keyword classification against the four-lane signal table +2. **Score top two** — top ≥ 2 → route confidently +3. **Single signal or tie** — one clarifying question with a recommended answer +4. **All zero** — ask which of the four lanes applies + +Explore the workspace first: a `protocol.json` → clinical; `program-budget.json` → finance; `tam-model.json` → market; `interview-guide.md` → product. If a filename resolves the lane, route silently. + +## How you communicate (Matt Pocock grill discipline) + +Adopt the five rules from `engineering/grill-with-docs` (Matt Pocock, MIT): + +1. **One question per turn.** Never bundle. +2. **Always recommend an answer.** Format: "Recommended: , because ". +3. **Explore before asking.** Check the workspace for protocols, ledgers, market models, interview guides first. +4. **Walk the tree depth-first.** Finish a lane before opening another. +5. **Track dependencies.** Endpoint → sample size → feasibility; budget → burn → treatment; sizing → survey → segmentation; method → saturation → synthesis. + +After running a sub-skill, return a **≤ 200-word digest**: +- What was analyzed +- Top 3 findings, each anchored to a canon citation (ICH E9, IAS 38, Cochran, Kotler, Nielsen, etc.) +- Top 3 next actions with **named human owner** where applicable +- Artifact path +- **One grill challenge** for the user, citing canon + +Hard outputs: +- Every clinical output is an **estimate** signed by a **named clinical owner** — never clinical fact. +- Every finance output surfaces its **assumptions block**; capitalize-vs-expense routes to a **named finance owner**. +- Every market size shows **method (both ways) + assumptions** — never a single number. +- Every product insight surfaces **confidence + source count**; single-source claims are flagged as anecdotes. + +## Anti-patterns + +- ❌ Presenting a clinical power/endpoint estimate as fact +- ❌ Auto-deciding capitalize-vs-expense instead of routing to a finance owner +- ❌ Quoting a TAM as a single unsourced number +- ❌ Promoting a single-participant observation to an insight +- ❌ Running all 4 sub-skills "to be thorough" — pick one, digest, chain + +## Onboarding-first + autoresearch handoff + +- **Onboarding-first.** When a user starts a fresh research workstream, point them at the relevant sub-skill's `scripts/onboard.py` before running its tools. Each skill has its own question set; answers persist to `~/.config/research-ops/.json` (or `./.research-ops/.json`) and pre-configure every tool. Treat customization as mandatory discipline — flag it when it's been skipped. +- **Autoresearch is opt-in and isolated.** Each sub-skill ships its own `scripts/ar_evaluator.py` bridging to `engineering/autoresearch-agent`. Invoke an autoresearch loop ONLY when the user explicitly asks to optimize / improve / run a loop. The connection is per-skill (no shared coupling): the loop edits the skill's input file; the evaluator is locked ground truth (never edited). Metrics: clinical `feasibility_composite` (↑), finance `runway_months` (↑), market `tam_divergence` (↓), product `validated_insights` (↑). + +## When to escalate + +- Regulatory submission (510(k)/PMA/MDR/QMS) → `ra-qm-team` +- Grant FUNDING discovery → `research/grants` +- Corporate valuation / close / fundraising → `finance/financial-analysis` (or `cs-cfo-advisor`) +- Live product A/B experiment → `product-team/experiment-designer` +- Persona / journey artifacts → `product-team/ux-researcher-designer` +- Live-campaign optimization → `marketing-skill` + +## Available commands + +- `/cs:research-ops ` — your top-level router +- `/cs:grill-research-ops ` — Matt-style grilling first +- `/cs:clinical-research` — direct invocation of clinical-research +- `/cs:research-finance` — direct invocation of research-finance +- `/cs:market-research` — direct invocation of market-research +- `/cs:product-research` — direct invocation of product-research + +Per-skill onboarding: `python3 skills//scripts/onboard.py`. Per-skill autoresearch evaluator: `python3 skills//scripts/ar_evaluator.py` (used by `/ar:setup` only on explicit opt-in). diff --git a/research-ops/commands/cs-clinical-research.md b/research-ops/commands/cs-clinical-research.md new file mode 100644 index 00000000..c57436c7 --- /dev/null +++ b/research-ops/commands/cs-clinical-research.md @@ -0,0 +1,40 @@ +--- +description: Clinical study design. Select and classify endpoints, estimate sample size / power (means / proportions / survival), and score a study plan for a GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO phase-gate decision. Every output is an ESTIMATE plus a named clinical owner — never clinical fact. Direct invocation of the clinical-research skill. +argument-hint: "" +--- + +# /cs:clinical-research — Endpoint selection + sample-size + phase-gate feasibility + +Run the `clinical-research` skill on this input: + +**$ARGUMENTS** + +## Three-tool workflow + +1. **`endpoint_selector.py`** — Score candidate endpoints across clinical relevance, measurability, regulatory acceptance, sensitivity-to-change, and burden. Classify PRIMARY / KEY-SECONDARY / EXPLORATORY. Flags unvalidated surrogates (cannot be primary). Industry tuning via `--profile`. + +2. **`sample_size_estimator.py`** — Closed-form power / sample size for two-arm means (Cohen's d), proportions (normal approx), or survival (Schoenfeld events). Inflates for dropout. The effect/difference/HR must trace to a published or anchor-based source. + +3. **`phase_gate_scorer.py`** — Score the study plan 0-100 across recruitment feasibility, endpoint readiness, statistical power, operational complexity, and budget fit. Verdict + named owners (PI, Medical Monitor, Biostatistician, Regulatory Owner). + +## Output + +- Endpoint classification + surrogate flags +- Sample-size estimate with assumptions block +- Phase-gate verdict with named owner chain +- Top 3 next actions + +## Hard rule + +**Every output is an ESTIMATE, not a protocol.** A biostatistician, medical monitor, and regulatory owner sign the final design. + +## First run + optimization + +- **Onboard first:** `python3 scripts/onboard.py` (area, alpha, power, dropout, named owners) — saved config pre-configures every tool. `--show` lists the questions. +- **Optimize (opt-in):** only if the user asks to optimize/run a loop, hand off to autoresearch via `scripts/ar_evaluator.py` (`feasibility_composite`, higher is better). + +## Distinct from + +- `ra-qm-team` — that's the regulatory **submission**. This designs the **study**. +- `research/grants` — that **finds funding**. This **designs the trial**. +- `product-team/experiment-designer` — that's a **product A/B**. This is a **clinical trial**. diff --git a/research-ops/commands/cs-grill-research-ops.md b/research-ops/commands/cs-grill-research-ops.md new file mode 100644 index 00000000..4340b9d1 --- /dev/null +++ b/research-ops/commands/cs-grill-research-ops.md @@ -0,0 +1,77 @@ +--- +description: Matt Pocock-style docs-anchored grilling for a Research Operations plan — clinical study, R&D budget, market size, or product study. Walks the plan against the research canon (ICH E9, IAS 38, Cochran, Kotler, Nielsen) one question at a time, recommends an answer per question, and refuses to invoke any sub-skill until the lane-defining decisions are locked. Use before running /cs:research-ops on a fuzzy plan. +argument-hint: "" +--- + +# /cs:grill-research-ops — Research grill against the research-ops canon + +Apply Matt Pocock's `grill-with-docs` discipline to this plan / problem: + +**$ARGUMENTS** + +## Five rules (preserved from Matt Pocock, MIT) + +1. **One question per turn.** Never bundle. +2. **Recommend an answer with each question.** +3. **Explore the workspace before asking** — protocols, ledgers, market models, interview guides. +4. **Walk depth-first.** +5. **Track dependencies** — endpoint → power → feasibility; budget → burn → treatment; sizing → survey → segmentation; method → saturation → synthesis. + +## The Research-Ops decision tree (depth-first) + +### Branch 1 — Which lane? + +- CLINICAL / RD_FINANCE / MARKET / PRODUCT + +### Branch 2 — The forcing question per lane + +**CLINICAL:** "Is your primary endpoint a clinical outcome or a surrogate — and if surrogate, is it validated for this indication?" +Recommended: clinical outcome unless the surrogate is on FDA's validated table. +Canon: FDA Surrogate Endpoint Table; BEST glossary; ICH E9. + +**RD_FINANCE:** "Is this spend in the research phase or the development phase, and can you evidence technical feasibility?" +Recommended: research = expense; development = capitalize-candidate only with feasibility evidence, routed to a named finance owner. +Canon: IAS 38; ASC 730. + +**MARKET:** "Is your TAM top-down or bottoms-up — and have you computed it both ways to triangulate?" +Recommended: both; reconcile the delta before quoting a number. +Canon: Bessemer / a16z market-sizing; Fermi estimation. + +**PRODUCT:** "Is this study generative (discover problems) or evaluative (test a solution)?" +Recommended: name it first; the method follows. +Canon: Rohrer's UX-research methods landscape (NN/g). + +### Branch 3 — Confidence & assumptions check + +"What's your confidence level, and what are the three assumptions the answer rests on?" +Recommended: state confidence (high/moderate/low) and surface the assumptions before any number. + +### Branch 4 — Named owner + +"Who is the human owner who signs this output?" +Recommended: a named clinician/biostatistician (clinical), a named finance controller (finance), a named decision-maker (market/product) — not "the team". + +### Branch 5 — Now invoke the sub-skill + +Only after branches 1-4 are locked, invoke `/cs:research-ops` with the synthesized inquiry. + +## Output format per turn + +``` +Q[i]/[total]: [precise question] +Recommended: [answer + canon-cited rationale] + +(Confirm, or override?) +``` + +## Stop conditions + +- All branches resolved → invoke `/cs:research-ops ` +- User says "stop grilling, just run it" → invoke with whatever's resolved, flag unresolved branches in digest +- User abandons → no sub-skill, save partial grill to `research-ops-grill-{timestamp}.md` + +## Distinct from + +- `engineering/grill-me` (Matt Pocock) — generic. +- `engineering/grill-with-docs` (Matt Pocock) — codebase + ADR-anchored for engineering. This is **Research-Ops-domain grilling** against the research canon. +- `/cs:research-ops` — **executes** routing. This **interrogates** first. diff --git a/research-ops/commands/cs-market-research.md b/research-ops/commands/cs-market-research.md new file mode 100644 index 00000000..43bab327 --- /dev/null +++ b/research-ops/commands/cs-market-research.md @@ -0,0 +1,40 @@ +--- +description: Market research methodology. Size a market as TAM/SAM/SOM computed BOTH top-down and bottoms-up (never a single number), plan a survey sample size with finite-population correction and per-segment minimums, and score candidate segments against Kotler's criteria. Outputs always show method + assumptions. Direct invocation of the market-research skill. +argument-hint: "" +--- + +# /cs:market-research — TAM/SAM/SOM + survey sampling + segmentation + +Run the `market-research` skill on this input: + +**$ARGUMENTS** + +## Three-tool workflow + +1. **`market_sizer.py`** — Compute TAM/SAM/SOM by BOTH top-down (total market value × fractions) and bottoms-up (customers × price × adoption) methods side-by-side. Reports divergence and flags failed triangulation. Industry tuning via `--profile`. Never returns a single number. + +2. **`sample_size_planner.py`** — Survey sample size from confidence, margin of error, and expected proportion, with the finite-population correction and per-segment minimums (a survey powered overall is not powered per reported segment). + +3. **`segmentation_scorer.py`** — Score candidate segments against Kotler's measurable / substantial / accessible / differentiable / actionable criteria. Enforces a substantiality + accessibility gate; drops demographic slices that are too small or unreachable. + +## Output + +- TAM/SAM/SOM both ways + triangulation flag + assumptions +- Survey n (overall + per-segment floors) +- Segment scores with TARGET / WATCH / DROP verdicts +- Top 3 next actions + +## Hard rule + +**A market size always travels with its method (both ways) and assumptions — never a single unsourced number.** + +## First run + optimization + +- **Onboard first:** `python3 scripts/onboard.py` (market profile, survey confidence, margin of error, sizing method) — saved config pre-configures every tool. `--show` lists the questions. +- **Optimize (opt-in):** only if the user asks to reconcile the sizing/run a loop, hand off to autoresearch via `scripts/ar_evaluator.py` (`tam_divergence`, lower is better). + +## Distinct from + +- `marketing-skill/campaign-analytics` — that measures a live campaign. This is upstream methodology. +- `marketing-skill/marketing-strategy-pmm` — that sets positioning/GTM. This sizes and segments the market. +- `commercial/pricing-strategist` — that sets price. This sizes the market. diff --git a/research-ops/commands/cs-product-research.md b/research-ops/commands/cs-product-research.md new file mode 100644 index 00000000..c6d213ac --- /dev/null +++ b/research-ops/commands/cs-product-research.md @@ -0,0 +1,41 @@ +--- +description: Product / user research methodology. Select the right method for the goal (generative vs evaluative vs validation), compute method-based saturation / sample size with an explicit confidence level, and synthesize coded observations into insights while flagging single-source anecdotes. Never fabricates insight. Direct invocation of the product-research skill. +argument-hint: "" +--- + +# /cs:product-research — Study design + saturation + insight synthesis + +Run the `product-research` skill on this input: + +**$ARGUMENTS** + +## Three-tool workflow + +1. **`study_designer.py`** — Map (research goal × product stage) to an appropriate method and emit a plan skeleton (objective, participant criteria, guide structure, success criteria). Redirects live A/B to `product-team/experiment-designer`. + +2. **`saturation_planner.py`** — Method-based sample guidance with an explicit confidence label: Nielsen problem-discovery (5/segment), Guest et al. thematic saturation (~12), evaluative coverage. Never claims a prevalence rate from a small-n usability test. + +3. **`insight_synthesizer.py`** — Cluster coded observations by tag, count distinct participants, rank by cross-participant recurrence, and flag any candidate below the source threshold as an ANECDOTE — never promoting it to an insight. + +## Output + +- Recommended method + plan skeleton (matched to the goal) +- Sample / saturation plan with confidence + limits +- Synthesized candidates: INSIGHT vs ANECDOTE with evidence +- Top 3 next actions + +## Hard rule + +**Method must match the goal, and an insight requires recurrence across independent participants.** A single quote is an anecdote, not a finding. + +## First run + optimization + +- **Onboard first:** `python3 scripts/onboard.py` (product profile, insight source-threshold, saturation method, high-stakes flag) — saved config pre-configures every tool. `--show` lists the questions. +- **Optimize (opt-in):** only if the user asks to optimize the synthesis/run a loop, hand off to autoresearch via `scripts/ar_evaluator.py` (`validated_insights`, higher is better). + +## Distinct from + +- `product-team/ux-researcher-designer` — that produces personas/journey artifacts. This is method + repository discipline. +- `product-team/product-discovery` — that plans discovery sprints. This designs and synthesizes the research. +- `product-team/experiment-designer` — that runs live A/B. This runs qualitative/evaluative research. +- `market-research` (sibling) — that studies the market. This studies users. diff --git a/research-ops/commands/cs-research-finance.md b/research-ops/commands/cs-research-finance.md new file mode 100644 index 00000000..23aeff3d --- /dev/null +++ b/research-ops/commands/cs-research-finance.md @@ -0,0 +1,39 @@ +--- +description: R&D program finance. Build a multi-period program budget with the F&A (indirect) split, track burn rate and runway against value-inflection milestones, and route R&D cost items to a capitalize-vs-expense determination. Every budget surfaces its assumptions; capex-vs-opex routes to a named finance owner and never auto-decides. Direct invocation of the research-finance skill. +argument-hint: "" +--- + +# /cs:research-finance — Program budget + burn/runway + capex-vs-opex routing + +Run the `research-finance` skill on this input: + +**$ARGUMENTS** + +## Three-tool workflow + +1. **`program_budget_planner.py`** — Build a multi-period budget from work-package lines, apply the F&A rate to an MTDC-style eligible base (excludes capital equipment + subaward portions over $25k), roll up direct / F&A / fully-loaded cost per period with an explicit assumptions block. Industry tuning via `--profile`. + +2. **`burn_runway_tracker.py`** — Compute average + trailing burn, runway in periods/months, and whether each value-inflection milestone is reachable before cash runs out. Flags accelerating burn and below-threshold runway. + +3. **`capex_vs_opex_router.py`** — Score each cost item against IAS 38 development-phase criteria (or flag ASC 730 expense-as-incurred under US GAAP). Route to CAPITALIZE-CANDIDATE / EXPENSE / FINANCE-OWNER-REVIEW with a named owner. Never books an entry. + +## Output + +- Budget rollup (direct / F&A / fully-loaded) with assumptions +- Runway + milestone verdicts + flags +- Per-item capex/opex routing with named owner +- Top 3 next actions + +## Hard rule + +**Every number carries its assumptions; accounting-treatment calls route to a named finance owner.** This skill never books an entry or decides treatment. + +## First run + optimization + +- **Onboard first:** `python3 scripts/onboard.py` (R&D area, F&A rate, runway threshold, accounting standard, finance owner) — saved config pre-configures every tool. `--show` lists the questions. +- **Optimize (opt-in):** only if the user asks to optimize/extend runway, hand off to autoresearch via `scripts/ar_evaluator.py` (`runway_months`, higher is better). + +## Distinct from + +- `finance/financial-analysis` — that's corporate DCF / close / valuation. This is R&D-program-level. +- `research/grants` — that **finds funding**. This **manages money already won**. diff --git a/research-ops/commands/cs-research-ops.md b/research-ops/commands/cs-research-ops.md new file mode 100644 index 00000000..b499768f --- /dev/null +++ b/research-ops/commands/cs-research-ops.md @@ -0,0 +1,43 @@ +--- +description: Top-level Research Operations router. Classifies an enterprise research inquiry (clinical study design / R&D finance / market research / product research) and forks context to the right sub-skill via the research-ops-skills orchestrator, returning a ≤200-word digest with a named owner and one grill challenge. +argument-hint: "" +--- + +# /cs:research-ops — Research Operations router + +Route this inquiry through the `research-ops-skills` orchestrator: + +**$ARGUMENTS** + +## Routing (deterministic, two-signal threshold) + +| Signal class | Keywords | Sub-skill | +|---|---|---| +| CLINICAL | clinical trial, study design, protocol, endpoint, sample size, power, phase, biostatistics, feasibility | `clinical-research` | +| RD_FINANCE | R&D budget, program budget, burn, runway, F&A, indirect rate, capitalize vs expense, portfolio ROI | `research-finance` | +| MARKET | TAM, SAM, SOM, market sizing, survey, sampling, margin of error, segmentation, competitive intelligence | `market-research` | +| PRODUCT | user interview, JTBD, usability, concept test, discovery research, research repository, insight, saturation | `product-research` | + +1. Explore the workspace first — a resolving filename routes silently. +2. Single signal or tie → one clarifying question with a recommended answer. +3. Genuine multi-lane → highest-confidence first, run in fork, ask before chaining. Never silently chain. + +## Output (≤200-word digest) + +- What was analyzed +- Top 3 findings, each anchored to a canon citation +- Top 3 next actions with a named human owner where applicable +- Artifact path +- One grill challenge for the user + +## Hard rules + +- Clinical output is an estimate + named clinical owner — never fact. +- Finance output surfaces assumptions; capex-vs-opex routes to a named finance owner. +- Market size shows method (both ways) + assumptions — never a single number. +- Product insight surfaces confidence + source count; singletons are anecdotes. + +## Distinct from + +- `research/` (academic) — finds literature/grants/patents. This plans/funds/scopes/synthesizes. +- `ra-qm-team` — regulatory submission. `finance/` — corporate close. `marketing-skill` — campaign analytics. diff --git a/research-ops/skills/clinical-research/SKILL.md b/research-ops/skills/clinical-research/SKILL.md new file mode 100644 index 00000000..54e8a25e --- /dev/null +++ b/research-ops/skills/clinical-research/SKILL.md @@ -0,0 +1,146 @@ +--- +name: clinical-research +description: Use when designing a prospective clinical study before submission — selecting and classifying endpoints (primary / key-secondary / exploratory, with surrogate-endpoint flagging), estimating sample size and power for two-arm designs (means / proportions / survival), or scoring a study plan for feasibility and a GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO phase-gate decision. Every output is an ESTIMATE plus a named human owner (clinician / biostatistician / regulatory owner) — never clinical fact, never a finished protocol. Distinct from ra-qm-team, which handles the regulatory/QM submission (ISO 13485, EU MDR, FDA 510(k)/PMA/QSR), not the study design. +version: 2.9.0 +author: claude-code-skills +license: MIT +tags: [research-ops, clinical-research, study-design, endpoint, sample-size, power, phase-gate, biostatistics] +compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli] +--- + +# clinical-research + +Prospective clinical study DESIGN: endpoints, sample size / power, and phase-gate feasibility. Every output is an **estimate with stated assumptions** routed to a **named human owner**. This skill never gives clinical advice as fact and never substitutes for a biostatistician or regulatory affairs. + +## Purpose + +R&D clinical teams, medical monitors, and biostatistics functions live at the moment between *we-have-a-hypothesis* and *we-have-a-protocol-ready-for-submission*. This skill structures three of the hardest design decisions: + +Three deterministic tools: + +1. `sample_size_estimator.py` — Closed-form power / sample-size for two-arm **means** (Cohen's d), **proportions** (normal approximation), and **survival** (Schoenfeld events). Inflates for dropout. Prints an "ESTIMATE — confirm with a biostatistician" banner. +2. `endpoint_selector.py` — Scores candidate endpoints across 5 weighted dimensions (clinical relevance, measurability, regulatory acceptance, sensitivity-to-change, burden) and classifies each as **PRIMARY / KEY-SECONDARY / EXPLORATORY**. Penalizes unvalidated surrogate endpoints. +3. `phase_gate_scorer.py` — Scores a study plan 0-100 across recruitment feasibility, endpoint readiness, statistical power, operational complexity, and budget fit; returns **GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO** plus the named owners who must sign. + +## When to use + +Invoke this skill when: + +- You are choosing a primary endpoint and need to defend it against surrogate-endpoint scrutiny. +- You need a defensible first sample-size estimate for a protocol synopsis. +- A study plan needs a feasibility read before a phase-gate review. +- You are pressure-testing whether the planned enrollment is achievable given the eligible population and sites. + +**Do NOT use this skill to**: prepare a regulatory submission or clinical evaluation report (use `ra-qm-team`), find or position a grant (use `research/grants`), design a live product A/B experiment (use `product-team/experiment-designer`), or replace a biostatistician's final sample-size justification. + +## Workflow + +1. **Draft the synopsis** — Fill `assets/protocol_synopsis_template.md` (objectives, design, population, endpoints, statistical plan placeholder, owners-to-sign). +2. **Select the endpoint** — Run `endpoint_selector.py --input endpoints.json --profile {drug|device|biologic|diagnostic|digital-therapeutic}`. Read the classification + surrogate flags. If >1 primary, plan multiplicity control. +3. **Estimate the sample size** — Run `sample_size_estimator.py --design {means|proportions|survival} ...`. Trace the effect/difference/HR to a published or anchor-based source; inflate for dropout. +4. **Score feasibility** — Run `phase_gate_scorer.py --input study.json --profile --phase {1|2|3|4}`. Read the verdict + blockers + named owners. +5. **Route for sign-off** — Assemble the synopsis + estimates into the gate packet. The packet is **a recommendation**; a biostatistician, medical monitor, and regulatory owner sign. + +## Scripts + +| Script | Purpose | Profiles | +|---|---|---| +| `scripts/sample_size_estimator.py` | Power / sample-size for means, proportions, survival | n/a (design-driven) | +| `scripts/endpoint_selector.py` | 5-dimension endpoint scoring + classification + surrogate flag | drug, device, biologic, diagnostic, digital-therapeutic | +| `scripts/phase_gate_scorer.py` | Feasibility 0-100 + GO/GO-WITH-CONDITIONS/REDESIGN/NO-GO + owners | drug, device, biologic, diagnostic, digital-therapeutic | + +All three: stdlib-only, `--help`, `--sample`, `--output {human,json}`. + +## Onboarding & customization + +Run the onboarding questionnaire **once before you start** — it captures your defaults and named owners so every tool in this skill is pre-configured. Customization is the point: the answers actually change tool behavior. + +```bash +python3 scripts/onboard.py # interactive (also: --defaults, --set key=value, --reset) +python3 scripts/onboard.py --show # see the questions + current effective config +``` + +Answers are saved to `~/.config/research-ops/clinical-research.json` (global) or `./.research-ops/clinical-research.json` (`--scope project`) and are read automatically by `config_loader.py`. They set the default development-area **profile**, default **alpha / power / dropout**, and the named **biostatistician / medical monitor / regulatory owner** printed on outputs. CLI flags always override saved config; `RESEARCH_OPS_NO_CONFIG=1` ignores it entirely. + +**The seven questions:** development area · alpha · power · dropout · biostatistician · medical monitor · regulatory owner. + +## Optimize with autoresearch (opt-in) + +This skill ships an **isolated, opt-in** bridge to `engineering/autoresearch-agent`. Only when you ask to "optimize" / "run a loop" does an autoresearch experiment iteratively improve a study plan against this skill's own feasibility score. `scripts/ar_evaluator.py` is the ground-truth evaluator; it prints `feasibility_composite: <0-100>` (higher is better). + +```bash +/ar:setup --domain custom --name trial-feasibility \ + --target study.json \ + --eval "python3 ar_evaluator.py --target study.json" \ + --metric feasibility_composite --direction higher +/ar:loop custom/trial-feasibility +``` + +Isolated: no hard dependency — autoresearch runs only on demand, and the loop edits `study.json`, never the evaluator (locked ground truth). + +## References + +- `references/study_design_canon.md` — ICH E8(R1) general considerations; ICH E9 + E9(R1) estimand addendum; CONSORT 2010; SPIRIT 2013; FDA Multiple Endpoints guidance (2022). +- `references/endpoint_and_power.md` — Cohen *Statistical Power Analysis*; Schoenfeld (1983) survival sample size; FDA Surrogate Endpoint Table / BEST glossary; FDA PRO guidance (2009); Chow, Shao & Wang *Sample Size Calculations in Clinical Research*. +- `references/trial_operations.md` — ICH E6(R2/R3) GCP; TransCelerate risk-based monitoring; FDA RBM guidance; CTTI recruitment best practices; site-feasibility scoring literature. + +## Assumptions + +- Sample-size formulas use normal approximations with a built-in z-table. They are first-pass **estimates**; a biostatistician produces the final justification (and may use simulation, adaptive designs, or exact methods). +- The endpoint scorer applies *customary* regulatory priors per development area via `--profile`. Company- or indication-specific precedent overrides the prior. +- The phase-gate scorer bakes in a profile cost-per-patient benchmark; pass a real budget to override the default. +- An unvalidated surrogate cannot anchor a PRIMARY endpoint — the scorer enforces this with a penalty. + +## Anti-patterns + +- **Presenting a power estimate as fact.** Every output is an estimate with a named owner who must sign. +- **Powering for a convenience effect size.** The effect must trace to a published or anchor-based MCID, not to the n you can afford. +- **Anchoring a primary on an unvalidated surrogate.** Surrogate endpoints need validation evidence for the indication. +- **Ignoring multiplicity.** More than one primary endpoint requires pre-specified alpha allocation. +- **Skipping dropout inflation.** Raw n undersizes the study; inflate by 1/(1 − dropout). + +## Distinct from + +| Sibling / neighbor | Scope | Difference | +|---|---|---| +| `ra-qm-team` | ISO 13485 QMS, ISO 14971 risk, EU MDR tech docs + clinical evaluation, FDA 510(k)/PMA/De Novo/QSR submission | That is the **submission**; clinical-research designs the **study** beforehand | +| `research/grants` | NIH funding discovery + positioning | That **finds funding**; this **designs the trial** | +| `product-team/experiment-designer` | Live product A/B hypothesis + sample size | That is a **product experiment**; this is a **clinical trial** | +| `research-finance` (sibling) | R&D program budget + burn | That **funds** the program; this **scopes** the study | + +## Quick examples + +```bash +python3 scripts/sample_size_estimator.py --sample +python3 scripts/sample_size_estimator.py --design proportions --p1 0.30 --p2 0.45 --dropout 0.15 +python3 scripts/endpoint_selector.py --sample +python3 scripts/phase_gate_scorer.py --sample --output json +``` + +The sample correctly flags an unvalidated serum-cytokine surrogate (cannot be primary) and ranks PASI-75 as the PRIMARY endpoint; the phase-gate sample returns a verdict with a named owner chain. + +## Forcing-question library (Matt Pocock grill discipline) + +Walked one at a time by `/cs:grill-research-ops` or the orchestrator. Recommended answer + canon citation per question. Never bundled. + +1. **"Is your primary endpoint a clinical outcome or a surrogate — and if surrogate, is it on FDA's validated table?"** + Recommended: clinical outcome unless the surrogate is validated for this indication. + Canon: FDA Surrogate Endpoint Table; BEST (Biomarkers, EndpointS, and other Tools) glossary. + +2. **"What's the minimal clinically important difference you're powering for — and where did that number come from?"** + Recommended: a published or anchor-based MCID, cited; never a convenience effect size. + Canon: ICH E9; Cohen *Statistical Power Analysis*. + +3. **"What dropout rate are you assuming, and is the sample size inflated for it?"** + Recommended: inflate n by 1/(1 − dropout) using a justified rate. + Canon: Chow, Shao & Wang; ICH E9(R1). + +4. **"Single primary endpoint or multiple — and if multiple, what's the multiplicity control?"** + Recommended: pre-specify alpha allocation (hierarchical / Bonferroni). + Canon: FDA Multiple Endpoints guidance (2022). + +5. **"Who is the named biostatistician / medical monitor / regulatory owner signing this synopsis?"** + Recommended: name them now — this output is a recommendation, not a protocol. + Canon: ICH E6(R2) GCP roles & responsibilities. + +Walk depth-first. Lock 1-2 before opening 3-5. After all are answered, invoke `endpoint_selector.py` → `sample_size_estimator.py` → `phase_gate_scorer.py`. diff --git a/research-ops/skills/clinical-research/assets/protocol_synopsis_template.md b/research-ops/skills/clinical-research/assets/protocol_synopsis_template.md new file mode 100644 index 00000000..37daeb69 --- /dev/null +++ b/research-ops/skills/clinical-research/assets/protocol_synopsis_template.md @@ -0,0 +1,57 @@ +# Protocol Synopsis — Template + +> Fill this before running the tools. This is a synopsis, not a full protocol. Every section +> ends in a named owner who must sign. Output of this skill is an ESTIMATE — a biostatistician, +> medical monitor, and regulatory owner sign the final protocol. + +## 1. Study identification +- Study ID: +- Sponsor / department: +- Phase: [1 | 2 | 3 | 4] +- Development area / profile: [drug | device | biologic | diagnostic | digital-therapeutic] + +## 2. Objectives +- Primary objective: +- Secondary objective(s): +- Estimand (population, treatment, endpoint, intercurrent-event strategy, summary measure): + +## 3. Design +- Type: [parallel-group RCT | crossover | adaptive | single-arm] +- Arms & allocation ratio: +- Randomization & stratification factors: +- Blinding: + +## 4. Population +- Indication: +- Key inclusion criteria: +- Key exclusion criteria: +- Estimated eligible population: +- Number of sites / countries: + +## 5. Endpoints +| Endpoint | Type (clinical / surrogate / PRO) | Validated? | Proposed class (PRIMARY / KEY-SECONDARY / EXPLORATORY) | +|---|---|---|---| +| | | | | + +- Multiplicity control (if >1 primary): + +## 6. Statistical plan (placeholder — biostatistician owns the final) +- Design for sample size: [means | proportions | survival] +- Assumed effect size / difference / HR + **source citation**: +- alpha (two-sided): ___ power: ___ dropout: ___ +- Estimated n (from `sample_size_estimator.py`): + +## 7. Feasibility & budget +- Target enrollment / enrollment months: +- Visits per patient / invasive procedures: +- Planned budget (USD): +- Phase-gate verdict (from `phase_gate_scorer.py`): + +## 8. Owners to sign (named, not roles) +- Principal Investigator: +- Medical Monitor: +- Biostatistician: +- Regulatory Owner: + +## 9. Assumptions register +- (List every assumption behind the effect size, dropout, eligible pool, and budget. Each must trace to a source or be flagged as an unverified planning assumption.) diff --git a/research-ops/skills/clinical-research/references/endpoint_and_power.md b/research-ops/skills/clinical-research/references/endpoint_and_power.md new file mode 100644 index 00000000..902e3f7e --- /dev/null +++ b/research-ops/skills/clinical-research/references/endpoint_and_power.md @@ -0,0 +1,37 @@ +# Endpoints and Statistical Power + +Reference for endpoint selection and sample-size estimation. Pairs with `endpoint_selector.py` and `sample_size_estimator.py`. + +## Endpoint hierarchy + +- **Clinical outcome** — directly measures how a patient feels, functions, or survives (mortality, stroke, symptom resolution). Strongest regulatory standing. +- **Surrogate endpoint** — a biomarker intended to substitute for a clinical outcome (LDL cholesterol, viral load, tumor response). Only acceptable if **validated** for the specific indication. FDA maintains a public Surrogate Endpoint Table listing surrogates that have supported approvals; the BEST glossary defines the validation hierarchy (candidate → reasonably likely → validated). +- **Patient-reported outcome (PRO)** — measured directly from the patient via a validated instrument. FDA's 2009 PRO guidance sets the bar for instrument validity, reliability, and content validity. + +The tool penalizes an **unvalidated surrogate** so it cannot anchor a PRIMARY endpoint — this mirrors the regulatory reality that an unvalidated surrogate carries approval risk. + +## Choosing the effect size (the hardest input) + +The single most consequential — and most abused — input is the assumed effect size. It must be **clinically meaningful** and **externally justified**, never reverse-engineered from the n you can afford. + +- For **means**, the effect is Cohen's d (standardized mean difference). Cohen's conventional small/medium/large (0.2 / 0.5 / 0.8) are last resorts, not anchors — prefer a published or anchor-based MCID. +- For **proportions**, specify the control and treatment rates from prior data; the absolute difference drives n. +- For **survival**, specify the target hazard ratio; required *events* (not patients) drive power via Schoenfeld's approximation, then n follows from the overall event probability. + +## Power formulas the tool implements + +- **Two-sample means:** n_per_arm = 2·((z_α + z_β)/d)², adjusted for allocation ratio k. +- **Two-sample proportions:** n = [z_α·√(2·p̄·q̄) + z_β·√(p₁q₁ + p₂q₂)]² / (p₁ − p₂)². +- **Survival (Schoenfeld):** required events E = 4·(z_α + z_β)² / (ln HR)²; n = E / P(event). + +All inflate for dropout by 1/(1 − dropout). These are estimates; a biostatistician produces the binding justification, possibly via simulation. + +## Sources + +1. Cohen, J., *Statistical Power Analysis for the Behavioral Sciences*, 2nd ed. (1988). +2. Schoenfeld, D., *Sample-size formula for the proportional-hazards regression model* — Biometrics 1983;39:499-503. +3. Chow, Shao, Wang & Lokhnygina, *Sample Size Calculations in Clinical Research*, 3rd ed. (CRC, 2017). +4. FDA, *Surrogate Endpoint Resources for Drug and Biologic Development* (public Surrogate Endpoint Table). +5. FDA-NIH BEST (Biomarkers, EndpointS, and other Tools) Resource glossary (2016, updated). +6. FDA, *Patient-Reported Outcome Measures: Use in Medical Product Development* (2009). +7. Fleming & DeMets, *Surrogate end points in clinical trials: are we being misled?* — Ann Intern Med 1996;125:605-613. diff --git a/research-ops/skills/clinical-research/references/study_design_canon.md b/research-ops/skills/clinical-research/references/study_design_canon.md new file mode 100644 index 00000000..a1c0bb82 --- /dev/null +++ b/research-ops/skills/clinical-research/references/study_design_canon.md @@ -0,0 +1,36 @@ +# Study Design Canon + +Reference knowledge base for prospective clinical study design. Use this when filling the protocol synopsis and defending design choices at a phase gate. + +## The estimand-first mindset (ICH E9(R1)) + +Before choosing an endpoint or a sample size, define the **estimand**: the precise treatment effect the trial will estimate. ICH E9(R1) defines five attributes — population, treatment, endpoint (variable), intercurrent-event handling strategy, and population-level summary. Skipping the estimand is the most common cause of a trial that "succeeds" statistically but answers the wrong question. Intercurrent events (treatment discontinuation, rescue medication, death) must have a pre-specified strategy (treatment-policy, hypothetical, composite, while-on-treatment, principal-stratum). + +## Design selection + +- **Parallel-group RCT** — the default for confirmatory efficacy. Two or more arms, randomized, concurrent controls. +- **Crossover** — each subject is their own control; only valid for chronic, stable, reversible conditions with adequate washout. +- **Adaptive designs** — pre-planned modifications (sample-size re-estimation, arm dropping, seamless phase 2/3). Powerful but require simulation and regulatory pre-agreement (FDA Adaptive Designs guidance, 2019). +- **Single-arm** — only defensible with a well-characterized natural history / external control, common in rare disease and oncology early phases. + +## Randomization & blinding + +Randomization removes selection bias; stratify on strong prognostic factors (and always on site in multicenter trials). Blinding (single / double / triple) removes ascertainment and analysis bias. Document the unblinding plan and the DSMB charter for any interim looks. + +## Multiplicity + +Any trial with more than one primary endpoint, more than two arms, or interim analyses inflates the family-wise type-I error. Pre-specify the control strategy: hierarchical (fixed-sequence) testing, Bonferroni / Holm, or a graphical (Bretz-Maurer) approach. The FDA Multiple Endpoints guidance (2022) is the operative reference. + +## Reporting standards as design checklists + +CONSORT 2010 (parallel-group RCT reporting) and SPIRIT 2013 (protocol content) are reporting standards — but used proactively they are design checklists. If you cannot fill a SPIRIT item, the design has a gap. + +## Sources + +1. ICH E8(R1), *General Considerations for Clinical Studies* (2021) — quality-by-design, fit-for-purpose study design. +2. ICH E9, *Statistical Principles for Clinical Trials* (1998) and the **E9(R1) Addendum on Estimands and Sensitivity Analysis** (2019). +3. Schulz, Altman & Moher, *CONSORT 2010 Statement* — BMJ 2010;340:c332. +4. Chan et al., *SPIRIT 2013 Statement: defining standard protocol items for clinical trials* — Ann Intern Med 2013;158:200-207. +5. FDA, *Multiple Endpoints in Clinical Trials: Guidance for Industry* (2022). +6. FDA, *Adaptive Designs for Clinical Trials of Drugs and Biologics* (2019). +7. Friedman, Furberg, DeMets, *Fundamentals of Clinical Trials*, 5th ed. (Springer, 2015). diff --git a/research-ops/skills/clinical-research/references/trial_operations.md b/research-ops/skills/clinical-research/references/trial_operations.md new file mode 100644 index 00000000..b1bbc842 --- /dev/null +++ b/research-ops/skills/clinical-research/references/trial_operations.md @@ -0,0 +1,33 @@ +# Trial Operations and Feasibility + +Reference for study feasibility and the phase-gate decision. Pairs with `phase_gate_scorer.py`. + +## Feasibility is the silent killer + +Most trials that fail do not fail on science — they fail on **enrollment**. A study powered for 240 patients across 18 sites assumes a recruitment rate per site per month that is often optimistic by 2-3×. The feasibility scorer enforces two reality checks: the **eligible pool ratio** (eligible population ÷ target enrollment should comfortably exceed 3×, ideally 10×) and **site capacity** (sites × nominal enroll rate × duration vs target). The "enrolling funnel" loses patients at screening, eligibility, and consent — Lasagna's Law (clinicians overestimate the eligible pool the moment a trial opens) is the operative caution. + +## Good Clinical Practice (GCP) + +ICH E6(R2) — and the in-progress E6(R3) — define the responsibilities of sponsors, investigators, and monitors; informed consent; protocol adherence; and the trial master file. A study design that cannot satisfy GCP roles is not gate-ready. Name the Principal Investigator, Medical Monitor, and Biostatistician before the gate. + +## Risk-based monitoring (RBM) + +Centralized, risk-based monitoring (FDA's 2013 guidance, expanded 2023; TransCelerate's RBM methodology) replaces 100% source-data verification with targeted monitoring of the data and processes that most affect patient safety and data integrity. Building RBM into the design lowers operational complexity (a scored dimension). + +## Operational complexity drivers + +Visits per patient, invasive procedures, central-lab logistics, imaging adjudication, and the number of countries all raise operational complexity and recruitment difficulty. The scorer inverts complexity (simpler design → higher score) because every added visit or procedure raises dropout and cost-per-patient. + +## Budget reality + +Cost-per-patient varies enormously by area (a digital-therapeutic at ~$6k/patient vs a biologic at ~$50k+/patient). The scorer compares planned budget to a profile benchmark and flags under-funding below 75% of benchmark. Research-finance (the sibling skill) owns the full program budget; this scorer only checks gate-level adequacy. + +## Sources + +1. ICH E6(R2), *Good Clinical Practice* (2016); ICH E6(R3) draft (2023). +2. FDA, *A Risk-Based Approach to Monitoring of Clinical Investigations* (2013; Q&A revision 2023). +3. TransCelerate BioPharma, *Risk-Based Monitoring Methodology* position papers. +4. CTTI (Clinical Trials Transformation Initiative), *Recruitment* and *Feasibility* recommendations. +5. Lasagna, L. — "Lasagna's Law" on the overestimation of eligible patients (clinical-trials folklore widely cited in feasibility literature). +6. Treweek et al., *Strategies to improve recruitment to randomised trials* — Cochrane Database Syst Rev 2018. +7. Getz & Campo, *Trial complexity and protocol design* — Tufts CSDD impact reports. diff --git a/research-ops/skills/clinical-research/scripts/ar_evaluator.py b/research-ops/skills/clinical-research/scripts/ar_evaluator.py new file mode 100644 index 00000000..de7770f5 --- /dev/null +++ b/research-ops/skills/clinical-research/scripts/ar_evaluator.py @@ -0,0 +1,74 @@ +#!/usr/bin/env python3 +"""ar_evaluator.py - Autoresearch evaluator for the clinical-research skill (OPT-IN). + +Stdlib-only. This is the ISOLATED bridge to engineering/autoresearch-agent. It does +NOT call autoresearch; it is the ground-truth evaluator that an autoresearch loop runs +after editing the target study plan. It reads a study-plan JSON (the file the loop +optimizes), scores it with phase_gate_scorer, and prints ONE metric line to stdout: + + feasibility_composite: <0-100> (higher is better) + +Usage inside autoresearch (the user opts in explicitly): + /ar:setup --domain custom --name trial-feasibility \\ + --target study.json --eval "python3 ar_evaluator.py --target study.json" \\ + --metric feasibility_composite --direction higher + +Direct use: + python3 ar_evaluator.py --sample + python3 ar_evaluator.py --target study.json --profile drug +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 +import phase_gate_scorer as pgs # noqa: E402 + +METRIC = "feasibility_composite" + + +def evaluate_target(study: dict, profile: str, phase: int) -> float: + result = pgs.evaluate(study, profile, phase) + return float(result["composite"]) + + +def main(argv: list[str] | None = None) -> int: + c = cfg.load_config() + p = argparse.ArgumentParser(description="Autoresearch evaluator: study-plan feasibility composite.") + p.add_argument("--target", help="path to study-plan JSON (or env AR_TARGET)") + p.add_argument("--profile", default=None, help="overrides onboarding default_profile") + p.add_argument("--phase", type=int, default=2, choices=[1, 2, 3, 4]) + p.add_argument("--sample", action="store_true", help="evaluate the embedded sample plan") + args = p.parse_args(argv) + + profile = args.profile or c.get("default_profile", "drug") + if args.sample: + study = pgs.SAMPLE + else: + target = args.target or os.environ.get("AR_TARGET") + if not target: + print("error: provide --target or set AR_TARGET", file=sys.stderr) + return 2 + try: + with open(target) as f: + study = json.load(f) + except (OSError, json.JSONDecodeError) as e: + # autoresearch treats a crash as DISCARD; emit N/A and non-zero. + print(f"{METRIC}: N/A") + print(f"error: {e}", file=sys.stderr) + return 1 + + phase = study.get("phase", args.phase) + value = evaluate_target(study, profile, phase) + # The single machine-readable metric line autoresearch parses: + print(f"{METRIC}: {value}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/clinical-research/scripts/config_loader.py b/research-ops/skills/clinical-research/scripts/config_loader.py new file mode 100644 index 00000000..cbfc194d --- /dev/null +++ b/research-ops/skills/clinical-research/scripts/config_loader.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""config_loader.py - Customization loader for the clinical-research skill. + +Stdlib-only. Importable from the skill's other scripts. Precedence (highest wins): + 1. Project config: /.research-ops/clinical-research.json + 2. Global config: ~/.config/research-ops/clinical-research.json + 3. Built-in DEFAULTS + +The onboarding answers (written by onboard.py) live in these files and are read +here so every tool in this skill picks up the user's customization automatically. +Set RESEARCH_OPS_NO_CONFIG=1 (or pass --no-config to a tool) to ignore saved config. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from pathlib import Path +from typing import Any + +SKILL = "clinical-research" + +GLOBAL_CONFIG_DIR = Path.home() / ".config" / "research-ops" +GLOBAL_CONFIG_PATH = GLOBAL_CONFIG_DIR / f"{SKILL}.json" +PROJECT_CONFIG_DIRNAME = ".research-ops" + +DEFAULTS: dict[str, Any] = { + "version": 1, + "skill": SKILL, + "default_profile": "drug", + "default_alpha": 0.05, + "default_power": 0.80, + "default_dropout": 0.15, + "owners": { + "biostatistician": None, + "medical_monitor": None, + "regulatory_owner": None, + }, + "setup_completed_at": None, +} + + +def project_config_path(cwd: Path | None = None) -> Path: + cwd = cwd or Path.cwd() + return cwd / PROJECT_CONFIG_DIRNAME / f"{SKILL}.json" + + +def _read_json(path: Path) -> dict[str, Any] | None: + try: + with path.open(encoding="utf-8") as f: + data = json.load(f) + return data if isinstance(data, dict) else None + except (FileNotFoundError, json.JSONDecodeError, OSError): + return None + + +def _deep_merge(base: dict[str, Any], override: dict[str, Any]) -> dict[str, Any]: + out = dict(base) + for k, v in override.items(): + if isinstance(v, dict) and isinstance(out.get(k), dict): + out[k] = _deep_merge(out[k], v) + else: + out[k] = v + return out + + +def load_config(cwd: Path | None = None) -> dict[str, Any]: + """Effective config = DEFAULTS <- global <- project. Honors RESEARCH_OPS_NO_CONFIG.""" + config = dict(DEFAULTS) + if os.environ.get("RESEARCH_OPS_NO_CONFIG") == "1": + return config + global_cfg = _read_json(GLOBAL_CONFIG_PATH) + if global_cfg: + config = _deep_merge(config, global_cfg) + project_cfg = _read_json(project_config_path(cwd)) + if project_cfg: + config = _deep_merge(config, project_cfg) + return config + + +def setup_completed() -> bool: + cfg = _read_json(GLOBAL_CONFIG_PATH) or _read_json(project_config_path()) + return bool(cfg and cfg.get("setup_completed_at")) + + +def write_config(config: dict[str, Any], scope: str = "global", cwd: Path | None = None) -> Path: + if scope == "project": + path = project_config_path(cwd) + else: + path = GLOBAL_CONFIG_PATH + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as f: + json.dump(config, f, indent=2, sort_keys=True) + return path + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Inspect {SKILL} customization config.") + p.add_argument("--show", action="store_true", help="Print the effective config") + p.add_argument("--status", action="store_true", help="Print setup status + paths") + p.add_argument("--sample", action="store_true", help="Print the built-in defaults") + args = p.parse_args(argv) + + if args.sample: + print(json.dumps(DEFAULTS, indent=2, sort_keys=True)) + elif args.status: + print(json.dumps({ + "skill": SKILL, + "global_config_path": str(GLOBAL_CONFIG_PATH), + "global_config_exists": GLOBAL_CONFIG_PATH.exists(), + "project_config_path": str(project_config_path()), + "project_config_exists": project_config_path().exists(), + "setup_completed": setup_completed(), + }, indent=2)) + else: + print(json.dumps(load_config(), indent=2, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/clinical-research/scripts/endpoint_selector.py b/research-ops/skills/clinical-research/scripts/endpoint_selector.py new file mode 100644 index 00000000..4d754482 --- /dev/null +++ b/research-ops/skills/clinical-research/scripts/endpoint_selector.py @@ -0,0 +1,183 @@ +#!/usr/bin/env python3 +"""endpoint_selector.py - Score candidate clinical endpoints and classify each. + +Stdlib-only. Deterministic. NO LLM calls. ESTIMATE / decision-support only — +endpoint selection must be confirmed by a clinician + biostatistician + regulatory owner. + +Each candidate endpoint is scored 0-100 across 5 weighted dimensions: + 1. clinical_relevance does it measure benefit patients care about? (weight 0.30) + 2. measurability validated instrument, low measurement error? (weight 0.20) + 3. regulatory_acceptance precedent acceptance by FDA/EMA for this indication (weight 0.25) + 4. sensitivity_to_change can it detect treatment effect in the trial window? (weight 0.15) + 5. burden patient/site burden (inverted: low burden = high) (weight 0.10) + +Classification: + - top composite -> PRIMARY + - composite >= 60 -> KEY-SECONDARY + - else -> EXPLORATORY +Surrogate endpoints flagged when is_surrogate=true and not validated. + +Profiles tune the regulatory-acceptance prior by development area. + +Usage: + python3 endpoint_selector.py --sample + python3 endpoint_selector.py --input endpoints.json --profile drug + python3 endpoint_selector.py --input endpoints.json --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +BANNER = "ESTIMATE ONLY — endpoint selection must be confirmed by clinician + biostatistician + regulatory owner." + +WEIGHTS = { + "clinical_relevance": 0.30, + "measurability": 0.20, + "regulatory_acceptance": 0.25, + "sensitivity_to_change": 0.15, + "burden": 0.10, +} + +# Per-area regulatory-acceptance multiplier applied to the regulatory_acceptance score. +PROFILES = { + "drug": 1.00, + "device": 0.95, + "biologic": 1.00, + "diagnostic": 0.90, + "digital-therapeutic": 0.80, +} + +SAMPLE = { + "indication": "moderate-to-severe plaque psoriasis", + "endpoints": [ + { + "name": "PASI-75 at week 16", + "is_surrogate": False, + "validated": True, + "scores": {"clinical_relevance": 90, "measurability": 85, "regulatory_acceptance": 95, + "sensitivity_to_change": 90, "burden": 80}, + }, + { + "name": "Serum cytokine level at week 4", + "is_surrogate": True, + "validated": False, + "scores": {"clinical_relevance": 40, "measurability": 90, "regulatory_acceptance": 30, + "sensitivity_to_change": 85, "burden": 50}, + }, + { + "name": "DLQI (quality of life) at week 16", + "is_surrogate": False, + "validated": True, + "scores": {"clinical_relevance": 75, "measurability": 70, "regulatory_acceptance": 70, + "sensitivity_to_change": 65, "burden": 75}, + }, + ], +} + + +def score_endpoint(ep: dict, profile_mult: float) -> dict: + raw = ep.get("scores", {}) + flags: list[str] = [] + composite = 0.0 + breakdown = {} + for dim, w in WEIGHTS.items(): + s = float(raw.get(dim, 0.0)) + if dim == "regulatory_acceptance": + s = min(100.0, s * profile_mult) + composite += s * w + breakdown[dim] = round(s, 1) + if ep.get("is_surrogate") and not ep.get("validated"): + flags.append("UNVALIDATED SURROGATE — not on a validated-surrogate table; confirm acceptability") + composite *= 0.7 # heavy penalty: unvalidated surrogate cannot anchor a primary endpoint + return { + "name": ep.get("name", "UNNAMED"), + "composite": round(composite, 1), + "breakdown": breakdown, + "is_surrogate": bool(ep.get("is_surrogate")), + "validated": bool(ep.get("validated")), + "flags": flags, + } + + +def classify(scored: list[dict]) -> list[dict]: + if not scored: + return scored + ordered = sorted(scored, key=lambda x: x["composite"], reverse=True) + top = ordered[0]["composite"] + for i, s in enumerate(ordered): + if i == 0 and not s["flags"]: + s["classification"] = "PRIMARY" + elif i == 0 and s["flags"]: + s["classification"] = "KEY-SECONDARY (flagged — cannot be primary)" + elif s["composite"] >= 60.0: + s["classification"] = "KEY-SECONDARY" + else: + s["classification"] = "EXPLORATORY" + return ordered + + +def evaluate(data: dict, profile: str) -> dict: + if profile not in PROFILES: + raise ValueError(f"Unknown profile '{profile}'. Choose from {list(PROFILES)}.") + mult = PROFILES[profile] + scored = [score_endpoint(ep, mult) for ep in data.get("endpoints", [])] + scored = classify(scored) + return { + "indication": data.get("indication", "UNSPECIFIED"), + "profile": profile, + "endpoints": scored, + "note": "Multiplicity control (e.g., hierarchical alpha allocation) required if >1 primary endpoint.", + } + + +def _render_human(result: dict) -> str: + lines = [f"!! {BANNER}", "", f"Indication: {result['indication']} (profile: {result['profile']})", ""] + for ep in result["endpoints"]: + lines.append(f"[{ep['classification']}] {ep['name']} — composite {ep['composite']}/100") + for dim, s in ep["breakdown"].items(): + lines.append(f" {dim:24s} {s}") + for f in ep["flags"]: + lines.append(f" ! {f}") + lines.append("") + lines.append(f"note: {result['note']}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Score and classify candidate clinical endpoints (ESTIMATE ONLY).") + p.add_argument("--input", help="Path to JSON with {indication, endpoints[]}") + p.add_argument("--profile", default=None, choices=list(PROFILES), + help="overrides onboarding default_profile") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + profile = args.profile or conf.get("default_profile", "drug") + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + try: + result = evaluate(data, profile) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + result["_banner"] = BANNER + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/clinical-research/scripts/onboard.py b/research-ops/skills/clinical-research/scripts/onboard.py new file mode 100644 index 00000000..69d9b5d3 --- /dev/null +++ b/research-ops/skills/clinical-research/scripts/onboard.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""onboard.py - Onboarding questionnaire for the clinical-research skill. + +Stdlib-only. Asks the user a short set of questions BEFORE they start designing a +study, then writes their answers to a customization config (read by every tool in +this skill via config_loader.py). Customization is the point: the answers become the +defaults for profile, alpha/power/dropout, and the named owners printed on outputs. + +Modes: + --show print the questions + the current effective config, then exit + --defaults write the built-in defaults without prompting (non-interactive) + --set key=value ... set specific answers non-interactively (repeatable) + --reset delete the saved config at the chosen scope + --scope {global,project} where to save (default: global = ~/.config/research-ops) + +With no flags and an interactive terminal, it walks the questions one at a time. +""" + +from __future__ import annotations + +import argparse +import datetime as _dt +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 + +# (key, prompt, choices_or_None, caster, default_key_in_DEFAULTS) +QUESTIONS = [ + ("default_profile", + "1. What development area are you working in?", + ["drug", "device", "biologic", "diagnostic", "digital-therapeutic"], str, "default_profile"), + ("default_alpha", + "2. Default two-sided significance level (alpha)?", + ["0.10", "0.05", "0.025", "0.01"], float, "default_alpha"), + ("default_power", + "3. Default target power (1 - beta)?", + ["0.80", "0.85", "0.90", "0.95"], float, "default_power"), + ("default_dropout", + "4. Default anticipated dropout fraction (for sample-size inflation)?", + None, float, "default_dropout"), + ("owner.biostatistician", + "5. Named biostatistician who signs the sample-size justification?", + None, str, None), + ("owner.medical_monitor", + "6. Named medical monitor for the study?", + None, str, None), + ("owner.regulatory_owner", + "7. Named regulatory owner who signs the gate decision?", + None, str, None), +] + + +def _apply(config: dict, key: str, value) -> None: + if key.startswith("owner."): + config.setdefault("owners", {})[key.split(".", 1)[1]] = value + else: + config[key] = value + + +def _print_questions() -> None: + print(f"Onboarding questions — {cfg.SKILL}:\n") + for _, prompt, choices, _c, _d in QUESTIONS: + line = f" {prompt}" + if choices: + line += f" [{ ' / '.join(choices) }]" + print(line) + + +def run_interactive(config: dict) -> dict: + print(f"Onboarding — {cfg.SKILL}. Press Enter to keep the current/default value.\n") + for key, prompt, choices, caster, dkey in QUESTIONS: + current = config.get("owners", {}).get(key.split(".", 1)[1]) if key.startswith("owner.") \ + else config.get(key) + suffix = f" [{ '/'.join(choices) }]" if choices else "" + cur = f" (current: {current})" if current is not None else "" + raw = input(f"{prompt}{suffix}{cur}: ").strip() + if not raw: + continue + try: + _apply(config, key, caster(raw)) + except ValueError: + print(f" ! invalid value for {key}, keeping current") + return config + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Onboarding for the {cfg.SKILL} skill.") + p.add_argument("--show", action="store_true", help="print questions + effective config") + p.add_argument("--defaults", action="store_true", help="write built-in defaults, no prompt") + p.add_argument("--set", action="append", default=[], metavar="key=value", + help="set an answer non-interactively (repeatable)") + p.add_argument("--reset", action="store_true", help="delete saved config at the scope") + p.add_argument("--scope", choices=["global", "project"], default="global") + args = p.parse_args(argv) + + if args.show: + _print_questions() + print("\nCurrent effective config:") + import json + print(json.dumps(cfg.load_config(), indent=2, sort_keys=True)) + return 0 + + if args.reset: + path = cfg.project_config_path() if args.scope == "project" else cfg.GLOBAL_CONFIG_PATH + if path.exists(): + path.unlink() + print(f"removed {path}") + else: + print(f"no config at {path}") + return 0 + + config = cfg.load_config() + + if args.set: + for item in args.set: + if "=" not in item: + print(f"error: --set expects key=value, got '{item}'", file=sys.stderr) + return 2 + k, v = item.split("=", 1) + # best-effort type coercion for known numeric keys + if k in ("default_alpha", "default_power", "default_dropout"): + try: + v = float(v) + except ValueError: + pass + _apply(config, k, v) + elif not args.defaults: + if sys.stdin.isatty(): + config = run_interactive(config) + else: + print("non-interactive shell: use --defaults or --set key=value. Showing questions:\n") + _print_questions() + return 0 + + config["setup_completed_at"] = _dt.datetime.now(_dt.timezone.utc).isoformat() + path = cfg.write_config(config, scope=args.scope) + print(f"saved {cfg.SKILL} customization -> {path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/clinical-research/scripts/phase_gate_scorer.py b/research-ops/skills/clinical-research/scripts/phase_gate_scorer.py new file mode 100644 index 00000000..4e54dc26 --- /dev/null +++ b/research-ops/skills/clinical-research/scripts/phase_gate_scorer.py @@ -0,0 +1,234 @@ +#!/usr/bin/env python3 +"""phase_gate_scorer.py - Score a study plan for feasibility and route a phase-gate verdict. + +Stdlib-only. Deterministic. NO LLM calls. ESTIMATE / decision-support only: the verdict +names the human owner(s) who must sign — it never authorizes a study on its own. + +Scores a study plan 0-100 across 5 dimensions: + 1. recruitment_feasibility eligible-population size vs target enrollment + timeline + 2. endpoint_readiness endpoint validated + instrument in place + 3. statistical_power is the planned n adequate for the stated effect? + 4. operational_complexity sites, visits, procedures (inverted: simpler = higher) + 5. budget_fit planned budget vs profile cost-per-patient benchmark + +Verdict: + - composite >= 80 and no blockers -> GO + - composite 65-79 -> GO-WITH-CONDITIONS + - composite 50-64 or 1 blocker -> REDESIGN + - composite < 50 or 2+ blockers -> NO-GO + +Profiles tune the cost-per-patient benchmark and recruitment difficulty. + +Usage: + python3 phase_gate_scorer.py --sample + python3 phase_gate_scorer.py --input study.json --profile device --phase 2 + python3 phase_gate_scorer.py --input study.json --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +BANNER = "ESTIMATE ONLY — a medical monitor + biostatistician + regulatory owner must sign the gate decision." + +WEIGHTS = { + "recruitment_feasibility": 0.25, + "endpoint_readiness": 0.20, + "statistical_power": 0.25, + "operational_complexity": 0.15, + "budget_fit": 0.15, +} + +# cost_per_patient_usd is the profile benchmark used for budget_fit scoring. +PROFILES = { + "drug": {"cost_per_patient_usd": 41000, "recruit_difficulty": 1.0}, + "device": {"cost_per_patient_usd": 28000, "recruit_difficulty": 0.9}, + "biologic": {"cost_per_patient_usd": 52000, "recruit_difficulty": 1.1}, + "diagnostic": {"cost_per_patient_usd": 12000, "recruit_difficulty": 0.8}, + "digital-therapeutic": {"cost_per_patient_usd": 6000, "recruit_difficulty": 0.7}, +} + +OWNERS = { + "GO": ["Principal Investigator", "Medical Monitor", "Biostatistician"], + "GO-WITH-CONDITIONS": ["Principal Investigator", "Medical Monitor", "Biostatistician", "Regulatory Owner"], + "REDESIGN": ["Medical Monitor", "Biostatistician", "Regulatory Owner", "Study Director"], + "NO-GO": ["Medical Monitor", "Biostatistician", "Regulatory Owner", "Study Director", "R&D Head"], +} + +SAMPLE = { + "study_id": "PSO-2026-P2", + "phase": 2, + "eligible_population": 4200, + "target_enrollment": 240, + "enrollment_months": 14, + "sites": 18, + "endpoint_validated": True, + "instrument_in_place": True, + "planned_n": 240, + "required_n": 260, + "visits_per_patient": 9, + "invasive_procedures": 2, + "planned_budget_usd": 8200000, +} + + +def _clamp(x, lo=0.0, hi=100.0): + return max(lo, min(hi, x)) + + +def score_plan(study: dict, profile: dict, phase: int) -> dict: + blockers: list[str] = [] + breakdown = {} + + # 1. recruitment feasibility: eligible pop must dwarf target; ~25 enroll/site/yr nominal + elig = float(study.get("eligible_population", 0)) + target = float(study.get("target_enrollment", 1)) or 1 + months = float(study.get("enrollment_months", 12)) or 12 + sites = float(study.get("sites", 1)) or 1 + pool_ratio = elig / target if target else 0 + nominal_capacity = sites * 25.0 * (months / 12.0) / profile["recruit_difficulty"] + capacity_ratio = nominal_capacity / target if target else 0 + recruit = _clamp(40.0 * min(pool_ratio / 10.0, 1.0) + 60.0 * min(capacity_ratio, 1.0)) + if pool_ratio < 3.0: + blockers.append("recruitment: eligible pool < 3x target enrollment") + breakdown["recruitment_feasibility"] = round(recruit, 1) + + # 2. endpoint readiness + er = 0.0 + er += 60.0 if study.get("endpoint_validated") else 0.0 + er += 40.0 if study.get("instrument_in_place") else 0.0 + if not study.get("endpoint_validated"): + blockers.append("endpoint: primary endpoint not validated") + breakdown["endpoint_readiness"] = round(er, 1) + + # 3. statistical power: planned_n vs required_n + planned_n = float(study.get("planned_n", 0)) + required_n = float(study.get("required_n", 0)) or 1 + ratio = planned_n / required_n if required_n else 0 + power = _clamp(100.0 * min(ratio, 1.0)) if ratio >= 1.0 else _clamp(100.0 * ratio - (1.0 - ratio) * 40.0) + if ratio < 0.9: + blockers.append(f"power: planned n ({planned_n:.0f}) < 90% of required n ({required_n:.0f})") + breakdown["statistical_power"] = round(power, 1) + + # 4. operational complexity (inverted: more visits/procedures = lower score) + visits = float(study.get("visits_per_patient", 6)) + procs = float(study.get("invasive_procedures", 0)) + complexity = _clamp(100.0 - (visits - 4) * 6.0 - procs * 10.0) + breakdown["operational_complexity"] = round(complexity, 1) + + # 5. budget fit: planned budget vs benchmark cost-per-patient * target + benchmark = profile["cost_per_patient_usd"] * target + planned_budget = float(study.get("planned_budget_usd", 0)) + if planned_budget <= 0: + budget = 0.0 + blockers.append("budget: no planned budget provided") + else: + coverage = planned_budget / benchmark if benchmark else 0 + # 100 if planned >= benchmark, sliding down if under-funded + budget = _clamp(100.0 * min(coverage, 1.0)) if coverage >= 1.0 else _clamp(coverage * 100.0) + if coverage < 0.75: + blockers.append("budget: planned budget < 75% of benchmark cost") + breakdown["budget_fit"] = round(budget, 1) + + composite = sum(breakdown[d] * w for d, w in WEIGHTS.items()) + verdict = _verdict(composite, blockers) + return { + "study_id": study.get("study_id", "UNSPECIFIED"), + "phase": phase, + "composite": round(composite, 1), + "verdict": verdict, + "named_owners": OWNERS[verdict], + "breakdown": breakdown, + "blockers": blockers, + "benchmark_cost_usd": round(benchmark, 0), + } + + +def _verdict(composite: float, blockers: list[str]) -> str: + n = len(blockers) + if n >= 2 or composite < 50.0: + return "NO-GO" + if n == 1 or composite < 65.0: + return "REDESIGN" + if composite < 80.0: + return "GO-WITH-CONDITIONS" + return "GO" + + +def evaluate(study: dict, profile_name: str, phase: int) -> dict: + if profile_name not in PROFILES: + raise ValueError(f"Unknown profile '{profile_name}'. Choose from {list(PROFILES)}.") + return score_plan(study, PROFILES[profile_name], phase) + + +def _apply_named_owners(roles: list[str], owners: dict) -> list[str]: + """Replace generic owner roles with 'Role (Name)' when onboarding named them.""" + role_to_key = { + "Biostatistician": "biostatistician", + "Medical Monitor": "medical_monitor", + "Regulatory Owner": "regulatory_owner", + } + out = [] + for r in roles: + name = owners.get(role_to_key.get(r, "")) + out.append(f"{r} ({name})" if name else r) + return out + + +def _render_human(r: dict) -> str: + lines = [f"!! {BANNER}", "", f"Study: {r['study_id']} (Phase {r['phase']})", + f"Composite feasibility: {r['composite']}/100", f"Verdict: {r['verdict']}", ""] + lines.append("Dimension breakdown:") + for d, s in r["breakdown"].items(): + lines.append(f" {d:26s} {s}") + lines.append("") + if r["blockers"]: + lines.append("Blockers (each can force a downgrade):") + for b in r["blockers"]: + lines.append(f" ! {b}") + lines.append("") + lines.append(f"Benchmark study cost (this profile): ${r['benchmark_cost_usd']:,.0f}") + lines.append("Named owners who must sign the gate decision: " + ", ".join(r["named_owners"])) + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Score study feasibility and route a phase-gate verdict (ESTIMATE ONLY).") + p.add_argument("--input", help="Path to JSON study plan") + p.add_argument("--profile", default=None, choices=list(PROFILES), + help="overrides onboarding default_profile") + p.add_argument("--phase", type=int, default=2, choices=[1, 2, 3, 4]) + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + profile = args.profile or conf.get("default_profile", "drug") + study = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + phase = study.get("phase", args.phase) if (args.sample or not args.input) else args.phase + try: + result = evaluate(study, profile, phase) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + result["named_owners"] = _apply_named_owners(result["named_owners"], conf.get("owners") or {}) + + if args.output == "json": + result["_banner"] = BANNER + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/clinical-research/scripts/sample_size_estimator.py b/research-ops/skills/clinical-research/scripts/sample_size_estimator.py new file mode 100644 index 00000000..596c193e --- /dev/null +++ b/research-ops/skills/clinical-research/scripts/sample_size_estimator.py @@ -0,0 +1,205 @@ +#!/usr/bin/env python3 +"""sample_size_estimator.py - Closed-form sample-size / power estimates for common trial designs. + +Stdlib-only. Deterministic. NO LLM calls. This is an ESTIMATE, not a protocol: +every output prints a banner instructing the user to confirm with a biostatistician. + +Supported designs (normal-approximation closed forms): + - means two-sample comparison of means (Cohen's d effect size) + - proportions two-sample comparison of proportions (arcsine-free normal approx) + - survival two-arm log-rank, Schoenfeld events approximation + +z-values come from a small built-in lookup table (no scipy dependency). + +Usage: + python3 sample_size_estimator.py --sample + python3 sample_size_estimator.py --design means --effect 0.5 --alpha 0.05 --power 0.8 + python3 sample_size_estimator.py --design proportions --p1 0.30 --p2 0.45 --dropout 0.15 + python3 sample_size_estimator.py --design survival --hr 0.65 --power 0.9 --output json +""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover - skill always ships config_loader + _cfg = None + +BANNER = "ESTIMATE ONLY — confirm with a biostatistician before finalizing the protocol." + +# Two-sided z for alpha, and one-sided z for power (1 - beta). Lookup avoids scipy. +Z_ALPHA_TWO_SIDED = {0.10: 1.6449, 0.05: 1.9600, 0.025: 2.2414, 0.01: 2.5758} +Z_POWER = {0.80: 0.8416, 0.85: 1.0364, 0.90: 1.2816, 0.95: 1.6449, 0.975: 1.9600} + + +def _z_alpha(alpha: float) -> float: + if alpha not in Z_ALPHA_TWO_SIDED: + raise ValueError(f"alpha must be one of {sorted(Z_ALPHA_TWO_SIDED)} (two-sided).") + return Z_ALPHA_TWO_SIDED[alpha] + + +def _z_power(power: float) -> float: + if power not in Z_POWER: + raise ValueError(f"power must be one of {sorted(Z_POWER)}.") + return Z_POWER[power] + + +def _inflate(n: float, dropout: float) -> int: + if not 0.0 <= dropout < 1.0: + raise ValueError("dropout must be in [0, 1).") + return math.ceil(n / (1.0 - dropout)) + + +def estimate_means(effect: float, alpha: float, power: float, allocation: float, dropout: float) -> dict: + """Two-sample means. effect = Cohen's d (standardized mean difference).""" + if effect <= 0: + raise ValueError("effect (Cohen's d) must be > 0.") + za, zb = _z_alpha(alpha), _z_power(power) + # Equal-n per-arm: n = 2 * ((za + zb) / d)^2 ; unequal handled via allocation ratio k. + k = allocation + base = ((za + zb) / effect) ** 2 + n1 = (1 + 1.0 / k) * base + n2 = (1 + k) * base + return { + "design": "means", + "effect_size_cohens_d": effect, + "alpha_two_sided": alpha, + "power": power, + "allocation_ratio_k": k, + "n_group1_raw": math.ceil(n1), + "n_group2_raw": math.ceil(n2), + "n_group1_with_dropout": _inflate(n1, dropout), + "n_group2_with_dropout": _inflate(n2, dropout), + "dropout_assumed": dropout, + "formula": "n_i = (1 + 1/k or k) * ((z_alpha + z_beta)/d)^2", + } + + +def estimate_proportions(p1: float, p2: float, alpha: float, power: float, dropout: float) -> dict: + """Two-sample proportions, normal approximation (pooled + unpooled variance term).""" + for p in (p1, p2): + if not 0.0 < p < 1.0: + raise ValueError("p1 and p2 must be in (0, 1).") + if p1 == p2: + raise ValueError("p1 and p2 must differ.") + za, zb = _z_alpha(alpha), _z_power(power) + pbar = (p1 + p2) / 2.0 + delta = abs(p1 - p2) + num = (za * math.sqrt(2 * pbar * (1 - pbar)) + zb * math.sqrt(p1 * (1 - p1) + p2 * (1 - p2))) ** 2 + n_per_arm = num / (delta ** 2) + return { + "design": "proportions", + "p1": p1, + "p2": p2, + "absolute_difference": round(delta, 4), + "alpha_two_sided": alpha, + "power": power, + "n_per_arm_raw": math.ceil(n_per_arm), + "n_per_arm_with_dropout": _inflate(n_per_arm, dropout), + "n_total_with_dropout": 2 * _inflate(n_per_arm, dropout), + "dropout_assumed": dropout, + "formula": "n = [z_a*sqrt(2*pbar*qbar) + z_b*sqrt(p1q1+p2q2)]^2 / (p1-p2)^2", + } + + +def estimate_survival(hr: float, alpha: float, power: float, prob_event: float, dropout: float) -> dict: + """Two-arm log-rank, Schoenfeld events approximation + n from event probability.""" + if hr <= 0 or hr == 1.0: + raise ValueError("hazard ratio must be > 0 and != 1.") + if not 0.0 < prob_event <= 1.0: + raise ValueError("prob_event (overall probability of event) must be in (0, 1].") + za, zb = _z_alpha(alpha), _z_power(power) + log_hr = math.log(hr) + # Schoenfeld: total events E = 4*(za+zb)^2 / (log HR)^2 (1:1 allocation) + events = 4.0 * ((za + zb) ** 2) / (log_hr ** 2) + n_total = events / prob_event + return { + "design": "survival", + "hazard_ratio": hr, + "alpha_two_sided": alpha, + "power": power, + "required_events_raw": math.ceil(events), + "overall_event_probability": prob_event, + "n_total_raw": math.ceil(n_total), + "n_total_with_dropout": _inflate(n_total, dropout), + "dropout_assumed": dropout, + "formula": "E = 4*(z_a+z_b)^2 / (ln HR)^2 ; n = E / P(event)", + } + + +def _render_human(result: dict) -> str: + lines = [f"!! {BANNER}", "", f"Design: {result['design']}", ""] + for k, v in result.items(): + if k == "design": + continue + lines.append(f" {k:28s} : {v}") + lines += [ + "", + "Assumptions block (state these in the protocol statistical section):", + f" - alpha (two-sided): {result.get('alpha_two_sided')}", + f" - power (1 - beta): {result.get('power')}", + f" - dropout inflation: {result.get('dropout_assumed')}", + " - The effect/difference/HR must trace to a published or anchor-based source.", + "", + f"Named owner required: {result.get('_biostatistician') or 'a biostatistician (run onboard.py to name one)'} " + "must sign the final sample-size justification.", + ] + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Closed-form clinical sample-size / power estimates (ESTIMATE ONLY).") + p.add_argument("--design", choices=["means", "proportions", "survival"], default="means") + p.add_argument("--alpha", type=float, default=None, help="two-sided alpha (0.10/0.05/0.025/0.01)") + p.add_argument("--power", type=float, default=None, help="target power (0.80/0.85/0.90/0.95/0.975)") + p.add_argument("--dropout", type=float, default=None, help="anticipated dropout fraction [0,1)") + # means + p.add_argument("--effect", type=float, default=0.5, help="Cohen's d (means design)") + p.add_argument("--allocation", type=float, default=1.0, help="allocation ratio k = n2/n1 (means)") + # proportions + p.add_argument("--p1", type=float, default=0.30, help="control proportion") + p.add_argument("--p2", type=float, default=0.45, help="treatment proportion") + # survival + p.add_argument("--hr", type=float, default=0.65, help="hazard ratio (survival)") + p.add_argument("--prob-event", type=float, default=0.60, help="overall probability of event (survival)") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="run the embedded sample (means design)") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + alpha = args.alpha if args.alpha is not None else conf.get("default_alpha", 0.05) + power = args.power if args.power is not None else conf.get("default_power", 0.80) + dropout = args.dropout if args.dropout is not None else conf.get("default_dropout", 0.0) + biostat = (conf.get("owners") or {}).get("biostatistician") + + try: + if args.sample: + result = estimate_means(0.5, alpha, power, 1.0, dropout if dropout else 0.15) + elif args.design == "means": + result = estimate_means(args.effect, alpha, power, args.allocation, dropout) + elif args.design == "proportions": + result = estimate_proportions(args.p1, args.p2, alpha, power, dropout) + else: + result = estimate_survival(args.hr, alpha, power, args.prob_event, dropout) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + result["_biostatistician"] = biostat + + if args.output == "json": + result["_banner"] = BANNER + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/market-research/SKILL.md b/research-ops/skills/market-research/SKILL.md new file mode 100644 index 00000000..8c0eb729 --- /dev/null +++ b/research-ops/skills/market-research/SKILL.md @@ -0,0 +1,146 @@ +--- +name: market-research +description: Use when doing upstream market-research methodology — sizing a market as TAM/SAM/SOM computed BOTH top-down and bottoms-up (never a single unsourced number), planning a survey sample size with finite-population correction and per-segment minimums, or scoring candidate market segments against Kotler's measurable/substantial/accessible/differentiable/actionable criteria. Outputs always show the method and the assumptions. For market-research analysts and product-marketing at the sizing/survey/segmentation moment. Distinct from marketing-skill (campaign analytics, attribution, demand-gen) — this is the evidence-building methodology, not live-campaign optimization. +version: 2.9.0 +author: claude-code-skills +license: MIT +tags: [research-ops, market-research, tam-sam-som, market-sizing, survey, sampling, segmentation, competitive-intelligence] +compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli] +--- + +# market-research + +Upstream market-research methodology: market sizing, survey/sampling design, and segmentation. The discipline here is **method + assumptions**: a TAM is never a single number, a survey is never powered only in aggregate, and a segment is never a demographic slice. + +## Purpose + +Market-research analysts, product marketers, and strategy teams need rigorous evidence *before* anyone optimizes a campaign or sets a strategy. This skill structures three methodology decisions: + +Three deterministic tools: + +1. `market_sizer.py` — Computes TAM/SAM/SOM by **both** top-down and bottoms-up methods side-by-side, reports the divergence, and flags failed triangulation. Never returns a single number. +2. `sample_size_planner.py` — Survey sample size from confidence, margin of error, and expected proportion, with the finite-population correction and **per-segment minimums** (a survey powered overall is not powered per reported segment). +3. `segmentation_scorer.py` — Scores candidate segments against Kotler's five criteria and enforces a substantiality + accessibility gate; a slice that is too small or unreachable is dropped. + +## When to use + +Invoke this skill when: + +- A board or exec asks "how big is this market?" and you need a defensible, triangulated answer. +- You are fielding a survey and need a sample size that holds up per segment, not just overall. +- You have a list of candidate segments and need to know which are real markets vs demographic slices. +- You are synthesizing competitive intelligence and need a methodological backbone. + +**Do NOT use this skill to**: measure a live campaign (attribution, ROAS, CPA → `marketing-skill/campaign-analytics`), build demand-gen / paid-media plans (`marketing-skill/marketing-demand-acquisition`), set positioning / GTM strategy (`marketing-skill/marketing-strategy-pmm`), or set pricing (`commercial/pricing-strategist`). + +## Workflow + +1. **Write the brief** — Fill `assets/market_research_brief_template.md` (objective, the decision this informs, sizing approach, sampling plan, assumptions register). +2. **Size the market** — Run `market_sizer.py --input market.json --method both --profile {b2b-saas|consumer|enterprise|marketplace|hardware|services}`. Reconcile the top-down/bottoms-up delta before quoting anything. +3. **Plan the survey** — Run `sample_size_planner.py --input survey.json`. Fund the per-segment floors, not just the overall n. +4. **Score the segments** — Run `segmentation_scorer.py --input segments.json --profile `. Drop segments failing the substantiality/accessibility gate. +5. **Assemble the evidence pack** — Combine into a brief. Every number carries its method + assumptions + confidence. + +## Scripts + +| Script | Purpose | Profiles | +|---|---|---| +| `scripts/market_sizer.py` | TAM/SAM/SOM top-down AND bottoms-up + triangulation flag | b2b-saas, consumer, enterprise, marketplace, hardware, services | +| `scripts/sample_size_planner.py` | Survey n + FPC + per-segment minima | n/a (parameter-driven) | +| `scripts/segmentation_scorer.py` | Kotler 5-criteria scoring + gate | b2b-saas, consumer, enterprise, marketplace, hardware, services | + +All three: stdlib-only, `--help`, `--sample`, `--output {human,json}`. + +## Onboarding & customization + +Run the onboarding questionnaire **once before you start** — it captures your defaults so every tool in this skill is pre-configured. Customization is the point: the answers actually change tool behavior. + +```bash +python3 scripts/onboard.py # interactive (also: --defaults, --set key=value, --reset) +python3 scripts/onboard.py --show # see the questions + current effective config +``` + +Answers are saved to `~/.config/research-ops/market-research.json` (global) or `./.research-ops/market-research.json` (`--scope project`) and are read automatically by `config_loader.py`. They set the default market **profile**, the default survey **confidence** and **margin of error**, and the default **sizing method**. CLI flags always override saved config; `RESEARCH_OPS_NO_CONFIG=1` ignores it. + +**The four questions:** market profile · survey confidence · margin of error · sizing method. + +## Optimize with autoresearch (opt-in) + +This skill ships an **isolated, opt-in** bridge to `engineering/autoresearch-agent`. Only when you ask to "optimize" / "reconcile the sizing" / "run a loop" does an autoresearch experiment iteratively reconcile your market model so top-down and bottoms-up triangulate. `scripts/ar_evaluator.py` is the ground-truth evaluator; it prints `tam_divergence: ` (**lower** is better). + +```bash +/ar:setup --domain custom --name tam-triangulation \ + --target market.json \ + --eval "python3 ar_evaluator.py --target market.json" \ + --metric tam_divergence --direction lower +/ar:loop custom/tam-triangulation +``` + +Isolated: no hard dependency — autoresearch runs only on demand, and the loop edits `market.json`, never the evaluator. + +## References + +- `references/market_sizing_canon.md` — TAM/SAM/SOM frameworks (Bessemer, a16z); top-down vs bottoms-up; Fermi estimation; market-model conventions; common sizing fallacies. +- `references/survey_methodology.md` — Cochran *Sampling Techniques*; Dillman *Tailored Design Method*; Groves *Survey Methodology*; question-wording bias (Schuman & Presser); AAPOR standards. +- `references/segmentation_and_ci.md` — Kotler segmentation criteria; needs-based vs firmographic; Porter Five Forces; SCIP ethics; Christensen JTBD; conjoint/MaxDiff primer. + +## Assumptions + +- The sizer reports both methods but cannot validate your inputs — a top-down "1% of a $40B market" is only as good as the cited source and the serviceable fraction. +- Sample-size uses the conservative p=0.5 (maximum variance) unless you supply an expected proportion. +- Segment scores are inputs you provide; the tool enforces the gates and the weighting, it does not gather the underlying evidence. +- Competitive intelligence must follow the SCIP code of ethics — no misrepresentation, no protected information. + +## Anti-patterns + +- **A single TAM number with no method.** Always triangulate top-down against bottoms-up. +- **Spurious precision.** Size to the decision's tolerance; "$3.7142B" implies a confidence you do not have. +- **Powering only the total.** Each reported segment needs its own sample floor. +- **Leading or double-barreled survey questions.** Pre-test wording against the bias literature. +- **Calling a demographic slice a segment.** It must be substantial AND accessible. + +## Distinct from + +| Neighbor | Scope | Difference | +|---|---|---| +| `marketing-skill/campaign-analytics` | Attribution, ROAS, CPA, funnel of a live campaign | That **measures spend deployed**; this is **upstream methodology** | +| `marketing-skill/marketing-demand-acquisition` | Demand-gen, paid media, channel mix | That **runs acquisition**; this **builds the evidence** | +| `marketing-skill/marketing-strategy-pmm` | Positioning, GTM, category | That **sets strategy**; this **sizes and segments the market** | +| `commercial/pricing-strategist` | Pricing model + WTP + packaging | That **sets price**; this **sizes the market** | +| `product-research` (sibling) | User/product discovery methods | That studies **users**; this studies **the market** | + +## Quick examples + +```bash +python3 scripts/market_sizer.py --sample +python3 scripts/sample_size_planner.py --population 62000 --confidence 0.95 --moe 0.05 +python3 scripts/segmentation_scorer.py --sample --output json +``` + +The sample market triangulates a ~$1.47B top-down SAM against the bottoms-up figure and flags the divergence; the segmentation sample drops the "solopreneurs who might want analytics" slice for failing the substantiality and accessibility gates. + +## Forcing-question library (Matt Pocock grill discipline) + +Walked one at a time by `/cs:grill-research-ops` or the orchestrator. Recommended answer + canon citation per question. Never bundled. + +1. **"Is your TAM top-down or bottoms-up — and have you computed it both ways to triangulate?"** + Recommended: both; reconcile the delta before quoting a number. + Canon: Bessemer / a16z market-sizing; Fermi estimation. + +2. **"What decision will this market size actually drive — and at what precision does it matter?"** + Recommended: size to the decision's tolerance, not to a spurious-precision number. + Canon: market-model conventions (Gartner/Forrester); decision-driven analysis. + +3. **"What's your target margin of error and confidence — and does your sample clear it per segment, not just overall?"** + Recommended: power each reported segment, not only the total. + Canon: Cochran *Sampling Techniques*; AAPOR standards. + +4. **"Are your survey questions free of leading and double-barreled wording?"** + Recommended: pre-test the wording; cite the bias source. + Canon: Schuman & Presser; Dillman *Tailored Design Method*. + +5. **"Do your segments pass measurable / substantial / accessible / actionable — or are they just demographic slices?"** + Recommended: drop segments that fail substantiality or accessibility. + Canon: Kotler segmentation criteria. + +Walk depth-first. Lock 1-2 before opening 3-5. After all are answered, invoke `market_sizer.py` → `sample_size_planner.py` → `segmentation_scorer.py`. diff --git a/research-ops/skills/market-research/assets/market_research_brief_template.md b/research-ops/skills/market-research/assets/market_research_brief_template.md new file mode 100644 index 00000000..c411459d --- /dev/null +++ b/research-ops/skills/market-research/assets/market_research_brief_template.md @@ -0,0 +1,36 @@ +# Market Research Brief — Template + +> Fill this before running the tools. Every number in the final brief must carry its method, +> assumptions, and confidence. Size to the decision's tolerance, not to false precision. + +## 1. Objective +- Research question: +- **The decision this informs** (and who makes it): +- Precision required (order-of-magnitude / ±20% / ±5%): + +## 2. Market sizing +- Approach: [top-down | bottoms-up | both — recommended both] +- Top-down inputs: total market value + source citation; serviceable fraction; reachable share. +- Bottoms-up inputs: total potential customers + source; annual price; serviceable fraction; realistic adoption. +- Triangulation result (from `market_sizer.py`) + divergence: + +## 3. Survey plan (if primary data) +- Population (N) + sampling frame: +- Confidence level / overall margin of error / expected proportion: +- Segments to report + per-segment margin of error: +- Recommended n (overall + per-segment floors, from `sample_size_planner.py`): +- Mode (online panel / phone / mixed) + coverage risk: + +## 4. Segmentation +- Candidate segments + scores across measurable / substantial / accessible / differentiable / actionable: +- Verdicts (from `segmentation_scorer.py`): TARGET / WATCH / DROP + +## 5. Competitive intelligence +- Sources (public, ethically obtained — SCIP code): +- Five Forces summary: + +## 6. Assumptions register +- (List every assumption behind the sizing, sampling, and segmentation. Each must trace to a source or be flagged as an unverified planning assumption.) + +## 7. Confidence statement +- Overall confidence in the headline numbers (high / moderate / low) and why: diff --git a/research-ops/skills/market-research/references/market_sizing_canon.md b/research-ops/skills/market-research/references/market_sizing_canon.md new file mode 100644 index 00000000..eecbf4b0 --- /dev/null +++ b/research-ops/skills/market-research/references/market_sizing_canon.md @@ -0,0 +1,36 @@ +# Market Sizing Canon + +Reference for TAM/SAM/SOM. Pairs with `market_sizer.py`. + +## The three numbers + +- **TAM (Total Addressable Market)** — total revenue if you captured 100% of the market for your category. +- **SAM (Serviceable Addressable Market)** — the portion of TAM you can serve given your geography, segment focus, and product scope. +- **SOM (Serviceable Obtainable Market)** — the realistic, capacity-constrained share of SAM you can win in the planning window. + +## Two methods — always do both + +- **Top-down**: start from a published total market value (analyst report, government statistic) and apply serviceable and reachable fractions. Fast, but only as good as the source and inherits its biases. The classic failure is "1% of a huge number" — a TAM that sounds enormous and means nothing. +- **Bottoms-up**: start from the number of potential customers × the price they would pay, then apply serviceable and adoption fractions. Slower but grounded in units you can defend. The discipline is that bottoms-up forces you to name customer counts and price points. + +When the two methods diverge by more than a tolerance, **triangulation has failed** — you do not yet have a defensible number. The tool flags this rather than averaging the two (averaging hides the disagreement). + +## Fermi discipline + +Good sizing is Fermi estimation: decompose the unknown into knowable factors, estimate each with a stated assumption, and carry the uncertainty through. A market size is a chain of assumptions; surfacing the chain is the deliverable, not the point estimate. + +## Common fallacies + +- **Double counting** — summing overlapping segments or counting the same revenue at multiple layers of the value chain. +- **Percent-of-a-big-number** — anchoring on "if we just get 1%" without a bottoms-up cross-check. +- **Confusing TAM growth with your growth** — a growing TAM does not entitle you to a fixed share. +- **Spurious precision** — quoting a TAM to four significant figures when the inputs are order-of-magnitude estimates. + +## Sources + +1. Bessemer Venture Partners, *State of the Cloud* and market-sizing memos (TAM/SAM/SOM discipline). +2. Andreessen Horowitz (a16z), *The truth about market sizing* and bottoms-up TAM essays. +3. Gartner / Forrester / IDC market-model methodology notes (forecast construction conventions). +4. Weinstein, L., & Adam, J., *Guesstimation* (Princeton, 2008) — Fermi estimation. +5. Blank, S., *The Four Steps to the Epiphany* — market-type and sizing in customer development. +6. Damodaran, A., *Narrative and Numbers* (Columbia, 2017) — disciplining market-size narratives with numbers. diff --git a/research-ops/skills/market-research/references/segmentation_and_ci.md b/research-ops/skills/market-research/references/segmentation_and_ci.md new file mode 100644 index 00000000..ac73d703 --- /dev/null +++ b/research-ops/skills/market-research/references/segmentation_and_ci.md @@ -0,0 +1,43 @@ +# Segmentation and Competitive Intelligence + +Reference for segmentation scoring and competitive-intelligence synthesis. Pairs with `segmentation_scorer.py`. + +## What makes a segment useful (Kotler) + +A market segment is only useful if it meets five criteria. The scorer weights and gates them: + +1. **Measurable** — you can size and identify it. +2. **Substantial** — it is large and profitable enough to be worth serving. (Gate: a tiny slice is not a market.) +3. **Accessible** — you can reach it through channels you can afford. (Gate: an unreachable segment is academic.) +4. **Differentiable** — it responds differently to your offer than other segments do; otherwise it is not a distinct segment. +5. **Actionable** — you can design and execute a program for it. + +Substantiality and accessibility are **gates** because they are the two that most often fail silently: teams fall in love with a precisely-described segment that is too small or has no viable channel. + +## Bases of segmentation + +- **Firmographic** (B2B): industry, size, geography — easy to measure, weak at predicting behavior. +- **Demographic** (B2C): age, income, role — same trade-off. +- **Needs-based / behavioral**: grouping by the job customers are trying to get done. Stronger predictor of response, harder to measure. Christensen's **Jobs-to-be-Done** reframes segmentation around the progress a customer is trying to make, not who they are. + +The strongest segmentations pair a needs-based core with a firmographic/demographic proxy you can actually target. + +## Competitive intelligence + +CI synthesis is structured, ethical analysis of the competitive landscape: + +- **Porter's Five Forces** frames structural attractiveness (rivalry, new entrants, substitutes, supplier power, buyer power). +- **SCIP (Strategic and Competitive Intelligence Professionals)** publishes a code of ethics: no misrepresentation of identity, no acquisition of protected/confidential information, full compliance with law. CI is built from public and ethically-obtained sources. + +## Advanced preference measurement + +When you need to quantify trade-offs, **conjoint analysis** and **MaxDiff** (best-worst scaling) estimate the relative importance of attributes and the willingness to trade one for another — far more reliable than directly asking "how important is X?" + +## Sources + +1. Kotler, P., & Keller, K., *Marketing Management*, 15th ed. — segmentation criteria. +2. Christensen, Hall, Dillon & Duncan, *Competing Against Luck* (2016) — Jobs-to-be-Done. +3. Porter, M., *Competitive Strategy* (1980) — Five Forces. +4. SCIP, *Code of Ethics for CI Professionals*. +5. Orme, B., *Getting Started with Conjoint Analysis*, 4th ed. (Sawtooth, Research Publishers). +6. Smith, W., *Product Differentiation and Market Segmentation as Alternative Marketing Strategies* — J Marketing 1956 (the founding segmentation paper). diff --git a/research-ops/skills/market-research/references/survey_methodology.md b/research-ops/skills/market-research/references/survey_methodology.md new file mode 100644 index 00000000..6c837291 --- /dev/null +++ b/research-ops/skills/market-research/references/survey_methodology.md @@ -0,0 +1,39 @@ +# Survey Methodology + +Reference for survey design and sampling. Pairs with `sample_size_planner.py`. + +## Sample size from first principles + +For estimating a proportion, the required sample is n₀ = z²·p·(1−p)/e², where z is the critical value for the confidence level, p the expected proportion, and e the margin of error. The maximum-variance choice p = 0.5 is the conservative default. When the sample is a meaningful fraction of a finite population N, apply the **finite-population correction**: n = n₀ / (1 + (n₀−1)/N). The planner does both. + +The most common analyst error: powering the **overall** sample but then reporting **per-segment** results that the sample cannot support. If you will report three segments at ±8%, each segment needs ~150 respondents — so the survey must be sized to the segment floors, not the aggregate. The planner computes per-segment minimums explicitly. + +## The total survey error framework + +Sample size only addresses **sampling error**. Groves' total-survey-error framework names the others, often larger: + +- **Coverage error** — the sampling frame omits part of the population (e.g., an email panel misses non-users). +- **Non-response error** — those who respond differ systematically from those who don't. +- **Measurement error** — the instrument itself biases answers (question wording, order, scale). + +A tight margin of error on a biased frame is precision without accuracy. + +## Question design + +- **Avoid leading questions** that imply a preferred answer. +- **Avoid double-barreled questions** that ask two things at once ("Is the product fast and reliable?"). +- **Watch scale and order effects** — response options and question sequence shift answers. +- **Pre-test** every instrument with a small cognitive-interview pass before fielding. + +## Standards + +AAPOR (American Association for Public Opinion Research) publishes disclosure standards and the standard definitions for response-rate calculation. Reputable market research follows them so results are comparable and auditable. + +## Sources + +1. Cochran, W.G., *Sampling Techniques*, 3rd ed. (Wiley, 1977). +2. Dillman, Smyth & Christian, *Internet, Phone, Mail, and Mixed-Mode Surveys: The Tailored Design Method*, 4th ed. (2014). +3. Groves et al., *Survey Methodology*, 2nd ed. (Wiley, 2009) — total survey error. +4. Schuman, H., & Presser, S., *Questions and Answers in Attitude Surveys* (1981) — wording/order effects. +5. AAPOR, *Standard Definitions: Final Dispositions of Case Codes and Outcome Rates for Surveys*. +6. Tourangeau, Rips & Rasinski, *The Psychology of Survey Response* (Cambridge, 2000). diff --git a/research-ops/skills/market-research/scripts/ar_evaluator.py b/research-ops/skills/market-research/scripts/ar_evaluator.py new file mode 100644 index 00000000..a2c3f079 --- /dev/null +++ b/research-ops/skills/market-research/scripts/ar_evaluator.py @@ -0,0 +1,77 @@ +#!/usr/bin/env python3 +"""ar_evaluator.py - Autoresearch evaluator for the market-research skill (OPT-IN). + +Stdlib-only. The ISOLATED bridge to engineering/autoresearch-agent. It does NOT call +autoresearch; it is the ground-truth evaluator an autoresearch loop runs after editing +the target market model. It reads a market-model JSON, runs market_sizer in "both" mode, +and prints ONE metric line: + + tam_divergence: (LOWER is better — top-down and bottoms-up should agree) + +Optimize a market model so the two sizing methods triangulate (reconcile assumptions), +while the agent edits the target. The user opts in explicitly: + /ar:setup --domain custom --name tam-triangulation \\ + --target market.json --eval "python3 ar_evaluator.py --target market.json" \\ + --metric tam_divergence --direction lower + +Direct use: + python3 ar_evaluator.py --sample + python3 ar_evaluator.py --target market.json --profile enterprise +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 +import market_sizer as ms # noqa: E402 + +METRIC = "tam_divergence" + + +def main(argv: list[str] | None = None) -> int: + c = cfg.load_config() + p = argparse.ArgumentParser(description="Autoresearch evaluator: TAM triangulation divergence.") + p.add_argument("--target", help="path to market-model JSON (or env AR_TARGET)") + p.add_argument("--profile", default=None, help="overrides onboarding default_profile") + p.add_argument("--sample", action="store_true") + args = p.parse_args(argv) + + profile = args.profile or c.get("default_profile", "b2b-saas") + if args.sample: + data = ms.SAMPLE + else: + target = args.target or os.environ.get("AR_TARGET") + if not target: + print("error: provide --target or set AR_TARGET", file=sys.stderr) + return 2 + try: + with open(target) as f: + data = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"{METRIC}: N/A") + print(f"error: {e}", file=sys.stderr) + return 1 + + try: + result = ms.size_market(data, "both", profile) + except ValueError as e: + print(f"{METRIC}: N/A") + print(f"error: {e}", file=sys.stderr) + return 1 + + div = result.get("tam_divergence") + if div is None: + print(f"{METRIC}: N/A") + print("error: need both top_down and bottoms_up blocks to triangulate", file=sys.stderr) + return 1 + print(f"{METRIC}: {div}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/market-research/scripts/config_loader.py b/research-ops/skills/market-research/scripts/config_loader.py new file mode 100644 index 00000000..dc32b7de --- /dev/null +++ b/research-ops/skills/market-research/scripts/config_loader.py @@ -0,0 +1,114 @@ +#!/usr/bin/env python3 +"""config_loader.py - Customization loader for the market-research skill. + +Stdlib-only. Importable from the skill's other scripts. Precedence (highest wins): + 1. Project config: /.research-ops/market-research.json + 2. Global config: ~/.config/research-ops/market-research.json + 3. Built-in DEFAULTS + +Onboarding answers (written by onboard.py) live in these files; every tool in this +skill reads them so the user's customization applies automatically. +Set RESEARCH_OPS_NO_CONFIG=1 to ignore saved config. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from pathlib import Path +from typing import Any + +SKILL = "market-research" + +GLOBAL_CONFIG_DIR = Path.home() / ".config" / "research-ops" +GLOBAL_CONFIG_PATH = GLOBAL_CONFIG_DIR / f"{SKILL}.json" +PROJECT_CONFIG_DIRNAME = ".research-ops" + +DEFAULTS: dict[str, Any] = { + "version": 1, + "skill": SKILL, + "default_profile": "b2b-saas", + "default_confidence": 0.95, + "default_moe": 0.05, + "sizing_method": "both", + "setup_completed_at": None, +} + + +def project_config_path(cwd: Path | None = None) -> Path: + cwd = cwd or Path.cwd() + return cwd / PROJECT_CONFIG_DIRNAME / f"{SKILL}.json" + + +def _read_json(path: Path) -> dict[str, Any] | None: + try: + with path.open(encoding="utf-8") as f: + data = json.load(f) + return data if isinstance(data, dict) else None + except (FileNotFoundError, json.JSONDecodeError, OSError): + return None + + +def _deep_merge(base: dict[str, Any], override: dict[str, Any]) -> dict[str, Any]: + out = dict(base) + for k, v in override.items(): + if isinstance(v, dict) and isinstance(out.get(k), dict): + out[k] = _deep_merge(out[k], v) + else: + out[k] = v + return out + + +def load_config(cwd: Path | None = None) -> dict[str, Any]: + config = dict(DEFAULTS) + if os.environ.get("RESEARCH_OPS_NO_CONFIG") == "1": + return config + global_cfg = _read_json(GLOBAL_CONFIG_PATH) + if global_cfg: + config = _deep_merge(config, global_cfg) + project_cfg = _read_json(project_config_path(cwd)) + if project_cfg: + config = _deep_merge(config, project_cfg) + return config + + +def setup_completed() -> bool: + cfg = _read_json(GLOBAL_CONFIG_PATH) or _read_json(project_config_path()) + return bool(cfg and cfg.get("setup_completed_at")) + + +def write_config(config: dict[str, Any], scope: str = "global", cwd: Path | None = None) -> Path: + path = project_config_path(cwd) if scope == "project" else GLOBAL_CONFIG_PATH + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as f: + json.dump(config, f, indent=2, sort_keys=True) + return path + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Inspect {SKILL} customization config.") + p.add_argument("--show", action="store_true", help="Print the effective config") + p.add_argument("--status", action="store_true", help="Print setup status + paths") + p.add_argument("--sample", action="store_true", help="Print the built-in defaults") + args = p.parse_args(argv) + + if args.sample: + print(json.dumps(DEFAULTS, indent=2, sort_keys=True)) + elif args.status: + print(json.dumps({ + "skill": SKILL, + "global_config_path": str(GLOBAL_CONFIG_PATH), + "global_config_exists": GLOBAL_CONFIG_PATH.exists(), + "project_config_path": str(project_config_path()), + "project_config_exists": project_config_path().exists(), + "setup_completed": setup_completed(), + }, indent=2)) + else: + print(json.dumps(load_config(), indent=2, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/market-research/scripts/market_sizer.py b/research-ops/skills/market-research/scripts/market_sizer.py new file mode 100644 index 00000000..ac028ea5 --- /dev/null +++ b/research-ops/skills/market-research/scripts/market_sizer.py @@ -0,0 +1,162 @@ +#!/usr/bin/env python3 +"""market_sizer.py - Compute TAM / SAM / SOM by BOTH top-down and bottoms-up methods. + +Stdlib-only. Deterministic. NO LLM calls. NEVER returns a single number: it computes both +methods side-by-side, reports the delta, and prints a mandatory method + assumptions block. + +Top-down: TAM = total_market_value ; SAM = TAM * serviceable_fraction ; SOM = SAM * reachable_share +Bottoms-up: TAM = total_potential_customers * annual_price + SAM = TAM * serviceable_fraction + SOM = SAM * realistic_adoption (capacity-constrained) + +If the two TAMs diverge by more than the tolerance, the tool flags it: triangulation failed. + +Usage: + python3 market_sizer.py --sample + python3 market_sizer.py --input market.json --method both + python3 market_sizer.py --input market.json --profile b2b-saas --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +# Profiles tune the divergence tolerance and a sanity note. +PROFILES = { + "b2b-saas": {"tolerance": 0.30}, + "consumer": {"tolerance": 0.40}, + "enterprise": {"tolerance": 0.25}, + "marketplace": {"tolerance": 0.40}, + "hardware": {"tolerance": 0.30}, + "services": {"tolerance": 0.35}, +} + +SAMPLE = { + "market_name": "Mid-market HR analytics SaaS (US)", + "top_down": { + "total_market_value": 4200000000, + "serviceable_fraction": 0.35, + "reachable_share": 0.04, + }, + "bottoms_up": { + "total_potential_customers": 62000, + "annual_price": 18000, + "serviceable_fraction": 0.35, + "realistic_adoption": 0.03, + }, +} + + +def top_down(td: dict) -> dict: + tam = float(td.get("total_market_value", 0.0)) + sam = tam * float(td.get("serviceable_fraction", 0.0)) + som = sam * float(td.get("reachable_share", 0.0)) + return {"method": "top-down", "TAM": round(tam, 0), "SAM": round(sam, 0), "SOM": round(som, 0)} + + +def bottoms_up(bu: dict) -> dict: + customers = float(bu.get("total_potential_customers", 0.0)) + price = float(bu.get("annual_price", 0.0)) + tam = customers * price + sam = tam * float(bu.get("serviceable_fraction", 0.0)) + som = sam * float(bu.get("realistic_adoption", 0.0)) + return {"method": "bottoms-up", "TAM": round(tam, 0), "SAM": round(sam, 0), + "SOM": round(som, 0), "implied_customers_at_SOM": round((sam / price) * float(bu.get("realistic_adoption", 0.0))) if price else None} + + +def size_market(data: dict, method: str, profile: str) -> dict: + if profile not in PROFILES: + raise ValueError(f"Unknown profile '{profile}'. Choose from {list(PROFILES)}.") + tol = PROFILES[profile]["tolerance"] + out = {"market_name": data.get("market_name", "UNSPECIFIED"), "profile": profile} + + td = top_down(data.get("top_down", {})) if method in ("top-down", "both") else None + bu = bottoms_up(data.get("bottoms_up", {})) if method in ("bottoms-up", "both") else None + if td: + out["top_down"] = td + if bu: + out["bottoms_up"] = bu + + flags = [] + if td and bu and td["TAM"] > 0: + delta = abs(td["TAM"] - bu["TAM"]) / td["TAM"] + out["tam_divergence"] = round(delta, 3) + if delta > tol: + flags.append(f"TRIANGULATION FAILED: top-down and bottoms-up TAM differ by {delta:.0%} " + f"(> {tol:.0%} tolerance). Reconcile before quoting a number.") + else: + flags.append(f"Triangulation OK: TAMs within {delta:.0%} (tolerance {tol:.0%}).") + out["flags"] = flags + out["method_and_assumptions"] = [ + "NEVER quote a single TAM number without stating the method and the assumptions below.", + "Top-down TAM = total market value (cite the source: analyst report, gov stat).", + "Bottoms-up TAM = total potential customers x annual price (cite both counts).", + "SAM = TAM x serviceable fraction (geography/segment you can actually serve).", + "SOM = SAM x realistic, capacity-constrained share you can win in the planning window.", + ] + return out + + +def _fmt(n): + return f"${n:,.0f}" if isinstance(n, (int, float)) else str(n) + + +def _render_human(r: dict) -> str: + lines = [f"Market Sizing: {r['market_name']} (profile: {r['profile']})", ""] + for key in ("top_down", "bottoms_up"): + if key in r: + m = r[key] + lines.append(f" [{m['method']}] TAM {_fmt(m['TAM'])} | SAM {_fmt(m['SAM'])} | SOM {_fmt(m['SOM'])}") + if m.get("implied_customers_at_SOM") is not None: + lines.append(f" implied customers at SOM: {m['implied_customers_at_SOM']:,}") + if "tam_divergence" in r: + lines.append(f" TAM divergence (top-down vs bottoms-up): {r['tam_divergence']:.1%}") + lines.append("") + for f in r["flags"]: + lines.append(f" ! {f}") + lines.append("") + lines.append("Method & assumptions (must travel with the number):") + for a in r["method_and_assumptions"]: + lines.append(f" - {a}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Compute TAM/SAM/SOM by top-down AND bottoms-up (never a single number).") + p.add_argument("--input", help="Path to JSON with top_down{} and bottoms_up{}") + p.add_argument("--method", choices=["top-down", "bottoms-up", "both"], default=None, + help="overrides onboarding sizing_method") + p.add_argument("--profile", default=None, choices=list(PROFILES), + help="overrides onboarding default_profile") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + method = args.method or conf.get("sizing_method", "both") + profile = args.profile or conf.get("default_profile", "b2b-saas") + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + try: + result = size_market(data, method, profile) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/market-research/scripts/onboard.py b/research-ops/skills/market-research/scripts/onboard.py new file mode 100644 index 00000000..e604b321 --- /dev/null +++ b/research-ops/skills/market-research/scripts/onboard.py @@ -0,0 +1,117 @@ +#!/usr/bin/env python3 +"""onboard.py - Onboarding questionnaire for the market-research skill. + +Stdlib-only. Asks the user a short set of questions BEFORE they size a market or field +a survey, then writes the answers to a customization config read by every tool in this +skill via config_loader.py. The answers become defaults for profile, survey confidence, +margin of error, and sizing method. + +Modes: --show | --defaults | --set key=value (repeatable) | --reset | --scope {global,project} +""" + +from __future__ import annotations + +import argparse +import datetime as _dt +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 + +NUMERIC_KEYS = {"default_confidence", "default_moe"} + +QUESTIONS = [ + ("default_profile", + "1. What market are you researching?", + ["b2b-saas", "consumer", "enterprise", "marketplace", "hardware", "services"], str), + ("default_confidence", + "2. Default survey confidence level?", + ["0.80", "0.85", "0.90", "0.95", "0.99"], float), + ("default_moe", + "3. Default survey margin of error (fraction, e.g. 0.05)?", + None, float), + ("sizing_method", + "4. Default market-sizing method?", + ["top-down", "bottoms-up", "both"], str), +] + + +def _print_questions() -> None: + print(f"Onboarding questions — {cfg.SKILL}:\n") + for _k, prompt, choices, _c in QUESTIONS: + line = f" {prompt}" + if choices: + line += f" [{' / '.join(choices)}]" + print(line) + + +def run_interactive(config: dict) -> dict: + print(f"Onboarding — {cfg.SKILL}. Press Enter to keep the current/default value.\n") + for key, prompt, choices, caster in QUESTIONS: + suffix = f" [{'/'.join(choices)}]" if choices else "" + cur = f" (current: {config.get(key)})" if config.get(key) is not None else "" + raw = input(f"{prompt}{suffix}{cur}: ").strip() + if not raw: + continue + try: + config[key] = caster(raw) + except ValueError: + print(f" ! invalid value for {key}, keeping current") + return config + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Onboarding for the {cfg.SKILL} skill.") + p.add_argument("--show", action="store_true") + p.add_argument("--defaults", action="store_true", help="write built-in defaults, no prompt") + p.add_argument("--set", action="append", default=[], metavar="key=value") + p.add_argument("--reset", action="store_true") + p.add_argument("--scope", choices=["global", "project"], default="global") + args = p.parse_args(argv) + + if args.show: + _print_questions() + print("\nCurrent effective config:") + print(json.dumps(cfg.load_config(), indent=2, sort_keys=True)) + return 0 + + if args.reset: + path = cfg.project_config_path() if args.scope == "project" else cfg.GLOBAL_CONFIG_PATH + if path.exists(): + path.unlink(); print(f"removed {path}") + else: + print(f"no config at {path}") + return 0 + + config = cfg.load_config() + + if args.set: + for item in args.set: + if "=" not in item: + print(f"error: --set expects key=value, got '{item}'", file=sys.stderr) + return 2 + k, v = item.split("=", 1) + if k in NUMERIC_KEYS: + try: + v = float(v) + except ValueError: + pass + config[k] = v + elif not args.defaults: + if sys.stdin.isatty(): + config = run_interactive(config) + else: + print("non-interactive shell: use --defaults or --set key=value. Showing questions:\n") + _print_questions() + return 0 + + config["setup_completed_at"] = _dt.datetime.now(_dt.timezone.utc).isoformat() + path = cfg.write_config(config, scope=args.scope) + print(f"saved {cfg.SKILL} customization -> {path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/market-research/scripts/sample_size_planner.py b/research-ops/skills/market-research/scripts/sample_size_planner.py new file mode 100644 index 00000000..de932645 --- /dev/null +++ b/research-ops/skills/market-research/scripts/sample_size_planner.py @@ -0,0 +1,183 @@ +#!/usr/bin/env python3 +"""sample_size_planner.py - Survey sample-size with finite-population correction + per-segment minima. + +Stdlib-only. Deterministic. NO LLM calls. + +Computes the classic proportion-estimate sample size: + n0 = z^2 * p * (1-p) / e^2 +then applies the finite-population correction (FPC) when a population N is given: + n = n0 / (1 + (n0 - 1)/N) +Also computes per-segment minimums and a proportional quota allocation, because a survey +powered overall is NOT powered per reported segment. + +Usage: + python3 sample_size_planner.py --sample + python3 sample_size_planner.py --population 62000 --confidence 0.95 --moe 0.05 --proportion 0.5 + python3 sample_size_planner.py --input survey.json --output json +""" + +from __future__ import annotations + +import argparse +import json +import math +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +Z = {0.80: 1.2816, 0.85: 1.4395, 0.90: 1.6449, 0.95: 1.9600, 0.99: 2.5758} + +SAMPLE = { + "population": 62000, + "confidence": 0.95, + "margin_of_error": 0.05, + "expected_proportion": 0.5, + "segments": [ + {"name": "1-50 employees", "population_share": 0.55}, + {"name": "51-250 employees", "population_share": 0.30}, + {"name": "251-1000 employees", "population_share": 0.15}, + ], + "segment_moe": 0.08, +} + + +def _z(conf: float) -> float: + if conf not in Z: + raise ValueError(f"confidence must be one of {sorted(Z)}.") + return Z[conf] + + +def base_n(conf: float, moe: float, p: float, population: float | None) -> dict: + if not 0.0 < p < 1.0: + raise ValueError("expected_proportion must be in (0,1).") + if not 0.0 < moe < 1.0: + raise ValueError("margin_of_error must be in (0,1).") + z = _z(conf) + n0 = (z ** 2) * p * (1 - p) / (moe ** 2) + if population and population > 0: + n = n0 / (1 + (n0 - 1) / population) + fpc_applied = True + else: + n = n0 + fpc_applied = False + return { + "confidence": conf, + "margin_of_error": moe, + "expected_proportion": p, + "population": population, + "n_unadjusted": math.ceil(n0), + "n_with_fpc": math.ceil(n), + "fpc_applied": fpc_applied, + "z": z, + } + + +def segment_plan(data: dict, overall: dict) -> dict: + seg_moe = float(data.get("segment_moe", data.get("margin_of_error", 0.05))) + conf = float(data.get("confidence", 0.95)) + p = float(data.get("expected_proportion", 0.5)) + z = _z(conf) + per_seg_min = math.ceil((z ** 2) * p * (1 - p) / (seg_moe ** 2)) + + segs = data.get("segments", []) + total_quota = max(overall["n_with_fpc"], per_seg_min * len(segs)) if segs else overall["n_with_fpc"] + out_segs = [] + for s in segs: + share = float(s.get("population_share", 0.0)) + proportional = math.ceil(total_quota * share) + quota = max(proportional, per_seg_min) + out_segs.append({ + "name": s.get("name", "UNNAMED"), + "population_share": share, + "proportional_quota": proportional, + "minimum_for_segment_moe": per_seg_min, + "recommended_quota": quota, + }) + return { + "segment_margin_of_error": seg_moe, + "minimum_per_segment": per_seg_min, + "recommended_total_with_segment_floors": sum(s["recommended_quota"] for s in out_segs) if out_segs else total_quota, + "segments": out_segs, + } + + +def plan(data: dict) -> dict: + overall = base_n( + float(data.get("confidence", 0.95)), + float(data.get("margin_of_error", 0.05)), + float(data.get("expected_proportion", 0.5)), + data.get("population"), + ) + result = {"overall": overall} + if data.get("segments"): + result["segmentation"] = segment_plan(data, overall) + result["notes"] = [ + "A survey powered overall is NOT powered per reported segment — fund the segment floors.", + "expected_proportion=0.5 is the conservative (maximum-variance) default.", + "FPC matters when the sample is a large fraction of the population (small N).", + ] + return result + + +def _render_human(r: dict) -> str: + o = r["overall"] + lines = ["Survey Sample-Size Plan", "", + f" Confidence: {o['confidence']:.0%} MoE: {o['margin_of_error']:.0%} p: {o['expected_proportion']}", + f" n (unadjusted): {o['n_unadjusted']}", + f" n (with FPC): {o['n_with_fpc']} (population {o['population']}, fpc_applied={o['fpc_applied']})"] + if "segmentation" in r: + s = r["segmentation"] + lines += ["", f" Per-segment MoE: {s['segment_margin_of_error']:.0%} => minimum {s['minimum_per_segment']} per segment", + f" Recommended total with segment floors: {s['recommended_total_with_segment_floors']}", ""] + for seg in s["segments"]: + lines.append(f" {seg['name']:24s} share {seg['population_share']:.0%} " + f"proportional {seg['proportional_quota']} recommended {seg['recommended_quota']}") + lines += ["", "Notes:"] + for n in r["notes"]: + lines.append(f" - {n}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Survey sample size with FPC + per-segment minima.") + p.add_argument("--input", help="Path to JSON survey spec") + p.add_argument("--population", type=float, default=None) + p.add_argument("--confidence", type=float, default=None, help="overrides onboarding default_confidence") + p.add_argument("--moe", type=float, default=None, help="margin of error (overrides onboarding default_moe)") + p.add_argument("--proportion", type=float, default=0.5, help="expected proportion") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + confidence = args.confidence if args.confidence is not None else conf.get("default_confidence", 0.95) + moe = args.moe if args.moe is not None else conf.get("default_moe", 0.05) + + if args.sample: + data = SAMPLE + elif args.input: + data = json.load(open(args.input)) + else: + data = {"population": args.population, "confidence": confidence, + "margin_of_error": moe, "expected_proportion": args.proportion} + + try: + result = plan(data) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/market-research/scripts/segmentation_scorer.py b/research-ops/skills/market-research/scripts/segmentation_scorer.py new file mode 100644 index 00000000..5f3f997c --- /dev/null +++ b/research-ops/skills/market-research/scripts/segmentation_scorer.py @@ -0,0 +1,139 @@ +#!/usr/bin/env python3 +"""segmentation_scorer.py - Score candidate market segments against Kotler's actionability criteria. + +Stdlib-only. Deterministic. NO LLM calls. + +Each candidate segment is scored 0-100 across the five Kotler criteria for a useful segment: + 1. measurable can you size and identify it? + 2. substantial is it large/profitable enough to serve? + 3. accessible can you reach it through channels? + 4. differentiable does it respond differently from other segments? + 5. actionable can you design and execute a program for it? + +Segments failing the substantiality or accessibility gates are flagged: a demographic slice +that is unreachable or too small is not a market segment. + +Usage: + python3 segmentation_scorer.py --sample + python3 segmentation_scorer.py --input segments.json --profile enterprise + python3 segmentation_scorer.py --input segments.json --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +CRITERIA = ["measurable", "substantial", "accessible", "differentiable", "actionable"] +WEIGHTS = {"measurable": 0.15, "substantial": 0.25, "accessible": 0.25, + "differentiable": 0.20, "actionable": 0.15} +GATE_FLOOR = 40.0 # substantiality + accessibility gate + +# Profiles nudge the substantiality expectation (enterprise tolerates smaller, higher-value segments). +PROFILES = { + "b2b-saas": 1.0, + "consumer": 1.0, + "enterprise": 0.85, + "marketplace": 1.0, + "hardware": 1.0, + "services": 0.9, +} + +SAMPLE = { + "segments": [ + {"name": "Mid-market HR teams (51-250 emp)", + "scores": {"measurable": 85, "substantial": 80, "accessible": 75, "differentiable": 70, "actionable": 80}}, + {"name": "Solopreneurs who 'might' want analytics", + "scores": {"measurable": 40, "substantial": 30, "accessible": 35, "differentiable": 30, "actionable": 40}}, + {"name": "Enterprise CHROs (1000+ emp)", + "scores": {"measurable": 90, "substantial": 95, "accessible": 50, "differentiable": 85, "actionable": 70}}, + ], +} + + +def score_segment(seg: dict, sub_mult: float) -> dict: + raw = seg.get("scores", {}) + breakdown = {} + composite = 0.0 + for c in CRITERIA: + s = float(raw.get(c, 0.0)) + if c == "substantial": + s = min(100.0, s / sub_mult) # enterprise: smaller segments still count (divide by <1 raises) + breakdown[c] = round(s, 1) + composite += s * WEIGHTS[c] + flags = [] + if breakdown["substantial"] < GATE_FLOOR: + flags.append("FAILS SUBSTANTIALITY GATE: too small/unprofitable to be a target segment.") + if breakdown["accessible"] < GATE_FLOOR: + flags.append("FAILS ACCESSIBILITY GATE: no viable channel to reach it.") + verdict = "DROP" if flags else ("TARGET" if composite >= 65 else "WATCH") + return { + "name": seg.get("name", "UNNAMED"), + "composite": round(composite, 1), + "breakdown": breakdown, + "flags": flags, + "verdict": verdict, + } + + +def evaluate(data: dict, profile: str) -> dict: + if profile not in PROFILES: + raise ValueError(f"Unknown profile '{profile}'. Choose from {list(PROFILES)}.") + mult = PROFILES[profile] + scored = sorted((score_segment(s, mult) for s in data.get("segments", [])), + key=lambda x: x["composite"], reverse=True) + return { + "profile": profile, + "segments": scored, + "note": "A demographic or firmographic slice is not a segment unless it is substantial AND accessible.", + } + + +def _render_human(r: dict) -> str: + lines = [f"Segmentation Scoring (profile: {r['profile']})", ""] + for s in r["segments"]: + lines.append(f"[{s['verdict']}] {s['name']} — composite {s['composite']}/100") + for c, v in s["breakdown"].items(): + lines.append(f" {c:16s} {v}") + for f in s["flags"]: + lines.append(f" ! {f}") + lines.append("") + lines.append(f"note: {r['note']}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Score market segments against Kotler's 5 criteria.") + p.add_argument("--input", help="Path to JSON with segments[]") + p.add_argument("--profile", default=None, choices=list(PROFILES), + help="overrides onboarding default_profile") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + profile = args.profile or conf.get("default_profile", "b2b-saas") + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + try: + result = evaluate(data, profile) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/product-research/SKILL.md b/research-ops/skills/product-research/SKILL.md new file mode 100644 index 00000000..1a1768c5 --- /dev/null +++ b/research-ops/skills/product-research/SKILL.md @@ -0,0 +1,145 @@ +--- +name: product-research +description: Use when planning and synthesizing product/user research as a method-and-repository discipline — selecting the right method for the goal (generative interviews vs usability test vs concept test vs validation), computing method-based saturation/sample size with an explicit confidence level, or synthesizing coded observations into insights while flagging single-source anecdotes. Never fabricates user insight; an insight requires recurrence across independent participants. Distinct from product-team/ux-researcher-designer (persona/journey artifacts), product-discovery (discovery-sprint planning), and experiment-designer (live A/B) — this is the research-ops method + insight-repository layer. +version: 2.9.0 +author: claude-code-skills +license: MIT +tags: [research-ops, product-research, ux-research, jtbd, usability, saturation, insight-synthesis, research-repository] +compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli] +--- + +# product-research + +Product / user research as an operational discipline: choosing the right method, sizing it honestly, and synthesizing findings into governed insights. The core rule: **method must match the goal**, and **an insight requires recurrence across independent participants** — a single quote is an anecdote. + +## Purpose + +Product researchers, ResearchOps teams, and PMs running discovery need method rigor and an insight repository they can trust. This skill structures three decisions: + +Three deterministic tools: + +1. `study_designer.py` — Maps (research goal × product stage) to an appropriate method and emits a method-matched plan skeleton (objective, participant criteria, guide structure, success criteria). Redirects live A/B to `product-team/experiment-designer`. +2. `saturation_planner.py` — Method-based sample guidance with an explicit **confidence label**: Nielsen problem-discovery (5/segment), Guest et al. thematic saturation (~12), and evaluative coverage. Never claims a prevalence rate from a small-n usability test. +3. `insight_synthesizer.py` — Clusters coded observations by tag, counts distinct participants, ranks by cross-participant recurrence, and flags any candidate below the source threshold as an **ANECDOTE**, never promoting it to an insight. + +## When to use + +Invoke this skill when: + +- You are planning a study and need the method to match the goal (generative vs evaluative vs validation). +- You need a defensible sample size / saturation rationale with a stated confidence. +- You have raw coded observations and need to synthesize insights without over-claiming. +- You are setting up or auditing a research repository and need the insight-vs-observation discipline. + +**Do NOT use this skill to**: generate personas / journey maps (use `product-team/ux-researcher-designer`), plan a discovery sprint or validate an opportunity (use `product-team/product-discovery`), design or analyze a live product A/B experiment (use `product-team/experiment-designer`), or do market sizing / surveys (use the `market-research` sibling). + +## Workflow + +1. **Frame the study** — Fill `assets/research_plan_template.md` (research questions, method rationale, participant criteria, analysis plan, repository tagging scheme). +2. **Pick the method** — Run `study_designer.py --goal {discovery|evaluative|validation} --stage {concept|prototype|beta|live} --profile {b2b-saas|consumer-app|enterprise|marketplace|hardware|platform}`. Honor the redirect if it routes to experiment-designer. +3. **Size it** — Run `saturation_planner.py --method {usability|thematic|evaluative-coverage} --segments N`. Record the confidence label and limits. +4. **Synthesize** — After fielding, code observations and run `insight_synthesizer.py --input observations.json --min-sources 3`. Treat ANECDOTE-flagged clusters as signals to probe, not findings to ship. +5. **File in the repository** — Tag insights to the atomic schema at synthesis time, with their evidence and confidence. + +## Scripts + +| Script | Purpose | Profiles | +|---|---|---| +| `scripts/study_designer.py` | (goal × stage) → method + plan skeleton | b2b-saas, consumer-app, enterprise, marketplace, hardware, platform | +| `scripts/saturation_planner.py` | Method-based sample guidance + confidence | n/a (method-driven) | +| `scripts/insight_synthesizer.py` | Cluster observations, flag anecdotes | n/a (evidence-driven) | + +All three: stdlib-only, `--help`, `--sample`, `--output {human,json}`. + +## Onboarding & customization + +Run the onboarding questionnaire **once before you start** — it captures your defaults so every tool in this skill is pre-configured. Customization is the point: the answers actually change tool behavior (e.g. the insight source-threshold). + +```bash +python3 scripts/onboard.py # interactive (also: --defaults, --set key=value, --reset) +python3 scripts/onboard.py --show # see the questions + current effective config +``` + +Answers are saved to `~/.config/research-ops/product-research.json` (global) or `./.research-ops/product-research.json` (`--scope project`) and are read automatically by `config_loader.py`. They set the default product **profile**, the **insight source-threshold** (how many independent participants make a finding an insight, not an anecdote), the default **saturation method**, and the **high-stakes** flag. CLI flags always override saved config; `RESEARCH_OPS_NO_CONFIG=1` ignores it. + +**The four questions:** product profile · insight source-threshold · saturation method · high-stakes flag. + +## Optimize with autoresearch (opt-in) + +This skill ships an **isolated, opt-in** bridge to `engineering/autoresearch-agent`. Only when you ask to "optimize the synthesis" / "run a loop" does an autoresearch experiment iteratively refine the coding/clustering of a fixed evidence set so more cross-participant patterns surface. `scripts/ar_evaluator.py` is the ground-truth evaluator; it prints `validated_insights: ` (higher is better). It optimizes the **coding**, never fabricates evidence. + +```bash +/ar:setup --domain custom --name insight-synthesis \ + --target observations.json \ + --eval "python3 ar_evaluator.py --target observations.json" \ + --metric validated_insights --direction higher +/ar:loop custom/insight-synthesis +``` + +Isolated: no hard dependency — autoresearch runs only on demand, and the loop edits `observations.json`, never the evaluator. + +## References + +- `references/research_methods_canon.md` — Portigal *Interviewing Users*; Christensen/Ulwick JTBD; Rohrer's UX-research methods landscape (NN/g); Sauro & Lewis *Quantifying the User Experience*; Goodman/Kuniavsky. +- `references/sampling_and_saturation.md` — Nielsen "test with 5 users"; Guest, Bunce & Johnson saturation; Faulkner on more-than-5; Sauro usability sample size; Braun & Clarke thematic analysis. +- `references/repository_and_synthesis.md` — ResearchOps / atomic research (Tomer Sharon "Polaris"); insight-vs-observation discipline; repository governance; affinity mapping; democratization guardrails. + +## Assumptions + +- Method selection assumes you can name the goal honestly; if the goal is fuzzy, grill it first (the goal drives everything). +- Saturation guidance is method-based, not a power calculation — usability tests find problems, not prevalence rates. +- The synthesizer counts evidence you provide; coding quality is upstream of it. Garbage tags → garbage clusters. +- The insight threshold (`--min-sources`) defaults to 3; raise it for high-stakes or heterogeneous populations. + +## Anti-patterns + +- **Mismatching method to goal.** A usability test cannot discover unmet needs; an interview cannot measure task success. +- **Reporting usability problems as percentages.** Small-n tests surface problems, not population rates. +- **Promoting an anecdote to an insight.** One participant is a signal to probe, not a finding. +- **Framing interview questions as feature reactions.** Probe the job-to-be-done and recent real behavior, not hypothetical opinions. +- **Synthesizing without a repository scheme.** Tag at synthesis time, or insights rot unfindable. + +## Distinct from + +| Neighbor | Scope | Difference | +|---|---|---| +| `product-team/ux-researcher-designer` | Personas, journey maps, usability frameworks tied to design output | That produces **artifacts**; this is **method + repository discipline** | +| `product-team/product-discovery` | Opportunity validation, discovery-sprint planning | That plans **discovery sprints**; this designs and synthesizes the **research** | +| `product-team/experiment-designer` | Live product A/B hypothesis + sample size | That runs **live experiments**; this runs **qualitative/evaluative research** | +| `market-research` (sibling) | Market sizing, surveys, segmentation | That studies **the market**; this studies **users** | + +## Quick examples + +```bash +python3 scripts/study_designer.py --sample +python3 scripts/saturation_planner.py --method thematic --segments 3 +python3 scripts/insight_synthesizer.py --sample --min-sources 3 +``` + +The synthesizer sample correctly promotes "import-confusion" (3 independent participants) to INSIGHT and flags "wants-slack" (1 participant) as an ANECDOTE. + +## Forcing-question library (Matt Pocock grill discipline) + +Walked one at a time by `/cs:grill-research-ops` or the orchestrator. Recommended answer + canon citation per question. Never bundled. + +1. **"Is this study generative (discover problems) or evaluative (test a solution)?"** + Recommended: name it first — the method follows from the goal. + Canon: Rohrer, *When to Use Which User-Experience Research Methods* (NN/g). + +2. **"What's your sample size and saturation rationale — and at what confidence?"** + Recommended: method-based n (5/segment usability; ~12 for thematic saturation), state the confidence. + Canon: Nielsen; Guest, Bunce & Johnson (2006); Faulkner (2003). + +3. **"How many independent participants support each insight — or is it a single-source anecdote?"** + Recommended: require recurrence across ≥3 sources before calling it an insight; flag singletons. + Canon: atomic research / ResearchOps; Braun & Clarke thematic analysis. + +4. **"Are your interview / usability tasks framed as outcomes (jobs) or as feature reactions?"** + Recommended: frame around the job-to-be-done and recent real behavior, not hypothetical opinion. + Canon: Christensen/Ulwick Jobs-to-be-Done; Portigal *Interviewing Users*. + +5. **"Where does this land in the repository, and how is it tagged for reuse?"** + Recommended: tag to the atomic schema at synthesis time, not later. + Canon: Tomer Sharon, *Polaris* / ResearchOps repository practice. + +Walk depth-first. Lock 1-2 before opening 3-5. After all are answered, invoke `study_designer.py` → `saturation_planner.py` → (after fielding) `insight_synthesizer.py`. diff --git a/research-ops/skills/product-research/assets/research_plan_template.md b/research-ops/skills/product-research/assets/research_plan_template.md new file mode 100644 index 00000000..f0b59165 --- /dev/null +++ b/research-ops/skills/product-research/assets/research_plan_template.md @@ -0,0 +1,50 @@ +# Product Research Plan — Template + +> Fill this before running the tools. Method must match the goal. An insight requires +> recurrence across independent participants — a single quote is an anecdote. + +## 1. Study identification +- Study name: +- Product / feature: +- Stage: [concept | prototype | beta | live] +- Profile: [b2b-saas | consumer-app | enterprise | marketplace | hardware | platform] + +## 2. Goal & questions +- Goal: [discovery (generative) | evaluative | validation] +- Research questions (3-5, answerable, not leading): +- The product decision this informs: + +## 3. Method (from `study_designer.py`) +- Recommended method: +- Why it matches the goal: +- (If live A/B → route to product-team/experiment-designer.) + +## 4. Participants +- Target segment(s) + screener (screen for the job, not a job title): +- Per-segment recruiting if reporting per segment? [yes/no] +- Exclusions (internal, biased, repeat): + +## 5. Sample & saturation (from `saturation_planner.py`) +- Method: [usability | thematic | evaluative-coverage] +- n per segment + total: +- Confidence label + limits: + +## 6. Study guide skeleton +1. +2. +3. +4. +5. + +## 7. Analysis & synthesis +- Coding / tagging scheme (atomic taxonomy): +- Insight threshold (min distinct participants): ___ (default 3) +- Synthesis tool: `insight_synthesizer.py` + +## 8. Repository +- Where insights are filed + tagging taxonomy: +- Evidence linked to each insight? [yes — required] +- Confidence field per insight? [yes — required] + +## 9. Confidence statement +- What this study can and cannot support: diff --git a/research-ops/skills/product-research/references/repository_and_synthesis.md b/research-ops/skills/product-research/references/repository_and_synthesis.md new file mode 100644 index 00000000..6df09fd8 --- /dev/null +++ b/research-ops/skills/product-research/references/repository_and_synthesis.md @@ -0,0 +1,28 @@ +# Research Repository and Synthesis + +Reference for turning observations into governed insights. Pairs with `insight_synthesizer.py`. + +## Observation vs insight + +The foundational discipline of ResearchOps is the distinction between an **observation** (a single piece of evidence — one participant did or said one thing) and an **insight** (a pattern that recurs across independent sources and carries an implication). Promoting an observation to an insight because it was vivid or confirmed a prior is the cardinal sin of synthesis. The synthesizer enforces a source threshold: a candidate supported by fewer than the threshold of distinct participants is labeled an ANECDOTE and is never promoted. + +## Atomic research + +Tomer Sharon's **atomic research** model (and the "Polaris" repository concept) decomposes research into reusable units: *Experiments → Facts (observations) → Insights → Recommendations*. Facts are tagged and stored so that insights can be traced back to evidence and reused across studies. The payoff is a repository where a claim can always be drilled down to the observations that support it — and where the same evidence can support future questions. + +## Affinity mapping + +The classic synthesis technique is affinity mapping: cluster observations into emergent themes bottom-up, then name the themes. The `insight_synthesizer.py` tool is a deterministic, tag-based proxy for this — it clusters by the codes you assign and ranks by cross-participant recurrence. The human still does the interpretive naming; the tool enforces the counting discipline. + +## Repository governance and democratization + +As organizations democratize research (PMs and designers running their own studies), the repository becomes the guardrail. Governance practices: a consistent tagging taxonomy, evidence linked to every insight, a confidence field, and a review step before an insight is marked "validated." Without governance, democratized research produces a pile of unsearchable anecdotes; with it, the repository compounds in value. + +## Sources + +1. Sharon, T., *Validating Product Ideas Through Lean User Research* (Rosenfeld, 2016) and the atomic-research / Polaris model. +2. ResearchOps Community, *Research Repositories* and *Democratization* working-group reports. +3. Braun, V., & Clarke, V., *Thematic Analysis: A Practical Guide* (Sage, 2022). +4. Beyer, H., & Holtzblatt, K., *Contextual Design* (1998) — affinity diagramming. +5. Dovetail / EnjoyHQ practitioner guides on insight repositories and tagging taxonomies. +6. Kaplan, K., *Taxonomy 101* and *Research Repositories* — Nielsen Norman Group. diff --git a/research-ops/skills/product-research/references/research_methods_canon.md b/research-ops/skills/product-research/references/research_methods_canon.md new file mode 100644 index 00000000..d54e566d --- /dev/null +++ b/research-ops/skills/product-research/references/research_methods_canon.md @@ -0,0 +1,34 @@ +# Product Research Methods Canon + +Reference for method selection. Pairs with `study_designer.py`. + +## The two-axis map + +UX/product research methods sort along two axes (Rohrer, NN/g): **attitudinal vs behavioral** (what people say vs what they do) and **qualitative vs quantitative** (why/how vs how-many). The single most important pre-method decision is the **goal**: + +- **Generative (discovery)** — you don't yet know the problem. Methods: semi-structured interviews, contextual inquiry, diary studies. Output: themes, unmet needs, jobs-to-be-done. +- **Evaluative** — you have a solution and want to know if it works. Methods: moderated/unmoderated usability tests, concept tests. Output: task-success, severity-rated problems. +- **Validation** — you want to confirm demand/desirability before building. Methods: surveys, preference tests, fake-door tests, and (when live) A/B experiments. + +Picking an evaluative method for a generative goal — "let's usability-test our way to product strategy" — is the most common and most expensive error. + +## Interviewing discipline + +Steve Portigal's *Interviewing Users* is the operative craft reference: ask about **recent, specific, real behavior** ("tell me about the last time you…"), not hypotheticals or opinions ("would you use…"). People are unreliable narrators of their future selves but good storytellers of their past. + +## Jobs-to-be-Done + +Christensen's and Ulwick's JTBD reframes research around the **progress a person is trying to make** in a circumstance, not their demographics or feature preferences. Outcome-Driven Innovation (Ulwick) operationalizes this into measurable desired outcomes — a bridge between qualitative discovery and quantitative validation. + +## Mixed methods + +Strong research triangulates: qualitative discovery surfaces hypotheses; quantitative validation sizes them. Sauro & Lewis (*Quantifying the User Experience*) provides the statistical backbone for turning usability observations into defensible metrics (task time, completion, SUS) without over-claiming from small samples. + +## Sources + +1. Portigal, S., *Interviewing Users*, 2nd ed. (Rosenfeld, 2023). +2. Christensen, Hall, Dillon & Duncan, *Competing Against Luck* (2016) — Jobs-to-be-Done. +3. Ulwick, A., *Jobs to Be Done: Theory to Practice* (2016) — Outcome-Driven Innovation. +4. Rohrer, C., *When to Use Which User-Experience Research Methods* — Nielsen Norman Group. +5. Sauro, J., & Lewis, J., *Quantifying the User Experience*, 2nd ed. (Morgan Kaufmann, 2016). +6. Goodman, Kuniavsky & Moed, *Observing the User Experience*, 2nd ed. (2012). diff --git a/research-ops/skills/product-research/references/sampling_and_saturation.md b/research-ops/skills/product-research/references/sampling_and_saturation.md new file mode 100644 index 00000000..ff13bb77 --- /dev/null +++ b/research-ops/skills/product-research/references/sampling_and_saturation.md @@ -0,0 +1,29 @@ +# Sampling and Saturation + +Reference for how many participants. Pairs with `saturation_planner.py`. + +## Usability: the "5 users" result + +Nielsen and Landauer's model says the proportion of usability problems found with n users is 1 − (1 − p)ⁿ, where p is the average probability that a single user surfaces a given problem (~0.31 in their data). At n = 5, that is ~85% of problems — hence "test with 5 users." Two crucial caveats the planner enforces: + +1. **Per segment.** The 5-user result holds *within a homogeneous user group*. If you have distinct segments that behave differently, you need ~5 per segment. +2. **Problems, not rates.** A small-n usability test finds *whether* a problem exists; it cannot estimate the *prevalence* of that problem in the population. Never report "60% of users struggled" from a 5-person test. + +Faulkner (2003) showed real variance: while the average across many 5-person samples is ~85%, individual 5-person runs ranged from ~55% to 100%. When stakes or heterogeneity are high, run more. + +## Qualitative: thematic saturation + +For interview-based thematic research, Guest, Bunce & Johnson (2006) found that **saturation** — the point where new interviews stop yielding new themes — typically occurs by ~12 interviews in a homogeneous group, with the basic elements present by ~6. Saturation is **observed, not guaranteed**: track the new-theme rate and stop when it flattens, rather than committing to a fixed n blindly. Heterogeneous populations need more, and per-group saturation applies just as in usability. + +## Reporting confidence honestly + +The planner attaches a confidence label (LOW / MODERATE / MODERATE-HIGH) and explicit limits to every plan, because the failure mode in product research is not too-small samples per se — it is **over-claiming** from whatever sample you ran. State the method, the n, and what the method can and cannot support. + +## Sources + +1. Nielsen, J., & Landauer, T., *A mathematical model of the finding of usability problems* — INTERCHI 1993. +2. Nielsen, J., *Why You Only Need to Test with 5 Users* — NN/g (2000). +3. Faulkner, L., *Beyond the five-user assumption* — Behavior Research Methods 2003;35:379-383. +4. Guest, G., Bunce, A., & Johnson, L., *How many interviews are enough?* — Field Methods 2006;18:59-82. +5. Braun, V., & Clarke, V., *Using thematic analysis in psychology* — Qual Res Psychol 2006;3:77-101. +6. Sauro, J., & Lewis, J., *Quantifying the User Experience*, 2nd ed. (2016) — confidence intervals for small samples. diff --git a/research-ops/skills/product-research/scripts/ar_evaluator.py b/research-ops/skills/product-research/scripts/ar_evaluator.py new file mode 100644 index 00000000..2f9026e4 --- /dev/null +++ b/research-ops/skills/product-research/scripts/ar_evaluator.py @@ -0,0 +1,68 @@ +#!/usr/bin/env python3 +"""ar_evaluator.py - Autoresearch evaluator for the product-research skill (OPT-IN). + +Stdlib-only. The ISOLATED bridge to engineering/autoresearch-agent. It does NOT call +autoresearch; it is the ground-truth evaluator an autoresearch loop runs after editing +the target coded-observations file. It reads an observations JSON, runs insight_synthesizer +at the configured source threshold, and prints ONE metric line: + + validated_insights: (higher is better — clusters that clear the source threshold) + +This optimizes the CODING/synthesis of a fixed evidence set (merging/splitting tags so +cross-participant patterns surface) — not the evidence itself. The user opts in explicitly: + /ar:setup --domain custom --name insight-synthesis \\ + --target observations.json --eval "python3 ar_evaluator.py --target observations.json" \\ + --metric validated_insights --direction higher + +Direct use: + python3 ar_evaluator.py --sample + python3 ar_evaluator.py --target observations.json --min-sources 3 +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 +import insight_synthesizer as isyn # noqa: E402 + +METRIC = "validated_insights" + + +def main(argv: list[str] | None = None) -> int: + c = cfg.load_config() + p = argparse.ArgumentParser(description="Autoresearch evaluator: count of validated insights.") + p.add_argument("--target", help="path to observations JSON (or env AR_TARGET)") + p.add_argument("--min-sources", type=int, default=None, help="overrides onboarding insight_min_sources") + p.add_argument("--sample", action="store_true") + args = p.parse_args(argv) + + min_sources = args.min_sources if args.min_sources is not None else int(c.get("insight_min_sources", 3)) + + if args.sample: + data = isyn.SAMPLE + else: + target = args.target or os.environ.get("AR_TARGET") + if not target: + print("error: provide --target or set AR_TARGET", file=sys.stderr) + return 2 + try: + with open(target) as f: + data = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"{METRIC}: N/A") + print(f"error: {e}", file=sys.stderr) + return 1 + + result = isyn.synthesize(data, min_sources) + count = sum(1 for c2 in result["candidates"] if c2["classification"] == "INSIGHT") + print(f"{METRIC}: {count}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/product-research/scripts/config_loader.py b/research-ops/skills/product-research/scripts/config_loader.py new file mode 100644 index 00000000..db5d4dfc --- /dev/null +++ b/research-ops/skills/product-research/scripts/config_loader.py @@ -0,0 +1,114 @@ +#!/usr/bin/env python3 +"""config_loader.py - Customization loader for the product-research skill. + +Stdlib-only. Importable from the skill's other scripts. Precedence (highest wins): + 1. Project config: /.research-ops/product-research.json + 2. Global config: ~/.config/research-ops/product-research.json + 3. Built-in DEFAULTS + +Onboarding answers (written by onboard.py) live in these files; every tool in this +skill reads them so the user's customization applies automatically. +Set RESEARCH_OPS_NO_CONFIG=1 to ignore saved config. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from pathlib import Path +from typing import Any + +SKILL = "product-research" + +GLOBAL_CONFIG_DIR = Path.home() / ".config" / "research-ops" +GLOBAL_CONFIG_PATH = GLOBAL_CONFIG_DIR / f"{SKILL}.json" +PROJECT_CONFIG_DIRNAME = ".research-ops" + +DEFAULTS: dict[str, Any] = { + "version": 1, + "skill": SKILL, + "default_profile": "b2b-saas", + "insight_min_sources": 3, + "default_method": "usability", + "stakes_high": False, + "setup_completed_at": None, +} + + +def project_config_path(cwd: Path | None = None) -> Path: + cwd = cwd or Path.cwd() + return cwd / PROJECT_CONFIG_DIRNAME / f"{SKILL}.json" + + +def _read_json(path: Path) -> dict[str, Any] | None: + try: + with path.open(encoding="utf-8") as f: + data = json.load(f) + return data if isinstance(data, dict) else None + except (FileNotFoundError, json.JSONDecodeError, OSError): + return None + + +def _deep_merge(base: dict[str, Any], override: dict[str, Any]) -> dict[str, Any]: + out = dict(base) + for k, v in override.items(): + if isinstance(v, dict) and isinstance(out.get(k), dict): + out[k] = _deep_merge(out[k], v) + else: + out[k] = v + return out + + +def load_config(cwd: Path | None = None) -> dict[str, Any]: + config = dict(DEFAULTS) + if os.environ.get("RESEARCH_OPS_NO_CONFIG") == "1": + return config + global_cfg = _read_json(GLOBAL_CONFIG_PATH) + if global_cfg: + config = _deep_merge(config, global_cfg) + project_cfg = _read_json(project_config_path(cwd)) + if project_cfg: + config = _deep_merge(config, project_cfg) + return config + + +def setup_completed() -> bool: + cfg = _read_json(GLOBAL_CONFIG_PATH) or _read_json(project_config_path()) + return bool(cfg and cfg.get("setup_completed_at")) + + +def write_config(config: dict[str, Any], scope: str = "global", cwd: Path | None = None) -> Path: + path = project_config_path(cwd) if scope == "project" else GLOBAL_CONFIG_PATH + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as f: + json.dump(config, f, indent=2, sort_keys=True) + return path + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Inspect {SKILL} customization config.") + p.add_argument("--show", action="store_true", help="Print the effective config") + p.add_argument("--status", action="store_true", help="Print setup status + paths") + p.add_argument("--sample", action="store_true", help="Print the built-in defaults") + args = p.parse_args(argv) + + if args.sample: + print(json.dumps(DEFAULTS, indent=2, sort_keys=True)) + elif args.status: + print(json.dumps({ + "skill": SKILL, + "global_config_path": str(GLOBAL_CONFIG_PATH), + "global_config_exists": GLOBAL_CONFIG_PATH.exists(), + "project_config_path": str(project_config_path()), + "project_config_exists": project_config_path().exists(), + "setup_completed": setup_completed(), + }, indent=2)) + else: + print(json.dumps(load_config(), indent=2, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/product-research/scripts/insight_synthesizer.py b/research-ops/skills/product-research/scripts/insight_synthesizer.py new file mode 100644 index 00000000..6cd86ff8 --- /dev/null +++ b/research-ops/skills/product-research/scripts/insight_synthesizer.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +"""insight_synthesizer.py - Cluster coded observations into candidate insights; flag anecdotes. + +Stdlib-only. Deterministic. NO LLM calls. NEVER fabricates an insight: it counts evidence, +clusters by tag, ranks by cross-participant recurrence, and flags any candidate supported by +fewer than --min-sources independent participants as an ANECDOTE, not an insight. + +Input: a list of observations, each with {participant, tag, note}. The synthesizer groups by +tag, counts distinct participants per tag, and ranks. This is the atomic-research discipline: +an observation is evidence; an insight requires recurrence across independent sources. + +Usage: + python3 insight_synthesizer.py --sample + python3 insight_synthesizer.py --input observations.json --min-sources 3 + python3 insight_synthesizer.py --input observations.json --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from collections import defaultdict + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +SAMPLE = { + "study": "Onboarding discovery (mid-market HR)", + "observations": [ + {"participant": "P1", "tag": "import-confusion", "note": "Couldn't find CSV import."}, + {"participant": "P2", "tag": "import-confusion", "note": "Expected import on the dashboard."}, + {"participant": "P3", "tag": "import-confusion", "note": "Gave up looking for bulk upload."}, + {"participant": "P1", "tag": "permissions-unclear", "note": "Unsure who could see reports."}, + {"participant": "P4", "tag": "permissions-unclear", "note": "Worried about data visibility."}, + {"participant": "P2", "tag": "wants-slack", "note": "Asked for a Slack integration."}, + ], +} + + +def synthesize(data: dict, min_sources: int) -> dict: + obs = data.get("observations", []) + by_tag_participants = defaultdict(set) + by_tag_notes = defaultdict(list) + for o in obs: + tag = o.get("tag", "untagged") + part = o.get("participant", "UNKNOWN") + by_tag_participants[tag].add(part) + by_tag_notes[tag].append({"participant": part, "note": o.get("note", "")}) + + candidates = [] + for tag, parts in by_tag_participants.items(): + n_sources = len(parts) + is_insight = n_sources >= min_sources + candidates.append({ + "tag": tag, + "distinct_participants": n_sources, + "observation_count": len(by_tag_notes[tag]), + "classification": "INSIGHT" if is_insight else "ANECDOTE (single/low-source — do not generalize)", + "evidence": by_tag_notes[tag], + }) + candidates.sort(key=lambda c: (c["distinct_participants"], c["observation_count"]), reverse=True) + + total_participants = len({o.get("participant") for o in obs}) + return { + "study": data.get("study", "UNSPECIFIED"), + "min_sources_for_insight": min_sources, + "total_participants": total_participants, + "candidates": candidates, + "note": "An observation is evidence; an insight requires recurrence across independent participants. " + "Anecdotes are surfaced, never promoted to insights.", + } + + +def _render_human(r: dict) -> str: + lines = [f"Insight Synthesis: {r['study']}", + f" total participants: {r['total_participants']} insight threshold: >= {r['min_sources_for_insight']} sources", ""] + for c in r["candidates"]: + lines.append(f"[{c['classification']}] {c['tag']} " + f"({c['distinct_participants']} participants, {c['observation_count']} observations)") + for e in c["evidence"]: + lines.append(f" {e['participant']}: {e['note']}") + lines.append("") + lines.append(f"note: {r['note']}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Cluster coded observations into insights; flag anecdotes.") + p.add_argument("--input", help="Path to JSON with observations[]") + p.add_argument("--min-sources", type=int, default=None, + help="min distinct participants to call it an insight (overrides onboarding)") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + min_sources = args.min_sources if args.min_sources is not None else int(conf.get("insight_min_sources", 3)) + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + result = synthesize(data, min_sources) + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/product-research/scripts/onboard.py b/research-ops/skills/product-research/scripts/onboard.py new file mode 100644 index 00000000..884ad6e9 --- /dev/null +++ b/research-ops/skills/product-research/scripts/onboard.py @@ -0,0 +1,124 @@ +#!/usr/bin/env python3 +"""onboard.py - Onboarding questionnaire for the product-research skill. + +Stdlib-only. Asks the user a short set of questions BEFORE they plan a study, then +writes the answers to a customization config read by every tool in this skill via +config_loader.py. The answers become defaults for profile, the insight source-threshold, +the default saturation method, and the high-stakes flag. + +Modes: --show | --defaults | --set key=value (repeatable) | --reset | --scope {global,project} +""" + +from __future__ import annotations + +import argparse +import datetime as _dt +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 + +INT_KEYS = {"insight_min_sources"} +BOOL_KEYS = {"stakes_high"} + +QUESTIONS = [ + ("default_profile", + "1. What kind of product is this?", + ["b2b-saas", "consumer-app", "enterprise", "marketplace", "hardware", "platform"], str), + ("insight_min_sources", + "2. How many independent participants must support a finding before it counts as an insight (not an anecdote)?", + None, int), + ("default_method", + "3. Default sample-saturation method?", + ["usability", "thematic", "evaluative-coverage"], str), + ("stakes_high", + "4. Is this high-stakes / high-heterogeneity research (raise sample sizes)?", + ["true", "false"], str), +] + + +def _coerce(key: str, value: str): + if key in INT_KEYS: + return int(value) + if key in BOOL_KEYS: + return str(value).strip().lower() in ("true", "yes", "y", "1") + return value + + +def _print_questions() -> None: + print(f"Onboarding questions — {cfg.SKILL}:\n") + for _k, prompt, choices, _c in QUESTIONS: + line = f" {prompt}" + if choices: + line += f" [{' / '.join(choices)}]" + print(line) + + +def run_interactive(config: dict) -> dict: + print(f"Onboarding — {cfg.SKILL}. Press Enter to keep the current/default value.\n") + for key, prompt, choices, _caster in QUESTIONS: + suffix = f" [{'/'.join(choices)}]" if choices else "" + cur = f" (current: {config.get(key)})" if config.get(key) is not None else "" + raw = input(f"{prompt}{suffix}{cur}: ").strip() + if not raw: + continue + try: + config[key] = _coerce(key, raw) + except ValueError: + print(f" ! invalid value for {key}, keeping current") + return config + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Onboarding for the {cfg.SKILL} skill.") + p.add_argument("--show", action="store_true") + p.add_argument("--defaults", action="store_true", help="write built-in defaults, no prompt") + p.add_argument("--set", action="append", default=[], metavar="key=value") + p.add_argument("--reset", action="store_true") + p.add_argument("--scope", choices=["global", "project"], default="global") + args = p.parse_args(argv) + + if args.show: + _print_questions() + print("\nCurrent effective config:") + print(json.dumps(cfg.load_config(), indent=2, sort_keys=True)) + return 0 + + if args.reset: + path = cfg.project_config_path() if args.scope == "project" else cfg.GLOBAL_CONFIG_PATH + if path.exists(): + path.unlink(); print(f"removed {path}") + else: + print(f"no config at {path}") + return 0 + + config = cfg.load_config() + + if args.set: + for item in args.set: + if "=" not in item: + print(f"error: --set expects key=value, got '{item}'", file=sys.stderr) + return 2 + k, v = item.split("=", 1) + try: + config[k] = _coerce(k, v) + except ValueError: + config[k] = v + elif not args.defaults: + if sys.stdin.isatty(): + config = run_interactive(config) + else: + print("non-interactive shell: use --defaults or --set key=value. Showing questions:\n") + _print_questions() + return 0 + + config["setup_completed_at"] = _dt.datetime.now(_dt.timezone.utc).isoformat() + path = cfg.write_config(config, scope=args.scope) + print(f"saved {cfg.SKILL} customization -> {path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/product-research/scripts/saturation_planner.py b/research-ops/skills/product-research/scripts/saturation_planner.py new file mode 100644 index 00000000..ba05ccea --- /dev/null +++ b/research-ops/skills/product-research/scripts/saturation_planner.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""saturation_planner.py - Method-based participant/sample guidance with a confidence label. + +Stdlib-only. Deterministic. NO LLM calls. NEVER fabricates insight: it gives method-based +sample guidance and an explicit confidence level, surfacing limits. + +Models: + - usability (Nielsen): ~5 users per segment uncovers ~85% of problems at typical p=0.31; + problems found = 1 - (1 - p)^n. + - thematic saturation (Guest et al.): ~12 interviews per homogeneous group typically + reaches saturation; >5 (Faulkner) when stakes/heterogeneity are high. + - evaluative coverage: detectable-problem coverage for a chosen per-problem detection rate. + +Usage: + python3 saturation_planner.py --sample + python3 saturation_planner.py --method usability --segments 2 --detection-rate 0.31 + python3 saturation_planner.py --method thematic --segments 3 --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +METHODS = ["usability", "thematic", "evaluative-coverage"] + + +def usability_plan(segments: int, p: float, target_coverage: float) -> dict: + # n per segment to reach target coverage: n = ln(1 - target) / ln(1 - p) + import math + if not 0.0 < p < 1.0: + raise ValueError("detection-rate must be in (0,1).") + n = math.ceil(math.log(1 - target_coverage) / math.log(1 - p)) + coverage_at_5 = 1 - (1 - p) ** 5 + return { + "method": "usability", + "per_problem_detection_rate": p, + "target_coverage": target_coverage, + "n_per_segment": n, + "segments": segments, + "total_participants": n * segments, + "coverage_at_5_per_segment": round(coverage_at_5, 3), + "confidence": "MODERATE" if n >= 5 else "LOW (small-n usability finds problems, not rates)", + "limits": "Usability tests surface problems, not their population prevalence. Do not report percentages.", + } + + +def thematic_plan(segments: int, stakes_high: bool) -> dict: + base = 12 # Guest et al. typical saturation for a homogeneous group + per_segment = base if not stakes_high else max(base, 15) + return { + "method": "thematic", + "n_per_segment": per_segment, + "segments": segments, + "total_participants": per_segment * segments, + "confidence": "MODERATE-HIGH" if per_segment >= 12 else "LOW", + "limits": "Saturation is observed, not guaranteed; track new-theme rate and stop when it flattens. " + "Faulkner (2003): more than 5 when heterogeneity or stakes are high.", + } + + +def evaluative_coverage_plan(segments: int, n_per_segment: int, p: float) -> dict: + coverage = 1 - (1 - p) ** n_per_segment + return { + "method": "evaluative-coverage", + "per_problem_detection_rate": p, + "n_per_segment": n_per_segment, + "segments": segments, + "expected_problem_coverage": round(coverage, 3), + "confidence": "MODERATE" if coverage >= 0.8 else "LOW", + "limits": "Coverage is for the assumed detection rate; rarer problems need more participants.", + } + + +def plan(method: str, segments: int, p: float, target: float, stakes_high: bool, n: int) -> dict: + if method == "usability": + out = usability_plan(segments, p, target) + elif method == "thematic": + out = thematic_plan(segments, stakes_high) + elif method == "evaluative-coverage": + out = evaluative_coverage_plan(segments, n, p) + else: + raise ValueError(f"method must be one of {METHODS}.") + out["disclaimer"] = "Method-based guidance with explicit confidence. This is not a power calculation; " \ + "it never claims an insight the data cannot support." + return out + + +def _render_human(r: dict) -> str: + lines = [f"Saturation / Sample Plan (method: {r['method']})", ""] + for k, v in r.items(): + if k in ("method",): + continue + lines.append(f" {k:32s} : {v}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Method-based product-research sample guidance with confidence.") + p.add_argument("--method", choices=METHODS, default=None, help="overrides onboarding default_method") + p.add_argument("--segments", type=int, default=1) + p.add_argument("--detection-rate", type=float, default=0.31, help="per-problem detection rate (usability)") + p.add_argument("--target-coverage", type=float, default=0.85, help="target problem coverage (usability)") + p.add_argument("--stakes-high", action="store_true", help="raise thematic n for high heterogeneity/stakes") + p.add_argument("--n-per-segment", type=int, default=8, help="n per segment (evaluative-coverage)") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + method = args.method or conf.get("default_method", "usability") + stakes_high = args.stakes_high or bool(conf.get("stakes_high", False)) + + if args.sample: + try: + result = plan("usability", 2, 0.31, 0.85, False, 8) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + else: + try: + result = plan(method, args.segments, args.detection_rate, + args.target_coverage, stakes_high, args.n_per_segment) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/product-research/scripts/study_designer.py b/research-ops/skills/product-research/scripts/study_designer.py new file mode 100644 index 00000000..58d48246 --- /dev/null +++ b/research-ops/skills/product-research/scripts/study_designer.py @@ -0,0 +1,139 @@ +#!/usr/bin/env python3 +"""study_designer.py - Select a product-research method from goal + stage, emit a plan skeleton. + +Stdlib-only. Deterministic. NO LLM calls. + +Maps (research goal x product stage) to an appropriate method and emits a method-matched +plan skeleton (objective framing, participant criteria, task/guide structure, success +criteria). The core discipline: GENERATIVE goals (discover problems) and EVALUATIVE goals +(test a solution) demand different methods — picking the wrong one is the most common error. + +Usage: + python3 study_designer.py --sample + python3 study_designer.py --goal discovery --stage concept --profile b2b-saas + python3 study_designer.py --goal evaluative --stage live --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +PROFILES = ["b2b-saas", "consumer-app", "enterprise", "marketplace", "hardware", "platform"] + +# (goal, stage) -> method. goal in {discovery, evaluative, validation}; stage in {concept, prototype, beta, live} +METHOD_MAP = { + ("discovery", "concept"): "generative interviews (semi-structured)", + ("discovery", "prototype"): "contextual inquiry", + ("discovery", "beta"): "diary study + follow-up interviews", + ("discovery", "live"): "behavioral analytics review + generative interviews", + ("evaluative", "concept"): "concept test (comprehension + desirability)", + ("evaluative", "prototype"): "moderated usability test", + ("evaluative", "beta"): "unmoderated usability test + task-success metrics", + ("evaluative", "live"): "benchmark usability study (SUS / task time)", + ("validation", "concept"): "survey (desirability + willingness signals)", + ("validation", "prototype"): "prototype A/B preference test", + ("validation", "beta"): "fake-door / feature-demand test", + ("validation", "live"): "live A/B experiment (route to product-team/experiment-designer)", +} + +GUIDE_SKELETONS = { + "generative": ["Warm-up + context", "Recent relevant experience (story, not opinion)", + "Workarounds + frustrations", "Jobs-to-be-done probe", "Magic-wand / wrap"], + "evaluative": ["Pre-task context", "Task 1 (representative)", "Task 2 (edge)", + "Observation: where do they hesitate/err?", "Post-task SUS / debrief"], + "validation": ["Screener", "Stimulus exposure", "Comprehension + desirability items", + "Trade-off / preference items", "Behavioral-intent item"], +} + + +def design(goal: str, stage: str, profile: str) -> dict: + if profile not in PROFILES: + raise ValueError(f"Unknown profile '{profile}'. Choose from {PROFILES}.") + key = (goal, stage) + if key not in METHOD_MAP: + raise ValueError(f"No method for goal={goal}, stage={stage}. " + f"goal in [discovery,evaluative,validation]; stage in [concept,prototype,beta,live].") + method = METHOD_MAP[key] + family = "generative" if goal == "discovery" else ("evaluative" if goal == "evaluative" else "validation") + redirect = None + if "experiment-designer" in method: + redirect = "Live A/B is a product experiment — use product-team/experiment-designer, not this skill." + return { + "goal": goal, + "stage": stage, + "profile": profile, + "method": method, + "method_family": family, + "objective_framing": f"A {family} study at the {stage} stage to {('discover unmet needs' if family=='generative' else 'evaluate the solution' if family=='evaluative' else 'validate demand/desirability')}.", + "participant_criteria": [ + "Recruit to the target segment (screen for the job, not a job title).", + "Exclude internal/biased participants and prior-study repeats unless longitudinal.", + "Recruit per-segment if results will be reported per-segment.", + ], + "guide_skeleton": GUIDE_SKELETONS[family], + "success_criteria": [ + "Generative: themes recur across independent participants (saturation).", + "Evaluative: task-success rate + severity-rated problem list.", + "Validation: pre-registered desirability / preference threshold.", + ], + "redirect": redirect, + "note": "Method must match the goal. A usability test cannot discover unmet needs; an interview cannot measure task success.", + } + + +def _render_human(r: dict) -> str: + lines = [f"Study Design: goal={r['goal']}, stage={r['stage']}, profile={r['profile']}", "", + f" Recommended method: {r['method']} (family: {r['method_family']})", + f" Objective: {r['objective_framing']}", "", " Participant criteria:"] + for c in r["participant_criteria"]: + lines.append(f" - {c}") + lines.append(" Guide skeleton:") + for i, g in enumerate(r["guide_skeleton"], 1): + lines.append(f" {i}. {g}") + lines.append(" Success criteria:") + for s in r["success_criteria"]: + lines.append(f" - {s}") + if r["redirect"]: + lines += ["", f" !! {r['redirect']}"] + lines += ["", f"note: {r['note']}"] + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Select a product-research method from goal + stage.") + p.add_argument("--goal", choices=["discovery", "evaluative", "validation"], default="discovery") + p.add_argument("--stage", choices=["concept", "prototype", "beta", "live"], default="prototype") + p.add_argument("--profile", default=None, choices=PROFILES, + help="overrides onboarding default_profile") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + profile_default = conf.get("default_profile", "b2b-saas") + goal, stage, profile = ("discovery", "prototype", profile_default) if args.sample \ + else (args.goal, args.stage, args.profile or profile_default) + try: + result = design(goal, stage, profile) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-finance/SKILL.md b/research-ops/skills/research-finance/SKILL.md new file mode 100644 index 00000000..5be64a20 --- /dev/null +++ b/research-ops/skills/research-finance/SKILL.md @@ -0,0 +1,146 @@ +--- +name: research-finance +description: Use when managing the money for an internal R&D program or portfolio — building a multi-period program budget with the F&A (indirect) split, tracking burn rate and runway against value-inflection milestones, or routing R&D cost items to a capitalize-vs-expense determination. Every budget output surfaces its assumptions block; capitalize-vs-expense is decision-support only and routes to a named finance owner — it never books an entry or decides accounting treatment. Distinct from finance/financial-analysis (corporate DCF, close, valuation) and research/grants (funding discovery — this manages money already won). +version: 2.9.0 +author: claude-code-skills +license: MIT +tags: [research-ops, research-finance, rd-budget, burn-rate, runway, fa-rate, capitalize-vs-expense, portfolio] +compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli] +--- + +# research-finance + +Financial management of internal R&D programs and portfolios: program budgeting with F&A, burn/runway tracking, and capitalize-vs-expense routing. Every number ships with its **assumptions block**, and accounting-treatment calls **route to a named finance owner** — this skill never books an entry. + +## Purpose + +R&D finance partners, program controllers, and operations leads manage money that has already been allocated or raised — not the corporate close, not the next funding round, not finding a grant. This skill structures three recurring decisions: + +Three deterministic tools: + +1. `program_budget_planner.py` — Builds a multi-period budget from work-package line items, applies the F&A (indirect) rate to an MTDC-style eligible base, and rolls up direct / F&A / fully-loaded cost per period with an explicit assumptions block. +2. `burn_runway_tracker.py` — Computes average + trailing burn, runway in periods/months, and whether each value-inflection milestone is reachable before cash runs out. Flags accelerating burn and below-threshold runway. +3. `capex_vs_opex_router.py` — Scores each R&D cost item against the IAS 38 development-phase criteria (or flags US GAAP ASC 730 expense-as-incurred) and routes it to **CAPITALIZE-CANDIDATE / EXPENSE / FINANCE-OWNER-REVIEW** with a named owner. Never auto-decides. + +## When to use + +Invoke this skill when: + +- You are building or revising an R&D program budget and need the F&A split made explicit. +- A program's runway is in question and you need a milestone-vs-cash read. +- Finance asks whether a development cost can be capitalized and you need a defensible first routing. +- You are preparing a portfolio review and need per-program burn consistency. + +**Do NOT use this skill to**: run corporate DCF / valuation / close (use `finance/financial-analysis`), discover or position grants (use `research/grants`), or make the final accounting determination (that is the controller's + auditor's call — this tool only routes). + +## Workflow + +1. **Lay out the program** — Fill `assets/rd_program_budget_template.md` with work-package lines, categories, and per-period amounts. +2. **Build the budget** — Run `program_budget_planner.py --input program.json --profile {pharma-rd|biotech|medtech|deep-tech|software-rd|university-lab} --fa-rate `. Read direct / F&A / fully-loaded rollups + assumptions. +3. **Track burn & runway** — Run `burn_runway_tracker.py --input ledger.json --threshold-months 6`. Read runway + milestone verdicts + flags. +4. **Route accounting treatment** — Run `capex_vs_opex_router.py --input costs.json --standard {ifrs|usgaap}`. Read the per-item routing; send CAPITALIZE-CANDIDATE and FINANCE-OWNER-REVIEW items to the named owner. +5. **Assemble the review** — Combine into a program-finance packet. Every number carries its assumptions; treatment calls carry a named owner. + +## Scripts + +| Script | Purpose | Profiles | +|---|---|---| +| `scripts/program_budget_planner.py` | Multi-period budget + F&A split + assumptions | pharma-rd, biotech, medtech, deep-tech, software-rd, university-lab | +| `scripts/burn_runway_tracker.py` | Burn, runway, milestone-vs-cash alignment | n/a (ledger-driven) | +| `scripts/capex_vs_opex_router.py` | IAS 38 / ASC 730 routing to named finance owner | pharma-rd, biotech, medtech, deep-tech, software-rd, university-lab | + +All three: stdlib-only, `--help`, `--sample`, `--output {human,json}`. + +## Onboarding & customization + +Run the onboarding questionnaire **once before you start** — it captures your defaults so every tool in this skill is pre-configured. Customization is the point: the answers actually change tool behavior. + +```bash +python3 scripts/onboard.py # interactive (also: --defaults, --set key=value, --reset) +python3 scripts/onboard.py --show # see the questions + current effective config +``` + +Answers are saved to `~/.config/research-ops/research-finance.json` (global) or `./.research-ops/research-finance.json` (`--scope project`) and are read automatically by `config_loader.py`. They set the default R&D-area **profile**, the default **F&A rate**, the **runway alert threshold**, the **accounting standard**, and the named **finance owner** printed on capitalize-vs-expense routing. CLI flags always override saved config; `RESEARCH_OPS_NO_CONFIG=1` ignores it. + +**The five questions:** R&D area · F&A rate · runway threshold · accounting standard · finance owner. + +## Optimize with autoresearch (opt-in) + +This skill ships an **isolated, opt-in** bridge to `engineering/autoresearch-agent`. Only when you ask to "optimize" / "extend runway" / "run a loop" does an autoresearch experiment iteratively improve a program plan against this skill's runway metric. `scripts/ar_evaluator.py` is the ground-truth evaluator; it prints `runway_months: ` (higher is better). + +```bash +/ar:setup --domain custom --name extend-runway \ + --target ledger.json \ + --eval "python3 ar_evaluator.py --target ledger.json" \ + --metric runway_months --direction higher +/ar:loop custom/extend-runway +``` + +Isolated: no hard dependency — autoresearch runs only on demand, and the loop edits `ledger.json`, never the evaluator. + +## References + +- `references/rd_program_finance_canon.md` — IAS 38 (research vs development); ASC 730 + ASC 985-20; Uniform Guidance 2 CFR 200 (F&A); FASB/IFRS capitalization criteria; NICRA basics. +- `references/burn_and_portfolio.md` — Cooper stage-gate; rNPV / real-options for R&D; risk-adjusted portfolio ROI; burn-rate / runway frameworks; milestone-based budgeting. +- `references/indirect_rate_modeling.md` — F&A pool composition (facilities + administration); MTDC base; de minimis 10%; fringe/overhead loading; CAS primer. + +## Assumptions + +- The F&A rate is the most error-prone input. The planner applies whatever rate you pass; it warns you to confirm it is a negotiated NICRA, not a guess. +- Burn/runway uses the trailing (recent-weighted) burn as the forward run-rate and assumes flat forward spend unless your ledger encodes a ramp. +- The capex router asserts criteria from your input; asserting "technical feasibility" does not make it true — the named finance owner and auditor validate it. +- Profiles annotate context (e.g., "most drug R&D is expensed") but do not change the accounting test. + +## Anti-patterns + +- **Stating a budget number without its assumptions.** F&A rate, escalation, and base must travel with the number. +- **Auto-deciding capitalize-vs-expense.** This tool routes; the controller (and auditor where required) decides. +- **Using lifetime-average burn for runway.** Recent burn is the honest forward run-rate; averages hide a slowdown or a ramp. +- **Applying F&A to the full base.** Capital equipment, large subaward portions, and certain categories are MTDC-exempt. +- **Confusing this with corporate finance.** Valuation, close, and fundraising live in `finance/`. + +## Distinct from + +| Sibling / neighbor | Scope | Difference | +|---|---|---| +| `finance/financial-analysis` | Corporate DCF, ratios, close, rolling forecast, SaaS metrics | That is **company-level**; this is **R&D-program-level** | +| `research/grants` | NIH funding discovery + positioning | That **finds funding**; this **manages money already won** | +| `clinical-research` (sibling) | Study design + feasibility + budget gate-check | That **scopes** the study; this **funds + tracks** the program | +| `ra-qm-team` | Regulatory/QM submission | Unrelated — no financial scope | + +## Quick examples + +```bash +python3 scripts/program_budget_planner.py --sample +python3 scripts/program_budget_planner.py --input program.json --profile university-lab --fa-rate 0.585 +python3 scripts/burn_runway_tracker.py --sample --output json +python3 scripts/capex_vs_opex_router.py --sample --standard ifrs +``` + +The sample budget excludes the sequencer (capital equipment) and CRO subaward from the F&A base; the capex router routes exploratory screening to EXPENSE, a fully-criteria'd pilot line to CAPITALIZE-CANDIDATE, and a partial-criteria software build to FINANCE-OWNER-REVIEW. + +## Forcing-question library (Matt Pocock grill discipline) + +Walked one at a time by `/cs:grill-research-ops` or the orchestrator. Recommended answer + canon citation per question. Never bundled. + +1. **"Is this spend in the research phase or the development phase — and can you evidence technical feasibility?"** + Recommended: research = expense; development = capitalize-candidate only with feasibility evidence, routed to a named finance owner. + Canon: IAS 38.54-57; ASC 730. + +2. **"What F&A / indirect rate are you applying, and is it your negotiated NICRA, a de minimis 10%, or an assumption?"** + Recommended: use the negotiated rate; if assumed, flag it explicitly. + Canon: 2 CFR 200 (Uniform Guidance); NICRA basics. + +3. **"What's runway in months at current burn, and does it clear the next value-inflection milestone?"** + Recommended: runway must cover the milestone plus a buffer; surface the gap. + Canon: Cooper stage-gate; SaaS/startup efficiency frameworks (a16z, Bessemer). + +4. **"Is portfolio ROI risk-adjusted (rNPV / probability-of-success weighted) or raw NPV?"** + Recommended: risk-adjusted; raw NPV overstates R&D value. + Canon: rNPV drug-development valuation; real-options literature. + +5. **"Who is the named finance / controller owner who signs the capitalize-vs-expense treatment?"** + Recommended: name them — this tool recommends, it never books the entry. + Canon: ASC 730 / IAS 38 governance; auditor sign-off requirements. + +Walk depth-first. Lock 1-2 before opening 3-5. After all are answered, invoke `program_budget_planner.py` → `burn_runway_tracker.py` → `capex_vs_opex_router.py`. diff --git a/research-ops/skills/research-finance/assets/rd_program_budget_template.md b/research-ops/skills/research-finance/assets/rd_program_budget_template.md new file mode 100644 index 00000000..27533c16 --- /dev/null +++ b/research-ops/skills/research-finance/assets/rd_program_budget_template.md @@ -0,0 +1,48 @@ +# R&D Program Budget — Template + +> Fill this before running `program_budget_planner.py`. Every number must travel with its +> assumptions. Capitalize-vs-expense calls route to a named finance owner — this is not the +> place to decide accounting treatment. + +## 1. Program identification +- Program name: +- R&D area / profile: [pharma-rd | biotech | medtech | deep-tech | software-rd | university-lab] +- Number of periods + period label (month / quarter / year): +- Funding source(s): + +## 2. F&A (indirect) basis +- F&A rate applied: ____% +- Rate type: [negotiated NICRA | de minimis 10% | internal assumption — FLAG IT] +- Fringe rate (loaded onto salaries before F&A): ____% + +## 3. Work packages (per-period amounts) +| Work package | Category | F&A-eligible? | P1 | P2 | P3 | P4 | +|---|---|---|---|---|---|---| +| Personnel (FTEs) | personnel | yes | | | | | +| Consumables / supplies | supplies | yes | | | | | +| Capital equipment | capital_equipment | NO (MTDC-exempt) | | | | | +| Subaward / CRO (>$25k) | subaward_over_25k | NO (over $25k exempt) | | | | | +| Travel | travel | yes | | | | | + +> Categories that are MTDC-exempt: capital_equipment, subaward_over_25k, tuition, patient_care. + +## 4. Milestones (for burn/runway) +| Milestone | Periods from now | Cumulative cash needed | +|---|---|---| +| | | | + +## 5. Capitalize-vs-expense candidates (for routing only) +| Cost item | Phase (research / development / software-development) | Standard (ifrs / usgaap) | +|---|---|---| +| | | | + +## 6. Assumptions register +- F&A rate basis: +- Escalation assumption: +- Forward burn assumption (flat / ramp): +- Probability-of-success weighting (for any portfolio ROI): + +## 7. Named owners +- R&D Finance Controller: +- External Auditor (if capitalization in play): +- Program Lead: diff --git a/research-ops/skills/research-finance/references/burn_and_portfolio.md b/research-ops/skills/research-finance/references/burn_and_portfolio.md new file mode 100644 index 00000000..b6b70919 --- /dev/null +++ b/research-ops/skills/research-finance/references/burn_and_portfolio.md @@ -0,0 +1,28 @@ +# Burn, Runway, and R&D Portfolio Management + +Reference for burn/runway tracking and risk-adjusted portfolio decisions. Pairs with `burn_runway_tracker.py`. + +## Burn and runway done honestly + +**Burn rate** is cash spent per period; **runway** is cash-on-hand ÷ forward run-rate. The honest forward run-rate is the **trailing** (recent-weighted) burn, not the lifetime average — averages mask both an accelerating spend and a funded ramp. The tracker uses trailing burn and flags when trailing exceeds 115% of the lifetime average (an acceleration signal). Runway must be measured against **value-inflection milestones**: cash that runs out one month before analytical validation is materially worse than the same runway that clears it, because reaching the milestone changes the program's financing options and valuation. + +## Stage-gate portfolio management + +Robert Cooper's **Stage-Gate** model structures R&D as a sequence of stages separated by go/kill **gates**. Each gate is a real-options decision: spend the next tranche, or kill and redeploy. The discipline is that money is committed one stage at a time, against pre-defined criteria — not as a lump sum at kickoff. This is why milestone-vs-cash alignment is the core runway question. + +## Risk-adjusted valuation + +Raw NPV systematically overstates R&D value because it ignores attrition. **Risk-adjusted NPV (rNPV)** weights each phase's cash flows by the cumulative probability of success of reaching it — in drug development, the product of per-phase success rates (which compound to single-digit percentages from preclinical to approval). **Real-options** valuation goes further, pricing the optionality of being able to abandon. For portfolio ROI, always state whether the number is raw NPV or risk-adjusted; the difference is often an order of magnitude. + +## Efficiency benchmarks + +Startup/SaaS efficiency frameworks (a16z's burn multiple, Bessemer's efficiency score) translate to R&D portfolios as "value created per dollar burned." They are blunt but useful for cross-program comparison when paired with milestone progress. + +## Sources + +1. Cooper, R.G., *Winning at New Products: Creating Value Through Innovation*, 5th ed. (2017) — Stage-Gate. +2. Stewart, Allison & Johnson, *Putting a price on biotechnology* — Nature Biotechnology 2001 (rNPV in drug development). +3. Trigeorgis, L., *Real Options: Managerial Flexibility and Strategy in Resource Allocation* (MIT Press). +4. DiMasi, Grabowski & Hansen, *Innovation in the pharmaceutical industry: New estimates of R&D costs* — J Health Econ 2016 (attrition / phase success rates). +5. a16z, *The burn multiple* and Bessemer State of the Cloud efficiency benchmarks. +6. Chan & Thornhill, *R&D portfolio management* — R&D Management literature. diff --git a/research-ops/skills/research-finance/references/indirect_rate_modeling.md b/research-ops/skills/research-finance/references/indirect_rate_modeling.md new file mode 100644 index 00000000..eb4ac078 --- /dev/null +++ b/research-ops/skills/research-finance/references/indirect_rate_modeling.md @@ -0,0 +1,42 @@ +# Indirect (F&A) Rate Modeling + +Deep reference for the F&A rate — the single most error-prone input in an R&D budget. Pairs with `program_budget_planner.py`. + +## What the F&A rate actually is + +The F&A rate recovers shared costs that cannot be traced to a single program. It is composed of two pools: + +- **Facilities** — depreciation on buildings and equipment, interest on facility debt, operations & maintenance, library, utilities. +- **Administration** — general administration, departmental administration, sponsored-projects administration, student services (in universities). + +The rate is computed as (indirect pool ÷ allocation base) and applied to that base on each program. + +## The base matters as much as the rate + +A 55% rate on a $1M total budget is *not* $550k of F&A — because the rate applies only to the **MTDC base**, which excludes: + +- Capital equipment (typically items > $5,000 with > 1-year life) +- The portion of **each** subaward exceeding $25,000 (the first $25k is in the base; the rest is exempt) +- Tuition remission +- Patient-care costs +- Rental of off-site facilities, scholarships, participant support + +So a budget heavy in equipment and large subawards has a much smaller F&A base than its headline total. The planner models this exclusion explicitly. + +## Negotiated vs de minimis + +- **NICRA** — the Negotiated Indirect Cost Rate Agreement, established with a cognizant federal agency. This is the authoritative rate for federally funded work. +- **De minimis 10%** — under 2 CFR 200.414(f), an entity that has never had a negotiated rate may elect a flat 10% of MTDC. Simpler, almost always lower than a negotiated research rate. + +## Fringe and the loading stack + +Personnel costs load in layers: base salary → **fringe** (benefits, often 25-35%) → then F&A applies to salary+fringe (both are in the MTDC base). Modeling fringe separately from F&A avoids double counting or under-recovery. + +## Sources + +1. 2 CFR 200.414, *Indirect (F&A) costs*, and Appendix III (IHEs) / Appendix IV (nonprofits). +2. 2 CFR 200.1, definition of *Modified Total Direct Cost (MTDC)*. +3. NIH Grants Policy Statement, indirect-cost chapter; DHHS Cost Allocation Services NICRA guidance. +4. Cost Accounting Standards Board, 48 CFR 9904 (CAS 410, 418 on allocation). +5. COGR (Council on Governmental Relations), *Indirect Cost / F&A* primers and white papers. +6. Federal Demonstration Partnership materials on subaward and MTDC treatment. diff --git a/research-ops/skills/research-finance/references/rd_program_finance_canon.md b/research-ops/skills/research-finance/references/rd_program_finance_canon.md new file mode 100644 index 00000000..9767a916 --- /dev/null +++ b/research-ops/skills/research-finance/references/rd_program_finance_canon.md @@ -0,0 +1,29 @@ +# R&D Program Finance Canon + +Reference for the accounting and budgeting rules that govern internal R&D spend. Pairs with `program_budget_planner.py` and `capex_vs_opex_router.py`. + +## The central question: research vs development + +The accounting treatment of R&D hinges on a phase distinction that the two major frameworks handle differently: + +- **IFRS (IAS 38)** — *Research* costs are always **expensed**. *Development* costs **must be capitalized** once all six conditions are met: (1) technical feasibility, (2) intention to complete, (3) ability to use or sell, (4) probable future economic benefit, (5) adequate resources to complete, (6) reliable measurement of expenditure. This is not optional under IFRS — if the criteria are met, capitalization is required. +- **US GAAP (ASC 730)** — R&D is **expensed as incurred**, full stop, with narrow exceptions. The main exception is software: **ASC 985-20** (software to be sold) capitalizes costs after *technological feasibility*; **ASC 350-40** (internal-use software) capitalizes during the application-development stage. + +This divergence is why the router takes a `--standard {ifrs,usgaap}` flag: the same cost item can be EXPENSE under US GAAP and CAPITALIZE-CANDIDATE under IFRS. + +## F&A / indirect cost (the budgeting half) + +Direct costs are traceable to the program (personnel, supplies). **Facilities & Administrative (F&A)**, a.k.a. indirect or overhead, covers shared costs (building, utilities, administration). For federally funded research, F&A is governed by **Uniform Guidance (2 CFR 200)**: organizations negotiate a rate (the NICRA — Negotiated Indirect Cost Rate Agreement) or use the **de minimis 10%** rate. F&A applies to the **Modified Total Direct Cost (MTDC)** base, which *excludes* capital equipment, the portion of each subaward over $25,000, tuition, and patient-care costs. The budget planner enforces this MTDC exclusion. + +## Why disclosure matters + +A budget number is only as trustworthy as its rate basis and escalation assumption. Two budgets for the same program can differ 40%+ purely on the F&A rate and the base. Every output of the planner ships an assumptions block for exactly this reason. + +## Sources + +1. IAS 38, *Intangible Assets* — IASB (research vs development, paragraphs 54-67). +2. FASB ASC 730, *Research and Development*; ASC 985-20, *Software — Costs of Software to Be Sold, Leased, or Marketed*; ASC 350-40, *Internal-Use Software*. +3. 2 CFR 200 (Uniform Guidance), Subpart E — Cost Principles, esp. §200.414 (Indirect F&A costs) and the MTDC definition (§200.1). +4. Cost Accounting Standards (CAS), 48 CFR 9904 — for federally funded R&D contractors. +5. KPMG / PwC / Deloitte IFRS-vs-US-GAAP comparison guides (R&D and intangibles chapters). +6. AICPA Accounting & Valuation Guide, *Research and Development*. diff --git a/research-ops/skills/research-finance/scripts/ar_evaluator.py b/research-ops/skills/research-finance/scripts/ar_evaluator.py new file mode 100644 index 00000000..010b950b --- /dev/null +++ b/research-ops/skills/research-finance/scripts/ar_evaluator.py @@ -0,0 +1,74 @@ +#!/usr/bin/env python3 +"""ar_evaluator.py - Autoresearch evaluator for the research-finance skill (OPT-IN). + +Stdlib-only. The ISOLATED bridge to engineering/autoresearch-agent. It does NOT call +autoresearch; it is the ground-truth evaluator an autoresearch loop runs after editing +the target ledger/budget. It reads a ledger JSON, computes runway via burn_runway_tracker, +and prints ONE metric line: + + runway_months: (higher is better) + +Optimize a program plan to maximize runway (e.g., resequencing spend) while the agent +edits the target. The user opts in explicitly: + /ar:setup --domain custom --name extend-runway \\ + --target ledger.json --eval "python3 ar_evaluator.py --target ledger.json" \\ + --metric runway_months --direction higher + +Direct use: + python3 ar_evaluator.py --sample + python3 ar_evaluator.py --target ledger.json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 +import burn_runway_tracker as brt # noqa: E402 + +METRIC = "runway_months" + + +def main(argv: list[str] | None = None) -> int: + c = cfg.load_config() + p = argparse.ArgumentParser(description="Autoresearch evaluator: R&D program runway in months.") + p.add_argument("--target", help="path to ledger JSON (or env AR_TARGET)") + p.add_argument("--threshold-months", type=float, default=None) + p.add_argument("--sample", action="store_true") + args = p.parse_args(argv) + + threshold = args.threshold_months if args.threshold_months is not None \ + else c.get("runway_threshold_months", 6) + + if args.sample: + data = brt.SAMPLE + else: + target = args.target or os.environ.get("AR_TARGET") + if not target: + print("error: provide --target or set AR_TARGET", file=sys.stderr) + return 2 + try: + with open(target) as f: + data = json.load(f) + except (OSError, json.JSONDecodeError) as e: + print(f"{METRIC}: N/A") + print(f"error: {e}", file=sys.stderr) + return 1 + + try: + result = brt.analyze(data, threshold) + except ValueError as e: + print(f"{METRIC}: N/A") + print(f"error: {e}", file=sys.stderr) + return 1 + + print(f"{METRIC}: {result['runway_months_approx']}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-finance/scripts/burn_runway_tracker.py b/research-ops/skills/research-finance/scripts/burn_runway_tracker.py new file mode 100644 index 00000000..59daab1b --- /dev/null +++ b/research-ops/skills/research-finance/scripts/burn_runway_tracker.py @@ -0,0 +1,153 @@ +#!/usr/bin/env python3 +"""burn_runway_tracker.py - Compute R&D program burn, runway, and milestone-vs-cash alignment. + +Stdlib-only. Deterministic. NO LLM calls. Surfaces the assumption behind every number. + +Given cash-on-hand, a period ledger of actual spend, and upcoming milestones (each with a +period index and the cash needed to reach it), computes: + - average + trailing burn rate + - runway in periods and (approx) months + - whether each value-inflection milestone is reachable before cash runs out + +Usage: + python3 burn_runway_tracker.py --sample + python3 burn_runway_tracker.py --input ledger.json --threshold-months 6 + python3 burn_runway_tracker.py --input ledger.json --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +SAMPLE = { + "program": "Next-Gen Assay Platform", + "cash_on_hand": 3200000, + "period_label": "month", + "actual_spend": [285000, 305000, 330000, 360000], + "milestones": [ + {"name": "Analytical validation", "period_from_now": 3, "cumulative_cash_needed": 1000000}, + {"name": "First-in-human readiness", "period_from_now": 9, "cumulative_cash_needed": 3400000}, + ], +} + + +def analyze(data: dict, threshold_months: float) -> dict: + spend = [float(x) for x in data.get("actual_spend", [])] + cash = float(data.get("cash_on_hand", 0.0)) + label = data.get("period_label", "month") + months_per_period = 1.0 if label == "month" else (3.0 if label == "quarter" else 1.0) + + if not spend: + raise ValueError("actual_spend must contain at least one period.") + + avg_burn = sum(spend) / len(spend) + trailing_n = min(3, len(spend)) + trailing_burn = sum(spend[-trailing_n:]) / trailing_n + # Use trailing burn (more recent) as the forward run-rate. + run_rate = trailing_burn if trailing_burn > 0 else avg_burn + runway_periods = cash / run_rate if run_rate > 0 else float("inf") + runway_months = runway_periods * months_per_period + + milestones_out = [] + for m in data.get("milestones", []): + needed = float(m.get("cumulative_cash_needed", 0.0)) + period_from_now = float(m.get("period_from_now", 0)) + reachable_cash = needed <= cash + reachable_time = period_from_now <= runway_periods + verdict = "REACHABLE" if (reachable_cash and reachable_time) else "AT-RISK" + milestones_out.append({ + "name": m.get("name", "UNNAMED"), + "period_from_now": period_from_now, + "cumulative_cash_needed": needed, + "cash_covers": reachable_cash, + "runway_covers_timing": reachable_time, + "verdict": verdict, + }) + + flags = [] + if runway_months < threshold_months: + flags.append(f"RUNWAY BELOW THRESHOLD: {runway_months:.1f} months < {threshold_months} month threshold.") + if any(m["verdict"] == "AT-RISK" for m in milestones_out): + flags.append("At least one value-inflection milestone is AT-RISK on current burn.") + if trailing_burn > avg_burn * 1.15: + flags.append(f"Burn accelerating: trailing burn ${trailing_burn:,.0f} > 115% of average ${avg_burn:,.0f}.") + + return { + "program": data.get("program", "UNSPECIFIED"), + "cash_on_hand": cash, + "average_burn_per_period": round(avg_burn, 2), + "trailing_burn_per_period": round(trailing_burn, 2), + "forward_run_rate_used": round(run_rate, 2), + "runway_periods": round(runway_periods, 2), + "runway_months_approx": round(runway_months, 1), + "milestones": milestones_out, + "flags": flags, + "assumptions": [ + f"Forward run-rate = trailing {trailing_n}-period burn (recent-weighted, not lifetime average).", + f"Period label '{label}' => {months_per_period} month(s) per period.", + "Runway assumes flat forward burn; a funded ramp or hiring plan changes this.", + "Milestone cash needs are cumulative-from-now as supplied; verify against the program budget.", + ], + } + + +def _render_human(r: dict) -> str: + lines = [f"Burn & Runway: {r['program']}", "", + f"Cash on hand: ${r['cash_on_hand']:,.0f}", + f"Average burn/period: ${r['average_burn_per_period']:,.0f}", + f"Trailing burn/period: ${r['trailing_burn_per_period']:,.0f}", + f"Forward run-rate used: ${r['forward_run_rate_used']:,.0f}", + f"Runway: {r['runway_periods']} periods (~{r['runway_months_approx']} months)", + ""] + lines.append("Milestones:") + for m in r["milestones"]: + lines.append(f" [{m['verdict']}] {m['name']} (+{m['period_from_now']:.0f} periods, " + f"needs ${m['cumulative_cash_needed']:,.0f})") + lines.append("") + if r["flags"]: + lines.append("Flags:") + for f in r["flags"]: + lines.append(f" ! {f}") + lines.append("") + lines.append("Assumptions:") + for a in r["assumptions"]: + lines.append(f" - {a}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Compute R&D program burn, runway, and milestone alignment.") + p.add_argument("--input", help="Path to JSON ledger") + p.add_argument("--threshold-months", type=float, default=None, help="runway alert threshold (months)") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + threshold = args.threshold_months if args.threshold_months is not None \ + else float(conf.get("runway_threshold_months", 6.0)) + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + try: + result = analyze(data, threshold) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-finance/scripts/capex_vs_opex_router.py b/research-ops/skills/research-finance/scripts/capex_vs_opex_router.py new file mode 100644 index 00000000..0bd627eb --- /dev/null +++ b/research-ops/skills/research-finance/scripts/capex_vs_opex_router.py @@ -0,0 +1,189 @@ +#!/usr/bin/env python3 +"""capex_vs_opex_router.py - Decision-SUPPORT for R&D capitalize-vs-expense treatment. + +Stdlib-only. Deterministic. NO LLM calls. This tool NEVER books an entry and NEVER +auto-decides accounting treatment. It scores each cost item against capitalization +criteria and ROUTES it to a named finance owner for the actual determination. + +Criteria reflect IAS 38 (development-phase capitalization test) and US GAAP ASC 730 +(R&D expensed as incurred) / ASC 985-20 (internal-use & sold software). The six IAS 38 +development-phase conditions: + 1. technical feasibility established + 2. intention to complete + 3. ability to use or sell + 4. probable future economic benefit + 5. adequate resources to complete + 6. reliable measurement of expenditure + +Verdicts: + - CAPITALIZE-CANDIDATE (development phase, all criteria met) -> still routes to finance owner + - EXPENSE (research phase, or criteria not met) + - FINANCE-OWNER-REVIEW (ambiguous / partial criteria) + +Usage: + python3 capex_vs_opex_router.py --sample + python3 capex_vs_opex_router.py --input costs.json --standard ifrs + python3 capex_vs_opex_router.py --input costs.json --standard usgaap --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +IAS38_CRITERIA = [ + "technical_feasibility", + "intention_to_complete", + "ability_to_use_or_sell", + "probable_future_benefit", + "adequate_resources", + "reliable_measurement", +] + +# Profiles only annotate context; they do not change the accounting test. +PROFILES = { + "pharma-rd": "Most drug R&D is expensed; capitalization rare pre-approval.", + "biotech": "Similar to pharma; pre-approval development typically expensed.", + "medtech": "Some development capitalizable post-feasibility under IFRS.", + "deep-tech": "Prototype-to-product transition is the key feasibility line.", + "software-rd": "ASC 985-20 / IAS 38: capitalize after technological feasibility / working model.", + "university-lab": "Grant-funded research almost always expensed per funder terms.", +} + +SAMPLE = { + "standard": "ifrs", + "items": [ + { + "name": "Exploratory target screening", + "phase": "research", + "criteria": {}, + }, + { + "name": "Pilot-line tooling for validated design", + "phase": "development", + "criteria": { + "technical_feasibility": True, "intention_to_complete": True, + "ability_to_use_or_sell": True, "probable_future_benefit": True, + "adequate_resources": True, "reliable_measurement": True, + }, + }, + { + "name": "Software build (post working-model, pre-release)", + "phase": "development", + "criteria": { + "technical_feasibility": True, "intention_to_complete": True, + "ability_to_use_or_sell": True, "probable_future_benefit": True, + "adequate_resources": False, "reliable_measurement": True, + }, + }, + ], +} + + +def route_item(item: dict, standard: str) -> dict: + phase = (item.get("phase") or "").lower() + crit = item.get("criteria", {}) or {} + met = [c for c in IAS38_CRITERIA if crit.get(c)] + missing = [c for c in IAS38_CRITERIA if not crit.get(c)] + + # US GAAP ASC 730: R&D expensed as incurred (software is the main exception via ASC 985-20). + if standard == "usgaap" and phase != "software-development": + verdict = "EXPENSE" + rationale = "ASC 730: R&D is expensed as incurred (non-software). Confirm software exceptions separately." + owner = "R&D Finance Controller" + elif phase == "research": + verdict = "EXPENSE" + rationale = "Research phase: cannot capitalize (IAS 38.54)." + owner = "R&D Finance Controller" + elif phase in ("development", "software-development") and not missing: + verdict = "CAPITALIZE-CANDIDATE" + rationale = "Development phase with all 6 IAS 38 criteria asserted. Routed for finance confirmation." + owner = "R&D Finance Controller + External Auditor sign-off" + else: + verdict = "FINANCE-OWNER-REVIEW" + rationale = f"Development phase but {len(missing)} criteria unmet/unstated: {', '.join(missing) or 'n/a'}." + owner = "R&D Finance Controller" + + return { + "name": item.get("name", "UNNAMED"), + "phase": phase or "UNSPECIFIED", + "criteria_met": met, + "criteria_missing": missing, + "verdict": verdict, + "rationale": rationale, + "named_owner": owner, + } + + +def route(data: dict, standard: str, profile: str) -> dict: + if profile not in PROFILES: + raise ValueError(f"Unknown profile '{profile}'. Choose from {list(PROFILES)}.") + items = [route_item(i, standard) for i in data.get("items", [])] + return { + "standard": standard, + "profile": profile, + "profile_note": PROFILES[profile], + "items": items, + "disclaimer": "DECISION SUPPORT ONLY. This tool does not book entries or decide treatment. " + "A named finance owner (and auditor where required) makes the determination.", + } + + +def _render_human(r: dict) -> str: + lines = [f"Capitalize-vs-Expense routing (standard: {r['standard']}, profile: {r['profile']})", + f" {r['profile_note']}", ""] + for it in r["items"]: + lines.append(f"[{it['verdict']}] {it['name']} (phase: {it['phase']})") + lines.append(f" {it['rationale']}") + if it["criteria_missing"]: + lines.append(f" missing/unstated: {', '.join(it['criteria_missing'])}") + lines.append(f" -> route to: {it['named_owner']}") + lines.append("") + lines.append(f"!! {r['disclaimer']}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Route R&D costs to capitalize/expense/review (DECISION SUPPORT ONLY).") + p.add_argument("--input", help="Path to JSON with items[]") + p.add_argument("--standard", default=None, choices=["ifrs", "usgaap"], + help="overrides onboarding accounting_standard") + p.add_argument("--profile", default=None, choices=list(PROFILES), + help="overrides onboarding default_profile") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + profile = args.profile or conf.get("default_profile", "biotech") + cli_standard = args.standard or conf.get("accounting_standard", "ifrs") + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + standard = data.get("standard", cli_standard) if (args.sample or not args.input) else cli_standard + try: + result = route(data, standard, profile) + except ValueError as e: + print(f"error: {e}", file=sys.stderr) + return 2 + finance_owner = conf.get("finance_owner") + if finance_owner: + for it in result["items"]: + it["named_owner"] = it["named_owner"].replace( + "R&D Finance Controller", f"R&D Finance Controller ({finance_owner})") + + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-finance/scripts/config_loader.py b/research-ops/skills/research-finance/scripts/config_loader.py new file mode 100644 index 00000000..15d13972 --- /dev/null +++ b/research-ops/skills/research-finance/scripts/config_loader.py @@ -0,0 +1,115 @@ +#!/usr/bin/env python3 +"""config_loader.py - Customization loader for the research-finance skill. + +Stdlib-only. Importable from the skill's other scripts. Precedence (highest wins): + 1. Project config: /.research-ops/research-finance.json + 2. Global config: ~/.config/research-ops/research-finance.json + 3. Built-in DEFAULTS + +Onboarding answers (written by onboard.py) live in these files; every tool in this +skill reads them so the user's customization applies automatically. +Set RESEARCH_OPS_NO_CONFIG=1 to ignore saved config. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from pathlib import Path +from typing import Any + +SKILL = "research-finance" + +GLOBAL_CONFIG_DIR = Path.home() / ".config" / "research-ops" +GLOBAL_CONFIG_PATH = GLOBAL_CONFIG_DIR / f"{SKILL}.json" +PROJECT_CONFIG_DIRNAME = ".research-ops" + +DEFAULTS: dict[str, Any] = { + "version": 1, + "skill": SKILL, + "default_profile": "biotech", + "default_fa_rate": None, # None => use the profile's default F&A rate + "runway_threshold_months": 6, + "accounting_standard": "ifrs", + "finance_owner": None, + "setup_completed_at": None, +} + + +def project_config_path(cwd: Path | None = None) -> Path: + cwd = cwd or Path.cwd() + return cwd / PROJECT_CONFIG_DIRNAME / f"{SKILL}.json" + + +def _read_json(path: Path) -> dict[str, Any] | None: + try: + with path.open(encoding="utf-8") as f: + data = json.load(f) + return data if isinstance(data, dict) else None + except (FileNotFoundError, json.JSONDecodeError, OSError): + return None + + +def _deep_merge(base: dict[str, Any], override: dict[str, Any]) -> dict[str, Any]: + out = dict(base) + for k, v in override.items(): + if isinstance(v, dict) and isinstance(out.get(k), dict): + out[k] = _deep_merge(out[k], v) + else: + out[k] = v + return out + + +def load_config(cwd: Path | None = None) -> dict[str, Any]: + config = dict(DEFAULTS) + if os.environ.get("RESEARCH_OPS_NO_CONFIG") == "1": + return config + global_cfg = _read_json(GLOBAL_CONFIG_PATH) + if global_cfg: + config = _deep_merge(config, global_cfg) + project_cfg = _read_json(project_config_path(cwd)) + if project_cfg: + config = _deep_merge(config, project_cfg) + return config + + +def setup_completed() -> bool: + cfg = _read_json(GLOBAL_CONFIG_PATH) or _read_json(project_config_path()) + return bool(cfg and cfg.get("setup_completed_at")) + + +def write_config(config: dict[str, Any], scope: str = "global", cwd: Path | None = None) -> Path: + path = project_config_path(cwd) if scope == "project" else GLOBAL_CONFIG_PATH + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as f: + json.dump(config, f, indent=2, sort_keys=True) + return path + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Inspect {SKILL} customization config.") + p.add_argument("--show", action="store_true", help="Print the effective config") + p.add_argument("--status", action="store_true", help="Print setup status + paths") + p.add_argument("--sample", action="store_true", help="Print the built-in defaults") + args = p.parse_args(argv) + + if args.sample: + print(json.dumps(DEFAULTS, indent=2, sort_keys=True)) + elif args.status: + print(json.dumps({ + "skill": SKILL, + "global_config_path": str(GLOBAL_CONFIG_PATH), + "global_config_exists": GLOBAL_CONFIG_PATH.exists(), + "project_config_path": str(project_config_path()), + "project_config_exists": project_config_path().exists(), + "setup_completed": setup_completed(), + }, indent=2)) + else: + print(json.dumps(load_config(), indent=2, sort_keys=True)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-finance/scripts/onboard.py b/research-ops/skills/research-finance/scripts/onboard.py new file mode 100644 index 00000000..bbf7a0b3 --- /dev/null +++ b/research-ops/skills/research-finance/scripts/onboard.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""onboard.py - Onboarding questionnaire for the research-finance skill. + +Stdlib-only. Asks the user a short set of questions BEFORE they build an R&D program +budget, then writes the answers to a customization config read by every tool in this +skill via config_loader.py. The answers become defaults for profile, F&A rate, runway +threshold, accounting standard, and the named finance owner printed on routing outputs. + +Modes: --show | --defaults | --set key=value (repeatable) | --reset | --scope {global,project} +""" + +from __future__ import annotations + +import argparse +import datetime as _dt +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import config_loader as cfg # noqa: E402 + +NUMERIC_KEYS = {"default_fa_rate", "runway_threshold_months"} + +QUESTIONS = [ + ("default_profile", + "1. What R&D area is this program?", + ["pharma-rd", "biotech", "medtech", "deep-tech", "software-rd", "university-lab"], str), + ("default_fa_rate", + "2. F&A / indirect rate as a fraction (e.g. 0.55), or blank to use the profile default?", + None, float), + ("runway_threshold_months", + "3. Runway alert threshold in months (warn below this)?", + None, float), + ("accounting_standard", + "4. Which accounting standard governs capitalize-vs-expense?", + ["ifrs", "usgaap"], str), + ("finance_owner", + "5. Named finance/controller owner who signs accounting treatment?", + None, str), +] + + +def _print_questions() -> None: + print(f"Onboarding questions — {cfg.SKILL}:\n") + for _k, prompt, choices, _c in QUESTIONS: + line = f" {prompt}" + if choices: + line += f" [{' / '.join(choices)}]" + print(line) + + +def run_interactive(config: dict) -> dict: + print(f"Onboarding — {cfg.SKILL}. Press Enter to keep the current/default value.\n") + for key, prompt, choices, caster in QUESTIONS: + suffix = f" [{'/'.join(choices)}]" if choices else "" + cur = f" (current: {config.get(key)})" if config.get(key) is not None else "" + raw = input(f"{prompt}{suffix}{cur}: ").strip() + if not raw: + continue + try: + config[key] = caster(raw) + except ValueError: + print(f" ! invalid value for {key}, keeping current") + return config + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description=f"Onboarding for the {cfg.SKILL} skill.") + p.add_argument("--show", action="store_true") + p.add_argument("--defaults", action="store_true", help="write built-in defaults, no prompt") + p.add_argument("--set", action="append", default=[], metavar="key=value") + p.add_argument("--reset", action="store_true") + p.add_argument("--scope", choices=["global", "project"], default="global") + args = p.parse_args(argv) + + if args.show: + _print_questions() + print("\nCurrent effective config:") + print(json.dumps(cfg.load_config(), indent=2, sort_keys=True)) + return 0 + + if args.reset: + path = cfg.project_config_path() if args.scope == "project" else cfg.GLOBAL_CONFIG_PATH + if path.exists(): + path.unlink(); print(f"removed {path}") + else: + print(f"no config at {path}") + return 0 + + config = cfg.load_config() + + if args.set: + for item in args.set: + if "=" not in item: + print(f"error: --set expects key=value, got '{item}'", file=sys.stderr) + return 2 + k, v = item.split("=", 1) + if k in NUMERIC_KEYS: + try: + v = float(v) + except ValueError: + pass + config[k] = v + elif not args.defaults: + if sys.stdin.isatty(): + config = run_interactive(config) + else: + print("non-interactive shell: use --defaults or --set key=value. Showing questions:\n") + _print_questions() + return 0 + + config["setup_completed_at"] = _dt.datetime.now(_dt.timezone.utc).isoformat() + path = cfg.write_config(config, scope=args.scope) + print(f"saved {cfg.SKILL} customization -> {path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-finance/scripts/program_budget_planner.py b/research-ops/skills/research-finance/scripts/program_budget_planner.py new file mode 100644 index 00000000..417aa76a --- /dev/null +++ b/research-ops/skills/research-finance/scripts/program_budget_planner.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 +"""program_budget_planner.py - Build a multi-period R&D program budget with F&A split. + +Stdlib-only. Deterministic. NO LLM calls. Every output surfaces an explicit assumptions +block: budget math without disclosed assumptions is theatre. + +Takes work-package line items, applies the F&A (indirect) rate to the F&A-eligible base +(MTDC-style: excludes capital equipment and the portion of subawards over $25k), computes +fully-loaded cost, and rolls up per period. + +Usage: + python3 program_budget_planner.py --sample + python3 program_budget_planner.py --input program.json --fa-rate 0.55 --periods 4 + python3 program_budget_planner.py --input program.json --profile biotech --output json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + import config_loader as _cfg +except ImportError: # pragma: no cover + _cfg = None + +# Profile default F&A (indirect) rate and escalation assumption when not supplied in input. +PROFILES = { + "pharma-rd": {"default_fa_rate": 0.50, "annual_escalation": 0.03}, + "biotech": {"default_fa_rate": 0.55, "annual_escalation": 0.04}, + "medtech": {"default_fa_rate": 0.45, "annual_escalation": 0.03}, + "deep-tech": {"default_fa_rate": 0.40, "annual_escalation": 0.03}, + "software-rd": {"default_fa_rate": 0.30, "annual_escalation": 0.04}, + "university-lab": {"default_fa_rate": 0.585, "annual_escalation": 0.025}, +} + +# Categories excluded from the F&A (MTDC) base. +FA_EXEMPT_CATEGORIES = {"capital_equipment", "subaward_over_25k", "tuition", "patient_care"} + +SAMPLE = { + "program": "Next-Gen Assay Platform", + "periods": 4, + "work_packages": [ + {"name": "Personnel (FTEs)", "category": "personnel", "amounts": [320000, 330000, 340000, 350000]}, + {"name": "Consumables", "category": "supplies", "amounts": [60000, 65000, 70000, 70000]}, + {"name": "Sequencer", "category": "capital_equipment", "amounts": [180000, 0, 0, 0]}, + {"name": "CRO subaward", "category": "subaward_over_25k", "amounts": [100000, 100000, 0, 0]}, + {"name": "Travel", "category": "travel", "amounts": [12000, 12000, 12000, 12000]}, + ], +} + + +def _period_sum(amounts: list, n: int, idx: int) -> float: + return float(amounts[idx]) if idx < len(amounts) else 0.0 + + +def plan_budget(data: dict, fa_rate: float, periods: int) -> dict: + wps = data.get("work_packages", []) + direct_by_period = [0.0] * periods + fa_base_by_period = [0.0] * periods + line_items = [] + + for wp in wps: + cat = wp.get("category", "other") + amounts = wp.get("amounts", []) + fa_eligible = cat not in FA_EXEMPT_CATEGORIES + wp_total = 0.0 + for i in range(periods): + amt = _period_sum(amounts, periods, i) + direct_by_period[i] += amt + if fa_eligible: + fa_base_by_period[i] += amt + wp_total += amt + line_items.append({ + "name": wp.get("name", "UNNAMED"), + "category": cat, + "fa_eligible": fa_eligible, + "total_direct": round(wp_total, 2), + }) + + fa_by_period = [round(b * fa_rate, 2) for b in fa_base_by_period] + loaded_by_period = [round(direct_by_period[i] + fa_by_period[i], 2) for i in range(periods)] + + return { + "program": data.get("program", "UNSPECIFIED"), + "periods": periods, + "fa_rate_applied": fa_rate, + "line_items": line_items, + "direct_by_period": [round(x, 2) for x in direct_by_period], + "fa_base_by_period": [round(x, 2) for x in fa_base_by_period], + "fa_by_period": fa_by_period, + "fully_loaded_by_period": loaded_by_period, + "total_direct": round(sum(direct_by_period), 2), + "total_fa": round(sum(fa_by_period), 2), + "total_fully_loaded": round(sum(loaded_by_period), 2), + "assumptions": [ + f"F&A (indirect) rate applied: {fa_rate:.1%}. Confirm this is your negotiated NICRA, not an assumption.", + f"F&A base excludes: {', '.join(sorted(FA_EXEMPT_CATEGORIES))} (MTDC-style base).", + "Amounts are taken as-entered per period; no escalation applied unless baked into inputs.", + "This is a planning estimate; a finance owner/controller validates the rate basis and booking.", + ], + } + + +def _render_human(r: dict) -> str: + lines = [f"R&D Program Budget: {r['program']} ({r['periods']} periods)", + f"F&A rate applied: {r['fa_rate_applied']:.1%}", ""] + lines.append("Line items:") + for li in r["line_items"]: + tag = "F&A-eligible" if li["fa_eligible"] else "F&A-EXEMPT" + lines.append(f" {li['name']:24s} {li['category']:20s} {tag:12s} ${li['total_direct']:,.0f}") + lines.append("") + hdr = " " + "".join(f"P{i+1:>14}" for i in range(r["periods"])) + lines.append("Per-period rollup:" ) + lines.append(hdr) + lines.append(" direct " + "".join(f"{v:>15,.0f}" for v in r["direct_by_period"])) + lines.append(" F&A " + "".join(f"{v:>15,.0f}" for v in r["fa_by_period"])) + lines.append(" loaded " + "".join(f"{v:>15,.0f}" for v in r["fully_loaded_by_period"])) + lines.append("") + lines.append(f"Total direct: ${r['total_direct']:,.0f}") + lines.append(f"Total F&A: ${r['total_fa']:,.0f}") + lines.append(f"Total fully-loaded: ${r['total_fully_loaded']:,.0f}") + lines.append("") + lines.append("Assumptions (state these alongside the number):") + for a in r["assumptions"]: + lines.append(f" - {a}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Build a multi-period R&D program budget with F&A split.") + p.add_argument("--input", help="Path to JSON program with work_packages[]") + p.add_argument("--profile", default=None, choices=list(PROFILES), + help="overrides onboarding default_profile") + p.add_argument("--fa-rate", type=float, default=None, help="Override F&A rate (fraction, e.g. 0.55)") + p.add_argument("--periods", type=int, default=None, help="Number of periods") + p.add_argument("--output", choices=["human", "json"], default="human") + p.add_argument("--sample", action="store_true", help="use the embedded sample") + args = p.parse_args(argv) + + conf = _cfg.load_config() if _cfg else {} + profile = args.profile or conf.get("default_profile", "biotech") + data = SAMPLE if (args.sample or not args.input) else json.load(open(args.input)) + periods = args.periods or int(data.get("periods", 4)) + # F&A precedence: CLI flag > onboarding default_fa_rate (if set) > profile default + if args.fa_rate is not None: + fa_rate = args.fa_rate + elif conf.get("default_fa_rate") is not None: + fa_rate = float(conf["default_fa_rate"]) + else: + fa_rate = PROFILES[profile]["default_fa_rate"] + + result = plan_budget(data, fa_rate, periods) + if args.output == "json": + print(json.dumps(result, indent=2)) + else: + print(_render_human(result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/research-ops/skills/research-ops-skills/SKILL.md b/research-ops/skills/research-ops-skills/SKILL.md new file mode 100644 index 00000000..ead536cf --- /dev/null +++ b/research-ops/skills/research-ops-skills/SKILL.md @@ -0,0 +1,156 @@ +--- +name: research-ops-skills +description: Use when planning, funding, scoping, or synthesizing enterprise research across workstreams — clinical study design, R&D program finance, market sizing/surveys, or product/user research. Triggers on "design this clinical study", "what sample size", "R&D budget", "burn rate", "capitalize or expense", "TAM SAM SOM", "market sizing", "survey design", "segment the market", "plan user interviews", "usability test", "synthesize research insights". Forks context to route to one of four Research-Operations sub-skills (clinical-research, research-finance, market-research, product-research) and returns a digest. Distinct from ra-qm-team (regulatory submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), and marketing-skill (campaign analytics). +context: fork +version: 2.9.0 +author: claude-code-skills +license: MIT +tags: [research-ops, clinical-research, research-finance, market-research, product-research, rd, orchestrator] +compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli] +--- + +# Research Operations — Domain Orchestrator + +The Research Operations surface is **how the enterprise plans, funds, scopes, and synthesizes research** across four workstreams: clinical R&D, R&D finance, market research, and product research. This orchestrator forks its context, routes your inquiry to one of four sub-skills, then returns a digest. Heavy intake (protocol drafts, program ledgers, survey exports, interview transcripts) stays in the forked context. + +This is the enterprise counterpart to the academic `research/` domain. If your question is about **finding** literature, grants, or patents, use `research/`. If it is about **planning, funding, scoping, or synthesizing** research as an operational discipline, you are in the right place. + +## When to invoke + +| Symptom | Sub-skill | +|---|---| +| "We're designing a Phase 2 trial — what's the endpoint and sample size?" | `clinical-research` | +| "What's our R&D program burn, and is this cost CapEx or OpEx?" | `research-finance` | +| "What's the TAM for this product, and how do we survey the segment?" | `market-research` | +| "How many users do we interview, and how do we synthesize the findings?" | `product-research` | + +## Routing logic (deterministic) + +Same two-signal threshold pattern as `commercial-skills`. Single-signal → clarifying question. Mixed signals → highest-confidence first, chain second in a follow-up turn. Never silently chain. + +### Signal table + +| Signal class | Keywords | Sub-skill | +|---|---|---| +| **CLINICAL** | clinical trial, study design, protocol, endpoint, sample size, power, phase 1/2/3, biostatistics, eligibility, feasibility, estimand | `clinical-research` | +| **RD_FINANCE** | R&D budget, program budget, burn, runway, F&A, indirect rate, overhead, capitalize vs expense, R&D capex, portfolio ROI, rNPV | `research-finance` | +| **MARKET** | TAM, SAM, SOM, market sizing, survey design, sampling, margin of error, segmentation, competitive intelligence, market research | `market-research` | +| **PRODUCT** | user interview, JTBD, usability test, concept test, prototype test, discovery research, research repository, insight synthesis, saturation | `product-research` | + +## Workflow (Matt Pocock grill discipline) + +Derived from Matt Pocock's `grill-with-docs` pattern: **explore-then-ask, one question per turn with a recommended answer, walk the decision tree depth-first, track dependencies, anchor every challenge in the research canon** (`references/` of each sub-skill). + +### Step 1 — Explore before asking + +Check the user's working directory first: +- Is there a protocol draft, program ledger, TAM model, or interview guide already in the workspace? +- Does the inquiry already disambiguate the lane (e.g., "what sample size for a two-arm trial" — that's `clinical-research`, no question needed)? +- Is there an artifact filename that resolves the lane (`protocol.json` → clinical; `program-budget.json` → finance; `tam-model.json` → market; `interview-guide.md` → product)? + +If the workspace resolves the lane, **route silently**. + +### Step 2 — If still ambiguous, ONE forcing question with a recommended answer + +Matt's rule: never bundle. Always recommend. + +Pattern: +``` +Q1/1: [precise question naming the two candidate lanes] +Recommended: [Lane X, because ] + +(Confirm, or override?) +``` + +### Step 3 — Decision-tree walk for multi-lane inquiries + +If the inquiry legitimately crosses two lanes (e.g., "design this trial AND budget it" = CLINICAL + RD_FINANCE), walk depth-first: + +1. Highest-confidence lane first → run sub-skill in forked context → digest +2. Ask: "Now run [second lane]? Recommended: yes, because [dependency]." +3. Confirm before chaining. + +Never silently chain. + +### Step 4 — Invoke sub-skill in forked context + +Forward original prompt + structured inputs (protocol JSON, program ledger CSV, market model, observation export). + +### Step 5 — Return digest with cited canon challenge + +≤ 200 words: analyzed, top 3 findings (anchored to a canon citation), top 3 next actions (named human owner where applicable), artifact path, and **one grill challenge** for the user. Examples: + +- "Your power calc assumes a 0.5 effect size with no published anchor. ICH E9 requires a justified, clinically meaningful difference. Where did 0.5 come from?" +- "Your TAM is a single top-down number (1% of a $40B market). Bessemer market-sizing discipline requires a bottoms-up cross-check. What's units × price × adoption?" + +## Forcing-question library (grill-with-docs pattern) + +Grill the user on lane-defining decisions before invoking the sub-skill. One per turn, recommended answer, canon citation: + +- **CLINICAL lane**: "Is your primary endpoint a clinical outcome or a surrogate — and if surrogate, is it validated for this indication? Recommended: clinical outcome unless the surrogate is on FDA's validated table. Canon: FDA Surrogate Endpoint Table; BEST glossary." +- **RD_FINANCE lane**: "Is this spend in the research phase or the development phase, and can you evidence technical feasibility? Recommended: research = expense; development = capitalize-candidate only with feasibility evidence, routed to a named finance owner. Canon: IAS 38; ASC 730." +- **MARKET lane**: "Is your TAM top-down or bottoms-up — and have you computed it both ways to triangulate? Recommended: both; reconcile the delta. Canon: Bessemer / a16z market-sizing; Fermi estimation." +- **PRODUCT lane**: "Is this study generative (discover problems) or evaluative (test a solution)? Recommended: name it first; the method follows. Canon: Rohrer's landscape of UX research methods (NN/g)." + +Never run a sub-skill until the lane-defining decision is locked. + +## Onboarding-first (per sub-skill) + +Before invoking a sub-skill for the first time in a workspace, point the user at that skill's onboarding questionnaire so the tools run pre-configured to their context: + +```bash +python3 skills//scripts/onboard.py # interactive Q&A +python3 skills//scripts/onboard.py --show # questions + current config +``` + +Each sub-skill has its **own** question set (clinical: area/alpha/power/dropout/owners · finance: area/F&A/runway/standard/owner · market: profile/confidence/MoE/method · product: profile/insight-threshold/method/stakes). Answers persist to `~/.config/research-ops/.json` (or `./.research-ops/.json` with `--scope project`) and are consumed automatically by every tool in that skill. Customization is mandatory discipline here, not decoration — surface the onboarding step when a user starts a fresh research workstream. + +## Autoresearch handoff (isolated, opt-in) + +Each sub-skill ships its own `scripts/ar_evaluator.py` — an **isolated** bridge to `engineering/autoresearch-agent`. Invoke autoresearch **only when the user explicitly asks** to "optimize", "improve", or "run a loop". The handoff is per-skill (no shared coupling): the loop edits the skill's input file and the evaluator scores it (clinical → `feasibility_composite` higher; finance → `runway_months` higher; market → `tam_divergence` lower; product → `validated_insights` higher). Never auto-start a loop; never let the loop edit the evaluator. + +## Assumptions + +1. User has research authority OR is preparing analysis for someone who does. +2. User wants **deterministic decision support**, not the final answer — a clinician approves the protocol, a controller books the entry, the human picks the market number. +3. Inputs may be partial — every sub-skill ships a templated sample so the user can see the shape before filling in their own. + +## Non-goals + +- Not an EDC, clinical-trial-management system, accounting system, survey platform, or research repository. +- Does not give clinical, accounting, or legal advice as fact. Every output is **a recommendation + named human owner**. +- Does not store research history across sessions. + +## Distinct from + +- **`research/` (academic)** — that domain **finds** literature, grants, and patents. This domain **plans, funds, scopes, and synthesizes** research. +- **`ra-qm-team`** — that's **regulatory/QM submission** (ISO 13485/14971, MDR, FDA 510(k)/PMA/QSR). clinical-research designs the **study**; it routes submission out to ra-qm-team. +- **`finance/financial-analysis`** — that's **corporate close + valuation**. research-finance manages **R&D program/portfolio spend**. +- **`research/grants`** — that's **funding discovery**. research-finance manages **money already won**. +- **`product-team`** — that's **persona/journey artifacts, discovery sprints, and live A/B experiments**. product-research is the **method + repository discipline**. +- **`marketing-skill`** — that's **campaign analytics and demand-gen**. market-research is **upstream methodology**. + +## Output artifacts + +| Sub-skill | Artifact | +|---|---| +| clinical-research | `protocol_synopsis.md` + `sample_size.json` | +| research-finance | `rd_program_budget.md` + `capex_opex_routing.json` | +| market-research | `market_sizing.md` + `sample_plan.json` | +| product-research | `research_plan.md` + `insight_synthesis.json` | + +## Anti-patterns (do not) + +- ❌ Present a clinical power/endpoint output as fact — it is an **estimate** with a named clinical owner +- ❌ Auto-decide capitalize-vs-expense — route to a **named finance owner** +- ❌ Report a market size as a single unsourced number — show **method + both-ways triangulation + assumptions** +- ❌ Assert a product insight from a single participant — flag it as an **anecdote** +- ❌ Run all 4 sub-skills "to be thorough" — pick one, digest, chain if needed + +## References + +- Clinical canon: ICH E8(R1)/E9/E9(R1), CONSORT, SPIRIT, FDA Multiple Endpoints +- R&D finance canon: IAS 38, ASC 730, 2 CFR 200, Cooper stage-gate +- Market canon: Cochran, Dillman, Kotler, Bessemer market-sizing +- Product canon: Nielsen, Guest et al., Christensen JTBD, ResearchOps/Polaris +- Path-B build pattern: `documentation/implementation/research-ops-expansion-plan.md` diff --git a/scripts/generate-docs.py b/scripts/generate-docs.py index df45e8bd..b75f292c 100644 --- a/scripts/generate-docs.py +++ b/scripts/generate-docs.py @@ -24,6 +24,7 @@ DOMAINS = { "research": ("Research", 12, ":material-magnify:", "research-skills"), "business-operations": ("Business Operations", 13, ":material-cog-outline:", "business-operations-skills"), "commercial": ("Commercial", 14, ":material-handshake-outline:", "commercial-skills"), + "research-ops": ("Research Operations", 15, ":material-flask-outline:", "research-ops-skills"), } # Skills to skip (nested assets, samples, etc.) @@ -629,6 +630,7 @@ description: "{agent_desc}" "finance": "finance", "business-operations": "business-operations", "commercial": "commercial", + "research-ops": "research-ops", } seen_slugs = {entry[1] for entry in agent_entries} for skill_domain in DOMAINS: diff --git a/scripts/sync-codex-skills.py b/scripts/sync-codex-skills.py index 1223ba6f..670973ec 100644 --- a/scripts/sync-codex-skills.py +++ b/scripts/sync-codex-skills.py @@ -75,6 +75,10 @@ SKILL_DOMAINS = { "commercial": { "category": "commercial", "description": "Per-deal-and-packaging Commercial skills (v2.8.0): pricing strategy, deal desk, partnerships, channel economics, commercial policy, RFP responder, commercial forecaster" + }, + "research-ops": { + "category": "research-ops", + "description": "Enterprise Research Operations skills (v2.9.0): clinical study design, R&D program finance, market research methodology, product/user research" } } diff --git a/scripts/sync-gemini-skills.py b/scripts/sync-gemini-skills.py index 8c2a2633..fda5eb72 100644 --- a/scripts/sync-gemini-skills.py +++ b/scripts/sync-gemini-skills.py @@ -33,7 +33,8 @@ DOMAIN_MAP = { "marketing": "marketing-top-level", "research": "research", "business-operations": "business-operations", - "commercial": "commercial" + "commercial": "commercial", + "research-ops": "research-ops" } diff --git a/scripts/sync-hermes-skills.py b/scripts/sync-hermes-skills.py index 6b2bee16..3837586c 100644 --- a/scripts/sync-hermes-skills.py +++ b/scripts/sync-hermes-skills.py @@ -48,6 +48,7 @@ DOMAIN_DIRS = [ "research", # v2.7.0 — pulse, litreview, grants, dossier, patent, syllabus, notebooklm, research orchestrator "business-operations", # v2.8.0 — process-mapper, vendor-management, capacity-planner, internal-comms, knowledge-ops, procurement-optimizer + orchestrator "commercial", # v2.8.0 — pricing-strategist, deal-desk, partnerships-architect, channel-economics, commercial-policy, rfp-responder, commercial-forecaster + orchestrator + "research-ops", # v2.9.0 — clinical-research, research-finance, market-research, product-research + orchestrator ] diff --git a/scripts/sync-vibe-skills.py b/scripts/sync-vibe-skills.py index 63dfaa36..c5436bdd 100755 --- a/scripts/sync-vibe-skills.py +++ b/scripts/sync-vibe-skills.py @@ -49,6 +49,7 @@ DOMAIN_DIRS = [ "research", # v2.7.0 — pulse, litreview, grants, dossier, patent, syllabus, notebooklm, research orchestrator "business-operations", # v2.8.0 — process-mapper, vendor-management, capacity-planner, internal-comms, knowledge-ops, procurement-optimizer + orchestrator "commercial", # v2.8.0 — pricing-strategist, deal-desk, partnerships-architect, channel-economics, commercial-policy, rfp-responder, commercial-forecaster + orchestrator + "research-ops", # v2.9.0 — clinical-research, research-finance, market-research, product-research + orchestrator ]