Merge pull request #899 from alirezarezvani/dev
Some checks are pending
Deploy Documentation to Pages / build (push) Waiting to run
Deploy Documentation to Pages / deploy (push) Blocked by required conditions
Sync Codex Skills Symlinks / sync (push) Waiting to run

This commit is contained in:
Alireza Rezvani 2026-07-06 07:48:49 +02:00 committed by GitHub
commit 2fb75e1af3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
82 changed files with 19812 additions and 172 deletions

View file

@ -4,12 +4,12 @@
"name": "Alireza Rezvani",
"url": "https://alirezarezvani.com"
},
"description": "348 production-ready skill packages for Claude AI across 18 domains: engineering advanced (78, incl. v2.9.0 workflow-builder for Claude Code Workflow-tool authoring), engineering core (51), marketing (46 \u2014 incl. AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), compliance-os (9), project management (9), business growth (5), finance (4), productivity (6), marketing top-level (1), research (8), research-ops (5, v2.9.0), business-operations (7), commercial (8), markdown-html (5, v2.10.3 \u2014 markdown-to-interactive-HTML converter complete: orchestrator + design-system + md-document long-form + md-review code-review + md-slides slide-deck), and loop-library (1 \u2014 vendored Forward Future Loop Library skill). Includes 586 Python tools, 713 reference documents, 94 agents, 100 slash commands across 79 marketplace plugins.",
"description": "348 production-ready skill packages for Claude AI across 18 domains: engineering advanced (78, incl. v2.9.0 workflow-builder for Claude Code Workflow-tool authoring), engineering core (51), marketing (46 incl. AEO/Answer Engine Optimization), c-level advisory (66), product (17), regulatory/QMS (18), compliance-os (9), project management (9), business growth (5), finance (4), productivity (6), marketing top-level (1), research (8), research-ops (5, v2.9.0), business-operations (7), commercial (8), markdown-html (5, v2.10.3 markdown-to-interactive-HTML converter complete: orchestrator + design-system + md-document long-form + md-review code-review + md-slides slide-deck), and loop-library (1 vendored Forward Future Loop Library skill). Includes 586 Python tools, 713 reference documents, 94 agents, 100 slash commands across 79 marketplace plugins.",
"homepage": "https://github.com/alirezarezvani/claude-skills",
"repository": "https://github.com/alirezarezvani/claude-skills",
"metadata": {
"description": "354 production-ready skills across 18 domains (engineering, engineering-core, marketing, product, c-level, compliance-os, project management, RA/QM, business growth, finance, productivity, marketing top-level, research, research-ops, business-operations, commercial, markdown-html, loop-library, plus standards). 593 Python tools, 722 reference guides, 96 agents (cs-* + personas), 102 slash commands across 82 marketplace plugins. v2.10.3 completes the markdown-html domain with md-slides \u2014 slide-deck converter (arrow-key / Space / PgDn / Home/End / P keyboard navigation + presenter mode with split-view clock + speaker notes + next-slide preview + URL-hash deep linking like #3 for direct slide jumps + @media print page-per-slide for browser-native PDF export). Reuses md-document's markdown parser; vanilla JS only (no framework runtime); Prism.js opt-in via --syntax. Joins md-review (v2.10.2 code-review converter), md-document (v2.10.1 long-form converter), and the v2.10.0 foundation (orchestrator + design-system). Compatible with Claude Code, Codex CLI, Gemini CLI, Cursor, OpenClaw, Hermes Agent, Mistral Vibe, and 5 more coding agents.",
"version": "2.10.3"
"description": "355 production-ready skills across 18 domains (engineering, engineering-core, marketing, product, c-level, compliance-os, project management, RA/QM, business growth, finance, productivity, marketing top-level, research, research-ops, business-operations, commercial, markdown-html, loop-library, plus standards). 602 Python tools, 731 reference guides, 99 agents (cs-* + personas), 109 slash commands across 83 marketplace plugins. v2.11.1 turns product-team and project-management into agent-harness domains: fork-orchestrators with deterministic goal routers, a Jira MCP snapshot bridge (Kanban flow metrics + Monte Carlo forecasting), a delegation-governance loop gate, a continuous-discovery cadence tracker, and an Opportunity Solution Tree linter, with /cs:pm and /cs:product command families. v2.10.3 completes the markdown-html domain with md-slides — slide-deck converter (arrow-key / Space / PgDn / Home/End / P keyboard navigation + presenter mode with split-view clock + speaker notes + next-slide preview + URL-hash deep linking like #3 for direct slide jumps + @media print page-per-slide for browser-native PDF export). Reuses md-document's markdown parser; vanilla JS only (no framework runtime); Prism.js opt-in via --syntax. Joins md-review (v2.10.2 code-review converter), md-document (v2.10.1 long-form converter), and the v2.10.0 foundation (orchestrator + design-system). Compatible with Claude Code, Codex CLI, Gemini CLI, Cursor, OpenClaw, Hermes Agent, Mistral Vibe, and 5 more coding agents.",
"version": "2.11.1"
},
"plugins": [
{
@ -61,7 +61,7 @@
{
"name": "c-level-agents",
"source": "./c-level-advisor/c-level-agents",
"description": "Founder-mode executive team plugin: 13 cs-* C-suite agents (CFO, CMO, CRO, CPO, COO, CHRO, CISO, Chief of Staff, General Counsel, Chief Data Officer, Chief AI Officer, Chief Customer Officer, VP of Engineering) with distinct cognitive voices, plus 21 /cs:* slash commands \u2014 forcing-question office hours (CFO/CMO/CPO/CRO/CTO/CISO/GC/CDO/CAIO/CCO/VPE reviews), strategic sprint pipeline (brief \u2192 boardroom \u2192 decide \u2192 execute \u2192 post-mortem), and meta routing (/cs:founder-mode auto-router, /cs:onboard, /cs:cross-eval multi-model consensus, /cs:freeze cooldown lock). Wraps the 33 c-level skills with cognitive gearing, persona voice, and artifact-driven handoffs. The business-domain answer to YC Garry Tan's gstack.",
"description": "Founder-mode executive team plugin: 13 cs-* C-suite agents (CFO, CMO, CRO, CPO, COO, CHRO, CISO, Chief of Staff, General Counsel, Chief Data Officer, Chief AI Officer, Chief Customer Officer, VP of Engineering) with distinct cognitive voices, plus 21 /cs:* slash commands forcing-question office hours (CFO/CMO/CPO/CRO/CTO/CISO/GC/CDO/CAIO/CCO/VPE reviews), strategic sprint pipeline (brief → boardroom → decide → execute → post-mortem), and meta routing (/cs:founder-mode auto-router, /cs:onboard, /cs:cross-eval multi-model consensus, /cs:freeze cooldown lock). Wraps the 33 c-level skills with cognitive gearing, persona voice, and artifact-driven handoffs. The business-domain answer to YC Garry Tan's gstack.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -111,7 +111,7 @@
{
"name": "general-counsel-advisor",
"source": "./c-level-advisor/general-counsel-advisor",
"description": "General Counsel advisory for startups: contract risk scanner (12 founder-killer patterns: auto-renew traps, uncapped indemnity, vague IP, MFN pricing, missing DPA, one-sided venue, broad non-solicit, perpetual license-back, etc.) and term sheet analyzer (0-100 founder-friendliness across 12 dimensions). 3 in-depth references: contracts playbook (7 startup contract types), IP + regulatory landscape mapping (HIPAA, GDPR, FDA, fintech, EU AI Act, SOC 2 \u2192 ISO sequencing), term sheet decoder (full glossary + founder-friendly defaults). Standalone-installable; also bundled in c-level-skills. Stdlib-only. NOT a substitute for licensed counsel.",
"description": "General Counsel advisory for startups: contract risk scanner (12 founder-killer patterns: auto-renew traps, uncapped indemnity, vague IP, MFN pricing, missing DPA, one-sided venue, broad non-solicit, perpetual license-back, etc.) and term sheet analyzer (0-100 founder-friendliness across 12 dimensions). 3 in-depth references: contracts playbook (7 startup contract types), IP + regulatory landscape mapping (HIPAA, GDPR, FDA, fintech, EU AI Act, SOC 2 ISO sequencing), term sheet decoder (full glossary + founder-friendly defaults). Standalone-installable; also bundled in c-level-skills. Stdlib-only. NOT a substitute for licensed counsel.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -133,7 +133,7 @@
{
"name": "arquiteto-de-empresa",
"source": "./c-level-advisor/arquiteto-de-empresa",
"description": "Arquiteto de Empresa (PT-BR): constr\u00f3i um neg\u00f3cio do zero como um bundle OKF (Open Knowledge Format) \u2014 \u00e1rvore de arquivos .md version\u00e1veis com frontmatter type, links formando grafo, e index.md/log.md reservados, leg\u00edvel por humanos e por agentes. Conduz o fundador por uma entrevista de 12 fases (funda\u00e7\u00e3o, estrat\u00e9gia, mercado, financeiro, comercial, marketing, produto, opera\u00e7\u00f5es, tech, pessoas, jur\u00eddico, governan\u00e7a), uma fase por vez, e gera os conceitos markdown conformantes. 3 ferramentas stdlib: scaffold_bundle (andaime), okf_linter (valida type/reservados/links), index_generator (regenera os index.md). Standalone-installable; tamb\u00e9m empacotado em c-level-skills. Em portugu\u00eas do Brasil.",
"description": "Arquiteto de Empresa (PT-BR): constrói um negócio do zero como um bundle OKF (Open Knowledge Format) — árvore de arquivos .md versionáveis com frontmatter type, links formando grafo, e index.md/log.md reservados, legível por humanos e por agentes. Conduz o fundador por uma entrevista de 12 fases (fundação, estratégia, mercado, financeiro, comercial, marketing, produto, operações, tech, pessoas, jurídico, governança), uma fase por vez, e gera os conceitos markdown conformantes. 3 ferramentas stdlib: scaffold_bundle (andaime), okf_linter (valida type/reservados/links), index_generator (regenera os index.md). Standalone-installable; também empacotado em c-level-skills. Em português do Brasil.",
"version": "2.10.3",
"author": {
"name": "leoal"
@ -155,7 +155,7 @@
{
"name": "chief-data-officer-advisor",
"source": "./c-level-advisor/chief-data-officer-advisor",
"description": "Chief Data Officer advisory for startups: AI training data audit (origin \u00d7 class \u00d7 use-case matrix with GDPR Art. 6 + EU AI Act citations), data product strategy picker (warehouse vs lakehouse vs mesh + 6-layer build-vs-buy + 12-month sequencing), data asset valuator (strategic value 0-10 + M&A multiplier with carve-out penalties + 3 ranked productization paths). 4 references answering one decision each: training rights, data product strategy, customer-data-as-asset, data team org evolution. Standalone-installable; also bundled in c-level-skills. Strategic only \u2014 does not duplicate engineering data skills.",
"description": "Chief Data Officer advisory for startups: AI training data audit (origin × class × use-case matrix with GDPR Art. 6 + EU AI Act citations), data product strategy picker (warehouse vs lakehouse vs mesh + 6-layer build-vs-buy + 12-month sequencing), data asset valuator (strategic value 0-10 + M&A multiplier with carve-out penalties + 3 ranked productization paths). 4 references answering one decision each: training rights, data product strategy, customer-data-as-asset, data team org evolution. Standalone-installable; also bundled in c-level-skills. Strategic only does not duplicate engineering data skills.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -177,7 +177,7 @@
{
"name": "vpe-advisor",
"source": "./c-level-advisor/vpe-advisor",
"description": "VP of Engineering advisory: delivery throughput analyzer (DORA 4 metrics + cycle-time bottleneck identification with typical fixes per stage), engineering hiring funnel calculator (7-stage conversion + pipeline gap + weakest-stage fixes from sourcing to offer-accept), engineering team structure designer (squad/tribe model + manager-trigger + director-trigger + span-of-control). 4 in-depth references citing DORA / Spotify / Conway / Google SRE / Larson / Fournier. Standalone-installable; also bundled in c-level-skills. NOT a CTO skill \u2014 VPE owns how the team ships; CTO owns what to build.",
"description": "VP of Engineering advisory: delivery throughput analyzer (DORA 4 metrics + cycle-time bottleneck identification with typical fixes per stage), engineering hiring funnel calculator (7-stage conversion + pipeline gap + weakest-stage fixes from sourcing to offer-accept), engineering team structure designer (squad/tribe model + manager-trigger + director-trigger + span-of-control). 4 in-depth references citing DORA / Spotify / Conway / Google SRE / Larson / Fournier. Standalone-installable; also bundled in c-level-skills. NOT a CTO skill VPE owns how the team ships; CTO owns what to build.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -201,7 +201,7 @@
{
"name": "chief-customer-officer-advisor",
"source": "./c-level-advisor/chief-customer-officer-advisor",
"description": "Chief Customer Officer advisory: retention decomposition analyzer (honest GRR vs NRR; 7-category churn taxonomy with preventable% scoring), customer segmentation designer (4-tier framework, ICP fit scoring across 7 weighted signals, kill list + upgrade candidates), CS coverage calculator (pooled vs named CSM ratio math + 12-month hiring plan with quarterly sequencing). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only \u2014 does not duplicate business-growth tactical CS skills.",
"description": "Chief Customer Officer advisory: retention decomposition analyzer (honest GRR vs NRR; 7-category churn taxonomy with preventable% scoring), customer segmentation designer (4-tier framework, ICP fit scoring across 7 weighted signals, kill list + upgrade candidates), CS coverage calculator (pooled vs named CSM ratio math + 12-month hiring plan with quarterly sequencing). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only does not duplicate business-growth tactical CS skills.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -223,7 +223,7 @@
{
"name": "chief-ai-officer-advisor",
"source": "./c-level-advisor/chief-ai-officer-advisor",
"description": "Chief AI Officer advisory for startups: model build-vs-buy calculator (API vs fine-tune vs build with 3-year TCO across 6 paths + breakeven that balances economics with practical feasibility), AI risk classifier (EU AI Act tier with 7 Article citations + US state patchwork: NYC LL 144, CO AI Act, IL HB 53, CA SB 1001, IL BIPA + industry overlays for FDA AI/ML, CFPB Circular 2023-03, NYDFS Reg 23, NAIC, ECOA, Fed SR 11-7), AI cost economics (API vs self-hosted breakeven with 2026 pricing across A100/H100, utilization reality, hidden costs). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only \u2014 does not duplicate engineering AI/ML skills.",
"description": "Chief AI Officer advisory for startups: model build-vs-buy calculator (API vs fine-tune vs build with 3-year TCO across 6 paths + breakeven that balances economics with practical feasibility), AI risk classifier (EU AI Act tier with 7 Article citations + US state patchwork: NYC LL 144, CO AI Act, IL HB 53, CA SB 1001, IL BIPA + industry overlays for FDA AI/ML, CFPB Circular 2023-03, NYDFS Reg 23, NAIC, ECOA, Fed SR 11-7), AI cost economics (API vs self-hosted breakeven with 2026 pricing across A100/H100, utilization reality, hidden costs). 4 in-depth references each citing 5+ authoritative sources. Standalone-installable; also bundled in c-level-skills. Strategic only does not duplicate engineering AI/ML skills.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -269,7 +269,7 @@
{
"name": "engineering-skills",
"source": "./engineering-team",
"description": "32 engineering skills: architecture, frontend, backend, fullstack, QA, DevOps, security, AI/ML, data engineering, Playwright (9 sub-skills), self-improving agent, Stripe integration, TDD guide, tech stack evaluator, Google Workspace CLI, a11y audit (WCAG 2.2), Azure cloud architect, GCP cloud architect, security pen testing, Snowflake development, adversarial-reviewer, ai-security, cloud-security, incident-response, red-team, threat-detection. v2.8.1 audits senior-fullstack / senior-frontend / senior-backend against karpathy-coder + Matt Pocock \u2014 each ships a 7-question forcing-question library, 4 customization profiles (JSON), deterministic decision engine, composition map into POWERFUL specialists, plus cs-fullstack-engineer / cs-frontend-engineer / cs-backend-engineer orchestrator agents (context: fork) + /cs:fullstack-review, /cs:frontend-review, /cs:backend-review, /cs:engineer-grill slash commands.",
"description": "32 engineering skills: architecture, frontend, backend, fullstack, QA, DevOps, security, AI/ML, data engineering, Playwright (9 sub-skills), self-improving agent, Stripe integration, TDD guide, tech stack evaluator, Google Workspace CLI, a11y audit (WCAG 2.2), Azure cloud architect, GCP cloud architect, security pen testing, Snowflake development, adversarial-reviewer, ai-security, cloud-security, incident-response, red-team, threat-detection. v2.8.1 audits senior-fullstack / senior-frontend / senior-backend against karpathy-coder + Matt Pocock each ships a 7-question forcing-question library, 4 customization profiles (JSON), deterministic decision engine, composition map into POWERFUL specialists, plus cs-fullstack-engineer / cs-frontend-engineer / cs-backend-engineer orchestrator agents (context: fork) + /cs:fullstack-review, /cs:frontend-review, /cs:backend-review, /cs:engineer-grill slash commands.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -316,8 +316,8 @@
{
"name": "product-skills",
"source": "./product-team",
"description": "13 product skills with 17 Python tools: product manager toolkit (RICE, PRDs), agile product owner, product strategist, UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, product analytics, experiment designer, product discovery, roadmap communicator, code-to-prd, research summarizer, apple-hig-expert.",
"version": "2.9.0",
"description": "13 bundled product skills with 22 Python tools: product-skills fork-orchestrator with continuous-discovery loop (deterministic 16-lane router, Torres cadence tracker, OST linter, /cs:product + /cs:grill-product + /cs:product-loop), product manager toolkit (RICE, PRDs), product strategist, UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, product analytics, experiment designer, product discovery, roadmap communicator, spec-to-repo. Companion standalone plugins: agile-product-owner, code-to-prd, apple-hig-expert, research-summarizer.",
"version": "2.11.1",
"author": {
"name": "Alireza Rezvani"
},
@ -343,8 +343,8 @@
{
"name": "pm-skills",
"source": "./project-management",
"description": "9 project management skills with 12 Python tools: senior PM, scrum master, Jira expert, Confluence expert, Atlassian admin, template creator.",
"version": "2.9.0",
"description": "9 project management skills with 15 Python tools: pm-skills fork-orchestrator with agentic delivery loop (deterministic 8-lane router, Jira MCP snapshot bridge to Kanban flow metrics + Monte Carlo forecasts, delegation-governance gate, /cs:pm + /cs:grill-pm + /cs:pm-loop), senior PM, scrum master, Jira expert, Confluence expert, Atlassian admin, template creator, meeting analyzer, team communications. Bundled Atlassian Remote MCP.",
"version": "2.11.1",
"author": {
"name": "Alireza Rezvani"
},
@ -437,7 +437,7 @@
{
"name": "autoresearch-agent",
"source": "./engineering/autoresearch-agent",
"description": "Autonomous experiment loop \u2014 optimize any file by a measurable metric. 5 slash commands (/ar:setup, /ar:run, /ar:loop, /ar:status, /ar:resume), 8 built-in evaluators, configurable loop intervals (10min to monthly).",
"description": "Autonomous experiment loop optimize any file by a measurable metric. 5 slash commands (/ar:setup, /ar:run, /ar:loop, /ar:status, /ar:resume), 8 built-in evaluators, configurable loop intervals (10min to monthly).",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -502,7 +502,7 @@
{
"name": "agenthub",
"source": "./engineering/agenthub",
"description": "Multi-agent collaboration \u2014 spawn N parallel subagents that compete on code optimization, content drafts, research approaches, or any task that benefits from diverse solutions. 7 slash commands (/hub:init, /hub:spawn, /hub:status, /hub:eval, /hub:merge, /hub:board, /hub:run), agent templates, DAG-based orchestration, LLM judge mode, message board coordination.",
"description": "Multi-agent collaboration spawn N parallel subagents that compete on code optimization, content drafts, research approaches, or any task that benefits from diverse solutions. 7 slash commands (/hub:init, /hub:spawn, /hub:status, /hub:eval, /hub:merge, /hub:board, /hub:run), agent templates, DAG-based orchestration, LLM judge mode, message board coordination.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -561,7 +561,7 @@
{
"name": "docker-development",
"source": "./engineering/docker-development",
"description": "Docker and container development \u2014 Dockerfile optimization, docker-compose orchestration, multi-stage builds, security hardening, and CI/CD container pipelines.",
"description": "Docker and container development Dockerfile optimization, docker-compose orchestration, multi-stage builds, security hardening, and CI/CD container pipelines.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -578,7 +578,7 @@
{
"name": "helm-chart-builder",
"source": "./engineering/helm-chart-builder",
"description": "Helm chart development \u2014 chart scaffolding, values design, template patterns, dependency management, and Kubernetes deployment strategies.",
"description": "Helm chart development chart scaffolding, values design, template patterns, dependency management, and Kubernetes deployment strategies.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -595,7 +595,7 @@
{
"name": "terraform-patterns",
"source": "./engineering/terraform-patterns",
"description": "Terraform infrastructure-as-code \u2014 module design patterns, state management, provider configuration, CI/CD integration, and multi-environment strategies.",
"description": "Terraform infrastructure-as-code module design patterns, state management, provider configuration, CI/CD integration, and multi-environment strategies.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -612,7 +612,7 @@
{
"name": "research-summarizer",
"source": "./product-team/research-summarizer",
"description": "Structured research summarization \u2014 summarize academic papers, market research, user interviews, and competitive analysis into actionable insights.",
"description": "Structured research summarization summarize academic papers, market research, user interviews, and competitive analysis into actionable insights.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -629,7 +629,7 @@
{
"name": "code-tour",
"source": "./engineering/code-tour",
"description": "Create CodeTour .tour files \u2014 persona-targeted, step-by-step walkthroughs that link to real files and line numbers. 10 developer personas, all CodeTour step types, SMIG description formula.",
"description": "Create CodeTour .tour files persona-targeted, step-by-step walkthroughs that link to real files and line numbers. 10 developer personas, all CodeTour step types, SMIG description formula.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -782,7 +782,7 @@
{
"name": "kubernetes-operator",
"source": "./engineering/kubernetes-operator",
"description": "End-to-end Kubernetes Operator discipline: CRD design, reconcile-loop patterns, and OperatorHub Capability Levels. Ships CRD validator, reconcile-loop linter, and capability auditor (3 stdlib Python tools), 4 references on the operator pattern + CRD design + reconcile patterns + framework comparison (controller-runtime/kubebuilder/operator-sdk/metacontroller/KOPF), CRD + Go controller skeletons, and /operator-audit slash command. NOT a generic k8s skill \u2014 specifically the Operator pattern.",
"description": "End-to-end Kubernetes Operator discipline: CRD design, reconcile-loop patterns, and OperatorHub Capability Levels. Ships CRD validator, reconcile-loop linter, and capability auditor (3 stdlib Python tools), 4 references on the operator pattern + CRD design + reconcile patterns + framework comparison (controller-runtime/kubebuilder/operator-sdk/metacontroller/KOPF), CRD + Go controller skeletons, and /operator-audit slash command. NOT a generic k8s skill specifically the Operator pattern.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -847,7 +847,7 @@
{
"name": "write-a-skill",
"source": "./engineering/write-a-skill",
"description": "Skill-author skill: create new agent skills with proper structure, progressive disclosure, and bundled resources. Derived from Matt Pocock's MIT-licensed write-a-skill with: (1) 3 stdlib Python validation tools (description validator, structure validator, review-checklist runner \u2014 all enforcing Matt's 6-item checklist), (2) 4 references citing 7-8 authoritative sources each (progressive disclosure principles, description design patterns, quality gates, companion tooling), (3) cs-skill-author persona agent + /cs:write-a-skill slash command. Matt's voice and 3-phase workflow (Gather \u2192 Draft \u2192 Review) preserved verbatim per MIT.",
"description": "Skill-author skill: create new agent skills with proper structure, progressive disclosure, and bundled resources. Derived from Matt Pocock's MIT-licensed write-a-skill with: (1) 3 stdlib Python validation tools (description validator, structure validator, review-checklist runner all enforcing Matt's 6-item checklist), (2) 4 references citing 7-8 authoritative sources each (progressive disclosure principles, description design patterns, quality gates, companion tooling), (3) cs-skill-author persona agent + /cs:write-a-skill slash command. Matt's voice and 3-phase workflow (Gather → Draft → Review) preserved verbatim per MIT.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -921,6 +921,27 @@
],
"category": "development"
},
{
"name": "agent-harness",
"source": "./engineering/agent-harness",
"description": "Turn any domain folder of skills into a bounded agentic loop: a manifest builder inventories a domain's skills/tools/checks, a goal compiler turns a goal into a verifiable task plan (refusing vague goals with forcing questions), and a JSON-backed loop controller drives execute->verify->close with retry caps, controller-run verification (no verification theater), human escalation on exhausted budgets, and a close gate that refuses while any task is unverified or unwaived. Ships 3 stdlib Python tools, 18 committed per-domain harness manifests + JSON schema, 3 references citing the 2024-2026 agent-harness canon (Anthropic long-running harnesses, verifier's law, SWE-agent, Ralph loop, Cognition), harness-runner agent + /cs:harness command. Use when an agent or subagent should pick up a goal, define its tasks, complete and verify them, and close the loop.",
"version": "1.0.0",
"author": {
"name": "Alireza Rezvani"
},
"keywords": [
"agent-harness",
"agentic-loop",
"verification-gate",
"goal-compiler",
"loop-controller",
"stop-conditions",
"escalation",
"multi-agent",
"engineering"
],
"category": "development"
},
{
"name": "grill-me",
"source": "./engineering/grill-me",
@ -943,7 +964,7 @@
{
"name": "handoff-engineering",
"source": "./engineering/handoff",
"description": "Conversation-handoff document generator. Compacts the current session into a markdown handoff for a fresh agent \u2014 references existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL instead of duplicating them. Derived from Matt Pocock's MIT-licensed handoff with: (1) 3 stdlib Python tools (template generator tailored to 5 next-session emphases, artifact deduplicator across 5 categories of duplication, skill recommender matching content to 14 skills in this repo), (2) 4 references citing 7-8 sources (handoff structure, deduplication discipline, next-session skill matching, companion tooling), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline + mktemp convention preserved verbatim per MIT.",
"description": "Conversation-handoff document generator. Compacts the current session into a markdown handoff for a fresh agent references existing artifacts (PRDs, plans, ADRs, issues, commits) by path/URL instead of duplicating them. Derived from Matt Pocock's MIT-licensed handoff with: (1) 3 stdlib Python tools (template generator tailored to 5 next-session emphases, artifact deduplicator across 5 categories of duplication, skill recommender matching content to 14 skills in this repo), (2) 4 references citing 7-8 sources (handoff structure, deduplication discipline, next-session skill matching, companion tooling), (3) cs-handoff-author persona agent + /cs:handoff slash command. Matt's no-duplication discipline + mktemp convention preserved verbatim per MIT.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -981,7 +1002,7 @@
{
"name": "capture-skill",
"source": "./productivity/capture",
"description": "Brain-dump-to-action workspace skill. Routes vague captures into discoverable actions via classify\u2192cluster\u2192connect\u2192clarify intake. Path-B from megaprompt 05.",
"description": "Brain-dump-to-action workspace skill. Routes vague captures into discoverable actions via classify→cluster→connect→clarify intake. Path-B from megaprompt 05.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -1075,7 +1096,7 @@
{
"name": "roast",
"source": "./productivity/roast",
"description": "Pressure-test a business idea before you build it. Convenes a 5-angle adversarial panel \u2014 The Critic (what kills this?), The Champion (the 10x upside?), The Analyst (does the logic hold?), The Investigator (what does the market say?), The Customer (would I actually pay?) \u2014 fired in parallel as independent reviewers, then a Judge synthesizes one GO / RESHAPE / KILL verdict with explicit confidence and the cheapest 48-hour test to de-risk it. Never averages the scores: a weighted synthesizer with demand/fatal-flaw/logic veto gates produces the call, backed by deterministic stdlib tools.",
"description": "Pressure-test a business idea before you build it. Convenes a 5-angle adversarial panel The Critic (what kills this?), The Champion (the 10x upside?), The Analyst (does the logic hold?), The Investigator (what does the market say?), The Customer (would I actually pay?) fired in parallel as independent reviewers, then a Judge synthesizes one GO / RESHAPE / KILL verdict with explicit confidence and the cheapest 48-hour test to de-risk it. Never averages the scores: a weighted synthesizer with demand/fatal-flaw/logic veto gates produces the call, backed by deterministic stdlib tools.",
"version": "2.10.3",
"author": {
"name": "Alireza Rezvani"
@ -1134,7 +1155,7 @@
{
"name": "deep-research",
"source": "./research/deep-research",
"description": "Disciplined multi-source meta-research for high-stakes questions \u2014 the heavyweight alternative to the fast research router. 9-phase pipeline (reframe into falsifiable hypotheses, plan, capability discovery, parallel sub-agent fan-out, score & triangulate, synthesize + adversarial pass, verify, refresh targets). Triangulates every thesis against >=3 independent differently-typed sources; per-source files with verbatim quotes; never fabricates a citation. Auditable, reusable folder + delta-update refresh protocol. Contributed via PR #851.",
"description": "Disciplined multi-source meta-research for high-stakes questions the heavyweight alternative to the fast research router. 9-phase pipeline (reframe into falsifiable hypotheses, plan, capability discovery, parallel sub-agent fan-out, score & triangulate, synthesize + adversarial pass, verify, refresh targets). Triangulates every thesis against >=3 independent differently-typed sources; per-source files with verbatim quotes; never fabricates a citation. Auditable, reusable folder + delta-update refresh protocol. Contributed via PR #851.",
"version": "2.10.3",
"author": {
"name": "Alireza Rezvani"
@ -1294,7 +1315,7 @@
{
"name": "aeo",
"source": "./marketing-skill/skills/aeo",
"description": "Answer Engine Optimization (AEO) skill \u2014 optimize content to be cited by AI language models (ChatGPT, Perplexity, Claude, Gemini, Mistral) as authoritative sources. Distinct from SEO (which optimizes for search rankings), AEO optimizes for citation in LLM-generated responses. 3 stdlib Python tools (aeo_audit, aeo_optimizer, citation_tracker), 3 references citing 8 sources each, industry-aware thresholds for 8 industries (saas/healthcare/finance/legal/ecommerce/b2b/media/education). Ported from alirezarezvani/aeo-box.",
"description": "Answer Engine Optimization (AEO) skill optimize content to be cited by AI language models (ChatGPT, Perplexity, Claude, Gemini, Mistral) as authoritative sources. Distinct from SEO (which optimizes for search rankings), AEO optimizes for citation in LLM-generated responses. 3 stdlib Python tools (aeo_audit, aeo_optimizer, citation_tracker), 3 references citing 8 sources each, industry-aware thresholds for 8 industries (saas/healthcare/finance/legal/ecommerce/b2b/media/education). Ported from alirezarezvani/aeo-box.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -1316,7 +1337,7 @@
{
"name": "security-guidance",
"source": "./engineering/security-guidance",
"description": "PreToolUse security reminder hook for Claude Code. Catches 12 common security anti-patterns in Edit/Write/MultiEdit operations BEFORE they happen \u2014 command injection (exec, os.system, subprocess shell=True), XSS (innerHTML, dangerouslySetInnerHTML, document.write), SQL injection (f-string queries, .format), unsafe deserialization (pickle, yaml.unsafe_load), code injection (eval, new Function), and GitHub Actions workflow injection. Session-state caching prevents duplicate warnings; 30-day auto-cleanup. Disable per-session with ENABLE_SECURITY_REMINDER=0. Ported from David Dworken at Anthropic.",
"description": "PreToolUse security reminder hook for Claude Code. Catches 12 common security anti-patterns in Edit/Write/MultiEdit operations BEFORE they happen command injection (exec, os.system, subprocess shell=True), XSS (innerHTML, dangerouslySetInnerHTML, document.write), SQL injection (f-string queries, .format), unsafe deserialization (pickle, yaml.unsafe_load), code injection (eval, new Function), and GitHub Actions workflow injection. Session-state caching prevents duplicate warnings; 30-day auto-cleanup. Disable per-session with ENABLE_SECURITY_REMINDER=0. Ported from David Dworken at Anthropic.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -1437,7 +1458,7 @@
{
"name": "research-ops-skills",
"source": "./research-ops",
"description": "Enterprise / cross-functional Research Operations domain \u2014 the managed counterpart to the academic research/ domain. v2.9.0 ships 5 skills: orchestrator (context: fork) + clinical-research (study design: protocol synopsis + endpoint selection + sample-size/power for means/proportions/survival + phase-gate feasibility) + research-finance (R&D program budgeting with F&A split + burn/runway + capitalize-vs-expense routing + portfolio ROI) + market-research (TAM/SAM/SOM computed both top-down and bottoms-up + survey sampling with FPC and per-segment minima + Kotler segmentation scoring) + product-research (goal-matched study design + method-based saturation with confidence + insight synthesis that flags single-source anecdotes). Hard rules: clinical outputs are estimates with a named clinical owner (never fact), finance outputs surface assumptions and route capex-vs-opex to a named finance owner (never auto-decide), market sizes show method + assumptions (never a single number), product insights require recurrence across independent participants. Each sub-skill ships per-skill onboarding questions (onboard.py), a customization config consumed by every tool, and an isolated opt-in autoresearch evaluator (ar_evaluator.py) bridging to engineering/autoresearch-agent. 24 stdlib Python tools (12 analysis + 12 onboarding/customization/autoresearch), 12 reference docs. Distinct from ra-qm-team (regulatory/QM submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), marketing-skill (campaign analytics).",
"description": "Enterprise / cross-functional Research Operations domain the managed counterpart to the academic research/ domain. v2.9.0 ships 5 skills: orchestrator (context: fork) + clinical-research (study design: protocol synopsis + endpoint selection + sample-size/power for means/proportions/survival + phase-gate feasibility) + research-finance (R&D program budgeting with F&A split + burn/runway + capitalize-vs-expense routing + portfolio ROI) + market-research (TAM/SAM/SOM computed both top-down and bottoms-up + survey sampling with FPC and per-segment minima + Kotler segmentation scoring) + product-research (goal-matched study design + method-based saturation with confidence + insight synthesis that flags single-source anecdotes). Hard rules: clinical outputs are estimates with a named clinical owner (never fact), finance outputs surface assumptions and route capex-vs-opex to a named finance owner (never auto-decide), market sizes show method + assumptions (never a single number), product insights require recurrence across independent participants. Each sub-skill ships per-skill onboarding questions (onboard.py), a customization config consumed by every tool, and an isolated opt-in autoresearch evaluator (ar_evaluator.py) bridging to engineering/autoresearch-agent. 24 stdlib Python tools (12 analysis + 12 onboarding/customization/autoresearch), 12 reference docs. Distinct from ra-qm-team (regulatory/QM submission), finance (corporate close/valuation), research/grants (funding discovery), product-team (persona/journey/live experiments), marketing-skill (campaign analytics).",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -1484,7 +1505,7 @@
{
"name": "markdown-html-skills",
"source": "./markdown-html",
"description": "Convert long markdown files into world-class single-file interactive HTML \u2014 DOMAIN COMPLETE at v2.10.3 (5 skills). v2.10.3 adds md-slides \u2014 the slide-deck converter (arrow-key / Space / PgDn / Home / End / P keyboard navigation + presenter mode with split-view clock + speaker notes + next-slide preview + URL-hash deep linking like #3 + @media print page-per-slide for browser-native PDF export; reuses md-document's markdown parser; vanilla JS only; Prism.js opt-in via --syntax for code-heavy decks). Joins md-review (v2.10.2 code-review converter: 2-col diff + severity-tagged margin annotations + WCAG-1.4.1 badges + mandatory named reviewer footer), md-document (v2.10.1 long-form converter: sticky TOC + scrollspy + search + code-copy + Prism autoloader), markdown-html-orchestrator (v2.10.0 context: fork; deterministic doc-type classifier; refuses < 100 lines per Shihipar; refuses without onboarding), and design-system (v2.10.0 10-question onboarding wizard; WCAG-AA 12-token palette; project > global > defaults precedence; MARKDOWN_HTML_NO_CONFIG=1 bypass). 15 stdlib-only Python tools, 15 references citing 5-7 sources each, 4 template/schema assets. Inspired by Thariq Shihipar's Claude Code HTML output essay (Medium, 2026).",
"description": "Convert long markdown files into world-class single-file interactive HTML — DOMAIN COMPLETE at v2.10.3 (5 skills). v2.10.3 adds md-slides — the slide-deck converter (arrow-key / Space / PgDn / Home / End / P keyboard navigation + presenter mode with split-view clock + speaker notes + next-slide preview + URL-hash deep linking like #3 + @media print page-per-slide for browser-native PDF export; reuses md-document's markdown parser; vanilla JS only; Prism.js opt-in via --syntax for code-heavy decks). Joins md-review (v2.10.2 code-review converter: 2-col diff + severity-tagged margin annotations + WCAG-1.4.1 badges + mandatory named reviewer footer), md-document (v2.10.1 long-form converter: sticky TOC + scrollspy + search + code-copy + Prism autoloader), markdown-html-orchestrator (v2.10.0 context: fork; deterministic doc-type classifier; refuses < 100 lines per Shihipar; refuses without onboarding), and design-system (v2.10.0 10-question onboarding wizard; WCAG-AA 12-token palette; project > global > defaults precedence; MARKDOWN_HTML_NO_CONFIG=1 bypass). 15 stdlib-only Python tools, 15 references citing 5-7 sources each, 4 template/schema assets. Inspired by Thariq Shihipar's Claude Code HTML output essay (Medium, 2026).",
"version": "2.10.3",
"author": {
"name": "Alireza Rezvani"
@ -1511,7 +1532,7 @@
{
"name": "youtube-full",
"source": "./marketing-skill/skills/youtube-full",
"description": "YouTube transcripts, video search, channel browsing, playlist extraction, and upload monitoring via TranscriptAPI. BYOK \u2014 100 free credits. OSS fallbacks: youtube-transcript-api / yt-dlp.",
"description": "YouTube transcripts, video search, channel browsing, playlist extraction, and upload monitoring via TranscriptAPI. BYOK 100 free credits. OSS fallbacks: youtube-transcript-api / yt-dlp.",
"version": "2.9.0",
"author": {
"name": "therohitdas"
@ -1532,7 +1553,7 @@
{
"name": "compliance-os",
"source": "./compliance-os",
"description": "Compliance OS \u2014 meta-orchestrator for multi-framework compliance programs spanning 9 frameworks (ISO 27001, ISO 13485, ISO 42001, ISO 14971, EU AI Act, MDR 745, GDPR, SOC 2, FDA QSR). Framework selector, cross-framework control mapper, audit simulator, and consolidated evidence-pool generator (stdlib Python), plus 3 cs-* compliance agents and 3 /cs:* readiness commands.",
"description": "Compliance OS meta-orchestrator for multi-framework compliance programs spanning 9 frameworks (ISO 27001, ISO 13485, ISO 42001, ISO 14971, EU AI Act, MDR 745, GDPR, SOC 2, FDA QSR). Framework selector, cross-framework control mapper, audit simulator, and consolidated evidence-pool generator (stdlib Python), plus 3 cs-* compliance agents and 3 /cs:* readiness commands.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -1572,7 +1593,7 @@
{
"name": "behuman",
"source": "./engineering/behuman",
"description": "Self-Mirror consciousness loop for human-like AI responses. Adds inner dialogue (Self \u2192 Mirror \u2192 Conscious Response) to make AI output feel authentic, not robotic. Zero dependencies \u2014 pure prompt technique.",
"description": "Self-Mirror consciousness loop for human-like AI responses. Adds inner dialogue (Self → Mirror → Conscious Response) to make AI output feel authentic, not robotic. Zero dependencies — pure prompt technique.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"
@ -1606,7 +1627,7 @@
{
"name": "grill-with-docs",
"source": "./engineering/grill-with-docs",
"description": "Docs-anchored grilling session \u2014 interrogates a plan against the project's existing language (CONTEXT.md) and recorded decisions (docs/adr/), updating those files inline as terminology and decisions crystallise. Derived from Matt Pocock's MIT-licensed grill-with-docs with stdlib validators (CONTEXT.md linter, ADR scanner, glossary-code consistency), reference docs, cs-grill-with-docs agent, and /cs:grill-with-docs command.",
"description": "Docs-anchored grilling session interrogates a plan against the project's existing language (CONTEXT.md) and recorded decisions (docs/adr/), updating those files inline as terminology and decisions crystallise. Derived from Matt Pocock's MIT-licensed grill-with-docs with stdlib validators (CONTEXT.md linter, ADR scanner, glossary-code consistency), reference docs, cs-grill-with-docs agent, and /cs:grill-with-docs command.",
"version": "2.9.0",
"author": {
"name": "Alireza Rezvani"

View file

@ -37,7 +37,7 @@ For each file with YAML frontmatter:
Run SEO checker on built HTML pages:
```bash
python3 marketing-skill/seo-audit/scripts/seo_checker.py --file site/{path}/index.html
python3 marketing-skill/skills/seo-audit/scripts/seo_checker.py --file site/{path}/index.html
```
## Phase 3: Content Quality
@ -46,13 +46,13 @@ python3 marketing-skill/seo-audit/scripts/seo_checker.py --file site/{path}/inde
**Readability:** Run content scorer:
```bash
python3 marketing-skill/content-production/scripts/content_scorer.py {file}
python3 marketing-skill/skills/content-production/scripts/content_scorer.py {file}
```
Target: readability ≥ 70, structure ≥ 60.
**AI detection** (on non-generated files only):
```bash
python3 marketing-skill/content-humanizer/scripts/humanizer_scorer.py {file}
python3 marketing-skill/skills/content-humanizer/scripts/humanizer_scorer.py {file}
```
Flag pages < 50. Fix AI clichés: "delve", "leverage", "it's important to note", "comprehensive".
@ -87,7 +87,7 @@ mkdocs build
Analyze the sitemap:
```bash
python3 marketing-skill/site-architecture/scripts/sitemap_analyzer.py site/sitemap.xml
python3 marketing-skill/skills/site-architecture/scripts/sitemap_analyzer.py site/sitemap.xml
```
Verify all pages appear, no duplicates, no broken URLs.

View file

@ -3,7 +3,7 @@
"name": "claude-code-skills",
"description": "Production-ready skill packages for AI agents - Marketing, Engineering, Product, C-Level, PM, and RA/QM",
"repository": "https://github.com/alirezarezvani/claude-skills",
"total_skills": 352,
"total_skills": 353,
"skills": [
{
"name": "business-growth-skills",
@ -935,6 +935,12 @@
"category": "engineering-advanced",
"description": "Use when the user asks to design a multi-agent system, pick an orchestration pattern (supervisor/swarm/pipeline), generate tool schemas for agents, or evaluate agent execution logs for cost, latency, and failure bottlenecks. Examples: 'design an agent architecture for research automation', 'generate Anthropic tool schemas from these tool descriptions', 'analyze these agent run logs for bottlenecks'. NOT for Claude Code workflow files (use workflow-builder) or single-agent prompt design (use agent-workflow-designer)."
},
{
"name": "agent-harness",
"source": "../../engineering/agent-harness/skills/agent-harness",
"category": "engineering-advanced",
"description": "Turn any domain folder of skills into a bounded agentic loop: compile a goal into a verifiable task plan, execute tasks with the domain's own tools, verify every task with machine-run checks, retry with caps, escalate to a human when budgets exhaust, and refuse to close until everything is verified or explicitly waived. Use when you want an agent or subagent to pick up a goal and drive it to a verified close across one of this repo's 18 domains ('run this goal through the engineering harness', 'set up an agentic loop for marketing work', 'make the finance domain self-verifying'). NOT for authoring Claude Code Workflow-tool .js scripts (workflow-builder), N-agent tournaments on one task (agenthub), single-file metric optimization (autoresearch-agent), or discovering published loop recipes (loop-library)."
},
{
"name": "agent-workflow-designer",
"source": "../../engineering/skills/agent-workflow-designer",
@ -1779,7 +1785,7 @@
"name": "product-skills",
"source": "../../product-team/skills/product-skills",
"category": "product",
"description": "Router/index for the 12 product skills bundled in this plugin (RICE prioritization, OKRs, UX research, design tokens, competitive teardown, analytics, experiments, discovery, roadmaps, spec-to-repo, landing pages, SaaS scaffolding). Use when a product request doesn't obviously match one skill and you need to pick the right one (e.g., 'help me prioritize features', 'plan a product experiment')."
"description": "Use when coordinating product work across the 12 bundled product sub-skills (RICE, OKRs, UX research, design tokens, competitive teardown, analytics, experiments, discovery, roadmaps, spec-to-repo, landing pages, SaaS scaffolding) or the 4 standalone product-team plugins (user stories, Apple HIG, code-to-PRD, research summarizer). Triggers on 'help me prioritize', 'plan a product experiment', 'we ship features nobody uses', 'run the discovery loop', 'is our OST sound'. Forks context to route to one sub-skill via a deterministic signal router and returns a digest; can also drive a continuous-discovery loop (Torres cadence tracker + OST linter as machine gates) or a full goal\u2192plan\u2192execute\u2192verify\u2192close run through the repo-wide agent-harness. Distinct from project-management (how to deliver vs what to build), marketing/landing (from-scratch pages), and engineering/agent-harness (the generic loop engine this orchestrator plugs into)."
},
{
"name": "product-strategist",
@ -1899,7 +1905,7 @@
"name": "pm-skills",
"source": "../../project-management/skills/pm-skills",
"category": "project-management",
"description": "Router/index for the 8 project-management skills bundled in this plugin (senior PM quant toolkit, scrum master, Jira/JQL, Confluence, Atlassian admin, Atlassian templates, meeting analyzer, team communications). Use when a PM request doesn't obviously match one skill and you need to pick the right one (e.g., 'our sprints feel off', 'audit our Jira permissions'). Bundles an Atlassian Remote MCP config (.mcp.json) for live Jira/Confluence access."
"description": "Use when coordinating project-delivery work across the 8 project-management sub-skills \u2014 sprint/velocity analytics, portfolio health, Jira/JQL, Confluence, Atlassian admin, templates, meeting analysis, team comms. Triggers on 'our sprints feel off', 'project health report', 'audit our Jira permissions', 'when will it be done', 'run the delivery loop'. Forks context to route to one sub-skill via a deterministic signal router and returns a digest; can also drive a full goal\u2192plan\u2192execute\u2192verify\u2192close delivery loop through the repo-wide agent-harness with Jira MCP data bridged into the domain's analytics tools. Distinct from product-team (what to build vs how to deliver it), business-operations (internal ops), and engineering/agent-harness (the generic loop engine this orchestrator plugs into)."
},
{
"name": "scrum-master",
@ -2155,7 +2161,7 @@
"description": "Software engineering and technical skills"
},
"engineering-advanced": {
"count": 79,
"count": 80,
"source": "../../engineering",
"description": "Advanced engineering skills - agents, RAG, MCP, CI/CD, databases, observability"
},

1
.codex/skills/agent-harness Symbolic link
View file

@ -0,0 +1 @@
../../engineering/agent-harness/skills/agent-harness

View file

@ -6,7 +6,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
This is a **comprehensive skills library** for Claude AI and Claude Code - reusable, production-ready skill packages that bundle domain expertise, best practices, analysis tools, and strategic frameworks. The repository provides modular skills that teams can download and use directly in their workflows.
**Current Scope:** 354 production-ready skills across 18 domains with 593 Python automation tools, 722 reference guides, 96 agents (cs-* + 7 personas), and 102 slash commands, distributed as 82 marketplace plugins. Headline counters are derived from the tree by `scripts/derive_counters.py` (run with `--check` to verify the docs still match). **v2.9.0 (complete)** added the **research-ops/** top-level domain — enterprise Research Operations (orchestrator + clinical-research + research-finance + market-research + product-research), the managed counterpart to the academic research/ domain, with `context: fork` orchestration and a Matt Pocock "Forcing-question library" in every SKILL.md plus `/cs:grill-research-ops`. **v2.8.0 (complete)** added 2 new top-level domains — **business-operations/** (7 internal-ops skills: orchestrator + process-mapper + vendor-management + capacity-planner + internal-comms + knowledge-ops + procurement-optimizer) and **commercial/** (8 per-deal-economics skills: orchestrator + pricing-strategist + deal-desk + partnerships-architect + channel-economics + commercial-policy + rfp-responder + commercial-forecaster) — with orchestrator skills using `context: fork` for chaining, Matt Pocock docs-anchored "Forcing-question library" in every SKILL.md, plus `/cs:grill-bizops` and `/cs:grill-commercial`. **v2.8.2** adds a productivity-shaped `handoff` skill (sibling to engineering/handoff) inspired by Matt Pocock — first-run setup with configurable save location, redaction linter, SessionStart + SessionEnd hooks, fidelity self-check, `--refresh` flag. **v2.8.1** upgraded the engineering role-skills (senior-fullstack / senior-frontend / senior-backend) with karpathy-coder + Matt Pocock decision engines + per-role forcing questions. v2.7.3 ports `alirezarezvani/aeo-box` — AEO (Answer Engine Optimization) skill into marketing-skill/ + security-guidance PreToolUse hook into engineering/. v2.7.0 added 13 Path-B skills across 3 top-level domains (productivity, marketing, research). v2.6.0 added 4 Matt Pocock-derived productivity skills.
**Current Scope:** 355 production-ready skills across 18 domains with 602 Python automation tools, 731 reference guides, 99 agents (cs-* + 7 personas), and 109 slash commands, distributed as 83 marketplace plugins. Headline counters are derived from the tree by `scripts/derive_counters.py` (run with `--check` to verify the docs still match). **v2.11.1 (current)** upgrades **product-team/** and **project-management/** into agent-harness domains: both prose routers rebuilt as `context: fork` orchestrators with deterministic goal routers (exit-code route/ask/refuse), a Jira MCP snapshot bridge (Kanban-Guide-2025 flow metrics + seeded Monte Carlo forecasts, verified end-to-end into velocity_analyzer), a delegation-governance loop gate (human owner / reviewer / machine-checkable acceptance / close refusal), a Torres continuous-discovery cadence tracker + Opportunity Solution Tree linter, cs-pm-orchestrator + cs-product-orchestrator agents, and /cs:pm|grill-pm|pm-loop + /cs:product|grill-product|product-loop commands — plus the public audit record `audit/pm-product-agentic-2026-07/` (AR-rubric scores for all 26 skills, research-backed improvement fields, executable verification criteria). **v2.9.0 (complete)** added the **research-ops/** top-level domain — enterprise Research Operations (orchestrator + clinical-research + research-finance + market-research + product-research), the managed counterpart to the academic research/ domain, with `context: fork` orchestration and a Matt Pocock "Forcing-question library" in every SKILL.md plus `/cs:grill-research-ops`. **v2.8.0 (complete)** added 2 new top-level domains — **business-operations/** (7 internal-ops skills: orchestrator + process-mapper + vendor-management + capacity-planner + internal-comms + knowledge-ops + procurement-optimizer) and **commercial/** (8 per-deal-economics skills: orchestrator + pricing-strategist + deal-desk + partnerships-architect + channel-economics + commercial-policy + rfp-responder + commercial-forecaster) — with orchestrator skills using `context: fork` for chaining, Matt Pocock docs-anchored "Forcing-question library" in every SKILL.md, plus `/cs:grill-bizops` and `/cs:grill-commercial`. **v2.8.2** adds a productivity-shaped `handoff` skill (sibling to engineering/handoff) inspired by Matt Pocock — first-run setup with configurable save location, redaction linter, SessionStart + SessionEnd hooks, fidelity self-check, `--refresh` flag. **v2.8.1** upgraded the engineering role-skills (senior-fullstack / senior-frontend / senior-backend) with karpathy-coder + Matt Pocock decision engines + per-role forcing questions. v2.7.3 ports `alirezarezvani/aeo-box` — AEO (Answer Engine Optimization) skill into marketing-skill/ + security-guidance PreToolUse hook into engineering/. v2.7.0 added 13 Path-B skills across 3 top-level domains (productivity, marketing, research). v2.6.0 added 4 Matt Pocock-derived productivity skills.
**Key Distinction**: This is NOT a traditional application. It's a library of skill packages meant to be extracted and deployed by users into their own Claude workflows.
@ -61,7 +61,7 @@ claude-code-skills/
├── agents/ # 32 standalone agents (cs-* + 7 personas); 51+ cs-* agents repo-wide
├── commands/ # slash commands (changelog, tdd, saas-health, prd, code-to-prd, plugin-audit, sprint-plan, slo-design, etc.); 87+ repo-wide
├── engineering-team/ # 51 core engineering skills + Playwright Pro + Self-Improving Agent + Security Suite
├── engineering/ # 78 POWERFUL-tier advanced skills (incl. AgentHub, autoresearch-agent, self-eval, llm-wiki, tc-tracker, ship-gate, slo-architect, write-a-skill, caveman, grill-me, handoff)
├── engineering/ # 81 POWERFUL-tier advanced skills (incl. AgentHub, autoresearch-agent, self-eval, llm-wiki, tc-tracker, ship-gate, slo-architect, write-a-skill, caveman, grill-me, handoff, agent-harness)
├── product-team/ # 17 product skills (incl. apple-hig-expert) + Python tools
├── marketing-skill/ # 46 marketing skills (8 pods) + Python tools
├── c-level-advisor/ # 66 C-level advisory skills (full C-suite + founder-mode agents + orchestration)
@ -159,6 +159,32 @@ See [standards/git/git-workflow-standards.md](standards/git/git-workflow-standar
## Current Version
**Version:** v2.11.1 (pm/product agent-harness domains — deep audit + orchestrated loops for product-team & project-management)
**v2.11.1 highlights — both PM/product routers become agent harnesses:**
Extends the v2.11.0 agent-harness layer to the two people-process domains. Public audit record at `audit/pm-product-agentic-2026-07/` (AR-rubric scores for all 26 skills, research-backed improvement fields, executable verification criteria).
- **project-management → delivery loop:** `pm-skills` rebuilt as a `context: fork` orchestrator with 3 stdlib tools — `pm_goal_router.py` (8 lanes, exit-code route/ask/refuse), `jira_snapshot_bridge.py` (saved `searchJiraIssuesUsingJql` output → Kanban-Guide-2025 flow metrics with SLE + aging-WIP alerts + seeded Monte Carlo forecasts that sample zero-throughput weeks, or the scrum-master sprint schema — verified end-to-end into `velocity_analyzer.py`), `delivery_loop_gate.py` (delegation governance G1G6: human owner, reviewer for agent tasks, machine-checkable acceptance, evidence-before-done, close refusal, exhausted-budget-is-escalation). Five reusable PM loops documented with named terminal states. Agent `cs-pm-orchestrator`; commands `/cs:pm`, `/cs:grill-pm`, `/cs:pm-loop`.
- **product-team → discovery loop:** `product-skills` rebuilt as a `context: fork` orchestrator with 3 stdlib tools — `product_goal_router.py` (16 lanes incl. the 4 standalone plugins), `discovery_cadence_tracker.py` (Torres weekly-habit health 0100 with named gaps + `next_loop_action`), `ost_linter.py` (Opportunity Solution Tree rules O1O5; exit 2 blocks an unsound tree from driving a roadmap). Agent `cs-product-orchestrator`; commands `/cs:product`, `/cs:grill-product`, `/cs:product-loop`.
- **6 new references** citing 67 sources each (flow/forecasting canon, agentic delivery governance, PM loop playbook, continuous discovery, product operating model, AI product evals) + pinned fixtures; fixed the two CLI-noncompliant product tools (`user_story_generator.py`, `persona_generator.py` — real argparse `--help`, seeded determinism); regenerated both domain harness manifests (orchestrators now score all five `agentic_signals`; manifest builder now truncates descriptions on word boundaries).
- **Counters:** tools 596 → 602; refs 725 → 731; agents 97 → 99; commands 103 → 109 (derived via `scripts/derive_counters.py --check`).
---
**Version:** v2.11.0 (agent-harness — turn any domain into a bounded, self-verifying agent loop + engineering agentic-readiness audit)
**v2.11.0 highlights — agent-harness skill + AR audit of both engineering folders:**
New `engineering/agent-harness/` skill — the thin unifying layer that lets an agent or subagent pick up a goal for any of the repo's 18 domains, decompose it into verifiable tasks, execute them with the domain's own tools, verify each with machine-run checks, retry within caps, escalate to a human on exhausted budgets, and refuse to close until every task is verified or explicitly waived.
- **3 stdlib tools:** `harness_manifest_builder.py` (scans a domain folder → `manifest.v1` JSON: skills, tools, exact `--help`/`--sample` checks, static agentic signals), `goal_compiler.py` (goal + manifest → `plan.v1` task plan via deterministic keyword scoring; refuses vague goals exit 3 with forcing questions, no-match exit 4 with nearest candidates), `loop_controller.py` (JSON-backed `init/next/record/verify/close/status` state machine — runs verification checks itself via subprocess to prevent verification theater, caps attempts + iterations with escalation, refuses close while any task is unverified; atomic state writes via `os.replace`).
- **18 committed per-domain manifests** under `assets/harnesses/` (the whole repo, machine-readable), a JSON schema, `harness-runner` agent, `/cs:harness <domain> <goal>` command, and 3 references citing the 20242026 harness canon (Anthropic long-running-agents harness, verifier's law, SWE-agent, Ralph loop, Cognition serialize-writers, plus the repo's own tc-tracker / autoresearch locked-evaluator / loop-library stop-state primitives — reuse, not reinvention).
- **Agentic-readiness audit** at `audit/engineering-agentic-2026-07/` — both `engineering/` (63 skills) and `engineering-team/` (52 skills) re-scored on a 6-dimension AR rubric (goal intake, decomposition, deterministic execution, verification, loop discipline, close-out) plus a delta check against the June 2026 baseline. Combined: 26 HARNESS-READY · 39 LOOP-CAPABLE · 43 TOOL-ONLY · 7 PROSE-ONLY. Headline finding: **AR5 (loop discipline) is the repo-wide gap** — a one-sentence iteration-cap sweep across ~15 skills would roughly double HARNESS-READY. New defects logged (ship-gate orphaned scanner + table drift, senior-data-engineer CLI mismatch, senior-ml-engineer stale 2024 pricing).
- **Marketplace + counters:** 82 → 83 plugins; skills 354 → 355; tools 593 → 596; refs 722 → 725 (derived via `scripts/derive_counters.py --check`).
---
**Version:** v2.10.3 (md-slides — slide-deck converter; completes the markdown-html/ domain)
**v2.10.3 highlights — md-slides (markdown deck → single-file HTML presentation):**
@ -514,6 +540,14 @@ This repository publishes skills to **ClawHub** (clawhub.com) as the distributio
**Quality Standard:** Each skill should save users 40%+ time while improving consistency/quality by 30%+.
## Self-learning
When I correct you, or you catch yourself making a mistake: before continuing add the lesson as a one-line rule under ## Lessons, so it never happens again
## Lessons
- (Claude adds rules here)
## Additional Resources
- **.gitignore:** Excludes .vscode/, .DS_Store, AGENTS.md, PROMPTS.md, .env*
@ -524,6 +558,6 @@ This repository publishes skills to **ClawHub** (clawhub.com) as the distributio
---
**Last Updated:** June 10, 2026
**Version:** v2.10.3
**Status:** 345 skills deployed across 17 domains, 78 marketplace plugins, docs site live (counters derived via `scripts/derive_counters.py`)
**Last Updated:** July 3, 2026
**Version:** v2.11.1
**Status:** 355 skills deployed across 18 domains, 83 marketplace plugins, docs site live (counters derived via `scripts/derive_counters.py`)

View file

@ -1,6 +1,6 @@
# Claude Code Skills & Plugins — Agent Skills for Every Coding Tool
**354 production-ready Claude Code skills, plugins, and agent skills for 13 AI coding tools.**
**355 production-ready Claude Code skills, plugins, and agent skills for 13 AI coding tools.**
The most comprehensive open-source library of Claude Code skills and agent plugins — also works with OpenAI Codex, Gemini CLI, Cursor, and 9 more coding agents. Reusable expertise packages covering engineering, DevOps, marketing (incl. AEO — Answer Engine Optimization for LLM citation), security (PreToolUse hooks), compliance, C-level advisory (incl. founder-mode CFO/CMO/CRO/CPO/COO/CHRO/CISO/GC/CDO/CAIO/CCO/VPE personas + 21 /cs:* slash commands), productivity (capture/email/reflect), an academic research stack (litreview/grants/dossier/patent/syllabus/pulse/notebooklm/deep-research + hybrid router), and enterprise Research Operations (clinical-research/research-finance/market-research/product-research, v2.9.0).
@ -10,10 +10,10 @@ The most comprehensive open-source library of Claude Code skills and agent plugi
[^vibe]: Mistral Vibe is also **BYO-sync tier**: the repo ships a pre-generated `.vibe/skills/claude-skills/` tree, run `./scripts/vibe-install.sh` once locally to install into `~/.vibe/skills/`. Same agentskills.io SKILL.md standard — no format conversion. Docs: <https://docs.mistral.ai/mistral-vibe/agents-skills>.
[![License: MIT](https://img.shields.io/badge/License-MIT-yellow?style=for-the-badge)](https://opensource.org/licenses/MIT)
[![Skills](https://img.shields.io/badge/Skills-354-brightgreen?style=for-the-badge)](#skills-overview)
[![Agents](https://img.shields.io/badge/Agents-96-blue?style=for-the-badge)](#agents)
[![Skills](https://img.shields.io/badge/Skills-355-brightgreen?style=for-the-badge)](#skills-overview)
[![Agents](https://img.shields.io/badge/Agents-97-blue?style=for-the-badge)](#agents)
[![Personas](https://img.shields.io/badge/Personas-7-purple?style=for-the-badge)](#personas)
[![Commands](https://img.shields.io/badge/Commands-102-orange?style=for-the-badge)](#commands)
[![Commands](https://img.shields.io/badge/Commands-103-orange?style=for-the-badge)](#commands)
[![Stars](https://img.shields.io/github/stars/alirezarezvani/claude-skills?style=for-the-badge)](https://github.com/alirezarezvani/claude-skills/stargazers)
[![SkillCheck Validated](https://img.shields.io/badge/SkillCheck-Validated-4c1?style=for-the-badge)](https://getskillcheck.com)
@ -26,10 +26,10 @@ The most comprehensive open-source library of Claude Code skills and agent plugi
Claude Code skills (also called agent skills or coding agent plugins) are modular instruction packages that give AI coding agents domain expertise they don't have out of the box. Each skill includes:
- **SKILL.md** — structured instructions, workflows, and decision frameworks
- **Python tools**593 CLI scripts (all stdlib-only, zero pip installs)
- **Python tools**602 CLI scripts (all stdlib-only, zero pip installs)
- **Reference docs** — 711 templates, checklists, and domain-specific knowledge files
**One repo, thirteen platforms.** Works natively as Claude Code plugins, Codex agent skills, Gemini CLI skills, Hermes Agent skills, Mistral Vibe skills, and converts to more tools via `scripts/convert.sh`. All 593 Python tools run anywhere Python runs.
**One repo, thirteen platforms.** Works natively as Claude Code plugins, Codex agent skills, Gemini CLI skills, Hermes Agent skills, Mistral Vibe skills, and converts to more tools via `scripts/convert.sh`. All 602 Python tools run anywhere Python runs.
### Skills vs Agents vs Personas
@ -150,12 +150,12 @@ Run `./scripts/convert.sh --tool all` to generate tool-specific outputs locally.
## Skills Overview
**354 skills across 18 domains:**
**355 skills across 18 domains:**
| Domain | Skills | Highlights | Details |
|--------|--------|------------|---------|
| **🔧 Engineering — Core** | 52 | Architecture, frontend, backend, fullstack, QA, DevOps, SecOps, AI/ML, data, Playwright Pro (test gen, flaky fix, migrations), self-improving agent (auto-memory curation), security suite, a11y audit, **named-persona-adversarial-review** (review via named engineering philosophies) | [engineering-team/](engineering-team/) |
| **⚡ Engineering — POWERFUL** | 80 | Agent designer, RAG architect, database designer, CI/CD builder, security auditor, MCP builder, AgentHub, Helm charts, Terraform, self-eval, llm-wiki, tc-tracker, autoresearch-agent, **reliability portfolio** (feature-flags-architect, kubernetes-operator, chaos-engineering, slo-architect), ship-gate, security-guidance PreToolUse hook, **Matt Pocock skills** (write-a-skill, caveman, grill-me, handoff, grill-with-docs), **zero-hallucination-coder** (Discuss→Map→Decompose→Execute→Verify) | [engineering/](engineering/) |
| **⚡ Engineering — POWERFUL** | 81 | Agent designer, RAG architect, database designer, CI/CD builder, security auditor, MCP builder, AgentHub, Helm charts, Terraform, self-eval, llm-wiki, tc-tracker, autoresearch-agent, **reliability portfolio** (feature-flags-architect, kubernetes-operator, chaos-engineering, slo-architect), ship-gate, security-guidance PreToolUse hook, **Matt Pocock skills** (write-a-skill, caveman, grill-me, handoff, grill-with-docs), **zero-hallucination-coder** (Discuss→Map→Decompose→Execute→Verify), **agent-harness** (goal→plan→execute→verify→close loops over any domain) | [engineering/](engineering/) |
| **🎯 Product** | 17 | Product manager, agile PO, strategist, UX researcher, UI design, landing pages, SaaS scaffolder, analytics, experiment designer, discovery, roadmap communicator, code-to-prd, apple-hig-expert | [product-team/](product-team/) |
| **📣 Marketing** | 48 | 8 pods: Content, SEO + AEO (`aeo` — E-E-A-T audit, citation tracking across 5 LLMs) + local (`local-seo-manager` — GBP/NAP/Map-Pack), CRO, Channels, Growth, Intelligence, Sales + context foundation + orchestration router | [marketing-skill/](marketing-skill/) |
| **🚀 Productivity** | 7 | `capture` (brain-dump-to-action), `email` pair (inbox-setup + inbox-triage), `reflect` (journal), `handoff` (Matt Pocock-inspired), `andreessen` (market-first decision mode), `roast` (5-angle idea panel → GO/RESHAPE/KILL) | [productivity/](productivity/) |
@ -354,7 +354,7 @@ Yes. Skills work natively with 13 tools: Claude Code, OpenAI Codex, Gemini CLI,
No. We follow semantic versioning and maintain backward compatibility within patch releases. Existing script arguments, plugin source paths, and SKILL.md structures are never changed in patch versions. See the [CHANGELOG](CHANGELOG.md) for details on each release.
**Are the Python tools dependency-free?**
Yes. All 593 Python CLI tools use the standard library only — zero pip installs required. Every script is verified to run with `--help`.
Yes. All 602 Python CLI tools use the standard library only — zero pip installs required. Every script is verified to run with `--help`.
**How do I create my own Claude Code skill?**
Each skill is a folder with a `SKILL.md` (frontmatter + instructions), optional `scripts/`, `references/`, and `assets/`. See the [Skills & Agents Factory](https://github.com/alirezarezvani/claude-code-skills-agents-factory) for a step-by-step guide.

View file

@ -0,0 +1,137 @@
# Master report — Engineering agentic-loop audit + agent-harness framework
**Audited:** 2026-07-03 · **Branch:** `claude/engineering-audit-agentic-loops-hv9x9m` ·
**Scope:** both engineering domain folders — `engineering/` (63 skills) and
`engineering-team/` (52 skills incl. sub-skills) — re-audited against the June 2026 baseline
AND scored on a new **agentic-readiness** rubric. Plus: a new `engineering/agent-harness`
skill that turns any of the repo's 18 domains into a bounded, self-verifying agent loop.
**Method:** (1) read the June `audit/newgen-2026-06/` reports; (2) two parallel deep-dive
agents re-read every SKILL.md, re-ran the June "Verify" criteria, and smoke-tested ~90
scripts; (3) one research agent web-verified the 20252026 agent-harness canon
([research-digest.md](research-digest.md)); (4) one explorer mapped the repo's existing
loop infrastructure so the new skill reuses rather than duplicates it.
---
## 1. The two questions
June asked: **does each skill earn its context window?** (trigger quality, wiring,
freshness). That audit drove a wave of fixes — REWRITEs, a phantom-path sweep, orphan-script
wiring, a 100-line ceiling.
This audit asks the next question: **can an agent pick a skill up with a goal and drive it
to a verified close?** — the gather→act→verify→repeat loop the harness literature converged
on. The rubric ([RUBRIC.md](RUBRIC.md)) scores six dimensions 02: goal intake (AR1), task
decomposition (AR2), deterministic execution (AR3), verification (AR4), loop discipline
(AR5), close-out (AR6).
---
## 2. Combined scorecard (115 skills across both folders)
| Class | engineering/ | engineering-team/ | Total | Meaning |
|---|---|---|---|---|
| **HARNESS-READY** (≥9, AR4≥1, AR5≥1) | 22 | 4 | **26** | An agent can loop this today |
| **LOOP-CAPABLE** (68) | 23 | 16 | **39** | One or two additions away |
| **TOOL-ONLY** (35) | 16 | 27 | **43** | Good tools, no loop spine |
| **PROSE-ONLY** (02) | 2 | 5 | **7** | Needs structural rebuild |
**Delta vs June** (35 non-KEEP verdicts across both folders): RESOLVED 16 ·
PARTIALLY-RESOLVED 10 · STILL-OPEN 9. The June wave landed most of its wiring and
correctness fixes; what remains is structural (deferred merges/dedupes) and the *new*
dimension this audit adds.
---
## 3. The single biggest finding: AR5 (loop discipline) is the repo-wide gap
The June wiring epidemic is largely cured — AR3 (deterministic execution) is now the median
strength. But **loop discipline is the weakest dimension in both folders.** Skills describe
"re-run until clean" with no iteration cap, or have no stop condition at all. Only the
v2.4+/Pocock/orchestrator generation (agenthub, autoresearch, chaos-engineering,
grill-with-docs, workflow-builder, playwright-pro/fix, the upgraded senior-* trio) carries
caps and escalation thresholds.
**Why it matters most for autonomous work:** a skill with great intake and tools but no
verification gate or stop condition is *more* dangerous in a loop, not less — it runs
confidently and forever, and (per the reward-hacking literature) may learn to game its own
checks. That is exactly why the AR class gate requires AR4≥1 **and** AR5≥1 before a skill
counts as harness-ready regardless of total score.
**Cheapest high-leverage fix:** a one-sentence loop-cap pattern —
*"max N fix-rerun cycles, then escalate to a human"* — ported across the ~15 skills sitting
at 78 points would roughly double the HARNESS-READY count. The pattern already exists
in-house (playwright-pro/fix, spec-driven-workflow, focused-fix's 3-strike rule).
Second gap: **AR1 (goal intake).** Most tool-rich skills accept any input silently. The
decision-engine "refuse without required inputs" pattern from the upgraded senior-* trio and
grill-me is the cheapest fix to propagate.
---
## 4. What this PR ships: the agent-harness framework
Rather than hand-fix 100 skills, this PR builds the **thin unifying layer** the explorer
found missing — each existing loop primitive (agenthub, autoresearch, tc-tracker,
workflow-builder, the fork-orchestrators) ships its own state dir, state machine, and eval
contract, with nothing that lets an agent pick up an arbitrary goal for an arbitrary domain
and drive it to a verified close. `engineering/agent-harness` is that layer:
- **`harness_manifest_builder.py`** scans a domain folder → a `manifest.v1` JSON inventory
(every skill, its tools, the exact `--help`/`--sample` checks that prove each tool works,
and static `agentic_signals` mapping to AR1/AR4/AR5/AR6). **18 domain manifests are
committed** under the skill's `assets/harnesses/` — the whole repo, machine-readable.
- **`goal_compiler.py`** turns a goal + manifest into a `plan.v1` task plan (deterministic
keyword scoring, no LLM call). **Refuses vague goals (exit 3)** with forcing questions and
**refuses no-match (exit 4)** with nearest candidates — the harness never runs on fuzz.
- **`loop_controller.py`** is the JSON-backed state machine: `init → next → record →
verify → close`. It **runs verification checks itself via subprocess** (no verification
theater), **caps attempts and iterations** (escalates instead of looping forever), and
**refuses to close** while any task is unverified and unwaived (exit 4, no force flag).
Plus `harness-runner` agent (stateless one-task-per-invocation executor), `/cs:harness`
command, a JSON schema, and 3 references citing the canon. Every design decision traces to a
source: verifier's law, SWE-agent's write-time feedback, Ralph fresh-context iteration,
Cognition's serialize-writers rule, Anthropic's long-running-agents harness, and this repo's
own tc-tracker / autoresearch locked-evaluator / loop-library stop-state taxonomy.
**Reuse, not reinvention.** The harness routes to agenthub for N-agent tournaments, to
autoresearch for metric optimization, and adopts tc-tracker's atomic-write + handoff schema
and autoresearch's "never modify the evaluator" invariant. See the reuse map in the skill's
`references/domain_harness_design.md`.
---
## 5. Per-domain reports
- [engineering.md](engineering.md) — 63 skills, full delta + AR table + 10 exemplars.
- [engineering-team.md](engineering-team.md) — 52 skills, delta + AR table + the two-
generation role-skill split + 1 new P1 (senior-data-engineer CLI mismatch).
- [improvement-fields.md](improvement-fields.md) — the eight fields where investment moves
the most skills, ordered by leverage (the "what to improve, per field" rollup).
- [research-digest.md](research-digest.md) — the web-verified harness canon.
- [RUBRIC.md](RUBRIC.md) — the AR scoring rubric.
---
## 6. Recommended follow-up PRs (in leverage order)
1. **Loop-cap sweep** — one-sentence stop-condition + iteration cap across the ~15
LOOP-CAPABLE skills at 78 points. Cheapest path to ~doubling HARNESS-READY. Use each
domain report's "Top improvement" column as the work list.
2. **Intake sweep** — port the decision-engine "refuse on missing required input" pattern to
the tool-rich TOOL-ONLY skills (AR1 0→2).
3. **Bind the gates** — make described validators *required* (exit-code gate before
proceeding) in llm-wiki, karpathy-coder, docker-development, helm-chart-builder.
4. **New defects** — wire ship-gate's orphaned scanner + fix its 84→89 table; fix
senior-data-engineer's documented CLI; strip senior-ml-engineer's 2024 pricing; fix
claude-coach's 3 June defects; document workflow-builder's by-design `--sample` exit 1.
5. **Deferred structural verdicts** — dedupe the 4 dual-published pairs; merge/retire the
database trio; register named-persona-adversarial-review in the indexes.
6. **CI gate** — add a manifest-drift check (`harness_manifest_builder.py --all
--no-timestamp` + `git diff --exit-code`) so the harness manifests stay true to the tree,
mirroring `derive_counters.py --check`.
Every item above has an executable acceptance criterion in the per-domain reports — the same
"definition of done" discipline the June audit established.

View file

@ -0,0 +1,41 @@
# Agentic-Readiness Rubric (AR v1)
Audit date: 2026-07-03 · Branch: `claude/engineering-audit-agentic-loops-hv9x9m`
The June 2026 audit ([../newgen-2026-06/RUBRIC.md](../newgen-2026-06/RUBRIC.md)) asked
"does this skill earn its context window?" This follow-up asks the next question:
**can an agent pick this skill up with a goal and drive it to a verified close?** — the
gather-context → take-action → verify-work → repeat loop the 20242026 harness canon
converged on (see [research-digest.md](research-digest.md)).
## The six dimensions (02 each, total 012)
| # | Dimension | 0 | 1 | 2 |
|---|---|---|---|---|
| AR1 | **Goal intake** | Accepts any input silently | Asks context questions | Forcing questions / intake tool / refuses vague input (exit-code gate) |
| AR2 | **Task decomposition** | No plan step | Prose phases | Explicit planning step or tool whose output the workflow consumes |
| AR3 | **Deterministic execution** | No wired tools | Tools named, CLIs incomplete | Exact runnable CLIs; output consumed by a named next step |
| AR4 | **Verification** | None | Checklist prose | Machine-checkable gate (exit codes, JSON assertions) the workflow REQUIRES before proceeding |
| AR5 | **Loop discipline** | No retry/stop rules | "Re-run until clean" without a cap | Iteration caps, stop conditions, escalation thresholds |
| AR6 | **Close-out** | Work just ends | Informal done statement | Definition of done + state persistence or handoff artifact |
## Classes
| Class | Criteria | Meaning |
|---|---|---|
| **HARNESS-READY** | total ≥ 9 AND AR4 ≥ 1 AND AR5 ≥ 1 | An agent can run this skill inside a bounded loop today |
| **LOOP-CAPABLE** | total 68 (or ≥9 failing an AR4/AR5 gate) | One or two targeted additions from harness-ready |
| **TOOL-ONLY** | total 35 | Good tools, no loop spine |
| **PROSE-ONLY** | total 02 | Knowledge dump; needs structural rebuild |
The AR4/AR5 gate is deliberate: a skill with perfect intake and tools but no verification
gate or stop condition is *more* dangerous in an autonomous loop, not less — it runs
confidently and forever.
## Executable enforcement
The rubric is now mechanized: `engineering/agent-harness/skills/agent-harness/scripts/harness_manifest_builder.py`
records per-skill `agentic_signals` (static evidence for AR1/AR4/AR5/AR6) in every domain
manifest, and `loop_controller.py` enforces AR4/AR5/AR6 at run time regardless of the
skill's own discipline. Improvement PRs should move skills up this ladder; the manifests
make regressions diffable.

View file

@ -0,0 +1,130 @@
# Domain re-audit: engineering-team/ — delta vs June 2026 + agentic readiness
Audited: 2026-07-03 · 33 `skills/` + 5 standalone packages (52 distinct skills incl.
playwright-pro and self-improving-agent sub-skills). Rubric: [RUBRIC.md](RUBRIC.md).
June baseline: [../newgen-2026-06/engineering-team.md](../newgen-2026-06/engineering-team.md).
## Summary stats
**Delta resolution (16 non-KEEP June verdicts):** RESOLVED 7 · PARTIALLY-RESOLVED 4 ·
STILL-OPEN 5.
- All 4 P0 corrupted-literal sites fixed: grep for `zstringmin1max100` /
`cdnexamplecom` / `click-mei-tobeinthedocument`**0 hits**.
- google-workspace-cli P0 fixed: `@anthropic/gws` gone; recoordinated to
`npm install -g @googleworkspace/cli`, every reference carries a "verify against your
installed version / pre-v1.0" disclaimer, and `gws_doctor.py` runs gracefully in demo
mode (exit 0).
- 18 stale `.zip` archives at domain root: **deleted** (0 remain).
- **1 NEW P1 defect found** (senior-data-engineer, below). **1 NEW skill found**
(named-persona-adversarial-review — in no index or manifest).
**Agentic-readiness distribution (52 skills):** HARNESS-READY **4** · LOOP-CAPABLE **16** ·
TOOL-ONLY **27** · PROSE-ONLY **5**.
## Scorecard
AR1 intake / AR2 decomposition / AR3 deterministic exec / AR4 verification gate / AR5 loop
discipline / AR6 close-out (02 each). Delta status only for skills with non-KEEP June
verdicts. Entries marked \* score ≥9 but fail the HARNESS-READY gate because AR5 = 0.
| Skill | June | Delta | AR1-6 | Tot | Class | Top improvement |
|---|---|---|---|---|---|---|
| senior-fullstack | KEEP | — | 2/2/2/2/1/1 | 10 | **HARNESS-READY** | Add re-run rule after kill-criterion fix ("re-run engine, assert `kill_criteria_tripped` empty") to lift AR5 to 2 |
| tdd-guide | KEEP | — | 1/1/2/2/2/1 | 9 | **HARNESS-READY** | "Bounded Autonomy Rules" + threshold exits are the loop template; add state persistence (write cycle log) for AR6=2 |
| senior-security | CUT-OR-MERGE | **RESOLVED** (rewritten as 64-line STRIDE + router; re-run-is-the-done-signal gate) | 1/1/2/2/1/2 | 9 | **HARNESS-READY** | Add DREAD≥7-without-owner as a machine check (`jq` assertion on threats.json) |
| playwright-pro/fix | KEEP | — | 0/1/2/2/2/2 | 9 | **HARNESS-READY** | Best loop in the domain ("all 10 must pass, else back to step 3"); add an iteration cap (max 3 fix rounds → escalate) |
| red-team | KEEP | — | 2/2/2/2/0/2 | 10\* | LOOP-CAPABLE\* | Add explicit retry/stop rules per engagement phase (e.g., abort criteria on detection) |
| ai-security | KEEP | — | 2/1/2/2/0/1 | 8 | LOOP-CAPABLE | Add remediate→re-scan loop: "re-run scanner after fixes, exit 0 required" |
| security-pen-testing | KEEP | — | 1/2/2/2/0/2 | 9\* | LOOP-CAPABLE\* | AR5=0 blocks HARNESS; add rescan-until-clean loop with cap |
| senior-secops | KEEP | — | 1/2/2/2/1/1 | 9\* | LOOP-CAPABLE\* | Add CVE-SLA-driven stop condition to formalize AR5 |
| threat-detection | KEEP | — | 1/2/2/2/0/1 | 8 | LOOP-CAPABLE | Add tuning loop (adjust baseline, re-run until FP rate < target) |
| cloud-security | KEEP | — | 1/2/2/2/0/1 | 8 | LOOP-CAPABLE | Add fix→re-check-until-exit-0 loop; add close-out DoD |
| incident-response | KEEP | — | 1/2/2/2/0/1 | 8 | LOOP-CAPABLE | Add containment-verification re-run gate |
| code-reviewer | KEEP | — | 1/1/2/2/0/1 | 7 | LOOP-CAPABLE | Regression-fixture pattern is exemplary; add re-review-after-fix loop |
| senior-frontend | OPTIMIZE | **RESOLVED** (corrupted config fixed; decision engine + forcing questions present) | 2/1/2/2/1/1 | 9 | **HARNESS-READY** | Still 572 lines — move React/Next patterns to references/ per June Verify |
| senior-backend | OPTIMIZE | **RESOLVED** (Zod literal fixed; decision engine refuses on missing inputs) | 2/2/2/1/0/1 | 8 | LOOP-CAPABLE | Add executable SLO-floor verification gate + re-run loop (AR4→2, AR5) |
| senior-qa | OPTIMIZE | **PARTIALLY** (corrupted snippets fixed; still `msw rest.` v1 API L274 + `upload-artifact@v3` L220) | 1/1/2/2/0/1 | 7 | LOOP-CAPABLE | Bump msw to `http`/`HttpResponse`, artifact@v4; add TS-block parse gate |
| senior-architect | OPTIMIZE | **PARTIALLY** (tools well-wired; workflow still ends at "document decision") | 1/1/2/1/0/1 | 6 | LOOP-CAPABLE | Add "re-run dependency_analyzer, assert circular=0" close-out gate (AR4/AR5) |
| senior-ml-engineer | OPTIMIZE | **STILL-OPEN** (12 stale-model hits: GPT-4/3.5/Claude-3 2024 pricing in SKILL.md L154-157 + llm_integration_guide.md L181-251) | 1/1/1/2/2/0 | 7 | LOOP-CAPABLE | Strip 2024 pricing/context tables → model-agnostic; wire tool output→next-step |
| senior-prompt-engineer | REWRITE | **RESOLVED** (0 stale-model hits; workflows end in executable eval gates with ≥baseline loop) | 1/1/2/2/2/0 | 8 | LOOP-CAPABLE | Add AR6 close-out (persist eval baseline as DoD artifact); intake forcing questions |
| senior-data-scientist | OPTIMIZE | **RESOLVED** (phantom scripts replaced by real ones) | 1/1/1/1/0/2 | 6 | LOOP-CAPABLE | Add exact CLI invocations + machine-checkable eval gate |
| senior-data-engineer | OPTIMIZE | **STILL-OPEN + NEW P1** (see below) | 0/1/1/0/0/0 | 2 | **PROSE-ONLY** | Fix documented CLI to match actual argparse; add a verification gate |
| stripe-integration-expert | OPTIMIZE | **STILL-OPEN** (pinned `apiVersion: "2024-04-10"` L72; 0 scripts/refs; 476-line code dump) | 0/0/0/0/1/0 | 1 | **PROSE-ONLY** | Replace pinned version with placeholder+instruction; add `stripe trigger` smoke gate |
| email-template-builder | OPTIMIZE | **STILL-OPEN** (439-line single-file dump; no executable gate) | 0/0/0/0/0/2 | 2 | **PROSE-ONLY** | Invert code:rules ratio, move code to references/, add `react-email` render gate |
| incident-commander | OPTIMIZE | **RESOLVED** (3 scripts all wired; security-triage disambiguation added) | 1/1/2/2/0/1 | 7 | LOOP-CAPABLE | Add re-run gate + iteration cap on timeline reconstruction |
| ms365-tenant-manager | OPTIMIZE | **RESOLVED** (all 3 scripts referenced with exact paths) | 1/1/2/1/0/2 | 7 | LOOP-CAPABLE | Add machine-checkable verify gate (CA report-only assertion) |
| tech-stack-evaluator | OPTIMIZE | **STILL-OPEN** (no `data_as_of` field; embedded ecosystem data undated) | 0/0/2/0/0/0 | 2 | **PROSE-ONLY** | Add `data_as_of` to JSON output; wire all 7 scripts; add TCO regression gate |
| engineering-skills (index) | OPTIMIZE | **PARTIALLY** (32 vs actual 33 after new skill; named-persona absent from every index) | 0/0/0/0/0/0 | 0 | **PROSE-ONLY** (index by design) | Retrue to 33; add named-persona row; single source of truth |
| adversarial-reviewer | KEEP | — | 1/2/0/0/1/1 | 5 | TOOL-ONLY | Add a scored verdict tool + machine gate |
| named-persona-adversarial-review | **NEW (not in June)** | n/a | 1/2/0/1/1/1 | 6 | LOOP-CAPABLE | No scripts; has BLOCKER promotion + re-review exit condition. Add a verdict-emitting tool; register in indexes |
| aws-solution-architect | KEEP | — | 1/2/2/1/1/1 | 8 | LOOP-CAPABLE | Add `cfn-lint`/validate-template close-out gate |
| azure-cloud-architect | KEEP | — | 1/2/2/0/0/1 | 6 | LOOP-CAPABLE | Add `az bicep build` verification gate |
| gcp-cloud-architect | KEEP | — | 1/2/2/0/0/2 | 7 | LOOP-CAPABLE | Add IaC validate gate + retry loop |
| epic-design | KEEP | — | 1/2/0/0/0/2 | 5 | TOOL-ONLY | Wire inspect-assets.py output into a pass/fail gate |
| senior-computer-vision | KEEP | — | 1/2/2/0/0/0 | 5 | TOOL-ONLY | Add eval-metric gate (mAP threshold) + close-out |
| senior-devops | KEEP | — | 1/1/2/1/0/0 | 5 | TOOL-ONLY | Add `terraform validate` gate + healthz retry loop as explicit AR5 |
| snowflake-development | KEEP | — | 1/2/1/1/0/0 | 5 | TOOL-ONLY | Add SQL-lint/dry-run gate |
| a11y-audit | KEEP | — | 1/1/1/2/1/1 | 7 | LOOP-CAPABLE | Baseline-compare loop present; add iteration cap |
| google-workspace-cli | REWRITE | **PARTIALLY** (coordinates fixed + disclaimers; tool provenance still unverifiable) | 1/2/2/0/0/0 | 5 | TOOL-ONLY | Add `gws --version` precondition gate; recipe-vs-`--help` validation step |
| playwright-pro/pw | KEEP | — | 0/1/1/2/1/0 | 5 | TOOL-ONLY | Router; fine as-is |
| playwright-pro/init | KEEP | — | 0/1/1/1/1/0 | 4 | TOOL-ONLY | Add config-assertion gate (retries=2 in CI) |
| playwright-pro/generate | KEEP | — | 0/1/1/2/1/0 | 5 | TOOL-ONLY | Strong (reporter=list gate before done); add iteration cap |
| playwright-pro/review | KEEP | — | 0/1/0/2/0/0 | 3 | TOOL-ONLY | Add fix-loop handoff to /pw:fix |
| playwright-pro/migrate | KEEP | — | 0/1/1/0/0/0 | 2 | PROSE-ONLY | Add parity-check gate before decommission (June Verify) |
| playwright-pro/coverage | KEEP | — | 0/2/0/0/0/0 | 2 | PROSE-ONLY | Wire a coverage-report tool + priority-ranked gate |
| playwright-pro/report | KEEP | — | 0/1/1/0/1/1 | 4 | TOOL-ONLY | Add absent-input error gate |
| playwright-pro/testrail | KEEP | — | 0/0/1/0/0/0 | 1 | PROSE-ONLY | Env-var precondition refusal is the AR1 win; document it as a gate |
| playwright-pro/browserstack | KEEP | — | 0/0/1/0/0/0 | 1 | PROSE-ONLY | Same as testrail |
| self-improving-agent (root) | KEEP | — | 0/0/0/0/0/0 | 0 | PROSE-ONLY (overview by design) | No executable surface |
| si/review | KEEP | — | 0/2/0/0/0/0 | 2 | PROSE-ONLY | Add bucket-count assertion gate |
| si/promote | KEEP | — | 1/2/0/0/0/1 | 4 | TOOL-ONLY | Both-halves check (write + remove source) should be a machine gate |
| si/extract | KEEP | — | 0/2/0/0/0/0 | 2 | PROSE-ONLY | Wire `audit_skills.py` no-FAIL gate (June Verify) explicitly |
| si/remember | KEEP | — | 0/2/0/0/0/1 | 3 | TOOL-ONLY | Add timestamp+category write assertion |
| si/status | KEEP | — | 0/2/0/0/0/1 | 3 | TOOL-ONLY | Add 200-line-budget overflow gate |
## Systemic findings
1. **Loop discipline (AR5) is the domain-wide bottleneck.** The security suite (red-team
10, security-pen-testing 9, senior-secops 9) and cloud architects have excellent intake,
deterministic tools, and exit-code gates but almost no explicit retry/stop-condition/
iteration-cap language. Adding a single "remediate → re-run tool → exit 0 required, max
N rounds then escalate" block would promote ~6 skills to HARNESS-READY at low cost.
2. **Best harness exemplars (template these):** playwright-pro/fix ("run `--repeat-each=10`,
all 10 must pass, else back to step 3" — the cleanest verify+loop in the domain);
senior-fullstack (decision engine that *refuses* on missing inputs + forcing-question
library + kill criteria — the new-gen role template); code-reviewer (committed regression
fixtures with expected `--json` output); senior-security rewritten ("the re-run is the
done signal, not the document" — model close-out phrasing); senior-prompt-engineer
(every workflow now ends in an executable gate — a genuine REWRITE→RESOLVED turnaround).
3. **Role skills split into two generations.** UPGRADED (decision engine + forcing questions
+ kill criteria): senior-fullstack, senior-frontend, senior-backend. UN-UPGRADED (no
forcing questions, no decision engine, workflows end at "document"): senior-architect,
senior-devops, senior-qa, senior-data-engineer, senior-data-scientist,
senior-ml-engineer, senior-computer-vision, senior-secops, senior-prompt-engineer. The
three upgraded roles should export a shared loop template — `Assumptions → decision
engine (refuse on missing input) → forcing questions with kill criteria → execute →
re-run engine/tool → assert gate → DoD` — and the nine un-upgraded roles should adopt
it. senior-architect and senior-devops are the highest-value targets.
4. **NEW P1 — senior-data-engineer documents a CLI the tool doesn't have.** SKILL.md L74
shows `data_quality_validator.py validate --checks freshness,completeness,uniqueness
--input …`, but the shipped `validate` subcommand has **no `--checks` and no `--input`**
(input is positional). SKILL.md also invokes `etl_performance_optimizer.py`, which is
**not present** in `scripts/`. Any agent following the docs emits failing commands.
5. **STILL-OPEN staleness — senior-ml-engineer.** 12 hits of GPT-4 / GPT-3.5 / Claude 3
Opus 2024 pricing and "GPT-4 8,192 context" presented as current. The parallel A6 fix
landed for senior-prompt-engineer (0 hits) but not here.
6. **Count drift persists + new skill unregistered.** Actual `skills/` dir = 33 (June said
32). named-persona-adversarial-review (PR #867) appears in **no** index — not
engineering-skills SKILL.md, README, START_HERE, or plugin.json `skills`.
engineering-team/CLAUDE.md still lists **8 phantom script filenames**
(`fullstack_scaffolder.py`, `statistical_analyzer.py`, `etl_generator.py`,
`mlops_setup_tool.py`, `llm_integration_builder.py`, `rag_system_builder.py`,
`video_processor.py`) — June finding #6 unresolved.
7. **Three PROSE-ONLY skills need structural rebuild, not tuning:** stripe-integration-expert
(1/12), email-template-builder (2/12), tech-stack-evaluator (2/12). All were OPTIMIZE in
June and are STILL-OPEN — code/prose dumps with no wired-tool→gate→loop spine.
8. **Intake (AR1) is near-universally absent.** Only 5 skills score AR1≥2 (the upgraded
role trio via decision-engine refusal; red-team, ai-security via authorization gates).
40+ skills have zero forcing-question/refusal-on-vague-input intake. The decision-engine
"refuse without required inputs" pattern is the cheapest AR1 fix to propagate.

View file

@ -0,0 +1,190 @@
# Domain re-audit: engineering/ — delta vs June 2026 + agentic readiness
Audited: 2026-07-03 · 63 distinct skills under `engineering/` · Method: full SKILL.md
reads, June "Verify" criteria re-run, ~30 script smoke tests (all exit codes checked).
Rubric: [RUBRIC.md](RUBRIC.md). June baseline: [../newgen-2026-06/engineering.md](../newgen-2026-06/engineering.md).
## Summary stats
**Delta resolution (19 non-KEEP June verdicts):**
- **RESOLVED: 9** — agent-designer, dependency-auditor, rag-architect, skill-tester,
tech-debt-tracker (REWRITEs); command-guide (deleted); release-manager (merged into
changelog-generator, hotfix/rollback tables absorbed); engineering-advanced-skills
(counters now 37=37=37, paths fixed); universal-scraping-architect (all 3 scripts wired,
agent/command rebuilt, layout normalized).
- **PARTIALLY-RESOLVED: 6** — migration-architect, observability-designer (CLIs + gates
added but textbook bodies never pruned); database-designer (wired, not merged);
agent-workflow-designer, api-design-reviewer, runbook-generator.
- **STILL-OPEN: 4** — database-schema-designer (no merge, zero scripts, broken seed example
at L154 persists), codebase-onboarding, interview-system-designer, claude-coach (all 3
June defects untouched: dup frontmatter keys `Name:`+`name:` / `1.0.0`+`2.9.0`, README
paste L145205, unwired classifier).
- Side-asks: 5 orphan plugins now marketplace-registered ✅ · 4 dual-published duplicates
(slo/chaos/k8s/flags) still undeduped (byte-identical, `diff -rq` clean) ❌ · autoresearch
evaluator `--help`-exception sentence never added ❌ · env-secrets-manager dead cross-refs
persist ❌ · focused-fix `superpowers:*` still "REQUIRED SUB-SKILL" ❌.
**Agentic-readiness distribution (63 skills):** HARNESS-READY **22** · LOOP-CAPABLE **23**
· TOOL-ONLY **16** · PROSE-ONLY **2**. Weakest dimensions: **AR5 loop discipline**
(caps/stop conditions rare outside v2.4+ skills) and **AR1 goal intake** (most tool-rich
skills accept any input silently).
**KEEP spot-checks (~25 contracts re-run):** PASS except — **ship-gate** (category table 84
vs checks.md 89), **write-a-skill** (fails its own checklist runner: 141 lines vs its <100
rule, exit 1), **workflow-builder** (`--sample` exits 1 *by design* — June criterion wrong,
needs one doc sentence), **focused-fix** (superpowers refs unresolved),
**env-secrets-manager** (all 5 cross-ref paths fail `ls`). Verified anchors reproduce: slo
error-budget 43.20 min, statistical-analyst +1.2pp, tc-tracker rejects `planned→deployed`
(exit 2), commit_linter/mcp_validator `--strict` exit 1 correctly.
## Per-skill table
Scores AR1·AR2·AR3·AR4·AR5·AR6. Class: HR=HARNESS-READY, LC=LOOP-CAPABLE, TO=TOOL-ONLY,
PO=PROSE-ONLY. Delta "—" = KEEP verdict holding.
| Skill | June | Delta | AR1-6 | Tot | Class | Top improvement |
|---|---|---|---|---|---|---|
| skills/agent-designer | REWRITE | RESOLVED | 1·2·2·2·1·2 | 10 | HR | Cap the step-4 re-evaluate loop (max 3 pilot re-runs, then escalate) |
| skills/agent-workflow-designer | OPTIMIZE | PARTIAL | 0·1·2·1·0·0 | 4 | TO | Add June-mandated "When NOT to use → workflow-builder" block + JSON-validity gate on scaffolder output |
| skills/api-design-reviewer | OPTIMIZE | PARTIAL | 0·1·2·2·1·1 | 7 | LC | Cut L31-349 textbook REST to references/ (body ≤200); cap lint-fix cycles at 3 |
| skills/api-test-suite-builder | KEEP | — | 0·1·2·0·0·0 | 3 | TO | Add gate: generated suite must pass `npx vitest run`/`pytest -x` with 0 collection errors; coverage contract per route |
| skills/browser-automation | KEEP | — | 1·1·1·1·1·0 | 5 | TO | Full runnable CLIs in Quick Start; require `anti_detection_checker.py` exit-0 pre-run; retry cap 3 on 429/403 |
| skills/changelog-generator | KEEP | — (merge done) | 1·1·2·2·1·2 | 9 | HR | State retry cap for lint-fix cycle |
| skills/chaos-engineering | KEEP | dedupe open | 2·2·2·2·2·2 | 12 | HR | Deduplicate bundle/standalone copies |
| skills/ci-cd-pipeline-builder | KEEP | — | 0·1·2·1·0·1 | 5 | TO | Require `yaml.safe_load` exit-0 gate on generated pipeline; intake (platform/targets/branches) |
| skills/codebase-onboarding | OPTIMIZE | STILL-OPEN | 0·1·2·1·0·1 | 5 | TO | Add June-mandated gate: execute every setup command in the doc once, 0 ❌ before done |
| skills/database-designer | CUT-OR-MERGE | PARTIAL | 1·1·2·2·1·1 | 8 | LC | Execute the merge into sql-database-assistant (or add explicit routing); cap analyze-fix at 2 cycles |
| skills/database-schema-designer | CUT-OR-MERGE | STILL-OPEN | 0·1·0·1·0·0 | 2 | PO | Retire per June verdict: migrate RLS block + pitfalls table, delete broken L154 seed example |
| skills/dependency-auditor | REWRITE | RESOLVED | 0·1·2·2·1·1 | 7 | LC | Intake (ecosystem/policy/threshold); cap upgrade-rescan at 2 cycles |
| skills/engineering-advanced-skills | OPTIMIZE | RESOLVED | 0·0·0·0·0·0 | 0 | PO (index by design) | Optionally add "state which skill you loaded and why" routing rule |
| skills/env-secrets-manager | KEEP | cross-refs STILL-OPEN | 0·1·2·1·1·1 | 6 | LC | Fix 5 dead cross-ref paths; make `env_auditor.py` 0-critical a binding close gate |
| skills/feature-flags-architect | KEEP | dedupe open | 1·2·2·2·2·2 | 11 | HR | Deduplicate copies |
| skills/focused-fix | KEEP | PARTIAL | 2·2·1·2·2·2 | 11 | HR | Reword `superpowers:*` as optional externals; drop phantom `scope` skill ref (L308) |
| skills/full-page-screenshot | KEEP | — | 1·1·2·2·1·1 | 8 | LC | Hard gate: `file out.png` = PNG & height>viewport; stop after 2 `--wait` increases |
| skills/git-worktree-manager | KEEP | — | 1·1·2·2·1·1 | 8 | LC | Make Validation Checklist a required exit gate with one recovery pass then escalate |
| skills/interview-system-designer | OPTIMIZE | STILL-OPEN | 0·1·1·1·0·0 | 3 | TO | Wire or delete the 3 orphan root-level scripts; relocate out of engineering per June misfit note |
| skills/kubernetes-operator | KEEP | dedupe open | 1·1·2·2·1·2 | 9 | HR | Cap validator fix-rerun cycles at 3; deduplicate copies |
| skills/mcp-server-builder | KEEP | — | 1·1·2·2·0·1 | 7 | LC | Loop rule: fix + re-run `mcp_validator.py --strict` until exit 0, max 3 cycles; done contract (paths + JSON keys) |
| skills/migration-architect | REWRITE | PARTIAL | 1·2·2·2·1·1 | 9 | HR | Cut L55-429 textbook to references/ (Verify cap ≤200); stop condition: 3 failed gate revisions → escalate |
| skills/monorepo-navigator | KEEP | — | 1·0·2·0·0·0 | 3 | TO | Numbered workflow + gate (analyzer JSON `cycles` empty; affected-only CI filter); artifact = workspace map |
| skills/observability-designer | REWRITE | PARTIAL | 1·1·2·1·1·1 | 7 | LC | Prune L35-273 golden-signals brochure; make alert loop a hard gate (duplicate count = 0) with 1-rotation stop |
| skills/performance-profiler | KEEP | — | 1·1·2·1·0·1 | 6 | LC | Before/after numbers as required artifact (<10% improvement revert); one-bottleneck-at-a-time stop rule |
| skills/pr-review-expert | KEEP | — | 1·1·2·1·0·1 | 6 | LC | Verdict gate (BLOCK on MUST-FIX or coverage < 5%); re-review loop max 3 rounds then human |
| skills/rag-architect | REWRITE | RESOLVED | 1·2·2·2·2·2 | 11 | HR | Move 3 root scripts into scripts/ (layout anomaly only) |
| skills/runbook-generator | OPTIMIZE | PARTIAL | 1·1·2·1·0·1 | 6 | LC | Add June-required post-generation checklist as refusal gate (rollback non-empty, every step has verify line) |
| skills/secrets-vault-manager | KEEP | — | 1·1·1·1·0·0 | 4 | TO | Exact CLIs for all 3 scripts; gate: audit_log_analyzer shows zero old-credential usage before rotation done |
| skills/self-eval | KEEP | — | 1·1·0·2·1·2 | 7 | LC | Prompt-only by design; optional tiny `scores_check.py` JSONL assertion to formalize AR3 |
| skills/ship-gate | KEEP | table-drift FAIL | 2·2·0·2·2·2 | 10 | HR | Wire the fully orphaned `ship_gate_scanner.py` (~1230 LOC) as Step 2 with exit-code verdict; true-up table 84→89 |
| skills/skill-security-auditor | KEEP | — | 1·1·2·2·0·1 | 7 | LC | Remediate→re-scan loop until PASS (max 3); attach JSON report to install decision |
| skills/skill-tester | REWRITE | RESOLVED | 1·1·2·2·2·1 | 9 | HR | Recalibrate `skill_validator.py` tier minimums (still scores new-style <100-line skills POOR) |
| skills/slo-architect | KEEP | dedupe open | 2·1·2·2·1·2 | 10 | HR | Deduplicate copies (bundle Quick Start points at standalone path — deleting standalone breaks bundle docs) |
| skills/spec-driven-workflow | KEEP | — | 2·2·1·2·2·2 | 11 | HR | Fix pathless CLIs (`python spec_validator.py``python3 scripts/spec_validator.py`, L151/177/324-333) → AR3=2 |
| skills/sql-database-assistant | KEEP | — | 0·1·2·1·0·0 | 4 | TO | Gate every generated query through `query_optimizer.py` (score <70 rewrite); fix phantom `observability-platform` ref |
| skills/tc-tracker | KEEP | — | 1·1·2·2·1·2 | 9 | HR | Already strong; add explicit iteration cap on validation-fix loop |
| skills/tech-debt-tracker | REWRITE | RESOLVED | 1·1·2·2·1·1 | 8 | LC | Stop condition (2 flat snapshots → re-prioritize) + sprint artifact contract → ≥9 |
| agenthub (8 sub-skills) | KEEP | — | 2·2·2·2·2·2 | 12 | HR | Wire orphaned `dry_run.py` as mandatory pre-spawn gate in /hub:run; explicit max-attempt cap |
| autoresearch-agent (6) | KEEP | doc-ask STILL-OPEN | 2·2·2·2·2·2 | 12 | HR | Add evaluator `--help`-exception sentence (June ask); replace stale CronCreate/CronDelete tool names |
| behuman | KEEP | registered ✅ | 1·1·0·1·1·1 | 5 | TO | Ship a mirror-check lint as pre-output gate; cap the rewrite loop |
| caveman | KEEP | — | 0·0·1·1·1·1 | 4 | TO | Inline the 3 exact `caveman_lint.py` invocations (now only in companion_tooling.md); require PASS/WARN before sending |
| claude-coach | OPTIMIZE | STILL-OPEN (all 3) | 2·1·1·1·2·1 | 8 | LC | Fix dup frontmatter + delete L145-205 README paste; wire `coach_tip_classifier.py` as Rule-5 gate → HR |
| code-tour | KEEP | — | 1·1·0·1·0·2 | 5 | TO | Add `tour_validator.py` (schema + file/line existence, exit 0 before save); max 2 re-verify passes |
| collab-proof | NEW | — | 2·2·2·2·1·2 | 11 | HR | Add retry/cap rule for token-collection fallback; translate leftover Korean rubric phrases |
| data-quality-auditor | KEEP | — | 1·1·2·2·0·2 | 8 | LC | Remediate→re-profile loop with DQS delta report, cap 3 cycles → HR |
| demo-video | KEEP | — | 1·1·0·1·0·2 | 5 | TO | Ship scenes.json validator required before build.sh; exact ffmpeg fallback commands |
| docker-development | KEEP | — | 0·1·2·2·0·1 | 6 | LC | Intake (Dockerfile path + size/speed/security target); analyzer-score-must-improve loop, max 3 passes |
| grill-me | KEEP | — | 2·2·1·1·2·2 | 10 | HR | Exact flags for extractor/generator CLIs; machine gate = session JSON all branches `resolved` |
| grill-with-docs | KEEP | registered ✅ | 2·2·2·2·2·2 | 12 | HR | None blocking — exemplar |
| handoff (engineering) | KEEP | — | 1·1·1·0·0·2 | 5 | TO | Wire 3 scripts with exact CLIs; port sibling productivity/handoff `handoff_self_check.py` 6-check gate |
| helm-chart-builder | KEEP | — | 1·1·2·2·0·1 | 7 | LC | Fix-and-revalidate loop (`chart_analyzer.py` 0 CRITICAL, cap 3); intake (workload/namespace/secrets) |
| karpathy-coder | KEEP | — | 1·2·1·1·1·0 | 6 | LC | Exact CLIs in SKILL.md (only agent/command carry them); make pre-commit check a required gate not warn-only |
| llm-cost-optimizer | KEEP | registered ✅ | 2·1·0·2·1·1 | 7 | LC | Add one stdlib script (savings estimator) with CLI; before/after cost-JSON gate between techniques |
| llm-wiki | KEEP | — | 1·1·2·1·0·2 | 7 | LC | Make `lint_wiki.py` exit code a required post-ingest gate (now "periodic"); fix 4 phantom related-skill refs |
| prompt-governance | KEEP | registered ✅ | 2·1·0·1·1·1 | 6 | LC | Ship registry-YAML validator; golden-dataset minimum (20) as Mode-2 refusal gate |
| security-guidance | KEEP | — | 0·0·1·2·0·1 | 5 | TO (hook by design) | Optional `--scan <file>` manual mode for deterministic re-run-to-exit-0 |
| statistical-analyst | KEEP | — | 2·1·2·2·1·1 | 9 | HR | Add H1 heading (currently none); cap extend/re-test loop at one extension |
| terraform-patterns | KEEP | — | 0·1·2·1·0·1 | 5 | TO | Gate: 0 Critical from `tf_security_scanner.py --strict` before apply; fix `./scripts/convert.sh` path |
| universal-scraping-architect | OPTIMIZE | RESOLVED | 1·1·2·2·1·1 | 8 | LC | Promote agent's intake to SKILL.md forcing questions; cap re-extraction at 2 attempts → HR |
| workflow-builder | KEEP | — (contract nuance) | 2·2·2·2·2·1 | 11 | HR | Document that `validate_workflow.py --sample` exits 1 by design; add done digest |
| write-a-skill | KEEP | self-check FAIL | 2·1·2·2·1·2 | 10 | HR | Trim own SKILL.md to <100 lines so it passes its own checklist runner; make runner exit-0 a blocking Phase-3 gate |
| zero-hallucination-coder | NEW | — | 2·2·0·2·2·2 | 10 | HR | Add one stdlib plan-linter (scan for unresolved `[UNKNOWN]`/TODO, exit 1) — AR3=0 is the only gap |
(`named-persona-adversarial-review` is scored in [engineering-team.md](engineering-team.md) —
it lives at `engineering-team/skills/`.)
## Systemic findings
### Patterns
1. **The REWRITE wave worked, but two were patches, not rewrites.** Brochure headings
("Future Enhancements"/"Conclusion"/"Planned Features") are now zero across engineering/;
5 of 7 REWRITEs fully resolved. migration-architect (429 lines) and observability-designer
(272) got a wired Quick Start + gate bolted onto an unpruned textbook body — the June
Verify line caps remain unmet.
2. **AR5 (loop discipline) is the domain's weakest muscle.** Only v2.4+/Pocock/orchestrator-
generation skills state iteration caps or stop conditions. ~20 skills have a "re-run until
clean" instruction with no cap; ~25 have none at all. A single sentence pattern ("max N
fix-rerun cycles, then escalate") would lift 8 skills sitting at 78 into HARNESS-READY.
3. **AR1 (goal intake) missing from tool-rich skills.** The wiring epidemic was fixed (AR3
median is now 2), but most wired skills run on whatever input arrives — no forcing
questions, no refusal on vague goals. The intake patterns already exist in-house
(workflow-builder, grill-me, zero-hallucination-coder) and just need porting.
4. **Gates exist but aren't binding.** Many skills *describe* a validator yet don't
*require* its exit code before proceeding (llm-wiki "periodic" lint, karpathy-coder
warn-only hook, docker/helm validate-steps without loop closure).
5. **Orphan scripts persist even in KEEP skills** — new cases surfaced: **ship-gate's
`ship_gate_scanner.py` (~1230 LOC, full exit-code contract, never mentioned in SKILL.md —
the model is told to scan manually)**, agenthub's `dry_run.py`, interview-system-designer's
3 root scripts, secrets-vault-manager's 3 unwired tools, handoff's 3 tools.
6. **Dead cross-references survived the phantom-path sweep** because they're skill-name
table refs, not file paths: env-secrets-manager (5 dead), sql-database-assistant
(`observability-platform`), llm-wiki (4 phantom related skills), focused-fix (`scope` +
`superpowers:*` as REQUIRED).
7. **Dual-published dedupe (June finding #4) not executed.** All 4 pairs remain
byte-identical (no divergence yet); trap: bundle copies' Quick Starts reference the
*standalone* paths, so naive deletion of standalone copies breaks the bundle's own docs.
8. **Database-trio merge (June finding #3a) not executed** — database-designer got wired
instead of merged; database-schema-designer remains the domain's worst skill (PROSE-ONLY,
broken seed example intact).
### New defects found this audit
- ship-gate: orphaned scanner + category-table drift (SKILL.md 84 vs checks.md 89) — its
KEEP contract now FAILS.
- write-a-skill: fails its own checklist runner (141 lines vs its <100 rule; exit 1)
the meta-skill doesn't dogfood.
- autoresearch loop sub-skill: instructs stale `CronCreate`/`CronDelete` tool names and a
10-min interval that conflicts with the current hourly-minimum trigger surface — broken
as written.
- claude-coach: zero progress on all 3 June defects (conflicting `Version: 1.0.0` /
`version: 2.9.0` still parses ambiguously).
- workflow-builder: `validate_workflow.py --sample` exits 1 *by design* (intentionally
broken sample) — the June KEEP criterion assumed 0; needs one documenting sentence.
- database-schema-designer: broken seed example at L154 persists (explicit June Verify item).
- collab-proof: untranslated Korean rubric phrases from the upstream Vela source.
- terraform-patterns: `./scripts/convert.sh` invocation resolves only from repo root; repo
version `2.9.0` leaked into an Infracost policy example.
- statistical-analyst: no H1 heading; "You are an expert…" opener also in
data-quality-auditor.
- skill-tester's `skill_validator.py` still penalizes new-style <100-line skills (scores
self-eval "POOR 33.3") despite the doc-level scope note.
- No broken scripts: all ~60 `--help`/`--sample`/pipe invocations exited per contract
(non-zero only where documented).
### Top-10 harness-ready exemplars
**agenthub (12)**, **autoresearch-agent (12)**, **chaos-engineering (12)**,
**grill-with-docs (12)**, **collab-proof (11)**, **feature-flags-architect (11)**,
**spec-driven-workflow (11)**, **workflow-builder (11)**, **rag-architect (11)**,
**focused-fix (11)**. Honorable mentions at 10: slo-architect, ship-gate, grill-me,
zero-hallucination-coder, write-a-skill — each one small fix from exemplar status.
### Highest-leverage next PRs
1. Wire ship-gate's scanner + fix its category table.
2. One-sentence loop-cap sweep across the eight 78-point LOOP-CAPABLE skills
(claude-coach, data-quality-auditor, universal-scraping-architect, docker-development,
helm-chart-builder, mcp-server-builder, tech-debt-tracker, skill-security-auditor) —
the cheapest path to ~30 HARNESS-READY.
3. Execute the deferred structural verdicts: dedupe the 4 dual-published pairs, merge/retire
the database trio, fix claude-coach.

View file

@ -0,0 +1,88 @@
# Improvement fields — what to fix, per field, across both engineering domains
The per-skill line items live in [engineering.md](engineering.md) and
[engineering-team.md](engineering-team.md). This file rolls them up into the **eight
fields** where investment moves the most skills, ordered by leverage (skills lifted per
unit of work). Distribution today, 115 distinct skills across both domains:
**26 HARNESS-READY · 39 LOOP-CAPABLE · 43 TOOL-ONLY · 7 PROSE-ONLY.**
## Field 1 — Loop discipline (AR5): the single biggest lever
~45 skills either say "re-run until clean" with no cap or have no retry/stop language at
all. The fix is one standardized block per skill:
> Remediate → re-run `<tool>` → exit 0 required. Max **3** cycles; a repeat failure for the
> same reason after 2 attempts means a structural assumption is wrong — stop and escalate
> with the evidence log.
- Cheapest wins (already 78 points, one sentence from HARNESS-READY):
claude-coach, data-quality-auditor, universal-scraping-architect, docker-development,
helm-chart-builder, mcp-server-builder, tech-debt-tracker, skill-security-auditor
(engineering/); ai-security, threat-detection, cloud-security, incident-response,
senior-secops, red-team, security-pen-testing (engineering-team/).
- Estimated movement: **~15 skills → HARNESS-READY** from this field alone.
## Field 2 — Goal intake (AR1): refuse to run on fuzz
40+ skills accept any input silently. Three proven in-house patterns to propagate:
1. Decision-engine refusal (senior-fullstack/frontend/backend): refuse when required
inputs are missing, list them.
2. Forcing-question library with recommended answers (grill-me, the fork-orchestrators,
zero-hallucination-coder).
3. Exit-code intake gates (agent-harness `goal_compiler.py` exits 3 on vague goals).
Priority targets: every security skill (authorization scope!), ci-cd-pipeline-builder,
docker-development, helm-chart-builder, dependency-auditor, sql-database-assistant.
## Field 3 — Binding verification (AR4): described ≠ required
Validators exist but their exit codes aren't load-bearing. Convert "run the validator"
into "the workflow does not proceed past step N until `<cmd>` exits 0":
- llm-wiki (`lint_wiki.py` is "periodic"), karpathy-coder (warn-only pre-commit),
docker/helm (validate steps without loop closure), senior-architect ("document decision"
instead of "re-run dependency_analyzer, assert circular=0"), cloud architects (no
`cfn-lint` / `az bicep build` / terraform-validate gates), api-test-suite-builder
(generated suite never executed).
## Field 4 — Orphan and mismatched tooling (AR3): the recurring A3 debt
- **New P1:** senior-data-engineer's documented CLI doesn't match the shipped argparse
(`--checks`/`--input` don't exist; `etl_performance_optimizer.py` missing).
- **Worst orphan:** ship-gate's `ship_gate_scanner.py` (~1230 LOC, full exit-code contract,
never mentioned in its SKILL.md).
- Others: agenthub `dry_run.py`, interview-system-designer (3 root scripts),
secrets-vault-manager (3), engineering/handoff (3), spec-driven-workflow (pathless CLIs).
- Prevention: extend CI gate G2 to assert every `scripts/*.py` basename appears in its
SKILL.md (the harness manifests already record `wired: true/false` per tool — a
one-line CI check over the manifests catches this class forever).
## Field 5 — Close-out & state (AR6): make "done" an artifact
Most skills just end. Adopt tc-tracker's handoff block or the agent-harness close contract
(`close` refuses while unverified; emits evidence log + waivers). Targets: performance-
profiler (before/after numbers as required artifact), senior-prompt-engineer (persist eval
baseline), monorepo-navigator (workspace map artifact), all nine un-upgraded senior-* roles.
## Field 6 — Structural verdicts deferred since June
1. Dedupe the 4 dual-published pairs (slo/chaos/k8s/flags) — mind the trap: bundle Quick
Starts reference standalone paths.
2. Merge/retire the database trio (database-schema-designer is still PROSE-ONLY with the
broken L154 seed example).
3. Rebuild the three engineering-team PROSE-ONLY dumps: stripe-integration-expert,
email-template-builder, tech-stack-evaluator.
4. claude-coach's three June defects (dup frontmatter, README paste, unwired classifier).
## Field 7 — Freshness & correctness spot-fixes
- senior-ml-engineer: 12 stale-model hits (2024 GPT-4/Claude-3 pricing as current).
- senior-qa: msw v1 API + `upload-artifact@v3`.
- autoresearch loop sub-skill: stale `CronCreate`/`CronDelete` tool names + interval below
the current hourly trigger minimum.
- stripe-integration-expert: pinned `apiVersion: "2024-04-10"`.
- collab-proof: untranslated Korean rubric phrases.
- Registry drift: named-persona-adversarial-review in zero indexes; engineering-team
CLAUDE.md lists 8 phantom script names; ship-gate table 84 vs 89.
- write-a-skill fails its own checklist runner (dogfooding gap).
## Field 8 — Role-skill loop template (engineering-team's bi

View file

@ -0,0 +1,85 @@
# Research digest: agent harnesses & agentic loops, 20252026 best practice
Compiled 2026-07-03 from web-verified sources. This digest informed the AR rubric
([RUBRIC.md](RUBRIC.md)) and the `engineering/agent-harness` skill's design; the full
per-source treatment lives in that skill's `references/` (3 docs, 21 citations).
## 1. The canonical loop
- **Anthropic, "Building Effective Agents" (Schluntz & Zhang, Dec 2024)** — workflows
(predefined code paths) vs agents (model directs its own process); patterns: prompt
chaining with gates, routing, parallelization, orchestrator-workers, evaluator-optimizer
("only when clear evaluation criteria exist"). Start simple; stopping conditions mandatory.
- **Anthropic, Claude Agent SDK (Sep 2025)** — the loop is **gather context → take action →
verify work → repeat**; filesystem as context store; verification ladder: rules-based >
visual > LLM-as-judge.
- **Anthropic, multi-agent research system (Jun 2025)** — subagent specs need objective,
output format, tool guidance, and boundaries; effort scaled by rule (simple = 1 agent,
310 calls) because early agents "spawned 50 subagents for simple queries".
- **Anthropic, "Effective harnesses for long-running agents" (Nov 2025)** — initializer
expands the goal into `feature-list.json` (description + acceptance criteria + status);
a worker wakes repeatedly, one feature per fresh-context session; all state on disk/git.
- **Anthropic, Agent Skills (Oct 2025)** — progressive disclosure (metadata → SKILL.md →
files on demand); deterministic scripts for anything reliably automatable; build skills
from observed agent failures.
## 2. Verification discipline
- **Jason Wei, "verifier's law" (Jul 2025)** — training/iterating AI on a task is
proportional to its verifiability; invest in checks before agents.
- **SWE-agent (NeurIPS 2024)** — the highest-value guardrail was a linter rejecting invalid
edits at write time; agents fail when the environment gives no feedback.
- **SWE-bench Verified (OpenAI 2024)** — even benchmark tests were too noisy without human
validation; checks need declared reliability classes.
- **Claude Code best practices (Cherny, Apr 2025)** — strongest loop is test-driven: write
the check first, confirm it fails, iterate against it.
- **Reflexion (Shinn 2023) + Huang et al. (ICLR 2024)** — self-critique helps only when
grounded in external feedback; intrinsic self-correction often degrades answers.
⇒ **Deterministic validators are the primary gate; LLM-as-judge is a fallback.**
- **Anthropic reward-hacking research (Nov 2025)** — agents that game their checks
generalize to worse behavior ⇒ the worker must never adjudicate or modify its own gates.
## 3. Loop patterns in production
- **Ralph Wiggum loop (Huntley, Jul 2025; now an official Claude Code plugin)** — same
prompt to a fresh-context agent in a `while true` loop; filesystem + TODO + git as memory.
Fresh context each iteration is the point; caps and completion criteria are added by
practice.
- **Cognition, "Don't Build Multi-Agents" (Jun 2025)** — conflicting parallel decisions are
the dominant multi-agent failure ⇒ **fan out readers/judges, serialize writers**.
- Caps as runtime errors: OpenAI Agents SDK `max_turns` / guardrail tripwires; LangGraph
`recursion_limit`; Anthropic effort budgets.
## 4. State + memory
- Single JSON state file, atomic writes, schema version; narrative handoff separate from
machine state; git as checkpoint layer; compaction with explicit preserve-lists
(Anthropic context-engineering, Sep 2025; LangGraph checkpointers).
## 5. Failure modes → mitigations
| Failure | Mitigation |
|---|---|
| Infinite loops / runaway effort | Triple cap: iterations, wall-clock, budget — breach = terminal state, never silent |
| Verification theater / reward hacking | Gates read-only to the worker; controller re-runs checks itself; diff-scan for edits to test/gate paths |
| Goal drift / conflicting decisions | Single-writer rule; full-context handoffs |
| Context rot / silent truncation | Fresh-context iterations against durable disk state |
## 6. Manifest designs (goals → skills → verifications)
- **AGENTS.md** (agents.md, Aug 2025; Agentic AI Foundation / Linux Foundation, Dec 2025) —
prose manifest for "how to build and verify here".
- **feature-list.json** (Anthropic long-running harness) — the closest published
goal→tasks→verification manifest.
- **MCP** — declared tool registries as the harness's action space.
- GitHub Agentic Workflows / claude-code-action — declarative agent jobs with permissions +
tool allowlists.
## Consensus (what this repo now implements)
Compile goals into explicit task lists with acceptance criteria; run stateless
fresh-context iterations against durable disk/git state; gate every promotion on
deterministic, agent-untouchable checks; serialize writes, parallelize reads; cap
everything; declare the goal→skill→verification mapping in a per-domain manifest.
Implemented as `engineering/agent-harness` (manifest builder + goal compiler + loop
controller, 18 committed domain manifests).

View file

@ -0,0 +1,123 @@
# Master report — Product & PM agentic-loop audit + domain harness upgrade
**Audited:** 2026-07-03 · **Branch:** `claude/pm-audit-agentic-loops-jxurlq` ·
**Scope:** both product/project domains — `product-team/` (17 skills) and
`project-management/` (9 skills) — deep-audited on quality AND scored on the
**agentic-readiness** rubric established by the engineering audit
([../engineering-agentic-2026-07/](../engineering-agentic-2026-07/00-MASTER.md)).
Plus: both domains upgraded into **agent harnesses** — fork-orchestrators with
deterministic routers, reusable loops, machine-checkable verification gates, and
integration with the repo-wide `engineering/agent-harness` framework.
**Method:** (1) two parallel deep-dive agents read every SKILL.md, smoke-tested all 31
scripts, and cross-referenced agents/commands/manifests; (2) one explorer mapped the
repo's harness conventions (loop-library contract, agent-harness state machine,
fork-orchestrator pattern) so the upgrade reuses rather than reinvents; (3) one research
agent web-verified the 20242026 PM/product/harness canon
([research-digest.md](research-digest.md)).
---
## 1. The two questions
The June 2026 audit asked: does each skill earn its context window? This audit asks the
engineering follow-up question for these two domains: **can an agent pick up a goal here
and drive it to a verified close?** — and additionally: **what should these domains
teach that the 20242026 canon now demands?**
([improvement-fields.md](improvement-fields.md) answers the second.)
## 2. Combined scorecard (26 skills, post-PR)
| Class | product-team | project-management | Total | Meaning |
|---|---|---|---|---|
| **HARNESS-READY** (≥9, AR4≥1, AR5≥1) | 2 | 1 | **3** | An agent can loop this today |
| **LOOP-CAPABLE** (68) | 4 | 4 | **8** | One or two additions away |
| **TOOL-ONLY** (35) | 11 | 3 | **14** | Good tools, no loop spine |
| **PROSE-ONLY** (02) | 0 | 1 | **1** | Needs structural rebuild |
Pre-PR both domain routers were PROSE-ONLY (score ≤ 1) and neither domain had a single
`context: fork`, forcing question, iteration cap, or `/cs:*` command — they predate every
v2.8+ convention. The weakest dimensions mirror engineering exactly: **AR5 loop
discipline** (zero caps anywhere pre-PR) and **AR1 goal intake** (most skills accept any
input silently).
## 3. The three biggest findings
1. **The MCP↔analytics gap (project-management).** The domain bundles a live Jira MCP
and ships real analytics tools, with no data path between them — sprint health and
velocity ran on hand-typed JSON. Fixed: `jira_snapshot_bridge.py` converts saved
`searchJiraIssuesUsingJql` results into the scrum-master schema (verified end-to-end
into velocity_analyzer) and computes the four Kanban-Guide-2025 flow metrics + seeded
Monte Carlo forecasts the domain never had.
2. **Verification exists but nothing binds it (both domains).** spec-to-repo's
validator, code-to-prd's golden outputs, scrum-master's pinned fixtures,
atlassian-admin's 7 VERIFY steps — good gates, all optional, none looped. Fixed at
the orchestration layer: plans are gated before execution and closes are refused
(exit 4) while tasks are unverified/unwaived; per-skill binding is follow-up F3.
3. **The canon moved (both domains).** No continuous-discovery cadence, no OST
discipline, no AI-feature evals, no flow metrics, no probabilistic forecasting, no
agentic-delegation governance, and 60 of 64 reference files cite zero sources. This
PR ships the two highest-leverage tool fields per domain plus six cited reference
docs; the remaining 14 fields are enumerated with tool specs in
[improvement-fields.md](improvement-fields.md).
## 4. What this PR ships: two domain harnesses
Both prose routers were rebuilt as `context: fork` orchestrators that plug into
`engineering/agent-harness` (manifests regenerated; both orchestrators now score all
five `agentic_signals`):
**project-management → `pm-skills`** — the *delivery loop*:
`pm_goal_router.py` (8 lanes, exit 0/2/3 — route/ask/refuse) ·
`jira_snapshot_bridge.py` (MCP snapshot → flow metrics | sprint schema; SLE conformance,
aging-WIP alerts, `--forecast` Monte Carlo, refuses thin history) ·
`delivery_loop_gate.py` (delegation governance G1G6: human owner, reviewer for agent
tasks, machine-checkable acceptance, evidence-before-done, close refusal,
exhausted-budget-is-escalation). Five reusable PM loops (sprint-flow, health,
retro-action, RAID-hygiene, comms) documented with terminal states. Agent
`cs-pm-orchestrator`; commands `/cs:pm`, `/cs:grill-pm`, `/cs:pm-loop`.
**product-team → `product-skills`** — the *discovery loop*:
`product_goal_router.py` (16 lanes incl. the 4 standalone plugins) ·
`discovery_cadence_tracker.py` (Torres weekly-habit scoring: streak, coverage, outcome
linkage, test throughput → health 0100 with named gaps and a `next_loop_action`) ·
`ost_linter.py` (O1O5: measurable outcome root, needs-not-features, ≥2 solutions per
target, tests per solution, no orphan solutions — exit 2 blocks the tree from driving a
roadmap). Graduation stop-states hand validated assumptions to experiment-designer/PRD.
Agent `cs-product-orchestrator`; commands `/cs:product`, `/cs:grill-product`,
`/cs:product-loop`.
Also fixed: the two CLI-noncompliant product tools (`user_story_generator.py`,
`persona_generator.py` — real argparse `--help`, seeded determinism, backward-compatible
positionals); domain CLAUDE.md counters; plugin manifests + marketplace descriptions.
All 8 new/changed tools pass `--help` and `--sample`; fixtures pinned
(`expected_flow_metrics.json`; sample OST with two planted violations). Every design
decision traces to the loop-library contract, the agent-harness invariants (locked
gates, evidence-before-status, budgets-as-terminal-states), and the cited canon.
## 5. Per-domain reports
- [product-team.md](product-team.md) — 17 skills, AR table, 7 domain findings,
executable verification criteria.
- [project-management.md](project-management.md) — 9 skills, AR table, 6 domain
findings, executable verification criteria.
- [improvement-fields.md](improvement-fields.md) — the per-field improvement rollup
(11 cross-domain/delivery/product fields shipped or specced + documentation debt).
- [research-digest.md](research-digest.md) — the web-verified 20242026 canon.
- [RUBRIC.md](RUBRIC.md) — the AR rubric as applied here.
## 6. Recommended follow-up PRs (in leverage order)
1. **Loop-cap sweep** — one-sentence caps in scrum-master, jira-expert, code-to-prd,
research-summarizer (~4 skills → HARNESS-READY).
2. **Bind the gates** — make spec-to-repo's validator and code-to-prd's goldens
*required*; name the bridge in scrum-master/senior-pm SKILL.mds.
3. **AI-evals tool** (Fh) — eval-spec linter + kappa calculator; the single
most-demanded missing PM competency.
4. **Path-B completion** — meeting-analyzer scripts (its spec is deterministic math),
team-communications linter, product-discovery references.
5. **Documentation truth** — product-team README (3 conflicting counts, 9 broken paths),
project-management legacy trio, citation back-fill (F9/F10).
6. **Remaining improvement fields** — DORA/EBM/pre-mortem/RACI (delivery); NSM/PLG
bands/WSJF/ODI/taxonomy linter (product), per the specs in improvement-fields.md.

View file

@ -0,0 +1,30 @@
# Agentic-Readiness Rubric (AR v1) — as applied to product-team + project-management
Audit date: 2026-07-03 · Branch: `claude/pm-audit-agentic-loops-jxurlq`
This audit applies the **same AR v1 rubric** established by the engineering agentic audit
([../engineering-agentic-2026-07/RUBRIC.md](../engineering-agentic-2026-07/RUBRIC.md)) —
six dimensions scored 02, answering: **can an agent pick this skill up with a goal and
drive it to a verified close?**
| # | Dimension | 2 means |
|---|---|---|
| AR1 | Goal intake | Forcing questions / intake tool / refuses vague input (exit-code gate) |
| AR2 | Task decomposition | Explicit planning step or tool whose output the workflow consumes |
| AR3 | Deterministic execution | Exact runnable CLIs; output consumed by a named next step |
| AR4 | Verification | Machine-checkable gate the workflow REQUIRES before proceeding |
| AR5 | Loop discipline | Iteration caps, stop conditions, escalation thresholds |
| AR6 | Close-out | Definition of done + state persistence or handoff artifact |
Classes: **HARNESS-READY** (total ≥ 9 AND AR4 ≥ 1 AND AR5 ≥ 1) · **LOOP-CAPABLE** (68,
or ≥ 9 failing the AR4/AR5 gate) · **TOOL-ONLY** (35) · **PROSE-ONLY** (02).
Baseline reference: the June 2026 quality audit of these domains lives at
[../newgen-2026-06/product-pm.md](../newgen-2026-06/product-pm.md); this audit scores the
new agentic dimension and ships the domain harness layer that the scores motivated.
Executable enforcement: the regenerated domain manifests
(`engineering/agent-harness/skills/agent-harness/assets/harnesses/{product-team,project-management}.json`)
record per-skill `agentic_signals`; the two new domain orchestrators (`product-skills`,
`pm-skills`) enforce AR1 (routers with exit-code gates), AR4 (delivery gate / OST linter),
and AR5/AR6 (harness budgets + close refusal) at run time.

View file

@ -0,0 +1,66 @@
# Improvement fields — where investment moves each domain most
Research-backed rollup (sources in [research-digest.md](research-digest.md)) of the
fields where the 20242026 canon moved past the two domains' current coverage, ordered
by leverage. Fields marked ✅ shipped in this PR; the rest are the follow-up work list,
each with what a deterministic stdlib tool computes.
## Cross-domain (the harness fields)
- **F1 — Loop discipline (AR5)***(orchestrator layer)* / open *(per skill)*. Both
domains had zero iteration caps or stop conditions. Shipped: the two orchestrators
carry budgets (3 attempts/task, 12 iterations/goal), named terminal states, and close
refusal. Open: the one-sentence cap pattern ("max N fix-rerun cycles, then escalate")
ported into scrum-master, jira-expert, code-to-prd, research-summarizer — the four
skills one sentence away from HARNESS-READY.
- **F2 — Goal intake (AR1)***(routers)* / open *(per skill)*. Both routers refuse
fuzz with exit codes (2/3); grill commands lock decisions before execution. Open:
refuse-on-missing-inputs blocks in the 14 tool-rich skills that still accept anything.
- **F3 — Bind existing gates** — open, cheapest wins: spec-to-repo's validator and
code-to-prd's goldens exist but aren't *required*; scrum-master/senior-pm SKILL.mds
should name the bridge as their data path (one line each).
## Project-management (delivery) fields
| # | Field | Status | Deterministic tool |
|---|---|---|---|
| F4 | Four Kanban flow metrics + SLE + aging-WIP alerts (Kanban Guide 2025) | ✅ `jira_snapshot_bridge.py --to flow` | WIP, throughput, cycle p50/85/95, work-item age, SLE conformance |
| F5 | Monte Carlo probabilistic forecasting (Vacanti; replaces story-point dates) | ✅ `--forecast N` (seeded; refuses < 10 items) | p50/70/85/95 week ranges |
| F6 | Agentic delegation governance (Linear/Rovo model) | ✅ `delivery_loop_gate.py` | owner/reviewer/acceptance/evidence/close rules G1G6 |
| Fa | DORA 2025 archetypes + AI-amplifier capabilities check | open | four keys from deploy/incident logs → archetype + enabling-capabilities score |
| Fb | EBM (Scrum.org) four Key Value Areas scorecard | open | map existing metrics → KVA coverage, flag empty CV/UV |
| Fc | Pre-mortem processor (Klein) + RAID hygiene linter | open (playbook documents the loops) | cluster failure reasons → owned risk entries; staleness/owner/mitigation lint |
| Fd | Derived project health vs self-reported RAG ("watermelon" diff) | partial (senior-pm dashboard + bridge signals) | composite from schedule variance, aging WIP, scope churn — diffed against RAG |
| Fe | Async-first meeting audit (GitLab canon) | open | calendar export → async-convertibility classes, recoverable hours |
| Ff | Agent-readiness audit of Jira hygiene (Rovo-era) | open | field completeness %, acceptance-criteria presence, stale statuses → delegation-readiness score |
| Fg | RACI validator | open | exactly-one-A, ≥1 R, overload histogram |
## Product-team fields
| # | Field | Status | Deterministic tool |
|---|---|---|---|
| F7 | Continuous-discovery cadence (Torres weekly habit) | ✅ `discovery_cadence_tracker.py` | streak, coverage, outcome linkage, test throughput → health 0100 |
| F8 | Opportunity Solution Tree structural linting | ✅ `ost_linter.py` | rules O1O5 (measurable root, needs-not-features, ≥2 solutions, tests, no orphans) |
| Fh | **AI-feature evals as the PRD quality contract** — the single biggest gap per every 20252026 source | reference shipped (`ai_product_evals.md`); tool open | eval-spec linter: golden-set floors, rubric pass-criteria, guardrail SLOs; Cohen's kappa on grader agreement |
| Fi | WSJF / cost-of-delay with rank-stability sensitivity (brackets RICE) | reference shipped (`product_operating_model.md`); tool open | CoD/duration ranking; ±1-step perturbation flags rank flips |
| Fj | Opportunity scoring (Ulwick ODI importancesatisfaction) | open | Opp Score per outcome statement from survey CSV |
| Fk | North Star Metric validator + input tree (Amplitude) | open | leading/value/not-vanity checks; input→NSM correlation |
| Fl | PLG funnel benchmark bands (ProductLed/OpenView) | open | stage conversion vs calibrated bands → weakest-stage verdict |
| Fm | Product operating model maturity (Cagan *Transformed*) | open | questionnaire → per-principle 0100 gap list |
| Fn | Event-taxonomy / tracking-plan linter (PostHog-era) | open | snake_case, verb allowlist, near-duplicate detection |
| Fo | JTBD switch-interview force coder (Moesta) | open | four-forces lexicon coding, force balance per interview |
| Fp | Story-map validator (Patton) | open | every story on the backbone; slices span end-to-end |
| Fq | Model-card completeness checker (Mitchell et al.) | open | nine canonical sections → completeness % |
## Documentation-debt fields
- **F9 — Source citations**: 60 of 64 pre-PR reference files across both domains cite
zero sources (six new references cite 67 each). Back-fill priority: scrum-master and
senior-pm references (the canon exists — Vacanti, Kanban Guide, DORA, PMBOK).
- **F10 — Counter/path truth**: product-team README (3 conflicting counts, 9 broken
paths), project-management legacy trio (README / IMPLEMENTATION_SUMMARY /
REAL_WORLD_SCENARIO all say "6 skills"), `.codex/instructions.md` broken paths in both
domains. Domain CLAUDE.mds fixed this PR.
- **F11 — Path-B completion**: meeting-analyzer (0 scripts/refs/assets — its own spec is
deterministic math), team-communications (0 scripts, 155 lines of references),
product-discovery (1 reference), 8 asset-less product skills.

View file

@ -0,0 +1,98 @@
# Domain audit: product-team/ — deep audit + agentic readiness
Audited: 2026-07-03 · 17 skills (13 under `skills/` incl. the router, 4 standalone
plugins) · 19 Python tools (all 19 pass `--help`, functional smoke tests on
sample_size_calculator and hig_checker reproduce correct math) · 5 agents + 1 persona ·
8 slash commands · 5 plugins (all `check_plugin_json.py`-clean).
Rubric: [RUBRIC.md](RUBRIC.md). Method: full SKILL.md reads, script smoke tests,
agent/command cross-referencing, counter verification.
## Summary
**Headline finding (now fixed in this PR): no orchestrator, no loop layer.** Before this
PR the domain had zero `context: fork`, zero forcing-question libraries, no `/cs:*`
router/grill commands, and its "router" (`product-skills`, 61 lines) shipped no tools —
the domain predated every v2.8+ convention. Three skills (spec-to-repo, code-to-prd,
research-summarizer) independently invented verification loops; nothing unified them.
**Agentic-readiness distribution (17 skills, post-PR):** HARNESS-READY **2**
(product-skills upgraded 1→12; spec-to-repo) · LOOP-CAPABLE **4** (product-manager-toolkit,
apple-hig-expert, code-to-prd, research-summarizer) · TOOL-ONLY **11** · PROSE-ONLY **0**.
Weakest dimensions domain-wide: **AR5 loop discipline** (only spec-to-repo has any
retry/stop language) and **AR1 goal intake** (14 of 17 accept any input silently).
## Per-skill table
Scores AR1·AR2·AR3·AR4·AR5·AR6 (post-PR where this PR changed the skill).
| Skill | AR1-6 | Tot | Class | Top improvement |
|---|---|---|---|---|
| skills/product-skills (orchestrator) | 2·2·2·2·2·2 | 12 | HR | (upgraded this PR: was a 61-line prose router, PROSE-ONLY) |
| skills/spec-to-repo | 1·2·2·2·1·1 | 9 | HR | Wire to an agent + command (currently orphaned from both) |
| skills/product-manager-toolkit | 1·1·2·1·0·1 | 6 | LC | Make the PRD checklist a blocking gate; add WSJF/CoD lane + eval-spec PRD section |
| code-to-prd (standalone) | 1·2·2·2·0·2 | 9 | LC (AR5 gate) | One sentence: max 2 analyze-fix cycles vs golden `expected_outputs/`, then escalate |
| research-summarizer (standalone) | 1·1·2·2·0·1 | 7 | LC | Cap the Verification Loop (2 re-extraction passes) → instant HR |
| apple-hig-expert (standalone) | 1·0·2·2·0·1 | 6 | LC | Numbered workflow; move `templates/``assets/`; add fix-recheck cap |
| skills/experiment-designer | 1·1·2·1·0·0 | 5 | TO | Make sample-size output a blocking gate on any test recommendation; 423 words is thin |
| skills/ux-researcher-designer | 0·1·2·1·0·1 | 5 | TO | (fake `--help` + unseeded RNG fixed this PR) Validation checklists → exit gates |
| skills/ui-design-system | 0·1·2·1·0·1 | 5 | TO | Add command; disambiguate vs markdown-html/design-system |
| skills/saas-scaffolder | 0·1·2·1·0·1 | 5 | TO | Make the 33-item checklist machine-checkable (reuse spec-to-repo's validator); dedupe references pair |
| agile-product-owner (standalone) | 0·1·2·1·0·1 | 5 | TO | (fake `--help` fixed this PR) INVEST checklist → gate |
| skills/product-analytics | 0·1·2·1·0·0 | 4 | TO | NSM validator + benchmark bands (see improvement-fields F7/F8); anti-patterns table exists, gate doesn't |
| skills/landing-page-generator | 0·1·2·1·0·0 | 4 | TO | Add `distinct_from` vs marketing/landing; render-check gate on emitted TSX |
| skills/product-strategist | 0·1·2·0·0·1 | 4 | TO | Alignment score exists but nothing requires it; outcome-vs-output OKR lint |
| skills/competitive-teardown | 0·1·2·0·0·0 | 3 | TO | Data-verification discipline (it synthesizes scraped claims with no source gate) |
| skills/product-discovery | 0·1·2·0·0·0 | 3 | TO | Wire to the new discovery loop (orchestrator now provides tracker + OST linter); 1 reference, no agent/command |
| skills/roadmap-communicator | 0·0·2·1·0·0 | 3 | TO | Merge-or-disambiguate vs `/changelog` + engineering/changelog-generator; 367 words |
## Domain-level findings
1. **Orchestration gap (fixed this PR).** `product-skills` is now a `context: fork`
orchestrator with a deterministic 16-lane router (`product_goal_router.py`, exit
0/2/3), a recurring discovery loop with two machine gates
(`discovery_cadence_tracker.py`, `ost_linter.py`), a forcing-question library, and
agent-harness integration. Agent `cs-product-orchestrator` + `/cs:product`,
`/cs:grill-product`, `/cs:product-loop` commands added.
2. **References cite no sources: 39 of 43 reference files contain zero URLs/citations.**
Only apple-hig-expert (3/3) and research-summarizer (1/2) meet the ≥5-sources bar. The
3 new orchestrator references cite 7 sources each; the other 39 remain open work.
3. **Two tools faked their `--help` (fixed this PR).** `user_story_generator.py` and
`persona_generator.py` exited 0 while ignoring the flag and running demos;
persona_generator was additionally non-deterministic (unseeded `random.choice`). Both
now use argparse; personas are seeded (default 42).
4. **Stale/contradictory counters + broken paths (open).** product-team/README.md holds 3
mutually inconsistent counts and 9 broken Quick Start paths (missing `skills/`
segment); `.codex/instructions.md` has 3 broken paths; CLAUDE.md says "13 skills" then
lists 16, claims 17 tools (actual 19). The domain plugin.json description claims
skills the bundle doesn't contain. CLAUDE.md counters updated this PR; README overhaul
is follow-up F10.
5. **Unmanaged overlaps (open):** landing-page-generator ↔ `marketing/landing`;
saas-scaffolder ↔ spec-to-repo; roadmap-communicator ↔ `/changelog` +
`engineering/changelog-generator`; ui-design-system ↔ `markdown-html/design-system`.
Only research-summarizer ships a "Distinct From" section. The orchestrator's routing
table now provides partial disambiguation; per-skill `distinct_from` notes remain.
6. **8 of 13 bundled skills ship 0 assets** despite the repo's template-heavy principle;
apple-hig-expert uses a nonstandard `templates/` dir.
7. **Agent/command coverage holes (open):** 6 skills map to no cs-* agent
(product-discovery, roadmap-communicator, spec-to-repo, code-to-prd, apple-hig-expert,
research-summarizer); 10 have no slash command. The orchestrator router reaches all 17
lanes, which mitigates but does not close this.
## Verification criteria (executable)
- **product-skills (orchestrator):** `python3 product-team/skills/product-skills/scripts/product_goal_router.py --sample`
exits 0 and routes to `product-discovery`; `--text "hello"` exits 3;
`discovery_cadence_tracker.py --input assets/sample_discovery_log.json` exits 0 with
`health_score` 62.0 and verdict `AT-RISK`; `ost_linter.py --input assets/sample_ost.json`
exits 2 with exactly one O2 and one O4 violation; `--sample` variants all exit 0.
- **spec-to-repo:** `validate_project.py --strict` on a scaffolded repo exits 0 before
the workflow may report done (existing contract, holds).
- **code-to-prd:** analyzer output diffs clean against `expected_outputs/` goldens
(existing contract, holds).
- **persona_generator (fixed):** `--help` prints argparse usage (not a demo); two `json`
runs produce byte-identical output.
- **user_story_generator (fixed):** `--help` prints argparse usage; `sprint 30` still
plans a 30-point sprint (backward-compatible positional).
- **Manifest truth:** `harness_manifest_builder.py --domain product-team --no-timestamp`
produces a diff-clean `product-team.json` with `product-skills` scoring all five
`agentic_signals` true and 3 wired, sample-supporting tools.

View file

@ -0,0 +1,102 @@
# Domain audit: project-management/ — deep audit + agentic readiness
Audited: 2026-07-03 · 9 skills · 12 Python tools pre-PR (all pass `--help`; end-to-end
runs of velocity_analyzer and project_health_dashboard reproduce documented fixtures),
15 post-PR · 1 agent pre-PR, 2 post · 3 commands pre-PR (+`/sprint-plan` generic), 6
post · plugin.json valid (`["./skills"]` canonical form).
Rubric: [RUBRIC.md](RUBRIC.md). Method: full SKILL.md reads, script smoke tests, MCP
tool-reference grepping, counter verification.
## Summary
**Headline finding #1 (fixed this PR): the MCP↔analytics gap.** The domain bundles a
live Atlassian Remote MCP (`.mcp.json`) and disciplined tool documentation
(`references/atlassian-mcp-tools.md`, verified live 2026-06-10), yet its two analytics
skills (senior-pm, scrum-master) had **zero** MCP references — nothing connected
`searchJiraIssuesUsingJql` output to the scripts' input schemas. Every sprint-health or
velocity run required hand-built JSON. `jira_snapshot_bridge.py` closes this: raw MCP
search results → scrum-master sprint schema (verified: piped output runs
velocity_analyzer clean) → plus the four Kanban flow metrics + seeded Monte Carlo
forecasting the domain never had.
**Headline finding #2 (fixed this PR): pre-modern agentics.** Zero `context: fork`, zero
forcing questions, no `/cs:*` namespace, no loop with named terminal states. What existed
was raw material: atlassian-admin's 7 concrete VERIFY steps, scrum-master's
data-sufficiency gates with pinned expected outputs (avg 20.2 pts, health 78.3,
action-item completion 46.7%), confluence/templates' verify-before-proceed steps.
**Agentic-readiness distribution (9 skills, post-PR):** HARNESS-READY **1** (pm-skills
upgraded 1→12) · LOOP-CAPABLE **4** (scrum-master, jira-expert, atlassian-admin,
atlassian-templates) · TOOL-ONLY **3** (senior-pm, confluence-expert, meeting-analyzer) ·
PROSE-ONLY **1** (team-communications).
## Per-skill table
Scores AR1·AR2·AR3·AR4·AR5·AR6 (post-PR where this PR changed the skill).
| Skill | AR1-6 | Tot | Class | Top improvement |
|---|---|---|---|---|
| pm-skills (orchestrator) | 2·2·2·2·2·2 | 12 | HR | (upgraded this PR: was a 50-line prose router, PROSE-ONLY) |
| scrum-master | 1·1·2·2·0·1 | 7 | LC | One sentence: cap re-analysis at 2 passes then escalate → instant HR; consume the bridge (`--to sprint`) instead of hand-built JSON |
| atlassian-admin | 0·1·2·2·0·2 | 7 | LC | Intake gate (refuse without approver named); its VERIFY steps are the domain's best — port the pattern to siblings |
| jira-expert | 0·1·2·2·0·1 | 6 | LC | Cap fix-revalidate cycles at 3; ship a sample workflow JSON asset (users must guess the validator's schema) |
| atlassian-templates | 0·1·2·1·1·1 | 6 | LC | Ship static template assets; expected-output fixture for the scaffolder |
| senior-pm | 0·1·2·1·0·1 | 5 | TO | Consume the bridge's flow output in the health dashboard; make KPI thresholds (on-time > 80% etc.) exit-code gates; portfolio-kpis.md is 32 lines |
| confluence-expert | 0·1·2·1·0·1 | 5 | TO | Make its Verify steps blocking; sample input for content_audit_analyzer |
| meeting-analyzer | 1·1·0·1·0·1 | 4 | TO | Ship the deterministic tools its own prose describes (speaking-ratio, filler counts = exactly the repo's "algorithm over AI" case); 0 scripts/refs/assets |
| team-communications | 1·1·0·0·0·0 | 2 | PO | References are 1565 lines (the skill's premise is "follow the reference exactly"); add a 3P-format linter script |
## Domain-level findings
1. **Orchestration + loop layer (fixed this PR).** `pm-skills` is now a `context: fork`
orchestrator: deterministic 8-lane router (`pm_goal_router.py`), the Jira bridge, and
a delegation-governance gate (`delivery_loop_gate.py` — G1 human owner, G2 reviewer
for agent tasks, G3 machine-checkable acceptance, G4 evidence-before-done, G5 close
refusal, G6 exhausted-budget-is-escalation), all wired to the repo harness
(`assets/harnesses/project-management.json`). Agent `cs-pm-orchestrator` +
`/cs:pm`, `/cs:grill-pm`, `/cs:pm-loop` added. Five reusable PM loops documented in
`references/pm_loop_playbook.md` (sprint-flow, health, retro-action, RAID-hygiene,
comms), each with machine gates and named terminal states.
2. **References cited zero sources (partially fixed).** 0 URLs across all 21 pre-PR
reference files — Schwaber/Sutherland, Vacanti, DORA, Kanban Guide all absent. The 3
new orchestrator references cite 67 sources each; back-filling the other 21 is
follow-up F9.
3. **Stale counters everywhere except plugin.json (open).** README ("6 world-class
skills"), IMPLEMENTATION_SUMMARY ("All 6", references `/mnt/user-data/outputs/` build
paths), REAL_WORLD_SCENARIO ("6 Expert Skills"), cs-project-manager agent ("six
skills"), CLAUDE.md (lists 6 of 9 — meeting-analyzer, team-communications, pm-skills
absent). CLAUDE.md updated this PR; the legacy trio (README /
IMPLEMENTATION_SUMMARY / REAL_WORLD_SCENARIO) should be rewritten or retired (F10).
4. **MCP integration is bimodal (structural, now bridged).** Concrete in 4 skills
(jira-expert 14 refs, confluence-expert 11, atlassian-templates 11, atlassian-admin 4
read-only-correct); zero in the 2 analytics skills. The bridge closes the data path;
the two SKILL.mds should still name it (one line each, F3).
5. **Two contributed skills violate the Path-B contract (open).** meeting-analyzer: zero
scripts/references/assets — its own spec (speaking-time %, filler-word counts) is
deterministic computation the repo mandates be scripted. team-communications: zero
scripts, 4 references totaling 155 lines.
6. **`/sprint-plan` counts against product-team but lives half in this domain** — the
sprint-planning integration pattern in CLAUDE.md calls product-team's
user_story_generator; fine, but the CLAUDE.md example used the old positional CLI
(still works — verified backward-compatible after this PR's argparse fix).
## Verification criteria (executable)
- **pm-skills (orchestrator):** `pm_goal_router.py --sample` exits 0 routing to
`scrum-master`; `--text "audit our jira permissions"` exits 2 (single signal → ask);
`--text "hello world"` exits 3. `jira_snapshot_bridge.py --input
assets/sample_jira_snapshot.json --to flow --forecast 20` exits 0 and matches
`assets/expected_flow_metrics.json` (p50=9, p85=14, p95=16 days; 90.9% SLE conformance;
aging alert on PHX-112; forecast p85 = 10 weeks, sampled over zero-filled observed
weeks); `--to sprint` output runs
`velocity_analyzer.py` to exit 0 (avg 11.8 pts over 4 sprints); a 2-sprint snapshot
exits 5. `delivery_loop_gate.py --sample` exits 0; sample plan passes `--mode plan`
(exit 0) and is refused by `--mode close` (exit 4, T2 in_progress).
- **scrum-master:** existing fixture contract holds — velocity_analyzer on
`assets/sample_sprint_data.json` reports avg 20.2 pts on 6 sprints.
- **atlassian-admin:** each VERIFY step names a concrete check (e.g. `GET
/rest/api/3/user?accountId=... returns "active": false`) — keep as the domain's AR4
exemplar.
- **Manifest truth:** `harness_manifest_builder.py --domain project-management
--no-timestamp` produces a diff-clean `project-management.json` with `pm-skills`
scoring all five `agentic_signals` true and 3 wired, sample-supporting tools.

View file

@ -0,0 +1,76 @@
# Research digest — the 20242026 canon behind this audit
Web-verified 2026-07-03. Full citations inline; this digest is the source layer for
[improvement-fields.md](improvement-fields.md) and the six new reference docs shipped
into the two orchestrators.
## Product management: where the canon moved
1. **Discovery became a weekly operating rhythm.** Torres (*Continuous Discovery
Habits*; producttalk.org/opportunity-solution-trees) reframed discovery as weekly
customer touchpoints anchored to one outcome, with the OST as the structural artifact
and assumption tests (Bland, *Testing Business Ideas*) as the unit of progress.
2. **The org-level frame is the product operating model.** Cagan's *Transformed* (SVPG,
2024): empowered teams, outcomes over output, innovation over predictability — 20
first principles (svpg.com/the-product-operating-model-an-introduction).
3. **Evals are the new PRD for AI features** — the consensus 2025 AI-PM competency:
golden set + rubric + guardrail SLOs before building (Lenny's Newsletter "Beyond vibe
checks"; Braintrust "Evals for PMs"; Aakash Gupta "AI Evals"); model cards for the
buyer-facing half (Mitchell et al., arxiv.org/abs/1810.03993).
4. **Metrics spine: North Star + input tree** (Amplitude, *The North Star Playbook*),
with PLG benchmark bands for verdicts (ProductLed; OpenView: activation median ~17%,
best-in-class 3350%+; free→paid median ~9%, PQL-driven 2530%).
5. **Prioritization is a bracket, not a framework**: RICE (steady state) + WSJF/cost of
delay (Reinertsen; SAFe WSJF + Yip's false-precision critique) + ODI opportunity
scoring (Ulwick). Sensitivity analysis counters the documented WSJF failure mode.
6. **Analytics practice is taxonomy-first** (PostHog product-analytics best practices):
naming discipline and tracking-plan review before any metric above it.
## Project management / delivery: where the canon moved
1. **Flow metrics are mandatory, not optional.** The Kanban Guide (May 2025,
kanbanguides.org) mandates exactly four measures — WIP, throughput, cycle time, work
item age — plus an SLE; age is the leading indicator.
2. **Forecasting went probabilistic.** Vacanti (*Actionable Agile Metrics*; *When Will
It Be Done?*; scrum.org Monte Carlo guidance): sample historical throughput, answer
with p50/70/85/95 ranges, never a date; refuse thin history.
3. **DORA 2025** (dora.dev/dora-report-2025) replaced elite/high/medium/low with seven
team archetypes over eight measures; core finding: AI **amplifies** existing org
strengths/dysfunctions (individual output up ~98% more merged PRs, org delivery flat
without enabling capabilities). SPACE (Forsgren/Storey, ACM Queue) remains the
multi-dimension corrective; EBM (scrum.org) the value-measurement frame.
4. **Risk practice:** Klein's pre-mortem (HBR 2007, ~30% better risk identification);
RAID hygiene as a linting problem; derived health vs self-reported RAG to catch
watermelon projects.
5. **Async-first delivery:** GitLab handbook (handbook.gitlab.com, asynchronous work) —
written 3-question standups 35 min vs 1530 sync; ~37% meeting-hour reduction. Moghe,
*The Async-First Playbook* (2023).
6. **The vendors shipped agentic PM.** Atlassian Rovo GA'd agents in Jira (assignable,
@mentionable, "every action logged and auditable", Teamwork Graph 150B+ connections,
MCP access — atlassian.com/software/rovo; Team '26 coverage, SiliconANGLE 2026-05-06).
Linear shipped the accountability pattern: agent as contributor, **human stays
primary assignee** (linear.app/agents; changelog 2026-03-24).
## Agentic harness design principles (applied in this PR)
1. **Workflows first, agents when needed** — Anthropic, "Building Effective Agents"
(anthropic.com/research/building-effective-agents): prompt chaining, routing,
parallelization, orchestrator-workers, evaluator-optimizer; the last "when there are
clear evaluation criteria and iterative refinement provides measurable value."
→ The routers are workflows; the loops engage only for goals with fresh feedback.
2. **Definition-of-done must be machine-checkable; never trust self-report** — the
plan→act→verify(deterministic)→reflect shape. → G3/G4 in `delivery_loop_gate.py`;
OST linter exit codes; agent-harness's evidence rule.
3. **Budgets and stop conditions are first-class** — max iterations, attempt caps,
escalation on confidence loss; otherwise reflection is infinite retry. → 3/12 caps,
G6, terminal-state taxonomy (loop-library: success, clean no-op, blocked,
approval-required, exhausted, stagnated).
4. **Human accountability stays attached to delegated work** (Linear; Rovo audit
discipline). → G1/G2; no un-reviewed Jira transitions; admin actions are
approval-required states.
5. **Context via structured interfaces, not prompt-stuffing** (Teamwork Graph / MCP;
Anthropic's ACI emphasis). → snapshot-file pattern: every loop iteration is
executable by a fresh session from files.
6. **Evaluator-optimizer pairs with PM-owned evals** — the golden set + rubric IS the
evaluator's criteria; the loop may never edit the gate it is judged by (this repo's
autoresearch locked-evaluator invariant, generalized).

View file

@ -22,7 +22,7 @@ description: "10 product agent skills and plugins for Claude Code, Codex, Gemini
### Claude Code
```
/read product-team/product-manager-toolkit/SKILL.md
/read product-team/skills/product-manager-toolkit/SKILL.md
```
### Codex CLI

View file

@ -0,0 +1,13 @@
{
"name": "agent-harness",
"description": "Turn any domain folder of skills into a bounded agentic loop: a manifest builder inventories a domain's skills/tools/checks, a goal compiler turns a goal into a verifiable task plan (refusing vague goals with forcing questions), and a JSON-backed loop controller drives execute->verify->close with retry caps, controller-run verification (no verification theater), human escalation on exhausted budgets, and a close gate that refuses while any task is unverified. Ships 3 stdlib Python tools, 18 committed per-domain harness manifests + JSON schema, 3 references citing the 2024-2026 agent-harness canon, harness-runner agent + /cs:harness command. Use when an agent should pick up a goal and drive it to a verified close across a domain.",
"version": "1.0.0",
"author": {
"name": "Alireza Rezvani",
"url": "https://alirezarezvani.com"
},
"homepage": "https://github.com/alirezarezvani/claude-skills/tree/main/engineering/agent-harness",
"repository": "https://github.com/alirezarezvani/claude-skills",
"license": "MIT",
"skills": ["./skills/agent-harness"]
}

View file

@ -0,0 +1,33 @@
# agent-harness
Turn any domain folder of this repo into a **bounded agentic loop**: pick up a goal,
compile it into tasks with machine-run verification, execute, verify, retry with caps,
escalate to a human when budgets exhaust, and close only when everything is verified.
```
GOAL → goal_compiler → PLAN → loop_controller: [execute → verify]* → CLOSE
↑ retry ≤ caps, changed approach
└ ESCALATE — never fake success
```
## What ships
| Piece | Purpose |
|---|---|
| `scripts/harness_manifest_builder.py` | Scan a domain folder → `manifest.v1` JSON (skills, tools, checks, agentic signals) |
| `scripts/goal_compiler.py` | Goal + manifest → `plan.v1` task plan; refuses vague goals (exit 3, forcing questions) |
| `scripts/loop_controller.py` | `init/next/record/verify/close/status` state machine; controller runs checks itself |
| `assets/harnesses/*.json` | 18 committed per-domain manifests (regenerable, diff-stable) |
| `assets/harness_manifest.schema.json` | Manifest schema |
| `references/` | Agentic-loop canon, verification discipline, domain-harness design (cited) |
| `agents/harness-runner.md` | Stateless one-task-per-invocation executor |
| `commands/cs-harness.md` | `/cs:harness <domain> <goal>` end-to-end driver |
All tools are stdlib-only, pass `--help` and `--sample`, and emit JSON.
## Design lineage
Anthropic's long-running-agents harness (feature-list + stateless shifts), verifier's law,
SWE-agent's environment-feedback lesson, Ralph-loop fresh-context iteration, Cognition's
serialize-writers rule, and this repo's own tc-tracker / autoresearch locked-evaluator /
loop-library stop-state primitives. See `skills/agent-harness/references/`.

View file

@ -0,0 +1,35 @@
---
name: harness-runner
description: Drives one agent-harness loop iteration to completion — reads the plan and state files, executes exactly one task with the task skill's own tools, lets the controller verify it, and reports the directive. Use when a goal has been compiled into an agent-harness plan and tasks need executing ("run the next harness task", "drive this loop until it escalates or closes"). Use PROACTIVELY after goal_compiler.py writes a plan. NOT for compiling goals (main session does that), authoring workflows (cs-workflow-architect), or tournaments (hub-coordinator).
tools: Read, Bash, Grep, Glob, Edit, Write
---
# Harness Runner
You execute ONE task per invocation from an agent-harness loop. You are a stateless shift
worker: everything you need is in the plan and state files; everything you learned goes back
into them via the controller. You never carry context between invocations.
## Workflow
1. `python3 <skill>/scripts/loop_controller.py next --state <state>` — obey the directive.
If it says `escalate` or `close`, report that verbatim and STOP.
2. For `execute T<n>`: open the task's `skill_path` SKILL.md, follow that skill's own
workflow with its own tools toward the task `objective`. Respect the goal's no-touch
constraints. Then `record --task T<n> --phase execute --exit-code <real code>`.
3. For `verify T<n>`: run `loop_controller.py verify --state <state> --task T<n> --cwd <repo-root>`.
If a `manual-evidence` check remains, gather the observable evidence and
`record --phase verify --exit-code 0 --evidence "<what you actually observed>"`.
4. Report: task id, resulting status, the controller's next directive, and (on failure)
the failing check's output tail plus what you will change on the retry.
## Hard rules
- Never edit a verification command, a manifest, or the plan to make a check pass.
- Never record a verify pass you did not observe. Fabricated evidence is the one
unforgivable failure mode.
- Never start a second task in the same invocation, even if the first finishes quickly —
serialized writes are the point.
- If the same check fails twice for the same reason, say what structural assumption is
wrong instead of trying a third cosmetic variation (3-strike rule, per focused-fix).
- On exit 2/5 from the controller: stop immediately and surface the evidence log path.

View file

@ -0,0 +1,34 @@
---
description: Compile a goal into a verified agent-harness loop for a domain and drive it to close — /cs:harness <domain> <goal>
argument-hint: <domain> <goal text>
---
# /cs:harness — run a goal through a domain's agent harness
Parse `$ARGUMENTS`: the first token is the domain (one of the 18 manifest names under
`engineering/agent-harness/skills/agent-harness/assets/harnesses/`); the rest is the goal.
If the domain token doesn't match a manifest file, list the available manifests and ask.
## Sequence (gates are blocking — never skip forward)
1. **Compile**
`python3 engineering/agent-harness/skills/agent-harness/scripts/goal_compiler.py --goal "<goal>" --manifest engineering/agent-harness/skills/agent-harness/assets/harnesses/<domain>.json --out .agent-harness/plan.json`
- Exit 3: relay the forcing questions to the user one at a time (recommended answer
first), then recompile with the enriched goal. Do not proceed on a vague goal.
- Exit 4: show `nearest_candidates`, ask whether to switch domain or refine the goal.
2. **Review the plan with the user** — show tasks, verifications, and caps. Confirm before
initializing: this is the only approval gate in the loop.
3. **Init**`python3 .../scripts/loop_controller.py init --plan .agent-harness/plan.json --state .agent-harness/state.json`
4. **Drive** — repeat: `next` → execute the task per its skill's SKILL.md → `record`
`verify`. For long goals, spawn the `harness-runner` agent per task instead of executing
inline, one at a time (writes stay serialized).
5. **On exit 2 or 5** — stop, show `status` and the failing evidence; the user decides:
fix and continue, waive with a reason, or abandon.
6. **Close**`close --state .agent-harness/state.json`; paste the handoff block
(tasks, statuses, evidence, waivers) as the deliverable summary.
## Rules
- Never edit checks, manifests, or the plan mid-loop to make verification pass.
- Never report an exhausted budget as success.
- `.agent-harness/` is git-ignorable working state; the handoff block is the record.

View file

@ -0,0 +1,130 @@
---
name: agent-harness
description: "Turn any domain folder of skills into a bounded agentic loop: compile a goal into a verifiable task plan, execute tasks with the domain's own tools, verify every task with machine-run checks, retry with caps, escalate to a human when budgets exhaust, and refuse to close until everything is verified or explicitly waived. Use when you want an agent or subagent to pick up a goal and drive it to a verified close across one of this repo's 18 domains ('run this goal through the engineering harness', 'set up an agentic loop for marketing work', 'make the finance domain self-verifying'). NOT for authoring Claude Code Workflow-tool .js scripts (workflow-builder), N-agent tournaments on one task (agenthub), single-file metric optimization (autoresearch-agent), or discovering published loop recipes (loop-library)."
---
# Agent Harness
You are a harness operator, not a hero. The loop — not your optimism — decides when work
is done. Your job: compile the goal into tasks with checks, execute one task at a time,
let the controller adjudicate verification, and stop when the state machine says stop.
## The contract
```
GOAL → goal_compiler → PLAN → loop_controller: [execute → verify]* → CLOSE
↑______retry (≤ max_attempts, changed approach)
└── ESCALATE on exhausted budgets — never fake success
```
Three layers, all JSON: a committed per-domain **manifest** (what skills/tools/checks
exist), a per-goal **plan** (which tasks, which verifications, what "done" means), and a
per-run **state file** (the single source of truth; a fresh session resumes from it alone).
## Quick start
```bash
# 0. Pick the domain manifest (18 committed under assets/harnesses/, e.g. engineering-team.json)
ls assets/harnesses/
# 1. Compile the goal (refuses vague goals with exit 3 + forcing questions)
python3 scripts/goal_compiler.py \
--goal "audit the payments service and design an SLO with an error budget" \
--manifest assets/harnesses/engineering.json --out plan.json
# 2. Initialize the loop state
python3 scripts/loop_controller.py init --plan plan.json --state .agent-harness/state.json
# 3. Drive the loop — repeat until directive is "close" or "escalate"
python3 scripts/loop_controller.py next --state .agent-harness/state.json
# → {"action": "execute", "task": "T1", ...}: open the task's skill (SKILL.md at
# skill_path), do the work with its tools, then:
python3 scripts/loop_controller.py record --state .agent-harness/state.json \
--task T1 --phase execute --exit-code 0
# → the controller runs the task's checks ITSELF (subprocess, timeout, evidence log):
python3 scripts/loop_controller.py verify --state .agent-harness/state.json --task T1 --cwd <repo-root>
# 4. Close — refused (exit 4) while any task is unverified and unwaived
python3 scripts/loop_controller.py close --state .agent-harness/state.json
```
Regenerate a manifest after skills change (diff-stable, CI-checkable):
```bash
python3 scripts/harness_manifest_builder.py --domain engineering-team \
--repo-root <repo-root> --out-dir assets/harnesses --no-timestamp
```
## Hard rules
1. **Never adjudicate your own verification.** `verify` runs the checks via subprocess;
a passing `record --phase verify` without `--evidence` is rejected (exit 6). You do not
get to declare a task verified.
2. **Never modify a gate you are judged by.** Check commands come from the manifest/plan.
Editing a check to make it pass is the reward-hacking failure mode
(see [references/verification_discipline.md](references/verification_discipline.md)) — same
invariant as autoresearch-agent's locked evaluator.
3. **One task at a time, writes serialized.** Parallelize reading and judging, never two
tasks writing the same artifact ([references/agentic_loop_canon.md](references/agentic_loop_canon.md)).
4. **Retry means a changed approach.** Same command + same input = same failure. The retry
directive says so; honor it.
5. **Budgets are terminal states, not suggestions.** `max_attempts_per_task` → escalated
(exit 2); `max_loop_iterations` → escalate (exit 5). Exhausted budgets are never
reported as success — a human waives (`close --waive T3 --reason "..."`), you don't.
6. **Fresh context beats long context.** Every `next` directive is executable by a new
session reading only the plan + state files. Long-running goals: run each iteration as
its own session against the durable state.
7. **State lives in `.agent-harness/`** — never in `.agenthub/`, `.autoresearch/`, or
`docs/TC/` (those belong to sibling skills).
8. **Plan and state files are a trust boundary.** `verify` shell-executes each task's
check command; only run the harness on plan/state files you or `goal_compiler.py`
produced, never on files from untrusted input (see
[references/verification_discipline.md](references/verification_discipline.md)).
## Forcing questions (ask before compiling; one per turn, with a recommended answer)
| # | Question | Recommended answer | Why (canon) |
|---|---|---|---|
| 1 | What single observable outcome means DONE? | A named artifact + a command that exits 0 against it | Verifier's law: invest in verifiability first |
| 2 | Which domain harness applies? | The domain whose skills name the deliverable; if two, run two sequential loops | Orchestrator-workers: scoped objectives beat mega-goals |
| 3 | What must NOT change? | List no-touch paths; put them in the goal text so the compiler's plan inherits them | Boundaries are part of a subagent spec |
| 4 | Who reviews escalations, and how fast? | A named human; escalations block the loop by design | Approval-required is a terminal state, not a nuisance |
| 5 | What is the iteration budget? | Default 12 loop iterations / 3 attempts per task; raise only with a reason | Caps are runtime errors, not advice (OpenAI SDK `max_turns`) |
## Exit codes (branch on these mechanically)
| Code | Tool | Meaning |
|---|---|---|
| 0 | all | OK / directive emitted |
| 2 | loop_controller | Escalation required — a human must review the evidence log |
| 3 | goal_compiler | Goal too vague — answer the forcing questions, recompile |
| 4 | goal_compiler / loop_controller | No skill matched / close refused (unverified tasks) |
| 5 | loop_controller | Global iteration cap reached |
| 6 | loop_controller | Invalid transition (recording on verified task, evidence missing, unknown task) |
## Verifiable success
- `python3 scripts/harness_manifest_builder.py --sample`, `scripts/goal_compiler.py --sample`,
and `scripts/loop_controller.py --sample` all exit 0.
- A vague goal (`--goal "make it better"`) exits 3 and prints forcing questions.
- `loop_controller.py close` on a state with an unverified task exits 4.
- The demo loop in `loop_controller.py --sample` shows a verify failure consuming an attempt
and the loop still closing only after a passing verify with evidence.
## Related skills
- **workflow-builder**: authoring deterministic `.js` scripts for Claude Code's Workflow
tool. NOT for goal-to-close loop state (this skill).
- **agenthub**: N parallel agents competing on ONE task in git worktrees. Use it *inside* a
harness task that wants competing attempts.
- **autoresearch-agent**: metric optimization of a single file against a locked evaluator.
Use it when a task's done_when is "metric improves".
- **tc-tracker**: per-code-change lifecycle records. Use for change bookkeeping; the harness
state file is per-goal, not per-change.
- **loop-library**: discover/audit published loop recipes conversationally. This skill is the
executable enforcement of that vocabulary.
- **ship-gate / self-eval / spec-driven-workflow**: plug in as close-time checks inside a
task's `verification[]`.
See [references/domain_harness_design.md](references/domain_harness_design.md) for the
three-layer architecture, the reuse map, and how to raise a domain's harness quality.

View file

@ -0,0 +1,65 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"title": "agent-harness domain manifest (agent-harness/manifest.v1)",
"description": "Machine-readable inventory of one domain folder: every skill, its tools, its verification checks, and its agentic signals. Produced by harness_manifest_builder.py; consumed by goal_compiler.py. Regenerate after any skill change — never hand-edit.",
"type": "object",
"required": ["schema", "domain", "skill_count", "loop_defaults", "skills"],
"properties": {
"schema": { "const": "agent-harness/manifest.v1" },
"domain": { "type": "string", "description": "Domain folder relative to repo root, e.g. 'engineering-team'." },
"skill_count": { "type": "integer", "minimum": 0 },
"generated_at": { "type": "string", "description": "UTC ISO-8601; omitted when built with --no-timestamp for diff-stable commits." },
"loop_defaults": {
"type": "object",
"required": ["max_attempts_per_task", "max_loop_iterations", "escalate_on"],
"properties": {
"max_attempts_per_task": { "type": "integer", "minimum": 1 },
"max_loop_iterations": { "type": "integer", "minimum": 1 },
"escalate_on": { "type": "array", "items": { "type": "string" } }
}
},
"skills": {
"type": "array",
"items": {
"type": "object",
"required": ["name", "path", "description", "tools", "agentic_signals"],
"properties": {
"name": { "type": "string" },
"path": { "type": "string", "description": "Skill dir (contains SKILL.md) relative to repo root." },
"description": { "type": "string", "maxLength": 600 },
"tools": {
"type": "array",
"items": {
"type": "object",
"required": ["script", "wired", "supports_sample", "verification"],
"properties": {
"script": { "type": "string" },
"wired": { "type": "boolean", "description": "True if the script basename appears in the skill's SKILL.md (A3 wiring)." },
"supports_sample": { "type": "boolean" },
"verification": {
"type": "array",
"items": {
"type": "object",
"required": ["cmd", "expect_exit", "kind"],
"properties": {
"cmd": { "type": "string" },
"expect_exit": { "type": "integer" },
"kind": { "enum": ["smoke", "sample", "manual-evidence"] }
}
}
}
}
}
},
"agentic_signals": {
"type": "object",
"description": "Static evidence of agentic structure in SKILL.md (maps to audit dimensions AR1/AR4/AR5/AR6).",
"required": ["goal_intake", "refusal_gate", "verification", "loop_discipline", "close_out"],
"additionalProperties": { "type": "boolean" }
},
"references": { "type": "array", "items": { "type": "string" } }
}
}
}
}
}

View file

@ -0,0 +1,220 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "business-growth",
"skill_count": 5,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "business-growth-skills",
"path": "business-growth/skills/business-growth-skills",
"description": "Router/index for the 4 business & growth skills bundled in this plugin: customer-success-manager (health scoring, churn risk, expansion), sales-engineer (RFP analysis, competitive matrices, PoC planning), revenue-operations (pipeline, forecast accuracy, GTM efficiency), and contract-and-proposal-writer. Use when a growth/revenue request doesn't obviously match one skill and you need to pick the right one (e.g., 'which accounts are at risk', 'should we bid on this RFP').",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "contract-and-proposal-writer",
"path": "business-growth/skills/contract-and-proposal-writer",
"description": "Generate professional, jurisdiction-aware business documents: freelance contracts, project proposals, SOWs, NDAs, and MSAs. Structured Markdown output with docx conversion instructions. Covers US (Delaware), EU (GDPR), UK, and DACH (German law) jurisdictions. Not a substitute for legal counsel \u2014 use as strong starting points. Use when drafting a freelance contract, preparing a client proposal, writing an SOW for a new engagement, or producing an NDA before sharing sensitive material.",
"tools": [],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": []
},
{
"name": "customer-success-manager",
"path": "business-growth/skills/customer-success-manager",
"description": "Monitors customer health, predicts churn risk, and identifies expansion opportunities using weighted scoring models for SaaS customer success. Use when analyzing customer accounts, reviewing retention metrics, scoring at-risk customers, or when the user mentions churn, customer health scores, upsell opportunities, expansion revenue, retention analysis, or customer analytics. Runs three Python CLI tools to produce deterministic health scores, churn risk tiers, and prioritized expansion recommendations across Enterprise, Mid-Market, and SMB segments.",
"tools": [
{
"script": "business-growth/skills/customer-success-manager/scripts/churn_risk_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/customer-success-manager/scripts/churn_risk_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "business-growth/skills/customer-success-manager/scripts/expansion_opportunity_scorer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/customer-success-manager/scripts/expansion_opportunity_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "business-growth/skills/customer-success-manager/scripts/health_score_calculator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-growth/skills/customer-success-manager/scripts/health_score_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-growth/skills/customer-success-manager/scripts/health_score_calculator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"cs-metrics-benchmarks.md",
"cs-playbooks.md",
"health-scoring-framework.md"
]
},
{
"name": "revenue-operations",
"path": "business-growth/skills/revenue-operations",
"description": "Analyzes sales pipeline health, revenue forecasting accuracy, and go-to-market efficiency metrics for SaaS revenue optimization. Use when analyzing sales pipeline coverage, forecasting revenue, evaluating go-to-market performance, reviewing sales metrics, assessing pipeline analysis, tracking forecast accuracy with MAPE, calculating GTM efficiency, or measuring sales efficiency and unit economics for SaaS teams.",
"tools": [
{
"script": "business-growth/skills/revenue-operations/scripts/forecast_accuracy_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/revenue-operations/scripts/forecast_accuracy_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "business-growth/skills/revenue-operations/scripts/gtm_efficiency_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/revenue-operations/scripts/gtm_efficiency_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "business-growth/skills/revenue-operations/scripts/pipeline_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-growth/skills/revenue-operations/scripts/pipeline_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-growth/skills/revenue-operations/scripts/pipeline_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"gtm-efficiency-benchmarks.md",
"pipeline-management-framework.md",
"revops-metrics-guide.md"
]
},
{
"name": "sales-engineer",
"path": "business-growth/skills/sales-engineer",
"description": "Analyzes RFP/RFI responses for coverage gaps, builds competitive feature comparison matrices, and plans proof-of-concept (POC) engagements for pre-sales engineering. Use when responding to RFPs, bids, or proposal requests; comparing product features against competitors; planning or scoring a customer POC or sales demo; preparing a technical proposal; or performing win/loss competitor analysis. Handles tasks described as 'RFP response', 'bid response', 'proposal response', 'competitor comparison', 'feature matrix', 'POC planning', 'sales demo prep', or 'pre-sales engineering'.",
"tools": [
{
"script": "business-growth/skills/sales-engineer/scripts/competitive_matrix_builder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/sales-engineer/scripts/competitive_matrix_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "business-growth/skills/sales-engineer/scripts/poc_planner.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/sales-engineer/scripts/poc_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "business-growth/skills/sales-engineer/scripts/rfp_response_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 business-growth/skills/sales-engineer/scripts/rfp_response_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"competitive-positioning-framework.md",
"poc-best-practices.md",
"rfp-response-guide.md"
]
}
]
}

View file

@ -0,0 +1,451 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "business-operations",
"skill_count": 7,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "business-operations-skills",
"path": "business-operations/skills/business-operations-skills",
"description": "Use when running, diagnosing, or designing internal business operations \u2014 process documentation, vendor SLAs, capacity planning, internal comms, SOP/runbook authoring, procurement spend. Triggers on \"BizOps review\", \"where's the bottleneck\", \"vendor health\", \"internal SOP\", \"all-hands deck\", \"spend categorization\", \"capacity for Q3\", \"process mapping\". Forks context to route to one of six BizOps sub-skills (process-mapper, vendor-management, capacity-planner, internal-comms, knowledge-ops, procurement-optimizer) and returns a digest. Distinct from business-growth (external sales motion) and \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": true
},
"references": []
},
{
"name": "capacity-planner",
"path": "business-operations/skills/capacity-planner",
"description": "Use when an ops leader (Director of CX, Head of Support, VP Ops, Head of BizOps, Head of IT ops, Head of Finance ops) is sizing ops capacity, building a headcount plan, modeling utilization risk, planning Q3 capacity or annual support capacity, or designing CS coverage \u2014 and needs Erlang-C queueing math, P90 demand sizing, shrinkage-adjusted FTE, manager-trigger thresholds, and a quarterly hiring sequence with ramp + attrition. Apply when sustained team utilization is above 80% or when the team is growing >50% in 12 months. Run before committing the headcount budget. This is NOT engineering \u2026",
"tools": [
{
"script": "business-operations/skills/capacity-planner/scripts/capacity_modeler.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/capacity-planner/scripts/capacity_modeler.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/capacity-planner/scripts/capacity_modeler.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/capacity-planner/scripts/hiring_sequencer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/capacity-planner/scripts/hiring_sequencer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/capacity-planner/scripts/hiring_sequencer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/capacity-planner/scripts/utilization_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/capacity-planner/scripts/utilization_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/capacity-planner/scripts/utilization_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"capacity_anti_patterns.md",
"ops_workforce_planning_canon.md",
"queueing_theory_canon.md"
]
},
{
"name": "internal-comms",
"path": "business-operations/skills/internal-comms",
"description": "Use when a Head of People Ops, BizOps lead, or Internal Communications owner needs to draft and sequence an internal-only change-management communication \u2014 a re-org announcement, a tool rollout, a policy change, a leadership transition, a layoff, an acquisition close, or an internal product launch \u2014 and the audience is employees (not customers). Pairs Prosci ADKAR and Kotter's 8-step change model with deterministic stdlib-only Python tools to produce a sequenced touchpoint calendar, a Kotter-compliant primary announcement, an audience-segmented FAQ, and manager cascade talking points \u2026",
"tools": [
{
"script": "business-operations/skills/internal-comms/scripts/change_announcement_builder.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/internal-comms/scripts/change_announcement_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/internal-comms/scripts/change_announcement_builder.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/internal-comms/scripts/comms_calendar_builder.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/internal-comms/scripts/comms_calendar_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/internal-comms/scripts/comms_calendar_builder.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/internal-comms/scripts/comms_template_filler.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/internal-comms/scripts/comms_template_filler.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/internal-comms/scripts/comms_template_filler.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"announcement_anti_patterns.md",
"change_management_canon.md",
"internal_comms_canon.md"
]
},
{
"name": "knowledge-ops",
"path": "business-operations/skills/knowledge-ops",
"description": "Use when a Head of Ops, Knowledge Manager, or TPM-Internal needs to author, validate, or clean up company SOPs and internal runbooks (procurement intake, vendor offboarding, incident-comms cascade, employee onboarding) \u2014 including 5W2H completeness checks (Who-What-When-Where-Why-How-HowMuch), cross-link and orphan-page validation across a sprawling Notion/Confluence/Obsidian wiki, KB ingestion + hygiene reporting, and runbook step verification (named owner, expected duration, observable success signal, rollback path, escalation contact). Pairs Ishikawa's 5W2H method, Gawande's *The \u2026",
"tools": [
{
"script": "business-operations/skills/knowledge-ops/scripts/kb_ingester.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/knowledge-ops/scripts/kb_ingester.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/knowledge-ops/scripts/kb_ingester.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/knowledge-ops/scripts/runbook_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/knowledge-ops/scripts/runbook_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/knowledge-ops/scripts/runbook_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/knowledge-ops/scripts/sop_generator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/knowledge-ops/scripts/sop_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/knowledge-ops/scripts/sop_generator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"5w2h_sop_canon.md",
"kb_hygiene_anti_patterns.md",
"runbook_canon.md"
]
},
{
"name": "process-mapper",
"path": "business-operations/skills/process-mapper",
"description": "Use when a BizOps lead, COO, or process-improvement owner needs to document an end-to-end business process (procurement, employee onboarding, incident handoff, customer-onboarding, claims adjudication) in BPMN-style notation, measure cycle times by stage, surface where work spends most of its time waiting vs. being worked, and quantify the gap between processing time and total elapsed time. Pairs Lean / Six Sigma / Theory-of-Constraints canon with deterministic stdlib-only Python tools to produce a process map, a ranked bottleneck list (with severity + root-cause hypothesis), and a \u2026",
"tools": [
{
"script": "business-operations/skills/process-mapper/scripts/bottleneck_detector.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/process-mapper/scripts/bottleneck_detector.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/process-mapper/scripts/bottleneck_detector.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/process-mapper/scripts/cycle_time_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/process-mapper/scripts/cycle_time_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/process-mapper/scripts/cycle_time_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/process-mapper/scripts/process_documenter.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/process-mapper/scripts/process_documenter.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/process-mapper/scripts/process_documenter.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"bottleneck_anti_patterns.md",
"bpmn_essentials.md",
"lean_six_sigma_canon.md"
]
},
{
"name": "procurement-optimizer",
"path": "business-operations/skills/procurement-optimizer",
"description": "Use when running an annual SaaS audit, doing category-level spend review, or rationalizing the supplier base \u2014 when the user needs a spend audit, spend categorization (UNSPSC-aligned with Pareto breakdown and industry profiles), purchasing-cycle analysis (bottleneck categories per Goldratt's Theory of Constraints), or risk-balanced supplier consolidation that refuses single-source recommendations for tier-1 categories without a documented break-glass plan. Triggers on \"spend audit\", \"SaaS audit\", \"spend categorization\", \"supplier rationalization\", \"supplier consolidation\", \"category \u2026",
"tools": [
{
"script": "business-operations/skills/procurement-optimizer/scripts/purchasing_cycle_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/procurement-optimizer/scripts/purchasing_cycle_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/procurement-optimizer/scripts/purchasing_cycle_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/procurement-optimizer/scripts/spend_categorizer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/procurement-optimizer/scripts/spend_categorizer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/procurement-optimizer/scripts/spend_categorizer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/procurement-optimizer/scripts/supplier_consolidation.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/procurement-optimizer/scripts/supplier_consolidation.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/procurement-optimizer/scripts/supplier_consolidation.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"procurement_anti_patterns.md",
"saas_management_canon.md",
"spend_management_canon.md"
]
},
{
"name": "vendor-management",
"path": "business-operations/skills/vendor-management",
"description": "Use when reviewing, scoring, or auditing third-party SaaS / vendor relationships \u2014 running a vendor scorecard with industry tuning, tracking SLA compliance with credit-claim flags, classifying third-party risk across 4 risk vectors, preparing a tier-1 vendor review, or auditing the SaaS portfolio. Forks context so large vendor catalogs (50-500 line items) and SLA logs don't pollute the parent thread. Triggers on \"vendor SLA\", \"vendor scorecard\", \"third-party risk\", \"TPRM\", \"vendor review\", \"supplier performance\", \"vendor health check\", \"renewal review\".",
"tools": [
{
"script": "business-operations/skills/vendor-management/scripts/sla_compliance_tracker.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/vendor-management/scripts/sla_compliance_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/vendor-management/scripts/sla_compliance_tracker.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/vendor-management/scripts/vendor_risk_classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/vendor-management/scripts/vendor_risk_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/vendor-management/scripts/vendor_risk_classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "business-operations/skills/vendor-management/scripts/vendor_scorer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 business-operations/skills/vendor-management/scripts/vendor_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 business-operations/skills/vendor-management/scripts/vendor_scorer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"sla_design_patterns.md",
"vendor_management_canon.md",
"vendor_risk_anti_patterns.md"
]
}
]
}

View file

@ -0,0 +1,521 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "commercial",
"skill_count": 8,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "channel-economics",
"path": "commercial/skills/channel-economics",
"description": "Use when reviewing or rebalancing direct vs. partner-led channel economics \u2014 computing fully-loaded cost-to-serve per channel, channel ROI with cash / LTV / marginal lenses, and optimal channel mix subject to constraints. For Head of Commercial, RevOps, and VP Sales doing quarterly channel review when pipeline is mixed (e.g., 60% direct + 40% partner-led) and nobody actually knows which channel makes money after CAC, support load, partner discount, deal-velocity differences, retention differential, and overhead allocation are all loaded in. Outputs cost to serve, channel ROI verdicts \u2026",
"tools": [
{
"script": "commercial/skills/channel-economics/scripts/channel_mix_optimizer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/channel-economics/scripts/channel_mix_optimizer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/channel-economics/scripts/channel_mix_optimizer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/channel-economics/scripts/channel_roi_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/channel-economics/scripts/channel_roi_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/channel-economics/scripts/channel_roi_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/channel-economics/scripts/cost_to_serve_calculator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/channel-economics/scripts/cost_to_serve_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/channel-economics/scripts/cost_to_serve_calculator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"channel_anti_patterns.md",
"channel_economics_canon.md",
"cost_to_serve_canon.md"
]
},
{
"name": "commercial-forecaster",
"path": "commercial/skills/commercial-forecaster",
"description": "Use when building a quarterly bookings forecast, ARR projection, pipeline forecast, NRR projection, or commit/best-case/pipe-only board number \u2014 especially when the CRO needs to walk the board through funnel math + cohort ARR + per-stage conversion assumptions without the theatre of a single undefended number. Decomposes pipeline into commit, best-case, and pipe-only tiers; projects cohort-level NRR/GRR to surface leaky cohorts before they show up in the consolidated number; scores per-stage funnel confidence so soft-floor stages get treated differently from high-confidence ones. Every \u2026",
"tools": [
{
"script": "commercial/skills/commercial-forecaster/scripts/bookings_forecaster.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/commercial-forecaster/scripts/bookings_forecaster.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/commercial-forecaster/scripts/bookings_forecaster.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/commercial-forecaster/scripts/cohort_arr_projector.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/commercial-forecaster/scripts/cohort_arr_projector.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/commercial-forecaster/scripts/cohort_arr_projector.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/commercial-forecaster/scripts/funnel_confidence_scorer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/commercial-forecaster/scripts/funnel_confidence_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/commercial-forecaster/scripts/funnel_confidence_scorer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"cohort_analysis_canon.md",
"forecast_anti_patterns.md",
"saas_forecasting_canon.md"
]
},
{
"name": "commercial-policy",
"path": "commercial/skills/commercial-policy",
"description": "Use when designing or revising a company's commercial policy \u2014 the rules of engagement governing discounts off list price, approver thresholds, exception flows, and the deal framework that Deal Desk and AEs operate under. Covers discount matrix design (ARR band x term length x payment terms x strategic value), commercial policy design, exception policy, discount governance, approval thresholds, deal framework structure, and policy linting (contradictions, gaps, cliff edges, gaming surfaces). For Head of Commercial, Head of Deal Desk, VP Sales, or RevOps at the policy-design moment \u2014 NOT \u2026",
"tools": [
{
"script": "commercial/skills/commercial-policy/scripts/discount_matrix_builder.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/commercial-policy/scripts/discount_matrix_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/commercial-policy/scripts/discount_matrix_builder.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/commercial-policy/scripts/exception_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/commercial-policy/scripts/exception_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/commercial-policy/scripts/exception_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/commercial-policy/scripts/policy_linter.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/commercial-policy/scripts/policy_linter.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/commercial-policy/scripts/policy_linter.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"discount_governance_canon.md",
"policy_anti_patterns.md",
"policy_design_canon.md"
]
},
{
"name": "commercial-skills",
"path": "commercial/skills/commercial-skills",
"description": "Use when reviewing, approving, or designing commercial motion \u2014 pricing models, deal review, discount approval, partnership economics, channel mix, commercial policy, RFP/RFI response, bookings forecast. Triggers on \"review this deal\", \"should we discount\", \"pricing model\", \"partner economics\", \"RFP response\", \"bookings forecast\", \"channel mix\". Forks context to route to one of seven Commercial sub-skills (pricing-strategist, deal-desk, partnerships-architect, channel-economics, commercial-policy, rfp-responder, commercial-forecaster) and returns a digest. Distinct from business-growth \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": []
},
{
"name": "deal-desk",
"path": "commercial/skills/deal-desk",
"description": "Use when reviewing a specific inbound deal before close \u2014 when sales has asked for a discount that exceeds AE authority, when the customer has redlined the MSA, when per-deal economics (margin after discount, multi-year payment shape, indemnity exposure) need to be quantified, or when discount approval needs to be routed to a named human approver (Sales Director, VP Sales, CFO, CRO, General Counsel). Covers deal review, discount approval routing, per-deal margin scoring, deal exception handling, MSA redline triage, contract landmine detection (uncapped indemnity, MFN, perpetual \u2026",
"tools": [
{
"script": "commercial/skills/deal-desk/scripts/deal_scorer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/deal-desk/scripts/deal_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/deal-desk/scripts/deal_scorer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/deal-desk/scripts/discount_approval_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/deal-desk/scripts/discount_approval_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/deal-desk/scripts/discount_approval_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/deal-desk/scripts/terms_redliner.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/deal-desk/scripts/terms_redliner.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/deal-desk/scripts/terms_redliner.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"contract_landmines.md",
"deal_desk_canon.md",
"discount_economics.md"
]
},
{
"name": "partnerships-architect",
"path": "commercial/skills/partnerships-architect",
"description": "Use when a startup is approached by a prospective partner and someone has to decide should we sign this partner, at what partner tier (referral / reseller / OEM / SI-consulting / strategic alliance), with what joint GTM commitment, and at what revshare. Classifies partner tier from independent-demand evidence vs. preferential-terms hunting, designs a 90-day joint GTM plan, models revshare against direct-sale margin, and surfaces kill criteria for unwinding under-performing partnerships. For Head of Partnerships, Head of BD, and Founder-CEOs doing reseller agreement, OEM deal, or strategic \u2026",
"tools": [
{
"script": "commercial/skills/partnerships-architect/scripts/joint_gtm_planner.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/partnerships-architect/scripts/joint_gtm_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/partnerships-architect/scripts/joint_gtm_planner.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/partnerships-architect/scripts/partner_tier_classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/partnerships-architect/scripts/partner_tier_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/partnerships-architect/scripts/partner_tier_classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/partnerships-architect/scripts/revshare_modeler.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/partnerships-architect/scripts/revshare_modeler.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/partnerships-architect/scripts/revshare_modeler.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"channel_partner_canon.md",
"joint_gtm_canon.md",
"partnership_anti_patterns.md"
]
},
{
"name": "pricing-strategist",
"path": "commercial/skills/pricing-strategist",
"description": "Use when designing or revisiting product pricing \u2014 selecting a pricing model (subscription seat-based, usage-based, value-based, freemium, or hybrid), running Van Westendorp Price Sensitivity Meter analysis on WTP survey data, or designing Good/Better/Best packaging tiers. Recommends a model and a price range with trade-offs, never a single number. For Commercial leads, Product Marketing, and CMOs at the pricing-design moment \u2014 not deal-by-deal discounting, not brand positioning.",
"tools": [
{
"script": "commercial/skills/pricing-strategist/scripts/packaging_designer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/pricing-strategist/scripts/packaging_designer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/pricing-strategist/scripts/packaging_designer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/pricing-strategist/scripts/pricing_model_picker.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/pricing-strategist/scripts/pricing_model_picker.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/pricing-strategist/scripts/pricing_model_picker.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/pricing-strategist/scripts/wtp_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/pricing-strategist/scripts/wtp_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/pricing-strategist/scripts/wtp_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"packaging_anti_patterns.md",
"saas_pricing_canon.md",
"van_westendorp_methodology.md"
]
},
{
"name": "rfp-responder",
"path": "commercial/skills/rfp-responder",
"description": "Use when an RFP, RFI, RFQ, security questionnaire, vendor questionnaire, or proposal request arrives and the team needs a structured response \u2014 parsing multi-section buyer-dictated requirements (MANDATORY vs WEIGHTED vs NICE-TO-HAVE), building a Shipley-method proof-point matrix mapping each requirement to a verifiable proof point, articulating 3-5 win-themes that ladder up across requirements, and producing a Shipley-derived winrate estimate that informs a bid / no-bid / partner-bid recommendation. For Bid Managers, Proposal Leads, Directors of Sales, and Sales Engineers at the \u2026",
"tools": [
{
"script": "commercial/skills/rfp-responder/scripts/response_drafter.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/rfp-responder/scripts/response_drafter.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/rfp-responder/scripts/response_drafter.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/rfp-responder/scripts/rfp_parser.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/rfp-responder/scripts/rfp_parser.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/rfp-responder/scripts/rfp_parser.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "commercial/skills/rfp-responder/scripts/winrate_predictor.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 commercial/skills/rfp-responder/scripts/winrate_predictor.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 commercial/skills/rfp-responder/scripts/winrate_predictor.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"rfp_anti_patterns.md",
"rfp_strategy_canon.md",
"shipley_method_canon.md"
]
}
]
}

View file

@ -0,0 +1,199 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "compliance-os",
"skill_count": 9,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "ai-act-readiness",
"path": "compliance-os/skills/ai-act-readiness",
"description": "/cs:ai-act-readiness <system> \u2014 EU AI Act 6-question forcing interrogation. Use during AI-system intake, before EU deployment, or during annual compliance refresh as Article 113 obligations phase in (2025-02-02 / 2025-08-02 / 2026-08-02 / 2027-08-02).",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "aims-audit",
"path": "compliance-os/skills/aims-audit",
"description": "/cs:aims-audit <scope> \u2014 ISO/IEC 42001 AIMS internal-audit 6-question forcing interrogation. Use before certification stage 1, before annual internal audit cycles, or when onboarding a new AI system into an existing AIMS.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "compliance-os",
"path": "compliance-os/skills/compliance-os",
"description": "Compliance OS \u2014 meta-orchestrator that lets compliance teams CONFIGURE which frameworks apply, COMPUTE cross-framework control overlap, SIMULATE internal audits, and CONSOLIDATE evidence across multiple frameworks. Four decisions: (1) Given a company profile, which of the 12 supported frameworks apply (ISO 27001/13485/42001/14971, EU AI Act, MDR 745, GDPR, SOC 2, FDA QSR, NIST CSF 2.0, NIS2, HIPAA)? (2) Across selected frameworks, which controls overlap and how much evidence reuses? (3) For a given framework + scope, what does a realistic mock audit produce \u2014 drawing from the 205-scenario \u2026",
"tools": [
{
"script": "compliance-os/skills/compliance-os/scripts/audit_simulator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 compliance-os/skills/compliance-os/scripts/audit_simulator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "compliance-os/skills/compliance-os/scripts/cross_framework_mapper.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 compliance-os/skills/compliance-os/scripts/cross_framework_mapper.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "compliance-os/skills/compliance-os/scripts/evidence_pool_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 compliance-os/skills/compliance-os/scripts/evidence_pool_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "compliance-os/skills/compliance-os/scripts/framework_selector.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 compliance-os/skills/compliance-os/scripts/framework_selector.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"audit_simulation_methodology.md",
"compliance_os_pattern.md",
"cross_framework_overlap.md",
"evidence_artifact_reuse_index.md",
"evidence_management.md",
"multi_framework_audit_playbook.md"
]
},
{
"name": "compliance-readiness",
"path": "compliance-os/skills/compliance-readiness",
"description": "/cs:compliance-readiness <program> \u2014 Multi-framework compliance officer 6-question forcing interrogation of any compliance program. Use before starting a new framework, planning the annual audit calendar, or preparing for certification stage 1.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "fda-qsr-audit-prep",
"path": "compliance-os/skills/fda-qsr-audit-prep",
"description": "/cs:fda-qsr-audit-prep <scope> \u2014 FDA 21 CFR 820 (QSR / QMSR) audit 6-question forcing interrogation. Post-Feb 2026 substantially harmonized with ISO 13485. Use before annual internal QSR audit, pre-FDA-inspection readiness, or Form 483 response.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "gdpr-audit-prep",
"path": "compliance-os/skills/gdpr-audit-prep",
"description": "/cs:gdpr-audit-prep <scope> \u2014 GDPR audit 6-question Article-cited forcing interrogation. Use before annual internal GDPR review, post-breach internal audit, DPA investigation readiness, or acquisition due diligence.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "iso13485-audit-prep",
"path": "compliance-os/skills/iso13485-audit-prep",
"description": "/cs:iso13485-audit-prep <scope> \u2014 ISO 13485 QMS audit 6-question forcing interrogation. Design controls + CAPA + post-market focused. Use before Clause 8.2.4 internal audit, MDR / FDA QSR alignment review, or product-launch DHF closure audit.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "iso27001-audit-prep",
"path": "compliance-os/skills/iso27001-audit-prep",
"description": "/cs:iso27001-audit-prep <scope> \u2014 ISO 27001 ISMS audit readiness 6-question forcing interrogation. Use before annual Clause 9.2 internal audit, surveillance audit prep, or stage 1 certification readiness.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "soc2-audit-prep",
"path": "compliance-os/skills/soc2-audit-prep",
"description": "/cs:soc2-audit-prep <scope> \u2014 SOC 2 Type II readiness 6-question forcing interrogation. Observation-period focused. Use before Type II observation begins, mid-period checkpoint, or pre-field-test month-10 readiness.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": []
}
]
}

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,167 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "finance",
"skill_count": 4,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "business-investment-advisor",
"path": "finance/business-investment-advisor/skills/business-investment-advisor",
"description": "Business investment analysis and capital allocation advisor. Use when evaluating whether to invest in equipment, real estate, a new business, hiring, technology, or any capital expenditure. Also use for ROI calculations, IRR, NPV, payback period, build vs buy decisions, lease vs buy analysis, vendor evaluation, or deciding where to allocate limited budget for maximum return.",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": []
},
{
"name": "finance-skills",
"path": "finance/skills/finance-skills",
"description": "Router/index for the 2 finance skills bundled in this plugin: financial-analyst (ratio analysis, DCF valuation, budget variance, rolling forecasts) and saas-metrics-coach (ARR/MRR, churn, CAC/LTV, NRR, quick ratio). Use when a finance request doesn't obviously match one skill and you need to pick the right one (e.g., 'analyze these financials', 'how healthy are my SaaS metrics').",
"tools": [],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": []
},
{
"name": "financial-analyst",
"path": "finance/skills/financial-analyst",
"description": "Performs financial ratio analysis, DCF valuation, budget variance analysis, and rolling forecast construction for strategic decision-making. Use when analyzing financial statements, building valuation models, assessing budget variances, or constructing financial projections and forecasts. Also applicable when users mention financial modeling, cash flow analysis, company valuation, financial projections, or spreadsheet analysis.",
"tools": [
{
"script": "finance/skills/financial-analyst/scripts/budget_variance_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/financial-analyst/scripts/budget_variance_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "finance/skills/financial-analyst/scripts/dcf_valuation.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/financial-analyst/scripts/dcf_valuation.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "finance/skills/financial-analyst/scripts/forecast_builder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/financial-analyst/scripts/forecast_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "finance/skills/financial-analyst/scripts/ratio_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/financial-analyst/scripts/ratio_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"financial-ratios-guide.md",
"forecasting-best-practices.md",
"industry-adaptations.md",
"valuation-methodology.md"
]
},
{
"name": "saas-metrics-coach",
"path": "finance/skills/saas-metrics-coach",
"description": "SaaS financial health advisor. Use when a user shares revenue or customer numbers, or mentions ARR, MRR, churn, LTV, CAC, NRR, or asks how their SaaS business is doing.",
"tools": [
{
"script": "finance/skills/saas-metrics-coach/scripts/metrics_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/saas-metrics-coach/scripts/metrics_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "finance/skills/saas-metrics-coach/scripts/quick_ratio_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/saas-metrics-coach/scripts/quick_ratio_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "finance/skills/saas-metrics-coach/scripts/unit_economics_simulator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 finance/skills/saas-metrics-coach/scripts/unit_economics_simulator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"benchmarks.md",
"formulas.md"
]
}
]
}

View file

@ -0,0 +1,34 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "loop-library",
"skill_count": 1,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "loop-library",
"path": "loop-library",
"description": "Discover, find, compare, audit, repair, adapt, and design repeatable AI-agent loops with explicit triggers, actions, verification, stopping conditions, guardrails, and handoffs. Use when a user asks to analyze a codebase for potential loops, mine coding-thread history for work done more than once, turn repeated engineering work into a loop, find or recommend a published loop, create a recurring agent workflow or automation cadence, turn an outcome into a bounded copy-ready loop, or review an existing loop for weak checks, unsafe authority, unbounded repetition, stale state, or unclear \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"audit.md",
"discover.md"
]
}
]
}

View file

@ -0,0 +1,357 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "markdown-html",
"skill_count": 5,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "design-system",
"path": "markdown-html/skills/design-system",
"description": "Captures the user's brand identity once via a 10-question onboarding wizard (primary/accent HEX + heading + body Google Fonts + design style editorial/technical/minimal/playful + default output directory + syntax theme + TOC behavior + optional logo/company), validates body-text and link contrast against WCAG 2.2 AA, derives 12 CSS custom properties in HSL space, and stores the result for every markdown-html converter to consume. Use before any markdown-html conversion. Triggers on first-run onboarding (\"set up the brand\", \"configure markdown-html\", \"run onboarding\"), on explicit reset \u2026",
"tools": [
{
"script": "markdown-html/skills/design-system/scripts/brand_palette_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/design-system/scripts/brand_palette_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/design-system/scripts/brand_palette_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/design-system/scripts/config_loader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/design-system/scripts/config_loader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/design-system/scripts/config_loader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/design-system/scripts/onboard.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 markdown-html/skills/design-system/scripts/onboard.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": [
"design_token_canon.md",
"typography_pairing.md",
"wcag_accessibility.md"
]
},
{
"name": "markdown-html-orchestrator",
"path": "markdown-html/skills/markdown-html-orchestrator",
"description": "Use when a user wants to convert any markdown file in their Claude project into a single-file, lightly-interactive HTML \u2014 long-form documents (specs, plans, RFCs, reports, explainers), code reviews with diffs and severity-tagged annotations, or slide decks. Triggers on \"convert this markdown to HTML\", \"make this an HTML file\", \"turn this into an interactive document\", \"render this report as HTML\", \"PR writeup as HTML\", \"slides from this markdown\". Forks context to route to one of three converter sub-skills (md-document, md-review, md-slides) based on a deterministic doctype classifier \u2026",
"tools": [
{
"script": "markdown-html/skills/markdown-html-orchestrator/scripts/doctype_classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/markdown-html-orchestrator/scripts/doctype_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/markdown-html-orchestrator/scripts/doctype_classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/markdown-html-orchestrator/scripts/output_path_resolver.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/markdown-html-orchestrator/scripts/output_path_resolver.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/markdown-html-orchestrator/scripts/output_path_resolver.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/markdown-html-orchestrator/scripts/route_explainer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 markdown-html/skills/markdown-html-orchestrator/scripts/route_explainer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": false,
"close_out": true
},
"references": [
"information_density_canon.md",
"orchestrator_routing_patterns.md",
"single_file_html_discipline.md"
]
},
{
"name": "md-document",
"path": "markdown-html/skills/md-document",
"description": "Converts long-form markdown (specs, RFCs, reports, plans, explainers) into a single-file, lightly-interactive HTML document with sticky TOC, scrollspy, search filter, code-copy buttons, and design-system-driven brand tokens. Triggers when the markdown-html-orchestrator classifies an input as DOCUMENT, or when invoked directly via /cs:md-document. Reads the design-system config via config_loader.py and inlines the user's 12 derived CSS custom properties; refuses to render if onboarding hasn't run. Single-file output \u2014 Google Fonts + Prism.js CDN are the only externals; no framework runtime \u2026",
"tools": [
{
"script": "markdown-html/skills/md-document/scripts/html_renderer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-document/scripts/html_renderer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-document/scripts/html_renderer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/md-document/scripts/interactivity_injector.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-document/scripts/interactivity_injector.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-document/scripts/interactivity_injector.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/md-document/scripts/markdown_parser.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-document/scripts/markdown_parser.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-document/scripts/markdown_parser.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"information_density_patterns.md",
"single_file_html_discipline.md",
"toc_and_nav_ux.md"
]
},
{
"name": "md-review",
"path": "markdown-html/skills/md-review",
"description": "Converts a markdown PR writeup or code review (one with ```diff fenced blocks and severity-tagged > [!BLOCKER]/[!MAJOR]/[!MINOR]/[!NIT] callouts) into a single-file 2-column HTML review \u2014 unified-diff on the left, severity-tagged annotation cards on the right, top jump-nav listing every finding, mandatory named reviewer footer. Triggers when the markdown-html-orchestrator classifies an input as REVIEW, or when invoked directly via /cs:md-review. Refuses without explicit --reviewer (a code review must name a human), refuses if no diff hunks present (route to md-document instead), and refuses \u2026",
"tools": [
{
"script": "markdown-html/skills/md-review/scripts/annotation_extractor.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-review/scripts/annotation_extractor.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-review/scripts/annotation_extractor.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/md-review/scripts/diff_parser.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-review/scripts/diff_parser.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-review/scripts/diff_parser.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/md-review/scripts/review_html_renderer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-review/scripts/review_html_renderer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-review/scripts/review_html_renderer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"diff_rendering_canon.md",
"pr_annotation_ux.md",
"severity_coding.md"
]
},
{
"name": "md-slides",
"path": "markdown-html/skills/md-slides",
"description": "Converts a markdown deck (slides separated by `---` HR boundaries or by `# ` H1 headings, with optional `<!-- notes: ... -->` presenter notes blocks) into a single-file HTML presentation with arrow-key / space / PgDn / PgUp / Home / End / P / Esc keyboard navigation, presenter mode (split view with current slide + speaker notes + clock + next-slide preview), URL-hash deep linking, and `@media print` page-per-slide for PDF export. Triggers when the markdown-html-orchestrator classifies an input as SLIDES, or when invoked directly via /cs:md-slides. Reuses md-document's markdown parser for \u2026",
"tools": [
{
"script": "markdown-html/skills/md-slides/scripts/deck_html_renderer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-slides/scripts/deck_html_renderer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-slides/scripts/deck_html_renderer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/md-slides/scripts/presenter_notes_parser.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-slides/scripts/presenter_notes_parser.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-slides/scripts/presenter_notes_parser.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "markdown-html/skills/md-slides/scripts/slide_splitter.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 markdown-html/skills/md-slides/scripts/slide_splitter.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 markdown-html/skills/md-slides/scripts/slide_splitter.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": [
"keyboard_nav_patterns.md",
"presentation_ux.md",
"single_file_deck_conventions.md"
]
}
]
}

View file

@ -0,0 +1,87 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "marketing",
"skill_count": 1,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "landing",
"path": "marketing/landing/skills/landing",
"description": "Generates a premium single-page HTML landing page with 3D CSS animations, GSAP scroll effects, and mouse-parallax depth. Forcing intake (product + elevator pitch, audience register, brand overrides, tone) locks down positioning before any copy or markup is written, so the page reflects the actual product rather than generic boilerplate. Use whenever the user says 'landing for X', 'create a landing page', 'build a landing page', 'make a landing page for X', 'I need a web page for Y', or provides product/service details and wants a polished website. Also triggers on 'promotional page' \u2026",
"tools": [
{
"script": "marketing/landing/skills/landing/scripts/brand_palette_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 marketing/landing/skills/landing/scripts/brand_palette_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 marketing/landing/skills/landing/scripts/brand_palette_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "marketing/landing/skills/landing/scripts/html_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 marketing/landing/skills/landing/scripts/html_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 marketing/landing/skills/landing/scripts/html_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "marketing/landing/skills/landing/scripts/kebab_slug_generator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 marketing/landing/skills/landing/scripts/kebab_slug_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 marketing/landing/skills/landing/scripts/kebab_slug_generator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"brand_system_design.md",
"gsap_animation_patterns.md",
"single_file_html_discipline.md"
]
}
]
}

View file

@ -0,0 +1,625 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "product-team",
"skill_count": 17,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "agile-product-owner",
"path": "product-team/agile-product-owner/skills/agile-product-owner",
"description": "Agile product ownership for backlog management and sprint execution. Covers user story writing, acceptance criteria, sprint planning, and velocity tracking. Use when writing user stories, creating acceptance criteria, planning sprints, estimating story points, breaking down epics, or prioritizing the backlog.",
"tools": [
{
"script": "product-team/agile-product-owner/skills/agile-product-owner/scripts/user_story_generator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 product-team/agile-product-owner/skills/agile-product-owner/scripts/user_story_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 product-team/agile-product-owner/skills/agile-product-owner/scripts/user_story_generator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"sprint-planning-guide.md",
"user-story-templates.md"
]
},
{
"name": "apple-hig-expert",
"path": "product-team/apple-hig-expert/skills/apple-hig-expert",
"description": "Audits and designs iOS/macOS/watchOS/visionOS interfaces against the Apple Human Interface Guidelines, including the Liquid Glass design language (announced WWDC25, shipped with iOS 26/macOS Tahoe, Sept 2025). Use when reviewing an Apple-platform mockup or app for HIG compliance, checking contrast or tap-target sizes, or designing native-feeling Apple UI (e.g., 'audit my iOS app against the HIG', 'is this text readable on Liquid Glass?').",
"tools": [
{
"script": "product-team/apple-hig-expert/skills/apple-hig-expert/scripts/hig_checker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/apple-hig-expert/skills/apple-hig-expert/scripts/hig_checker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"accessibility.md",
"platform-specifics.md",
"visual-design.md"
]
},
{
"name": "code-to-prd",
"path": "product-team/code-to-prd/skills/code-to-prd",
"description": "Reverse-engineer any codebase into a complete Product Requirements Document (PRD). Analyzes routes, components, state management, API integrations, and user interactions to produce business-readable documentation detailed enough for engineers or AI agents to fully reconstruct every page and endpoint. Works with frontend frameworks (React, Vue, Angular, Svelte, Next.js, Nuxt), backend frameworks (NestJS, Django, Express, FastAPI), and fullstack applications. Use when users mention: generate PRD, reverse-engineer requirements, code to documentation, extract product specs from code, document \u2026",
"tools": [
{
"script": "product-team/code-to-prd/skills/code-to-prd/scripts/codebase_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/code-to-prd/skills/code-to-prd/scripts/codebase_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "product-team/code-to-prd/skills/code-to-prd/scripts/prd_scaffolder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/code-to-prd/skills/code-to-prd/scripts/prd_scaffolder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"framework-patterns.md",
"prd-quality-checklist.md"
]
},
{
"name": "research-summarizer",
"path": "product-team/research-summarizer/skills/research-summarizer",
"description": "Structured research summarization agent skill for non-dev users. Handles academic papers, web articles, reports, and documentation. Extracts key findings, generates comparative analyses, and produces properly formatted citations. Use when: user wants to summarize a research paper, compare multiple sources, extract citations from documents, or create structured research briefs. Plugin for Claude Code, Codex, Gemini CLI, and OpenClaw.",
"tools": [
{
"script": "product-team/research-summarizer/skills/research-summarizer/scripts/extract_citations.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/research-summarizer/skills/research-summarizer/scripts/extract_citations.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "product-team/research-summarizer/skills/research-summarizer/scripts/format_summary.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/research-summarizer/skills/research-summarizer/scripts/format_summary.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"citation-formats.md",
"summary-templates.md"
]
},
{
"name": "competitive-teardown",
"path": "product-team/skills/competitive-teardown",
"description": "Analyzes competitor products and companies by synthesizing data from pricing pages, app store reviews, job postings, SEO signals, and social media into structured competitive intelligence. Produces feature comparison matrices scored across 12 dimensions, SWOT analyses, positioning maps, UX audits, pricing model breakdowns, action item roadmaps, and stakeholder presentation templates. Use when conducting competitor analysis, comparing products against competitors, researching the competitive landscape, building battle cards for sales, preparing for a product strategy or roadmap session \u2026",
"tools": [
{
"script": "product-team/skills/competitive-teardown/scripts/competitive_matrix_builder.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/competitive-teardown/scripts/competitive_matrix_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"analysis-templates.md",
"competitive-analysis-frameworks.md",
"data-collection-guide.md",
"scoring-rubric.md"
]
},
{
"name": "experiment-designer",
"path": "product-team/skills/experiment-designer",
"description": "Use when planning product experiments, writing testable hypotheses, estimating sample size, prioritizing tests, or interpreting A/B outcomes with practical statistical rigor.",
"tools": [
{
"script": "product-team/skills/experiment-designer/scripts/sample_size_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/experiment-designer/scripts/sample_size_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"experiment-playbook.md",
"statistics-reference.md"
]
},
{
"name": "landing-page-generator",
"path": "product-team/skills/landing-page-generator",
"description": "Generates high-converting landing pages as complete Next.js/React (TSX) components with Tailwind CSS. Creates hero sections, feature grids, pricing tables, FAQ accordions, testimonial blocks, and CTA sections using proven copy frameworks (PAS, AIDA, BAB). Outputs SEO meta tags, structured data, and performance-optimised code targeting Core Web Vitals (LCP < 1s, CLS < 0.1). Use when the user asks to create a landing page, marketing page, homepage, single-page site, lead capture page, campaign page, promo page, or conversion-optimised web page \u2014 or when they want to A/B test landing page \u2026",
"tools": [
{
"script": "product-team/skills/landing-page-generator/scripts/landing_page_scaffolder.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/landing-page-generator/scripts/landing_page_scaffolder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"conversion-patterns.md",
"copy-frameworks.md",
"landing-page-patterns.md",
"seo-checklist.md"
]
},
{
"name": "product-analytics",
"path": "product-team/skills/product-analytics",
"description": "Use when defining product KPIs, building metric dashboards, running cohort or retention analysis, or interpreting feature adoption trends across product stages.",
"tools": [
{
"script": "product-team/skills/product-analytics/scripts/metrics_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/product-analytics/scripts/metrics_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": [
"dashboard-templates.md",
"metrics-frameworks.md"
]
},
{
"name": "product-discovery",
"path": "product-team/skills/product-discovery",
"description": "Use when validating product opportunities, mapping assumptions, planning discovery sprints, or testing problem-solution fit before committing delivery resources.",
"tools": [
{
"script": "product-team/skills/product-discovery/scripts/assumption_mapper.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/product-discovery/scripts/assumption_mapper.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": [
"discovery-frameworks.md"
]
},
{
"name": "product-manager-toolkit",
"path": "product-team/skills/product-manager-toolkit",
"description": "Comprehensive toolkit for product managers including RICE prioritization, customer interview analysis, PRD templates, discovery frameworks, and go-to-market strategies. Use when prioritizing features, synthesizing user research, writing requirement documentation, or developing product strategy.",
"tools": [
{
"script": "product-team/skills/product-manager-toolkit/scripts/customer_interview_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/product-manager-toolkit/scripts/customer_interview_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "product-team/skills/product-manager-toolkit/scripts/rice_prioritizer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/product-manager-toolkit/scripts/rice_prioritizer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"frameworks.md",
"input-output-examples.md",
"prd_templates.md"
]
},
{
"name": "product-skills",
"path": "product-team/skills/product-skills",
"description": "Use when coordinating product work across the 12 bundled product sub-skills (RICE, OKRs, UX research, design tokens, competitive teardown, analytics, experiments, discovery, roadmaps, spec-to-repo, landing pages, SaaS scaffolding) or the 4 standalone product-team plugins (user stories, Apple HIG, code-to-PRD, research summarizer). Triggers on 'help me prioritize', 'plan a product experiment', 'we ship features nobody uses', 'run the discovery loop', 'is our OST sound'. Forks context to route to one sub-skill via a deterministic signal router and returns a digest; can also drive a \u2026",
"tools": [
{
"script": "product-team/skills/product-skills/scripts/discovery_cadence_tracker.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 product-team/skills/product-skills/scripts/discovery_cadence_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 product-team/skills/product-skills/scripts/discovery_cadence_tracker.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "product-team/skills/product-skills/scripts/ost_linter.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 product-team/skills/product-skills/scripts/ost_linter.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 product-team/skills/product-skills/scripts/ost_linter.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "product-team/skills/product-skills/scripts/product_goal_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 product-team/skills/product-skills/scripts/product_goal_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 product-team/skills/product-skills/scripts/product_goal_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"ai_product_evals.md",
"continuous_discovery_canon.md",
"product_operating_model.md"
]
},
{
"name": "product-strategist",
"path": "product-team/skills/product-strategist",
"description": "Strategic product leadership toolkit for Head of Product covering OKR cascade generation, quarterly planning, competitive landscape analysis, product vision documents, and team scaling proposals. Use when creating quarterly OKR documents, defining product goals or KPIs, building product roadmaps, running competitive analysis, drafting team structure or hiring plans, aligning product strategy across engineering and design, or generating cascaded goal hierarchies from company to team level.",
"tools": [
{
"script": "product-team/skills/product-strategist/scripts/okr_cascade_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/product-strategist/scripts/okr_cascade_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": [
"examples",
"okr_framework.md",
"strategy_types.md"
]
},
{
"name": "roadmap-communicator",
"path": "product-team/skills/roadmap-communicator",
"description": "Use when preparing roadmap narratives, release notes, changelogs, or stakeholder updates tailored for executives, engineering teams, and customers.",
"tools": [
{
"script": "product-team/skills/roadmap-communicator/scripts/changelog_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/roadmap-communicator/scripts/changelog_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"communication-templates.md",
"roadmap-templates.md"
]
},
{
"name": "saas-scaffolder",
"path": "product-team/skills/saas-scaffolder",
"description": "Generates complete, production-ready SaaS project boilerplate including authentication, database schemas, billing integration, API routes, and a working dashboard using Next.js 14+ App Router, TypeScript, Tailwind CSS, shadcn/ui, Drizzle ORM, and Stripe. Use when the user wants to create a new SaaS app, start a subscription-based web project, scaffold a Next.js application, or mentions terms like starter template, boilerplate, new project, or wiring up auth and payments.",
"tools": [
{
"script": "product-team/skills/saas-scaffolder/scripts/project_bootstrapper.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/saas-scaffolder/scripts/project_bootstrapper.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"architecture-patterns.md",
"auth-billing-guide.md",
"saas-architecture-patterns.md",
"tech-stack-comparison.md"
]
},
{
"name": "spec-to-repo",
"path": "product-team/skills/spec-to-repo",
"description": "Use when the user says 'build me an app', 'create a project from this spec', 'scaffold a new repo', 'generate a starter', 'turn this idea into code', 'bootstrap a project', 'I have requirements and need a codebase', or provides a natural-language project specification and expects a complete, runnable repository. Stack-agnostic: Next.js, FastAPI, Rails, Go, Rust, Flutter, and more.",
"tools": [
{
"script": "product-team/skills/spec-to-repo/scripts/validate_project.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/spec-to-repo/scripts/validate_project.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"spec-parsing-guide.md",
"stack-templates.md"
]
},
{
"name": "ui-design-system",
"path": "product-team/skills/ui-design-system",
"description": "UI design system toolkit for Senior UI Designer including design token generation, component documentation, responsive design calculations, and developer handoff tools. Use when creating design systems, generating design tokens, maintaining visual consistency, or facilitating design-dev collaboration and developer handoff.",
"tools": [
{
"script": "product-team/skills/ui-design-system/scripts/design_token_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 product-team/skills/ui-design-system/scripts/design_token_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"component-architecture.md",
"developer-handoff.md",
"responsive-calculations.md",
"token-generation.md"
]
},
{
"name": "ux-researcher-designer",
"path": "product-team/skills/ux-researcher-designer",
"description": "UX research and design toolkit for Senior UX Designer/Researcher including data-driven persona generation, journey mapping, usability testing frameworks, and research synthesis. Use when conducting user research, creating personas, mapping user journeys, planning usability tests, or validating designs.",
"tools": [
{
"script": "product-team/skills/ux-researcher-designer/scripts/persona_generator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 product-team/skills/ux-researcher-designer/scripts/persona_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 product-team/skills/ux-researcher-designer/scripts/persona_generator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"example-personas.md",
"journey-mapping-guide.md",
"persona-methodology.md",
"usability-testing-frameworks.md"
]
}
]
}

View file

@ -0,0 +1,568 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "productivity",
"skill_count": 7,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "andreessen",
"path": "productivity/andreessen/skills/andreessen",
"description": "Marc Andreessen-mode decision and productivity skill. A blunt, market-first operator that pressure-tests ideas, ventures, features, and career bets through Andreessen's actual frameworks \u2014 market dominates team and product; the only milestone that matters is product/market fit; bias to build over deliberate. Use when the user says 'andreessen', 'pmarca mode', 'should I build this', 'is there a market', 'are we at product/market fit', 'pmf check', 'pressure-test this idea', 'be brutal about this venture', 'market-first take', or wants a no-disclaimers, no-hedging, confidence-leveled verdict \u2026",
"tools": [
{
"script": "productivity/andreessen/skills/andreessen/scripts/anti_todo_card.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/andreessen/skills/andreessen/scripts/anti_todo_card.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/andreessen/skills/andreessen/scripts/anti_todo_card.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/andreessen/skills/andreessen/scripts/market_first_evaluator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/andreessen/skills/andreessen/scripts/market_first_evaluator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/andreessen/skills/andreessen/scripts/market_first_evaluator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/andreessen/skills/andreessen/scripts/pmf_signal_scorer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/andreessen/skills/andreessen/scripts/pmf_signal_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/andreessen/skills/andreessen/scripts/pmf_signal_scorer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"market_first_canon.md",
"operating_prompt.md",
"personal_productivity_system.md",
"pmf_and_build_canon.md"
]
},
{
"name": "capture",
"path": "productivity/capture/skills/capture",
"description": "Captures and organizes chaotic brain dumps into a structured, actionable system with zero information loss. Use this skill whenever the user says 'capture this', 'brain dump', 'let me dump some ideas', 'I've got a bunch of thoughts', 'here's everything on my mind', 'idea dump', 'let me get this out of my head', 'I need to organize my thoughts', 'here's what I'm thinking', or any variation where someone is unloading a messy stream of ideas, tasks, thoughts, and plans wanting them turned into something coherent. Also trigger when the user pastes or dictates a long, unstructured block of mixed \u2026",
"tools": [
{
"script": "productivity/capture/skills/capture/scripts/complexity_estimator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/capture/skills/capture/scripts/complexity_estimator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/capture/skills/capture/scripts/complexity_estimator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/capture/skills/capture/scripts/dump_classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/capture/skills/capture/scripts/dump_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/capture/skills/capture/scripts/dump_classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/capture/skills/capture/scripts/workspace_inventory.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/capture/skills/capture/scripts/workspace_inventory.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/capture/skills/capture/scripts/workspace_inventory.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"complexity_matching.md",
"voice_preservation.md",
"workspace_detection.md"
]
},
{
"name": "inbox-setup",
"path": "productivity/email/skills/inbox-setup",
"description": "One-time setup skill that builds a personalized inbox triage knowledge base via interactive interview. Interviews the user about their email patterns, business context, reply style, and priorities using grill-me discipline (one question at a time, forcing format where possible, dependency-ordered, each question explains why I'm asking), then generates the knowledge base files that power the companion 'inbox-triage' skill. Run this once before using inbox-triage for the first time. Re-run when business, pricing, or priorities change significantly. Triggers: 'set up my inbox', 'configure \u2026",
"tools": [
{
"script": "productivity/email/skills/inbox-setup/scripts/kb_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/email/skills/inbox-setup/scripts/kb_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/email/skills/inbox-setup/scripts/kb_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/email/skills/inbox-setup/scripts/section_progress_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 productivity/email/skills/inbox-setup/scripts/section_progress_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "productivity/email/skills/inbox-setup/scripts/voice_sample_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/email/skills/inbox-setup/scripts/voice_sample_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/email/skills/inbox-setup/scripts/voice_sample_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": true
},
"references": [
"grill_me_section_walk.md",
"kb_file_contract.md",
"voice_calibration.md"
]
},
{
"name": "inbox-triage",
"path": "productivity/email/skills/inbox-triage",
"description": "Runs a full inbox triage using the knowledge base created by the 'inbox-setup' skill. Light-intake by design (most invocations skip questions and run with KB-default preferences); asks at most 2 grill-me override questions when invocation is outside normal cadence or includes category-skip intent. Searches recent emails, classifies them via the user's taxonomy, researches new senders, generates recommendations, drafts replies (NEVER sends), delivers a report in the user's preferred format, and updates the knowledge base with learnings. Designed to run on a recurring schedule (1-3x daily) or \u2026",
"tools": [
{
"script": "productivity/email/skills/inbox-triage/scripts/draft_safety_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/email/skills/inbox-triage/scripts/draft_safety_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/email/skills/inbox-triage/scripts/draft_safety_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/email/skills/inbox-triage/scripts/kb_reader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/email/skills/inbox-triage/scripts/kb_reader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/email/skills/inbox-triage/scripts/kb_reader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/email/skills/inbox-triage/scripts/search_window_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 productivity/email/skills/inbox-triage/scripts/search_window_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"drafts_only_safety.md",
"kb_file_contract.md",
"triage_decision_framework.md"
]
},
{
"name": "handoff",
"path": "productivity/handoff/skills/handoff",
"description": "Compact the current conversation into a handoff document for another agent to pick up. Save to a user-configured location (OS temp, home folder, or per-project .handoff/), redact secrets before write, suggest skills for the next session, and auto-load the latest handoff on the next SessionStart. First-run setup asks where to save so the project folder never gets cluttered. Use when the user says 'hand this off', 'handoff doc', 'summarize this for a new session', 'compact this conversation', 'I'm ending this session', 'pick this up later', or any variation signaling intent to pass work to a \u2026",
"tools": [
{
"script": "productivity/handoff/skills/handoff/scripts/cleanup.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/cleanup.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/cleanup.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/handoff/skills/handoff/scripts/config_loader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/config_loader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/config_loader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/handoff/skills/handoff/scripts/handoff_self_check.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/handoff_self_check.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/handoff_self_check.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/handoff/skills/handoff/scripts/handoff_template_generator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/handoff_template_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/handoff_template_generator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/handoff/skills/handoff/scripts/redaction_linter.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/redaction_linter.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/redaction_linter.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/handoff/skills/handoff/scripts/setup.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/setup.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/setup.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/handoff/skills/handoff/scripts/skill_recommender.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/skill_recommender.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/handoff/skills/handoff/scripts/skill_recommender.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": true,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"configuration.md",
"deduplication_discipline.md",
"handoff_prompt.md",
"handoff_structure.md",
"redaction_checklist.md"
]
},
{
"name": "reflect",
"path": "productivity/reflect/skills/reflect",
"description": "Mid-conversation reflection skill that pauses execution and zooms out from detail-mode to honestly reassess direction, assumptions, and bias. Use when the user says 'reflect', 'take a step back', 'step back', 'zoom out', 'are we missing something', 'bigger picture', 'sanity check this', 'are we on track', 'are we overthinking this', 'forest for the trees', or any variation signaling intent to break out of detail-mode and reassess. Also trigger when the conversation has gone deep on implementation details without strategic check-in, or when the user shows signs of being stuck \u2014 that's often \u2026",
"tools": [
{
"script": "productivity/reflect/skills/reflect/scripts/bias_pattern_detector.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/reflect/skills/reflect/scripts/bias_pattern_detector.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/reflect/skills/reflect/scripts/bias_pattern_detector.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/reflect/skills/reflect/scripts/conversation_depth_analyzer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/reflect/skills/reflect/scripts/conversation_depth_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/reflect/skills/reflect/scripts/conversation_depth_analyzer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/reflect/skills/reflect/scripts/directional_recommendation_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/reflect/skills/reflect/scripts/directional_recommendation_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/reflect/skills/reflect/scripts/directional_recommendation_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"cognitive_bias_canon.md",
"conversation_reflection_practice.md",
"honest_output_discipline.md"
]
},
{
"name": "roast",
"path": "productivity/roast/skills/roast",
"description": "Use when someone asks to roast an idea, pressure-test or stress-test an idea, validate a business idea, \"convene the panel\", get a brutal second opinion before building something, or says \"/roast\". Spins up a 5-angle panel (Critic, Champion, Analyst, Investigator, Customer) that attacks the idea from every angle, then a Judge returns one GO / RESHAPE / KILL verdict with the cheapest test to de-risk it.",
"tools": [
{
"script": "productivity/roast/skills/roast/scripts/brief_builder.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/roast/skills/roast/scripts/brief_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/roast/skills/roast/scripts/brief_builder.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/roast/skills/roast/scripts/cheapest_test_designer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/roast/skills/roast/scripts/cheapest_test_designer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/roast/skills/roast/scripts/cheapest_test_designer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "productivity/roast/skills/roast/scripts/verdict_synthesizer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 productivity/roast/skills/roast/scripts/verdict_synthesizer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 productivity/roast/skills/roast/scripts/verdict_synthesizer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"adversarial_panel_canon.md",
"cheapest_test_canon.md",
"verdict_synthesis_method.md"
]
}
]
}

View file

@ -0,0 +1,377 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "project-management",
"skill_count": 9,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "atlassian-admin",
"path": "project-management/skills/atlassian-admin",
"description": "Atlassian Administrator for managing and organizing Atlassian products (Jira, Confluence, Bitbucket, Trello), users, permissions, security, integrations, system configuration, and org-wide governance. Use when asked to add users to Jira, change Confluence permissions, configure access control, update admin settings, manage Atlassian groups, set up SSO, install marketplace apps, review security policies, or handle any org-wide Atlassian administration task.",
"tools": [
{
"script": "project-management/skills/atlassian-admin/scripts/permission_audit_tool.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/atlassian-admin/scripts/permission_audit_tool.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"security-hardening-guide.md",
"user-provisioning-checklist.md"
]
},
{
"name": "atlassian-templates",
"path": "project-management/skills/atlassian-templates",
"description": "Atlassian Template and Files Creator/Modifier expert for creating, modifying, and managing Jira and Confluence templates, blueprints, custom layouts, reusable components, and standardized content structures. Use when building org-wide templates, custom blueprints, page layouts, and automated content generation.",
"tools": [
{
"script": "project-management/skills/atlassian-templates/scripts/template_scaffolder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/atlassian-templates/scripts/template_scaffolder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"governance-framework.md",
"template-design-patterns.md"
]
},
{
"name": "confluence-expert",
"path": "project-management/skills/confluence-expert",
"description": "Atlassian Confluence expert for creating and managing spaces, knowledge bases, and documentation. Configures space permissions and hierarchies, creates page templates with macros, sets up documentation taxonomies, designs page layouts, and manages content governance. Use when users need to build or restructure a Confluence space, design page hierarchies with permission structures, author or standardise documentation templates, embed Jira reports in pages, run knowledge base audits, or establish documentation standards and collaborative workflows.",
"tools": [
{
"script": "project-management/skills/confluence-expert/scripts/content_audit_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/confluence-expert/scripts/content_audit_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "project-management/skills/confluence-expert/scripts/space_structure_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/confluence-expert/scripts/space_structure_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"macro-cheat-sheet.md",
"space-architecture-patterns.md",
"templates.md"
]
},
{
"name": "jira-expert",
"path": "project-management/skills/jira-expert",
"description": "Atlassian Jira expert for creating and managing projects, planning, product discovery, JQL queries, workflows, custom fields, automation, reporting, and all Jira features. Use when setting up or configuring Jira projects, writing JQL and advanced searches, creating dashboards, designing workflows, or performing technical Jira operations.",
"tools": [
{
"script": "project-management/skills/jira-expert/scripts/jql_query_builder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/jira-expert/scripts/jql_query_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "project-management/skills/jira-expert/scripts/workflow_validator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/jira-expert/scripts/workflow_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"AUTOMATION.md",
"WORKFLOWS.md",
"automation-examples.md",
"jql-examples.md"
]
},
{
"name": "meeting-analyzer",
"path": "project-management/skills/meeting-analyzer",
"description": "Analyzes meeting transcripts and recordings to surface behavioral patterns, communication anti-patterns, and actionable coaching feedback. Use this skill whenever the user uploads or points to meeting transcripts (.txt, .md, .vtt, .srt, .docx), asks about their communication habits, wants feedback on how they run meetings, requests speaking ratio analysis, mentions filler words or conflict avoidance, or wants to compare their communication across time periods. Also trigger when users mention tools like Granola, Otter, Fireflies, or Zoom transcripts. Even if the user just says \"look at my \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "pm-skills",
"path": "project-management/skills/pm-skills",
"description": "Use when coordinating project-delivery work across the 8 project-management sub-skills \u2014 sprint/velocity analytics, portfolio health, Jira/JQL, Confluence, Atlassian admin, templates, meeting analysis, team comms. Triggers on 'our sprints feel off', 'project health report', 'audit our Jira permissions', 'when will it be done', 'run the delivery loop'. Forks context to route to one sub-skill via a deterministic signal router and returns a digest; can also drive a full goal\u2192plan\u2192execute\u2192verify\u2192close delivery loop through the repo-wide agent-harness with Jira MCP data bridged into the domain's \u2026",
"tools": [
{
"script": "project-management/skills/pm-skills/scripts/delivery_loop_gate.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 project-management/skills/pm-skills/scripts/delivery_loop_gate.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 project-management/skills/pm-skills/scripts/delivery_loop_gate.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "project-management/skills/pm-skills/scripts/jira_snapshot_bridge.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 project-management/skills/pm-skills/scripts/jira_snapshot_bridge.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 project-management/skills/pm-skills/scripts/jira_snapshot_bridge.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "project-management/skills/pm-skills/scripts/pm_goal_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 project-management/skills/pm-skills/scripts/pm_goal_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 project-management/skills/pm-skills/scripts/pm_goal_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"agentic_delivery_governance.md",
"flow_forecasting_canon.md",
"pm_loop_playbook.md"
]
},
{
"name": "scrum-master",
"path": "project-management/skills/scrum-master",
"description": "Advanced Scrum Master skill for data-driven agile team analysis and coaching. Use when the user asks about sprint planning, velocity tracking, retrospectives, standup facilitation, backlog grooming, story points, burndown charts, blocker resolution, or agile team health. Runs Python scripts to analyse sprint JSON exports from Jira or similar tools: velocity_analyzer.py for Monte Carlo sprint forecasting, sprint_health_scorer.py for multi-dimension health scoring, and retrospective_analyzer.py for action-item and theme tracking. Produces confidence-interval forecasts, health grade reports \u2026",
"tools": [
{
"script": "project-management/skills/scrum-master/scripts/retrospective_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/scrum-master/scripts/retrospective_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "project-management/skills/scrum-master/scripts/sprint_health_scorer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/scrum-master/scripts/sprint_health_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "project-management/skills/scrum-master/scripts/velocity_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/scrum-master/scripts/velocity_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": true
},
"references": [
"retro-formats.md",
"team-dynamics-framework.md",
"velocity-forecasting-guide.md"
]
},
{
"name": "senior-pm",
"path": "project-management/skills/senior-pm",
"description": "Senior Project Manager for enterprise software, SaaS, and digital transformation projects. Specializes in portfolio management, quantitative risk analysis, resource optimization, stakeholder alignment, and executive reporting. Uses advanced methodologies including EMV analysis, Monte Carlo simulation, WSJF prioritization, and multi-dimensional health scoring. Use when a user needs help with project plans, project status reports, risk assessments, resource allocation, project roadmaps, milestone tracking, team capacity planning, portfolio health reviews, program management, or \u2026",
"tools": [
{
"script": "project-management/skills/senior-pm/scripts/project_health_dashboard.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/senior-pm/scripts/project_health_dashboard.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "project-management/skills/senior-pm/scripts/resource_capacity_planner.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/senior-pm/scripts/resource_capacity_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "project-management/skills/senior-pm/scripts/risk_matrix_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 project-management/skills/senior-pm/scripts/risk_matrix_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": true
},
"references": [
"portfolio-kpis.md",
"portfolio-prioritization-models.md",
"risk-management-framework.md"
]
},
{
"name": "team-communications",
"path": "project-management/skills/team-communications",
"description": "Write internal company communications \u2014 3P updates (Progress/Plans/Problems), company-wide newsletters, FAQ roundups, incident reports, leadership updates, status reports, project updates, and general internal comms. Use this skill any time the user asks to draft, edit, or format something meant for internal audiences. Trigger on keywords like \"3P\", \"weekly update\", \"newsletter\", \"FAQ\", \"internal comms\", \"status report\", \"company update\", \"team update\", \"incident report\", or any request to summarize work for leadership, teammates, or the broader company. Even casual requests like \"write my \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": false,
"close_out": false
},
"references": [
"3p-updates.md",
"company-newsletter.md",
"faq-answers.md",
"general-comms.md"
]
}
]
}

View file

@ -0,0 +1,848 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "ra-qm-team",
"skill_count": 19,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "eu-ai-act-specialist",
"path": "ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist",
"description": "EU AI Act (Regulation (EU) 2024/1689) operational compliance for compliance teams. Three Article-level decisions: (1) What's the risk tier of this AI system \u2014 prohibited (Art. 5), high-risk (Art. 6 + Annex III), limited-risk (Art. 50), or minimal-risk? (2) For high-risk systems, what's the Article 43 conformity assessment route (Module A internal control vs Module H full QMS + notified body) and what goes in the Annex IV technical documentation? (3) Per organizational role (provider / deployer / importer / distributor / authorized representative), what are the active obligations and \u2026",
"tools": [
{
"script": "ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist/scripts/ai_act_obligation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist/scripts/ai_act_obligation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist/scripts/ai_system_risk_classifier.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist/scripts/ai_system_risk_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist/scripts/conformity_assessment_planner.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/compliance-team-eu-ai-act/skills/eu-ai-act-specialist/scripts/conformity_assessment_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"cross_framework_mapping_ai_act.md",
"eu_ai_act_titles.md",
"gpai_obligations.md",
"high_risk_systems_annex_iii.md"
]
},
{
"name": "iso42001-specialist",
"path": "ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist",
"description": "ISO/IEC 42001:2023 AI Management System (AIMS) specialist for compliance teams running internal audits. Three decisions: (1) Where are the gaps against Clauses 4-10 and what do we close first? (2) What goes in the AI risk register and which Annex A controls treat each risk? (3) What's the 12-month internal audit plan that satisfies Clause 9.2? Use when preparing for certification, scoping internal audit cycles, or onboarding AI systems into an existing ISMS (27001) / QMS (13485) program. NOT an executive AI strategy skill (see chief-ai-officer-advisor). NOT EU AI Act compliance (see \u2026",
"tools": [
{
"script": "ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist/scripts/ai_risk_register_builder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist/scripts/ai_risk_register_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist/scripts/aims_audit_scheduler.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist/scripts/aims_audit_scheduler.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist/scripts/aims_gap_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/compliance-team-iso42001/skills/iso42001-specialist/scripts/aims_gap_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"aims_controls_annex_a.md",
"aims_implementation_guide.md",
"cross_framework_mapping_ai.md",
"iso42001_clauses.md"
]
},
{
"name": "agent-decision-receipts",
"path": "ra-qm-team/skills/agent-decision-receipts",
"description": "Mint a tamper-evident, post-quantum-signed receipt for a consequential agent action (deploy, delete, pay, grant-access, model decision) so it can be verified later from the certificate alone. Use when an autonomous agent takes a side-effecting action that may need to be proven later, or when satisfying EU AI Act Article 12 record-keeping. Three decisions: whether an action needs a receipt, minting it, verifying it. Signing is delegated to the open-source OpenAgentOntology package. Not after-the-fact log analysis; not a hosted notary; not a legal opinion.",
"tools": [
{
"script": "ra-qm-team/skills/agent-decision-receipts/scripts/build_action_manifest.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/agent-decision-receipts/scripts/build_action_manifest.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": true,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"receipt-fields.md"
]
},
{
"name": "capa-officer",
"path": "ra-qm-team/skills/capa-officer",
"description": "CAPA system management for medical device QMS. Covers root cause analysis, corrective action planning, effectiveness verification, and CAPA metrics. Use when running CAPA investigations, 5-Why analysis, fishbone diagrams, root cause determination, corrective action tracking, effectiveness verification, or CAPA program optimization.",
"tools": [
{
"script": "ra-qm-team/skills/capa-officer/scripts/capa_tracker.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/capa-officer/scripts/capa_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 ra-qm-team/skills/capa-officer/scripts/capa_tracker.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "ra-qm-team/skills/capa-officer/scripts/root_cause_analyzer.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/capa-officer/scripts/root_cause_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"effectiveness-verification-guide.md",
"rca-methodologies.md"
]
},
{
"name": "eu-ai-act-specialist",
"path": "ra-qm-team/skills/eu-ai-act-specialist",
"description": "EU AI Act (Regulation (EU) 2024/1689) operational compliance for compliance teams. Three Article-level decisions: (1) What's the risk tier of this AI system \u2014 prohibited (Art. 5), high-risk (Art. 6 + Annex III), limited-risk (Art. 50), or minimal-risk? (2) For high-risk systems, what's the Article 43 conformity assessment route (Module A internal control vs Module H full QMS + notified body) and what goes in the Annex IV technical documentation? (3) Per organizational role (provider / deployer / importer / distributor / authorized representative), what are the active obligations and \u2026",
"tools": [
{
"script": "ra-qm-team/skills/eu-ai-act-specialist/scripts/ai_act_obligation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/eu-ai-act-specialist/scripts/ai_act_obligation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/eu-ai-act-specialist/scripts/ai_system_risk_classifier.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/eu-ai-act-specialist/scripts/ai_system_risk_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/eu-ai-act-specialist/scripts/conformity_assessment_planner.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/eu-ai-act-specialist/scripts/conformity_assessment_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"cross_framework_mapping_ai_act.md",
"eu_ai_act_titles.md",
"gpai_obligations.md",
"high_risk_systems_annex_iii.md"
]
},
{
"name": "fda-consultant-specialist",
"path": "ra-qm-team/skills/fda-consultant-specialist",
"description": "FDA regulatory consultant for medical device companies. Provides 510(k)/PMA/De Novo pathway guidance, QMSR (21 CFR 820, which incorporates ISO 13485:2016 by reference since 2026-02-02; formerly QSR) compliance, HIPAA assessments, and device cybersecurity. Use when user mentions FDA submission, 510(k), PMA, De Novo, QMSR, QSR, ISO 13485 for FDA, premarket, predicate device, substantial equivalence, HIPAA medical device, or FDA cybersecurity.",
"tools": [
{
"script": "ra-qm-team/skills/fda-consultant-specialist/scripts/fda_submission_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/fda-consultant-specialist/scripts/fda_submission_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/fda-consultant-specialist/scripts/hipaa_risk_assessment.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/fda-consultant-specialist/scripts/hipaa_risk_assessment.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/fda-consultant-specialist/scripts/qsr_compliance_checker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/fda-consultant-specialist/scripts/qsr_compliance_checker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"device_cybersecurity_guidance.md",
"fda_capa_requirements.md",
"fda_submission_guide.md",
"hipaa_compliance_framework.md",
"qsr_compliance_requirements.md"
]
},
{
"name": "gdpr-dsgvo-expert",
"path": "ra-qm-team/skills/gdpr-dsgvo-expert",
"description": "GDPR and German DSGVO compliance automation. Scans codebases for privacy risks, generates DPIA documentation, tracks data subject rights requests with Art. 12(3) one-month deadlines. Use when running GDPR compliance assessments, privacy audits, data protection planning, DPIA generation, or data subject rights (DSAR) management (e.g., 'check this service for GDPR risks', 'track an access request deadline'). Final compliance determinations route to the DPO or legal counsel.",
"tools": [
{
"script": "ra-qm-team/skills/gdpr-dsgvo-expert/scripts/data_subject_rights_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/gdpr-dsgvo-expert/scripts/data_subject_rights_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/gdpr-dsgvo-expert/scripts/dpia_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/gdpr-dsgvo-expert/scripts/dpia_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/gdpr-dsgvo-expert/scripts/gdpr_compliance_checker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/gdpr-dsgvo-expert/scripts/gdpr_compliance_checker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"dpia_methodology.md",
"gdpr_audit_playbook.md",
"gdpr_compliance_guide.md",
"german_bdsg_requirements.md"
]
},
{
"name": "information-security-manager-iso27001",
"path": "ra-qm-team/skills/information-security-manager-iso27001",
"description": "ISO 27001 ISMS implementation and cybersecurity governance for HealthTech and MedTech companies. Use when designing an ISMS, running security risk assessments, implementing controls, pursuing ISO 27001 certification, preparing security audits, responding to security incidents, or verifying compliance. Covers ISO 27001, ISO 27002, healthcare security, and medical device cybersecurity.",
"tools": [
{
"script": "ra-qm-team/skills/information-security-manager-iso27001/scripts/compliance_checker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/information-security-manager-iso27001/scripts/compliance_checker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/information-security-manager-iso27001/scripts/risk_assessment.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/information-security-manager-iso27001/scripts/risk_assessment.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"incident-response.md",
"iso27001-controls.md",
"risk-assessment-guide.md"
]
},
{
"name": "isms-audit-expert",
"path": "ra-qm-team/skills/isms-audit-expert",
"description": "Information Security Management System (ISMS) audit expert for ISO 27001 compliance verification, security control assessment, and certification support. Use when the user mentions ISO 27001, ISMS audit, Annex A controls, Statement of Applicability (SOA), gap analysis, nonconformity management, internal audit, surveillance audit, or security certification preparation. Helps review control implementation evidence, document audit findings, classify nonconformities, generate risk-based audit plans, map controls to Annex A requirements, prepare Stage 1 and Stage 2 audit documentation, and \u2026",
"tools": [
{
"script": "ra-qm-team/skills/isms-audit-expert/scripts/isms_audit_scheduler.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/isms-audit-expert/scripts/isms_audit_scheduler.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"cloud-security-audit.md",
"iso27001-audit-methodology.md",
"iso27001_audit_playbook.md",
"security-control-testing.md"
]
},
{
"name": "iso42001-specialist",
"path": "ra-qm-team/skills/iso42001-specialist",
"description": "ISO/IEC 42001:2023 AI Management System (AIMS) specialist for compliance teams running internal audits. Three decisions: (1) Where are the gaps against Clauses 4-10 and what do we close first? (2) What goes in the AI risk register and which Annex A controls treat each risk? (3) What's the 12-month internal audit plan that satisfies Clause 9.2? Use when preparing for certification, scoping internal audit cycles, or onboarding AI systems into an existing ISMS (27001) / QMS (13485) program. NOT an executive AI strategy skill (see chief-ai-officer-advisor). NOT EU AI Act compliance (see \u2026",
"tools": [
{
"script": "ra-qm-team/skills/iso42001-specialist/scripts/ai_risk_register_builder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/iso42001-specialist/scripts/ai_risk_register_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/iso42001-specialist/scripts/aims_audit_scheduler.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/iso42001-specialist/scripts/aims_audit_scheduler.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/iso42001-specialist/scripts/aims_gap_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/iso42001-specialist/scripts/aims_gap_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"aims_controls_annex_a.md",
"aims_implementation_guide.md",
"cross_framework_mapping_ai.md",
"iso42001_clauses.md"
]
},
{
"name": "mdr-745-specialist",
"path": "ra-qm-team/skills/mdr-745-specialist",
"description": "EU MDR 2017/745 compliance specialist for medical device classification, technical documentation, clinical evidence, and post-market surveillance. Covers Annex VIII classification rules, Annex II/III technical files, Annex XIV clinical evaluation, Art. 86 PSUR schedules, and EUDAMED integration. Use when classifying a medical device under MDR, building or gap-checking a technical file, planning clinical evaluation or PMS/PSUR cadence, or preparing for notified body review (e.g., 'what class is my device under MDR', 'review my PSUR schedule').",
"tools": [
{
"script": "ra-qm-team/skills/mdr-745-specialist/scripts/mdr_gap_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/mdr-745-specialist/scripts/mdr_gap_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"clinical-evidence-requirements.md",
"mdr-classification-guide.md",
"technical-documentation-templates.md"
]
},
{
"name": "qms-audit-expert",
"path": "ra-qm-team/skills/qms-audit-expert",
"description": "ISO 13485 internal audit expertise for medical device QMS. Covers audit planning, execution, nonconformity classification, and CAPA verification. Use when planning internal audits, executing audits, classifying findings, preparing for external audits, or managing an audit program.",
"tools": [
{
"script": "ra-qm-team/skills/qms-audit-expert/scripts/audit_schedule_optimizer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/qms-audit-expert/scripts/audit_schedule_optimizer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"iso13485-audit-guide.md",
"iso13485_audit_playbook.md",
"nonconformity-classification.md"
]
},
{
"name": "quality-documentation-manager",
"path": "ra-qm-team/skills/quality-documentation-manager",
"description": "Document control system management for medical device QMS. Covers document numbering, version control, change management, and 21 CFR Part 11 compliance. Use when working on document control procedures, change control workflows, document numbering, version management, electronic signature compliance, or regulatory documentation review.",
"tools": [
{
"script": "ra-qm-team/skills/quality-documentation-manager/scripts/document_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/quality-documentation-manager/scripts/document_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 ra-qm-team/skills/quality-documentation-manager/scripts/document_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "ra-qm-team/skills/quality-documentation-manager/scripts/document_version_control.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/quality-documentation-manager/scripts/document_version_control.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"21cfr11-compliance-guide.md",
"document-control-procedures.md"
]
},
{
"name": "quality-manager-qmr",
"path": "ra-qm-team/skills/quality-manager-qmr",
"description": "Senior Quality Manager Responsible Person (QMR) for HealthTech and MedTech companies. Provides quality system governance, management review leadership, regulatory compliance oversight, and quality performance monitoring per ISO 13485 Clause 5.5.2. Use when leading management reviews, setting quality policy and objectives, monitoring quality KPIs and cost of quality, or exercising QMR governance and regulatory oversight responsibilities.",
"tools": [
{
"script": "ra-qm-team/skills/quality-manager-qmr/scripts/management_review_tracker.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/quality-manager-qmr/scripts/management_review_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 ra-qm-team/skills/quality-manager-qmr/scripts/management_review_tracker.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "ra-qm-team/skills/quality-manager-qmr/scripts/quality_effectiveness_monitor.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/quality-manager-qmr/scripts/quality_effectiveness_monitor.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"management-review-guide.md",
"quality-kpi-framework.md"
]
},
{
"name": "quality-manager-qms-iso13485",
"path": "ra-qm-team/skills/quality-manager-qms-iso13485",
"description": "ISO 13485 Quality Management System implementation and maintenance for medical device organizations. Provides QMS design, documentation control, internal auditing, CAPA management, and certification support. Use when working with medical device quality systems, preparing for ISO 13485 audits, managing regulatory compliance documentation, setting up corrective actions, or building audit preparation programs. Useful for quality management, audit preparation, regulatory compliance, medical device documentation, and corrective action workflows.",
"tools": [
{
"script": "ra-qm-team/skills/quality-manager-qms-iso13485/scripts/qms_audit_checklist.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/quality-manager-qms-iso13485/scripts/qms_audit_checklist.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"iso13485-clause-requirements.md",
"qms-process-templates.md"
]
},
{
"name": "ra-qm-skills",
"path": "ra-qm-team/skills/ra-qm-skills",
"description": "Router/index for the 15 regulatory & quality-management skills bundled in this plugin (ISO 13485 QMS, EU MDR 2017/745, FDA submissions under QMSR, ISO 14971 risk, CAPA, document control, ISO 27001/ISMS, ISO 42001 AIMS, EU AI Act, GDPR/DSGVO, SOC 2, auditing). Use when a compliance request doesn't obviously match one skill and you need to pick the right one (e.g., 'prepare us for an ISO 13485 audit', 'is my AI system high-risk under the AI Act').",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": []
},
{
"name": "regulatory-affairs-head",
"path": "ra-qm-team/skills/regulatory-affairs-head",
"description": "Senior Regulatory Affairs Manager for HealthTech and MedTech companies. Prepares FDA 510(k), De Novo, and PMA submission packages; analyzes regulatory pathways for new medical devices; drafts responses to FDA deficiency letters and Notified Body queries; develops CE marking technical documentation under EU MDR 2017/745; coordinates multi-market approval strategies across FDA, EU, Health Canada, PMDA, and NMPA; and maintains regulatory intelligence on evolving standards. Use when users need to plan or execute FDA submissions, navigate 510(k) or PMA approval processes, achieve CE marking \u2026",
"tools": [
{
"script": "ra-qm-team/skills/regulatory-affairs-head/scripts/regulatory_pathway_analyzer.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/regulatory-affairs-head/scripts/regulatory_pathway_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/regulatory-affairs-head/scripts/regulatory_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/regulatory-affairs-head/scripts/regulatory_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"eu-mdr-submission-guide.md",
"fda-submission-guide.md",
"global-regulatory-pathways.md",
"iso-regulatory-requirements.md"
]
},
{
"name": "risk-management-specialist",
"path": "ra-qm-team/skills/risk-management-specialist",
"description": "Medical device risk management specialist implementing ISO 14971 throughout product lifecycle. Provides risk analysis, risk evaluation, risk control, and post-production information analysis. Use when user mentions risk management, ISO 14971, risk analysis, FMEA, fault tree analysis, hazard identification, risk control, risk matrix, benefit-risk analysis, residual risk, risk acceptability, or post-market risk.",
"tools": [
{
"script": "ra-qm-team/skills/risk-management-specialist/scripts/fmea_analyzer.py",
"wired": false,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/risk-management-specialist/scripts/fmea_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/risk-management-specialist/scripts/risk_matrix_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/risk-management-specialist/scripts/risk_matrix_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"iso14971-implementation-guide.md",
"risk-analysis-methods.md",
"risk-assessment-templates.md"
]
},
{
"name": "soc2-compliance",
"path": "ra-qm-team/skills/soc2-compliance",
"description": "Use when the user asks to prepare for SOC 2 audits, map Trust Service Criteria, build control matrices, collect audit evidence, perform gap analysis, or assess SOC 2 Type I vs Type II readiness.",
"tools": [
{
"script": "ra-qm-team/skills/soc2-compliance/scripts/control_matrix_builder.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/soc2-compliance/scripts/control_matrix_builder.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/soc2-compliance/scripts/evidence_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/soc2-compliance/scripts/evidence_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "ra-qm-team/skills/soc2-compliance/scripts/gap_analyzer.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 ra-qm-team/skills/soc2-compliance/scripts/gap_analyzer.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": true
},
"references": [
"evidence_collection_guide.md",
"soc2_audit_playbook.md",
"trust_service_criteria.md",
"type1_vs_type2.md"
]
}
]
}

View file

@ -0,0 +1,495 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "research-ops",
"skill_count": 5,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "clinical-research",
"path": "research-ops/skills/clinical-research",
"description": "Use when designing a prospective clinical study before submission \u2014 selecting and classifying endpoints (primary / key-secondary / exploratory, with surrogate-endpoint flagging), estimating sample size and power for two-arm designs (means / proportions / survival), or scoring a study plan for feasibility and a GO / GO-WITH-CONDITIONS / REDESIGN / NO-GO phase-gate decision. Every output is an ESTIMATE plus a named human owner (clinician / biostatistician / regulatory owner) \u2014 never clinical fact, never a finished protocol. Distinct from ra-qm-team, which handles the regulatory/QM submission \u2026",
"tools": [
{
"script": "research-ops/skills/clinical-research/scripts/ar_evaluator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/ar_evaluator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/ar_evaluator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/clinical-research/scripts/config_loader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/config_loader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/config_loader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/clinical-research/scripts/endpoint_selector.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/endpoint_selector.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/endpoint_selector.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/clinical-research/scripts/onboard.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/onboard.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research-ops/skills/clinical-research/scripts/phase_gate_scorer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/phase_gate_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/phase_gate_scorer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/clinical-research/scripts/sample_size_estimator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/sample_size_estimator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/clinical-research/scripts/sample_size_estimator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"endpoint_and_power.md",
"study_design_canon.md",
"trial_operations.md"
]
},
{
"name": "market-research",
"path": "research-ops/skills/market-research",
"description": "Use when doing upstream market-research methodology \u2014 sizing a market as TAM/SAM/SOM computed BOTH top-down and bottoms-up (never a single unsourced number), planning a survey sample size with finite-population correction and per-segment minimums, or scoring candidate market segments against Kotler's measurable/substantial/accessible/differentiable/actionable criteria. Outputs always show the method and the assumptions. For market-research analysts and product-marketing at the sizing/survey/segmentation moment. Distinct from marketing-skill (campaign analytics, attribution, demand-gen) \u2014 \u2026",
"tools": [
{
"script": "research-ops/skills/market-research/scripts/ar_evaluator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/market-research/scripts/ar_evaluator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/market-research/scripts/ar_evaluator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/market-research/scripts/config_loader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/market-research/scripts/config_loader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/market-research/scripts/config_loader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/market-research/scripts/market_sizer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/market-research/scripts/market_sizer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/market-research/scripts/market_sizer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/market-research/scripts/onboard.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research-ops/skills/market-research/scripts/onboard.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research-ops/skills/market-research/scripts/sample_size_planner.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/market-research/scripts/sample_size_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/market-research/scripts/sample_size_planner.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/market-research/scripts/segmentation_scorer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/market-research/scripts/segmentation_scorer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/market-research/scripts/segmentation_scorer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"market_sizing_canon.md",
"segmentation_and_ci.md",
"survey_methodology.md"
]
},
{
"name": "product-research",
"path": "research-ops/skills/product-research",
"description": "Use when planning and synthesizing product/user research as a method-and-repository discipline \u2014 selecting the right method for the goal (generative interviews vs usability test vs concept test vs validation), computing method-based saturation/sample size with an explicit confidence level, or synthesizing coded observations into insights while flagging single-source anecdotes. Never fabricates user insight; an insight requires recurrence across independent participants. Distinct from product-team/ux-researcher-designer (persona/journey artifacts), product-discovery (discovery-sprint \u2026",
"tools": [
{
"script": "research-ops/skills/product-research/scripts/ar_evaluator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/product-research/scripts/ar_evaluator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/product-research/scripts/ar_evaluator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/product-research/scripts/config_loader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/product-research/scripts/config_loader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/product-research/scripts/config_loader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/product-research/scripts/insight_synthesizer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/product-research/scripts/insight_synthesizer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/product-research/scripts/insight_synthesizer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/product-research/scripts/onboard.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research-ops/skills/product-research/scripts/onboard.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research-ops/skills/product-research/scripts/saturation_planner.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/product-research/scripts/saturation_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/product-research/scripts/saturation_planner.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/product-research/scripts/study_designer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/product-research/scripts/study_designer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/product-research/scripts/study_designer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"repository_and_synthesis.md",
"research_methods_canon.md",
"sampling_and_saturation.md"
]
},
{
"name": "research-finance",
"path": "research-ops/skills/research-finance",
"description": "Use when managing the money for an internal R&D program or portfolio \u2014 building a multi-period program budget with the F&A (indirect) split, tracking burn rate and runway against value-inflection milestones, or routing R&D cost items to a capitalize-vs-expense determination. Every budget output surfaces its assumptions block; capitalize-vs-expense is decision-support only and routes to a named finance owner \u2014 it never books an entry or decides accounting treatment. Distinct from finance/financial-analysis (corporate DCF, close, valuation) and research/grants (funding discovery \u2014 this \u2026",
"tools": [
{
"script": "research-ops/skills/research-finance/scripts/ar_evaluator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/research-finance/scripts/ar_evaluator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/research-finance/scripts/ar_evaluator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/research-finance/scripts/burn_runway_tracker.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/research-finance/scripts/burn_runway_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/research-finance/scripts/burn_runway_tracker.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/research-finance/scripts/capex_vs_opex_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/research-finance/scripts/capex_vs_opex_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/research-finance/scripts/capex_vs_opex_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/research-finance/scripts/config_loader.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/research-finance/scripts/config_loader.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/research-finance/scripts/config_loader.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research-ops/skills/research-finance/scripts/onboard.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research-ops/skills/research-finance/scripts/onboard.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research-ops/skills/research-finance/scripts/program_budget_planner.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research-ops/skills/research-finance/scripts/program_budget_planner.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research-ops/skills/research-finance/scripts/program_budget_planner.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"burn_and_portfolio.md",
"indirect_rate_modeling.md",
"rd_program_finance_canon.md"
]
},
{
"name": "research-ops-skills",
"path": "research-ops/skills/research-ops-skills",
"description": "Use when planning, funding, scoping, or synthesizing enterprise research across workstreams \u2014 clinical study design, R&D program finance, market sizing/surveys, or product/user research. Triggers on \"design this clinical study\", \"what sample size\", \"R&D budget\", \"burn rate\", \"capitalize or expense\", \"TAM SAM SOM\", \"market sizing\", \"survey design\", \"segment the market\", \"plan user interviews\", \"usability test\", \"synthesize research insights\". Forks context to route to one of four Research-Operations sub-skills (clinical-research, research-finance, market-research, product-research) and \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": false,
"verification": false,
"loop_discipline": true,
"close_out": true
},
"references": []
}
]
}

View file

@ -0,0 +1,560 @@
{
"schema": "agent-harness/manifest.v1",
"domain": "research",
"skill_count": 9,
"loop_defaults": {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected"
]
},
"skills": [
{
"name": "deep-research",
"path": "research/deep-research/skills/deep-research",
"description": "Run a disciplined, multi-source research investigation for a high-stakes question or decision \u2014 fan-out web search across many channels, parallel sub-agents, source triangulation (each claim backed by \u22653 independent sources), an adversarial review pass, and every source saved to its own file with verbatim quotes for reuse. Use when a low-quality answer is expensive: strategy work, comparing N products/methods/markets, validating a hypothesis with external data, or mapping how a field works. NOT for quick fact-checks (answer directly), structured 12-dimension competitor scoring (use \u2026",
"tools": [],
"agentic_signals": {
"goal_intake": false,
"refusal_gate": false,
"verification": true,
"loop_discipline": false,
"close_out": false
},
"references": [
"full-catalog.md"
]
},
{
"name": "dossier",
"path": "research/dossier/skills/dossier",
"description": "Decision-grade entity research skill \u2014 produces a hypothesis-tested dossier on a specific company, person, nonprofit, or government org, not a generic profile. Forcing intake makes the user state their hypothesis upfront (what they already believe and want to verify or disprove) so the dossier tests it rather than confirms it. Output is an editable Word document (.docx) with verdict on the hypothesis, identity facts, 12-month activity timeline, network and reputation signals, red flags, conversation hooks tied to specific findings, and source-provenance audit log. Uses WebSearch + WebFetch \u2026",
"tools": [
{
"script": "research/dossier/skills/dossier/scripts/citation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/dossier/skills/dossier/scripts/citation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/dossier/skills/dossier/scripts/disconfirming_evidence_balance.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/dossier/skills/dossier/scripts/disconfirming_evidence_balance.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/dossier/skills/dossier/scripts/disconfirming_evidence_balance.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/dossier/skills/dossier/scripts/source_tier_classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/dossier/skills/dossier/scripts/source_tier_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/dossier/skills/dossier/scripts/source_tier_classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"conversation_hook_quality.md",
"hypothesis_testing_discipline.md",
"subject_type_source_matrix.md"
]
},
{
"name": "grants",
"path": "research/grants/skills/grants",
"description": "NIH grant research skill for clinical researchers. Grill-me intake (research idea + career stage + preliminary data + environment + submission posture + known institute targets) locks down the funding strategy before any search runs. Runs a 5-facet Consensus positioning analysis (with draft Significance/Innovation language), maps the research to the right NIH institutes and study sections via RePORTER, finds NOSIs and funded overlap, and produces an editable Word document (.docx) with budget/scope-aware mechanism recommendations, submission timelines, and a mandatory program officer \u2026",
"tools": [
{
"script": "research/grants/skills/grants/scripts/citation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/grants/skills/grants/scripts/citation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/grants/skills/grants/scripts/fiscal_year_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/grants/skills/grants/scripts/fiscal_year_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/grants/skills/grants/scripts/mechanism_matcher.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/grants/skills/grants/scripts/mechanism_matcher.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/grants/skills/grants/scripts/mechanism_matcher.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"docx_9_sections.md",
"nih_mechanism_matching.md",
"reporter_post_patterns.md"
]
},
{
"name": "litreview",
"path": "research/litreview/skills/litreview",
"description": "Academic literature orientation skill that searches papers via free keyless APIs (PubMed E-utilities + OpenAlex) by default \u2014 with the Consensus MCP as an optional enhancement lane when connected \u2014 builds a strategic search plan using PICO (default) or SPIDER / Decomposition / hybrid as fallbacks, and synthesizes findings into a formatted Word (.docx) research guide. Grill-me intake (research question specificity + framework hint + tentative depth) before the recon search; a second forcing checkpoint after Phase 2 confirms framework + sub-areas + depth before searches consume budget. \u2026",
"tools": [
{
"script": "research/litreview/skills/litreview/scripts/citation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/litreview/skills/litreview/scripts/citation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/litreview/skills/litreview/scripts/cross_search_aggregator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/litreview/skills/litreview/scripts/cross_search_aggregator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/litreview/skills/litreview/scripts/cross_search_aggregator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/litreview/skills/litreview/scripts/framework_recommender.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/litreview/skills/litreview/scripts/framework_recommender.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/litreview/skills/litreview/scripts/framework_recommender.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/litreview/skills/litreview/scripts/free_search.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/litreview/skills/litreview/scripts/free_search.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"docx_8_sections.md",
"framework_selection.md",
"search_budget_allocation.md"
]
},
{
"name": "notebooklm",
"path": "research/notebooklm/skills/notebooklm",
"description": "Browser automation skill for controlling Google's NotebookLM. Use when the user wants anything done in NotebookLM (e.g., 'open NotebookLM', 'check my [name] notebook', 'ask my notebook about X', 'add [source] to NotebookLM', 'generate a Video Overview from my notebook', 'use NotebookLM Studio'). Handles reading and querying notebooks, adding sources (URLs, text, files, YouTube links, synthesized content), generating Studio outputs (Audio/Video Overviews, Mind Maps, Reports incl. Briefing Doc/Study Guide/FAQ, Flashcards, Quiz, slide decks, infographics \u2014 discover the exact set from the live \u2026",
"tools": [
{
"script": "research/notebooklm/skills/notebooklm/scripts/action_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/notebooklm/skills/notebooklm/scripts/action_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/notebooklm/skills/notebooklm/scripts/action_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/notebooklm/skills/notebooklm/scripts/async_action_classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/notebooklm/skills/notebooklm/scripts/async_action_classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/notebooklm/skills/notebooklm/scripts/async_action_classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/notebooklm/skills/notebooklm/scripts/custom_prompt_template_generator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/notebooklm/skills/notebooklm/scripts/custom_prompt_template_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/notebooklm/skills/notebooklm/scripts/custom_prompt_template_generator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": true
},
"references": [
"async_action_discipline.md",
"browser_automation_canon.md",
"studio_output_custom_prompts.md"
]
},
{
"name": "patent",
"path": "research/patent/skills/patent",
"description": "Patent prior-art and landscape intelligence skill \u2014 not generic patent help. Commits to one of five sub-use-cases via forcing intake (novelty search / freedom-to-operate / competitive landscape / acquisition diligence / litigation prior-art) before any search runs. Searches Google Patents, Espacenet, USPTO, and optionally Lens.org for citation-graph signals. Output is an editable Word document (.docx) with verdict, ranked closest art (claim-text extracted), CPC-class-aware landscape, family-resolved hits, geographic coverage, FTO flags where applicable, strategy recommendations, and full \u2026",
"tools": [
{
"script": "research/patent/skills/patent/scripts/citation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/patent/skills/patent/scripts/citation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/patent/skills/patent/scripts/family_resolver.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/patent/skills/patent/scripts/family_resolver.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/patent/skills/patent/scripts/family_resolver.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/patent/skills/patent/scripts/sub_use_case_router.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/patent/skills/patent/scripts/sub_use_case_router.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/patent/skills/patent/scripts/sub_use_case_router.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": true,
"loop_discipline": true,
"close_out": false
},
"references": [
"cpc_classification_canon.md",
"legal_disclaimer_discipline.md",
"sub_use_case_routing.md"
]
},
{
"name": "pulse",
"path": "research/pulse/skills/pulse",
"description": "Multi-source recency research skill that takes the pulse of any topic across Reddit, Hacker News, the open web, and optionally X/Twitter within a configurable recent window (default 30 days). Forcing intake clarifies topic specificity, angle (trend/sentiment/problems/opportunities/comparison), time window, and platform scope before searching. Returns a synthesized briefing with citations, engagement metrics, and cross-platform pattern analysis. Use when the user requests multi-source recency intelligence on a topic (e.g., 'pulse on [topic]', 'what's happening with [topic]', 'what are people \u2026",
"tools": [
{
"script": "research/pulse/skills/pulse/scripts/citation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/pulse/skills/pulse/scripts/citation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/pulse/skills/pulse/scripts/time_window_calculator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/pulse/skills/pulse/scripts/time_window_calculator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/pulse/skills/pulse/scripts/topic_slug_generator.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/pulse/skills/pulse/scripts/topic_slug_generator.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"cross_platform_synthesis.md",
"parallel_execution_discipline.md",
"research_pack_conventions.md"
]
},
{
"name": "research",
"path": "research/research/skills/research",
"description": "Default entry point for any research request \u2014 a hybrid router that classifies the question deterministically and either delegates to a specialist research skill (pulse for trends/sentiment, grants for NIH funding, litreview for academic literature, syllabus for course reading, patent for prior-art + IP landscape, dossier for entity research) or runs its own plan-decompose-multi-source-search-synthesize-cite fallback workflow when no specialist matches. Always surfaces the routing decision so users can override. Use when the user makes any research request that doesn't obviously match a \u2026",
"tools": [
{
"script": "research/research/skills/research/scripts/classifier.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/research/skills/research/scripts/classifier.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/research/skills/research/scripts/classifier.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/research/skills/research/scripts/fallback_decomposer.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/research/skills/research/scripts/fallback_decomposer.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/research/skills/research/scripts/fallback_decomposer.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/research/skills/research/scripts/routing_transparency_logger.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/research/skills/research/scripts/routing_transparency_logger.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/research/skills/research/scripts/routing_transparency_logger.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": true
},
"references": [
"deterministic_classification_canon.md",
"fallback_workflow_canon.md",
"hybrid_router_architecture.md"
]
},
{
"name": "syllabus",
"path": "research/syllabus/skills/syllabus",
"description": "Generates a curated supplementary reading list from any course syllabus using Consensus academic search. Grill-me intake (syllabus input format + course audience + year range) plus a grouping forcing-options checkpoint before any search runs \u2014 so the reading list matches the course's level and recency need. Parses the syllabus to extract topics and learning outcomes, searches Consensus for recent peer-reviewed papers per topic, and produces a professionally formatted .docx with clickable Consensus links, plain-language summaries calibrated to audience level, and Bloom-higher-order \u2026",
"tools": [
{
"script": "research/syllabus/skills/syllabus/scripts/citation_tracker.py",
"wired": true,
"supports_sample": false,
"verification": [
{
"cmd": "python3 research/syllabus/skills/syllabus/scripts/citation_tracker.py --help",
"expect_exit": 0,
"kind": "smoke"
}
]
},
{
"script": "research/syllabus/skills/syllabus/scripts/discussion_question_validator.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/syllabus/skills/syllabus/scripts/discussion_question_validator.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/syllabus/skills/syllabus/scripts/discussion_question_validator.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
},
{
"script": "research/syllabus/skills/syllabus/scripts/topic_grouper.py",
"wired": true,
"supports_sample": true,
"verification": [
{
"cmd": "python3 research/syllabus/skills/syllabus/scripts/topic_grouper.py --help",
"expect_exit": 0,
"kind": "smoke"
},
{
"cmd": "python3 research/syllabus/skills/syllabus/scripts/topic_grouper.py --sample",
"expect_exit": 0,
"kind": "sample"
}
]
}
],
"agentic_signals": {
"goal_intake": true,
"refusal_gate": true,
"verification": false,
"loop_discipline": true,
"close_out": false
},
"references": [
"applied_domain_weaving.md",
"audience_calibration.md",
"bundled_script_pattern.md"
]
}
]
}

View file

@ -0,0 +1,89 @@
# The Agentic Loop Canon
What the 20242026 practitioner literature agrees an agent loop is, and the design
decisions this skill inherits from it. Every rule in `SKILL.md` traces to one of
these sources.
## Sources
1. **Erik Schluntz & Barry Zhang (Anthropic), "Building Effective Agents", Dec 2024**
https://www.anthropic.com/research/building-effective-agents. The reference taxonomy:
*workflows* (LLM steps orchestrated through predefined code paths) vs *agents* (the LLM
directs its own process). Patterns: prompt chaining with programmatic gates, routing,
parallelization, orchestrator-workers, evaluator-optimizer. Rule inherited: **compile the
goal into a workflow — explicit ordered tasks with checks — and let the model be dynamic
only inside a task**, because evaluator-optimizer loops only pay off "when clear evaluation
criteria exist."
2. **Anthropic, "Building agents with the Claude Agent SDK", Sep 2025**
https://claude.com/blog/building-agents-with-the-claude-agent-sdk. Canonizes the loop as
**gather context → take action → verify work → repeat**, with the filesystem as the context
store and a verification-reliability ladder: rules-based checks > visual inspection >
LLM-as-judge. Rule inherited: every task record in the plan carries a `verification` array;
deterministic checks outrank judgment.
3. **Anthropic, "How we built our multi-agent research system", Jun 2025**
https://www.anthropic.com/engineering/multi-agent-research-system. Production
orchestrator-workers: subagent specs need **objective, output format, tool guidance, and
task boundaries** or workers duplicate and wander; effort must be scaled by rule (simple
query = 1 agent, 310 calls). Rule inherited: `goal_compiler.py` emits per-task objective +
suggested tools + done_when, and caps tasks with `--max-tasks`.
4. **Anthropic, "Effective harnesses for long-running agents", Nov 2025**
https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents. The
flagship harness: an initializer expands the goal into a structured `feature-list.json`;
a worker is woken repeatedly, each fresh-context session doing ONE item: read progress →
implement → run tests → write progress → commit. **All state lives on disk/git; sessions
are stateless shifts.** Rule inherited: the plan file + state file ARE the loop; never
assume conversational carryover between iterations.
5. **Geoffrey Huntley, "Ralph Wiggum as a 'software engineer'", Jul 2025**
https://ghuntley.com/ralph/ (now an official Claude Code plugin). A `while true` loop
feeding the same prompt to a fresh-context agent, with the filesystem + TODO file + git as
memory. Load-bearing insight: **fresh context each iteration is the point** — quality
degrades as the window fills, so restart against durable state instead of continuing.
Community practice adds iteration caps and completion criteria. Rule inherited:
`max_loop_iterations` is mandatory and enforced by the controller, not the agent.
6. **Walden Yan (Cognition), "Don't Build Multi-Agents", Jun 2025**
https://cognition.com/blog/dont-build-multi-agents. The counterweight to fan-out
enthusiasm: parallel actors making conflicting decisions on partial context is the dominant
multi-agent failure. Synthesis with source 3: **fan out readers and judges; serialize
writers.** Rule inherited: the default loop order is `sequential`; parallel execution is an
explicit opt-in and only for non-writing tasks.
7. **Anthropic, "Equipping agents for the real world with Agent Skills", Oct 2025**
https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills.
Skills load by **progressive disclosure** (metadata → SKILL.md → referenced files on
demand); ship deterministic scripts for anything reliably automatable. Rule inherited: the
manifest carries one-paragraph skill descriptors only; the agent opens a skill's SKILL.md
when — and only when — its task starts.
## The loop this skill implements
```
GOAL ──goal_compiler──▶ PLAN (tasks × verification × caps)
┌─────────────▼──────────────┐
│ loop_controller next │◀────────────┐
│ → execute ONE task │ │
│ → record --phase execute │ │
│ → loop_controller verify │ retry ≤ max_attempts,
│ (controller runs checks) │ changed approach only
└──────┬──────────────┬───────┘ │
verified failed ───────────────────┘
│ │ (attempts exhausted)
▼ ▼
close ESCALATE to a human
(refuses while any task unverified)
```
This is the six-step Observe→Choose→Act→Verify→Record→Repeat-or-stop cycle from the
vendored `loop-library/SKILL.md` (Forward Future, MIT), with the terminal-state taxonomy it
defines — success · clean no-op · blocked · approval-required · exhausted · stagnated —
mapped onto controller states: `closed` (success/no-op), `escalated`
(approval-required/exhausted), and the global iteration cap (stagnated).
## What the canon says NOT to do
- **Don't run the loop inside one ever-growing context.** (Sources 4, 5.) Each `next`
directive is designed to be executable by a fresh session reading only the state file.
- **Don't let two tasks write the same artifact in parallel.** (Source 6.)
- **Don't hand the model an open-ended goal without acceptance criteria.** (Sources 1, 3.)
`goal_compiler.py` refuses vague goals (exit 3) with forcing questions instead.
- **Don't treat subagent enthusiasm as progress.** (Source 3: early agents "spawned 50
subagents for simple queries.") Task count is capped; effort is budgeted up front.

View file

@ -0,0 +1,93 @@
# Domain Harness Design
How a *domain folder full of skills* becomes an *agent harness*: a declarative manifest
mapping goals → skills → verifications, plus the repo primitives this skill deliberately
reuses instead of rebuilding.
## Sources
1. **AGENTS.md convention** — https://agents.md/ (launched Aug 2025; adopted by Codex,
Cursor, Devin, Gemini CLI, Copilot; stewarded by the Agentic AI Foundation under the Linux
Foundation since Dec 2025). The de-facto standard for declaring "how to build and verify
work here" in a file agents read first. The harness manifest is the same idea made
machine-readable per domain.
2. **Anthropic, "Effective harnesses for long-running agents", Nov 2025**
https://www.anthropic.com/engineering/effective-harnesses-for-long-running-agents. Its
`feature-list.json` — each item carrying description + acceptance criteria + status, looped
until all verified — is the closest published goal→tasks→verification manifest and the
direct model for `plan.v1` / `state.v1`.
3. **Model Context Protocol** — https://modelcontextprotocol.io/ (Anthropic, Nov 2024;
multi-vendor stewardship 2025). JSON-schema'd tool registries as the declared action space
of a harness. The manifest's `tools[]` blocks follow the same declare-don't-discover
philosophy: an agent should read what a skill can do, not grep for it mid-loop.
4. **LangGraph checkpointers** — https://langchain-ai.github.io/langgraph/. Durable
execution: every super-step persisted, enabling pause/resume/replay and human-in-the-loop
interrupts. The stdlib equivalent here: an atomically-written JSON state file
(`os.replace`), append-only evidence entries, and git as the checkpoint layer.
5. **OpenAI Agents SDK** — https://openai.github.io/openai-agents-python/. `max_turns`
raising `MaxTurnsExceeded` and tripwire guardrails: caps are *runtime errors*, not
suggestions. Mirrored by controller exit codes 2 (escalate), 4 (close refused),
5 (iteration cap) — a caller script can branch on them mechanically.
6. **Forward Future, "Loop Library" (MIT, vendored at `loop-library/` in this repo)** — loop
anatomy (Observe/Choose/Act/Verify/Record/Repeat-or-stop) and the terminal-state taxonomy
(success · clean no-op · blocked · approval-required · exhausted · stagnated). The harness
adopts this vocabulary; "errors and exhausted budgets are never reported as success" is
implemented as the no-force `close` gate.
7. **Anthropic, "Effective context engineering for AI agents", Sep 2025**
https://www.anthropic.com/engineering/effective-context-engineering-for-ai-agents.
Compaction, structured note-taking, and subagent summaries for long horizons. The state
file's `evidence` log + the close-time handoff block are the structured notes; a fresh
session needs only `plan.json` + `state.json` to continue the loop.
## The three-layer architecture
```
Layer 1 — INVENTORY (per domain, committed, regenerated by tooling)
assets/harnesses/<domain>.json what skills exist, what tools they ship,
what checks prove each tool works,
which agentic signals the skill already has
Layer 2 — PLAN (per goal, generated at run time)
plan.json ordered tasks: skill × objective ×
verification[] × max_attempts × done_when
Layer 3 — STATE (per run, the single source of truth)
.agent-harness/state.json task statuses, attempts, evidence log,
iteration counter, close/handoff record
```
Layer 1 is diff-stable (`--no-timestamp`) so CI can regenerate and `git diff --exit-code`
it — manifest drift against the tree becomes a machine-checkable gate, exactly like this
repo's `derive_counters.py --check` discipline.
## Reuse map (import the pattern, don't rebuild the mechanism)
| Need | Reuse from this repo | The harness adds |
|---|---|---|
| Task/state persistence + handoff schema | `engineering/skills/tc-tracker` (atomic writes, append-only history, session handoff block) | A goal-scoped, domain-agnostic state file (`.agent-harness/`), not per-code-change records |
| Anti-overfit locked evaluator | `engineering/autoresearch-agent` (never modify `evaluate.py`; KEEP/DISCARD/CRASH) | The same invariant generalized: `verify` re-runs gates itself |
| N-agent tournament on one task | `engineering/agenthub` (worktrees, DAG, result ranker) | Nothing — route there when the task wants competing attempts |
| Deterministic fan-out/pipeline topologies | `engineering/workflow-builder` (Workflow-tool .js + validator) | Nothing — route there when orchestrating Claude Code's Workflow tool |
| Loop design/audit vocabulary | `loop-library/` (six-step cycle, stop-state taxonomy) | An executable enforcement of that vocabulary |
| Severity-gated ship decision | `engineering/skills/ship-gate` (CRITICAL/HIGH/ADVISORY verdict) | Use as a close-time check inside a task's `verification[]` |
| Honest self-scoring | `engineering/skills/self-eval` (matrix-locked composite) | Optional close-out step before the handoff |
| Bounded-autonomy STOP triggers | `engineering/skills/spec-driven-workflow` + `focused-fix` 3-strike rule | `escalate_on` defaults in every manifest |
## Namespacing (collisions this skill deliberately avoids)
- State directory is **`.agent-harness/`** — never `.agenthub/`, `.autoresearch/`, or
`docs/TC/`, which belong to sibling skills.
- Command is **`/cs:harness`** — `/hub:*`, `/ar:*`, `/si:*`, `/tc` are taken.
- The runtime agent is **`harness-runner`** — "orchestrator" already denotes the
`context: fork` domain routers, and `hub-coordinator` / `experiment-runner` are taken.
- The bare name "loop" is overloaded in this repo (the `/loop` scheduler skill,
`loop-library`, workflow loop templates) — this skill never claims it.
## Extending a domain's harness
1. Author or improve skills so they carry the agentic signals (intake, refusal gates,
verification, loop discipline, close-out — see the July 2026 audit's AR1AR6 rubric in
`audit/engineering-agentic-2026-07/RUBRIC.md`).
2. Regenerate the manifest:
`python3 scripts/harness_manifest_builder.py --domain <domain> --repo-root . --out-dir assets/harnesses --no-timestamp`
(run from this skill's directory).
3. The richer the skill's tools and `--sample` support, the more executable checks its tasks
get for free — `manual-evidence` tasks are the fallback, not the goal.

View file

@ -0,0 +1,85 @@
# Verification Discipline
Why the harness adjudicates its own gates, and why an agent's claim of success is
never evidence. The controller's design decisions trace to these sources.
## Sources
1. **Jason Wei, "Asymmetry of verification and verifier's law", Jul 2025**
https://www.jasonwei.net/blog/asymmetry-of-verification-and-verifiers-law. "The ease of
training AI to solve a task is proportional to how verifiable the task is." Tasks easy to
check but hard to do are exactly where iteration works. Design consequence: **invest in
making the task verifiable before investing in the agent** — a task in a harness plan with
no executable check is a liability, which is why `goal_compiler.py` marks such tasks
`manual-evidence` and the controller refuses to auto-verify them.
2. **John Yang, Carlos E. Jimenez et al., "SWE-agent: Agent-Computer Interfaces Enable
Automated Software Engineering", NeurIPS 2024** — https://arxiv.org/abs/2405.15793.
Agents fail when the environment gives no feedback on bad actions; the single highest-value
guardrail was a linter that **rejects invalid edits at write time**. Design consequence:
gates run cheap→expensive and fail fast; a failed check returns the failing command's
output tail so the next attempt has signal, not vibes.
3. **OpenAI, "SWE-bench Verified", 2024** — https://www.swebench.com/verified.html. Even
benchmark test suites were noisy enough to need human validation before scores meant
anything. Design consequence: every check in a manifest declares its `kind`
(smoke/sample/manual-evidence); only deterministic kinds can flip a task to `verified`
without a human-authored evidence line.
4. **Boris Cherny (Anthropic), "Claude Code: Best practices for agentic coding", Apr 2025**
https://www.anthropic.com/engineering/claude-code-best-practices. The strongest loop is
test-driven: write the check first, confirm it fails, then iterate work against it —
"Claude performs best when it has a clear target to iterate against." Design consequence:
the harness's recommended flow is gate-first (run the verification before the work; a gate
that already passes pre-work is invalid as evidence of progress).
5. **Noah Shinn et al., "Reflexion: Language Agents with Verbal Reinforcement Learning",
NeurIPS 2023** — https://arxiv.org/abs/2303.11366 — self-critique improves outcomes **only
when grounded in external feedback signals**; and **Jie Huang et al., "Large Language Models
Cannot Self-Correct Reasoning Yet", ICLR 2024** — https://arxiv.org/abs/2310.01798 —
intrinsic self-correction without external feedback often makes answers worse. Design
consequence: retries are only granted after a *recorded external failure* (nonzero exit),
and the retry directive explicitly demands a changed approach.
6. **Anthropic, "From shortcuts to sabotage: natural emergent misalignment from reward
hacking", Nov 2025** — https://www.anthropic.com/research/emergent-misalignment-reward-hacking.
Agents that learn to game their checks (hard-coding expected values, editing tests)
generalize to worse behavior. Design consequence — the harness's central invariant:
**the worker must not adjudicate or modify the gates it is judged by.**
`loop_controller.py verify` re-runs the check commands itself via subprocess; a passing
`record --phase verify` without `--evidence` is rejected outright ("no verification
theater"); and the same invariant already ships in this repo as autoresearch-agent's
locked-evaluator rule ("`evaluate.py` is ground truth — never modify it").
7. **Google SRE Workbook (Beyer et al., 2018), ch. 2 "Implementing SLOs"**
https://sre.google/workbook/implementing-slos/. Error budgets are the production-grade
version of the same idea: a numeric, pre-agreed threshold decides whether you ship or stop,
not the operator's optimism. Design consequence: attempts and iterations are budgets
(`max_attempts_per_task`, `max_loop_iterations`); exhausting a budget is a *terminal,
reportable state* — never silently absorbed.
## The verification ladder (most → least trustworthy)
| Rank | Check type | Harness treatment |
|---|---|---|
| 1 | Deterministic command, exit-code contract (`kind: smoke`/`sample`) | `verify` subcommand runs it; pass can auto-flip state |
| 2 | Deterministic command with output assertion (JSON keys, thresholds) | Same, encode the assertion in the command (`... | python3 -c "assert ..."`) |
| 3 | Human-readable evidence written by the agent (`kind: manual-evidence`) | Requires `record --phase verify --evidence "<observation>"`; controller refuses empty evidence |
| 4 | Agent asserting "done" | **Never accepted.** Not a state transition in the machine. |
## Anti-gaming rules the controller enforces
- `verify` executes checks itself (subprocess, timeout, output tail captured to the evidence
log) — recorded exit codes are for the *execute* phase only.
- A passing verify record without evidence text is exit 6, not a pass.
- Failure at `max_attempts` escalates (exit 2); the loop cannot convert an exhausted task
into a success, only a human can waive it — and `close --waive` demands a reason that is
written permanently into the handoff.
- `close` with any unverified, unwaived task is exit 4. There is no force flag.
## Trust boundary: plan and state files
`loop_controller.py verify` shell-executes each task's `verification[].cmd` string via
`subprocess.run(..., shell=True)`. In the documented flow those commands are template-
generated from repo-scanned script paths (`harness_manifest_builder.py``goal_compiler.py`),
so they are not attacker-reachable. But the controller does **not** re-validate a `--state`
or `--plan` file's contents before shelling out — a hand-crafted or tampered plan/state file
is therefore effectively arbitrary local command execution, the same trust model as a
Makefile or a CI config. **Treat `plan.json` and `state.json` as a trust boundary: only
run the harness on plan/state files you (or the `goal_compiler`) produced, never on files
sourced from untrusted input.** This matters because the harness is designed to be driven by
an agent (`harness-runner`) that could in principle be handed a malicious plan.

View file

@ -0,0 +1,221 @@
#!/usr/bin/env python3
"""Compile a goal into a verifiable task plan against a domain harness manifest.
Deterministic keyword scoring (no LLM calls): the goal is tokenized, each skill
in the manifest is scored on name/description overlap, and the top matches
become ordered tasks each with the verification checks the manifest recorded
for that skill and a done_when contract loop_controller.py can enforce.
Refusal gates (the harness never runs on fuzz):
exit 3 goal too vague (< 4 content tokens): emits forcing questions.
exit 4 no skill scores above --min-score: emits nearest candidates.
Usage:
python3 goal_compiler.py --goal "audit our API design and ship an SLO" \
--manifest assets/harnesses/engineering.json --out plan.json
python3 goal_compiler.py --sample
"""
import argparse
import datetime
import json
import re
import sys
SCHEMA = "agent-harness/plan.v1"
STOPWORDS = {
"a", "an", "and", "are", "as", "at", "be", "by", "can", "do", "for",
"from", "get", "have", "how", "i", "in", "is", "it", "make", "me", "my",
"of", "on", "or", "our", "please", "set", "should", "so", "that", "the",
"then", "this", "to", "up", "us", "want", "we", "what", "when", "will",
"with", "you", "your", "need", "needs", "into", "using", "use",
}
FORCING_QUESTIONS = [
"What is the single observable outcome that means this goal is DONE "
"(a file, a passing check, a published artifact)?",
"Which system, repo, or artifact does the work act on?",
"What must NOT change (constraints, no-touch zones, budgets)?",
"Who reviews the result, and what evidence do they need to accept it?",
"What is the deadline or iteration budget before a human takes over?",
]
def tokenize(text):
words = re.findall(r"[a-z0-9][a-z0-9\-]+", text.lower())
out = []
for w in words:
out.extend(w.split("-"))
return [w for w in out if len(w) > 2 and w not in STOPWORDS]
def score_skill(goal_tokens, skill):
name_tokens = set(tokenize(skill.get("name", "")))
desc_tokens = set(tokenize(skill.get("description", "")))
score = 0
hits = []
for t in set(goal_tokens):
if t in name_tokens:
score += 3
hits.append(t)
elif t in desc_tokens:
score += 1
hits.append(t)
return score, sorted(hits)
def build_task(idx, skill, goal, defaults):
checks = []
tool_cmds = []
for tool in skill.get("tools", []):
tool_cmds.append("python3 %s --help # discover flags first" % tool["script"])
checks.extend(tool.get("verification", []))
if not checks:
checks.append({
"cmd": "MANUAL: state the observable evidence that this task met "
"its objective; a task with no check cannot be closed, only "
"escalated.",
"expect_exit": 0,
"kind": "manual-evidence",
})
return {
"id": "T%d" % idx,
"skill": skill["name"],
"skill_path": skill["path"],
"objective": "Apply skill '%s' toward goal: %s" % (skill["name"], goal),
"suggested_tools": tool_cmds,
"verification": checks,
"done_when": "every verification check meets expect_exit AND the "
"output is consistent with the task objective",
"max_attempts": defaults.get("max_attempts_per_task", 3),
"status": "pending",
}
SAMPLE_PLAN = {
"schema": SCHEMA,
"goal": "design an SLO and error budget for the payments API",
"domain": "engineering",
"tasks": [{
"id": "T1",
"skill": "slo-architect",
"skill_path": "engineering/slo-architect/skills/slo-architect",
"objective": "Apply skill 'slo-architect' toward goal: design an SLO "
"and error budget for the payments API",
"suggested_tools": [
"python3 .../scripts/error_budget_calculator.py --help # discover flags first",
],
"verification": [
{"cmd": "python3 .../error_budget_calculator.py --target 99.9 "
"--window-days 30", "expect_exit": 0, "kind": "sample"},
],
"done_when": "every verification check meets expect_exit AND the "
"output is consistent with the task objective",
"max_attempts": 3,
"status": "pending",
}],
"loop": {
"order": "sequential",
"max_loop_iterations": 12,
"escalate_on": ["attempts_exhausted", "no_verification_available"],
},
"close": {
"requires": "all tasks verified (or explicitly waived with a reason)",
"handoff": "loop_controller.py close emits the evidence log + summary",
},
}
def main():
ap = argparse.ArgumentParser(
description="Compile a goal into a verifiable task plan from a domain "
"harness manifest.")
ap.add_argument("--goal", help="The goal statement to compile.")
ap.add_argument("--manifest", help="Path to a harness manifest JSON.")
ap.add_argument("--max-tasks", type=int, default=5)
ap.add_argument("--min-score", type=int, default=2,
help="Minimum match score for a skill to become a task.")
ap.add_argument("--out", help="Write the plan JSON here.")
ap.add_argument("--json", action="store_true", help="Print plan to stdout.")
ap.add_argument("--sample", action="store_true",
help="Print an example plan and exit 0.")
args = ap.parse_args()
if args.sample:
print(json.dumps(SAMPLE_PLAN, indent=2))
return 0
if not args.goal or not args.manifest:
ap.error("--goal and --manifest are required (or use --sample)")
goal_tokens = tokenize(args.goal)
if len(goal_tokens) < 4:
print(json.dumps({
"verdict": "REFUSED-VAGUE-GOAL",
"reason": "goal has %d content tokens (< 4); the harness never "
"runs on fuzz" % len(goal_tokens),
"forcing_questions": FORCING_QUESTIONS,
}, indent=2))
return 3
with open(args.manifest, encoding="utf-8") as f:
manifest = json.load(f)
defaults = manifest.get("loop_defaults", {})
scored = []
for skill in manifest.get("skills", []):
s, hits = score_skill(goal_tokens, skill)
if s > 0:
scored.append((s, skill["name"], hits, skill))
scored.sort(key=lambda x: (-x[0], x[1]))
eligible = [x for x in scored if x[0] >= args.min_score]
if not eligible:
print(json.dumps({
"verdict": "REFUSED-NO-MATCH",
"reason": "no skill in domain '%s' scored >= %d for this goal"
% (manifest.get("domain"), args.min_score),
"nearest_candidates": [
{"skill": n, "score": s, "matched": h}
for s, n, h, _ in scored[:5]
],
"forcing_questions": FORCING_QUESTIONS[:2],
}, indent=2))
return 4
tasks = [build_task(i + 1, sk, args.goal, defaults)
for i, (_, _, _, sk) in enumerate(eligible[: args.max_tasks])]
plan = {
"schema": SCHEMA,
"goal": args.goal,
"domain": manifest.get("domain"),
"compiled_at": datetime.datetime.now(datetime.timezone.utc)
.strftime("%Y-%m-%dT%H:%M:%SZ"),
"skill_match_report": [
{"skill": n, "score": s, "matched": h} for s, n, h, _ in scored[:10]
],
"tasks": tasks,
"loop": {
"order": "sequential",
"max_loop_iterations": defaults.get("max_loop_iterations", 12),
"escalate_on": defaults.get("escalate_on", ["attempts_exhausted"]),
},
"close": {
"requires": "all tasks verified (or explicitly waived with a reason)",
"handoff": "loop_controller.py close emits the evidence log + summary",
},
}
out = json.dumps(plan, indent=2)
if args.out:
with open(args.out, "w", encoding="utf-8") as f:
f.write(out + "\n")
print("wrote %s (%d tasks)" % (args.out, len(tasks)), file=sys.stderr)
if args.json or not args.out:
print(out)
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,257 @@
#!/usr/bin/env python3
"""Build a domain harness manifest: scan a domain folder's skills and emit the
machine-readable inventory (skills, tools, verification checks, agentic signals)
that goal_compiler.py and loop_controller.py consume.
Stdlib-only. Deterministic: same tree in, same manifest out (modulo the
generated_at stamp, which --no-timestamp suppresses for diff-stable output).
Usage:
python3 harness_manifest_builder.py --domain engineering --repo-root . --json
python3 harness_manifest_builder.py --all --repo-root . --out-dir assets/harnesses --no-timestamp
python3 harness_manifest_builder.py --sample
"""
import argparse
import datetime
import json
import os
import re
import sys
SCHEMA = "agent-harness/manifest.v1"
# Folders that are never skill content.
SKIP_DIRS = {".git", ".github", "node_modules", "__pycache__", ".claude-plugin",
"expected_outputs", ".codex", ".gemini", ".hermes", ".vibe"}
# Signal regexes: cheap, static evidence that a skill already carries agentic
# structure. Matched case-insensitively against the SKILL.md body.
SIGNALS = {
"goal_intake": r"forcing[- ]question|before starting|intake|clarify(?:ing)? question",
"refusal_gate": r"refus(?:e|al)|exit(?:s|ed)? (?:code )?[1-9]|hard rule|NO-GO",
"verification": r"verif(?:y|ication|iable)|checklist|--sample|exit 0|definition of done",
"loop_discipline": r"\bretry\b|\biterat(?:e|ion)|stop condition|max attempts|budget|until",
"close_out": r"close[- ]?(?:the[- ])?loop|handoff|hand-off|state persist|done when|completion",
}
LOOP_DEFAULTS = {
"max_attempts_per_task": 3,
"max_loop_iterations": 12,
"escalate_on": [
"attempts_exhausted",
"no_verification_available",
"destructive_or_irreversible_action",
"goal_drift_detected",
],
}
def read_text(path):
try:
with open(path, encoding="utf-8", errors="replace") as f:
return f.read()
except OSError:
return ""
def truncate_words(text, limit):
"""Cap at `limit` chars, cutting on a word boundary with an ellipsis marker."""
if len(text) <= limit:
return text
cut = text[:limit - 2]
if " " in cut:
cut = cut.rsplit(" ", 1)[0]
return cut.rstrip(",;:") + ""
def parse_frontmatter(text):
"""Extract name/description from YAML frontmatter without a YAML dep."""
meta = {"name": "", "description": ""}
if not text.startswith("---"):
return meta
end = text.find("\n---", 3)
if end == -1:
return meta
block = text[3:end]
m = re.search(r"^name:\s*(.+)$", block, re.MULTILINE)
if m:
meta["name"] = m.group(1).strip().strip("\"'")
m = re.search(r"^description:\s*(.+)$", block, re.MULTILINE)
if m:
desc = m.group(1).strip()
# Fold simple multi-line continuations (indented lines).
idx = block.find(m.group(0)) + len(m.group(0))
for line in block[idx:].splitlines():
if line.startswith((" ", "\t")) and not re.match(r"^\s*\w+:", line):
desc += " " + line.strip()
elif line.strip():
break
meta["description"] = desc.strip().strip("\"'")
return meta
def find_skills(domain_path):
"""Yield (skill_dir, skill_md_path) for every SKILL.md under the domain."""
hits = []
for root, dirs, files in os.walk(domain_path):
dirs[:] = sorted(d for d in dirs if d not in SKIP_DIRS)
if "SKILL.md" in files:
hits.append((root, os.path.join(root, "SKILL.md")))
return sorted(hits)
def scan_skill(skill_dir, skill_md, repo_root):
text = read_text(skill_md)
meta = parse_frontmatter(text)
body = text.lower()
rel_dir = os.path.relpath(skill_dir, repo_root)
tools = []
scripts_dir = os.path.join(skill_dir, "scripts")
script_paths = []
if os.path.isdir(scripts_dir):
for fn in sorted(os.listdir(scripts_dir)):
if fn.endswith(".py"):
script_paths.append(os.path.join(scripts_dir, fn))
# Root-level scripts (older layout).
for fn in sorted(os.listdir(skill_dir)):
if fn.endswith(".py"):
script_paths.append(os.path.join(skill_dir, fn))
for sp in script_paths:
rel = os.path.relpath(sp, repo_root)
src = read_text(sp)
tools.append({
"script": rel,
"wired": os.path.basename(sp) in text,
"supports_sample": "--sample" in src,
"verification": build_checks(rel, src),
})
signals = {k: bool(re.search(rx, body)) for k, rx in SIGNALS.items()}
return {
"name": meta["name"] or os.path.basename(skill_dir),
"path": rel_dir,
"description": truncate_words(meta["description"], 600),
"tools": tools,
"agentic_signals": signals,
"references": sorted(os.listdir(os.path.join(skill_dir, "references")))
if os.path.isdir(os.path.join(skill_dir, "references")) else [],
}
def build_checks(rel_script, src):
checks = [{"cmd": "python3 %s --help" % rel_script, "expect_exit": 0,
"kind": "smoke"}]
if "--sample" in src:
checks.append({"cmd": "python3 %s --sample" % rel_script,
"expect_exit": 0, "kind": "sample"})
return checks
def build_manifest(domain_path, repo_root, timestamp=True):
domain = os.path.relpath(domain_path, repo_root)
skills = [scan_skill(d, s, repo_root) for d, s in find_skills(domain_path)]
manifest = {
"schema": SCHEMA,
"domain": domain,
"skill_count": len(skills),
"loop_defaults": LOOP_DEFAULTS,
"skills": skills,
}
if timestamp:
manifest["generated_at"] = (
datetime.datetime.now(datetime.timezone.utc)
.strftime("%Y-%m-%dT%H:%M:%SZ"))
return manifest
SAMPLE_MANIFEST = {
"schema": SCHEMA,
"domain": "engineering",
"skill_count": 1,
"loop_defaults": LOOP_DEFAULTS,
"skills": [{
"name": "slo-architect",
"path": "engineering/slo-architect/skills/slo-architect",
"description": "Design SLOs/SLIs and error budgets per the Google SRE Workbook...",
"tools": [{
"script": "engineering/slo-architect/skills/slo-architect/scripts/error_budget_calculator.py",
"wired": True,
"supports_sample": True,
"verification": [
{"cmd": "python3 .../error_budget_calculator.py --help",
"expect_exit": 0, "kind": "smoke"},
{"cmd": "python3 .../error_budget_calculator.py --sample",
"expect_exit": 0, "kind": "sample"},
],
}],
"agentic_signals": {
"goal_intake": True, "refusal_gate": True, "verification": True,
"loop_discipline": True, "close_out": True,
},
"references": ["slo_canon.md"],
}],
}
def main():
ap = argparse.ArgumentParser(
description="Scan a domain folder and emit its agent-harness manifest.")
ap.add_argument("--domain", action="append", default=[],
help="Domain folder relative to --repo-root (repeatable).")
ap.add_argument("--all", action="store_true",
help="Build manifests for every top-level domain folder "
"containing at least one SKILL.md.")
ap.add_argument("--repo-root", default=".")
ap.add_argument("--out-dir", help="Write <domain>.json per domain here.")
ap.add_argument("--json", action="store_true",
help="Print manifest(s) to stdout as JSON.")
ap.add_argument("--no-timestamp", action="store_true",
help="Omit generated_at for diff-stable committed manifests.")
ap.add_argument("--sample", action="store_true",
help="Print an example manifest and exit 0.")
args = ap.parse_args()
if args.sample:
print(json.dumps(SAMPLE_MANIFEST, indent=2))
return 0
repo_root = os.path.abspath(args.repo_root)
targets = list(args.domain)
if args.all:
for entry in sorted(os.listdir(repo_root)):
p = os.path.join(repo_root, entry)
if (os.path.isdir(p) and entry not in SKIP_DIRS
and not entry.startswith(".")
and find_skills(p)):
targets.append(entry)
if not targets:
ap.error("provide --domain, --all, or --sample")
results = []
for t in sorted(set(targets)):
dp = os.path.join(repo_root, t)
if not os.path.isdir(dp):
print("ERROR: no such domain folder: %s" % t, file=sys.stderr)
return 2
manifest = build_manifest(dp, repo_root, timestamp=not args.no_timestamp)
results.append(manifest)
if args.out_dir:
os.makedirs(args.out_dir, exist_ok=True)
slug = t.rstrip("/").replace(os.sep, "-")
out = os.path.join(args.out_dir, "%s.json" % slug)
with open(out, "w", encoding="utf-8") as f:
json.dump(manifest, f, indent=2, sort_keys=False)
f.write("\n")
print("wrote %s (%d skills)" % (out, manifest["skill_count"]),
file=sys.stderr)
if args.json or not args.out_dir:
print(json.dumps(results if len(results) > 1 else results[0], indent=2))
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,408 @@
#!/usr/bin/env python3
"""JSON-backed loop state machine: the harness's execute→verify→close enforcer.
Takes the plan from goal_compiler.py and drives a bounded loop. The agent asks
`next` for a directive, does the work, then `record`s the outcome with an exit
code. The controller enforces the discipline the agent must not be trusted to
enforce on itself:
* a task is only VERIFIED by recording a passing verify phase never by
recording execution success alone (no verification theater);
* failed attempts increment a counter; at max_attempts the task ESCALATES
to a human instead of retrying forever;
* a global iteration cap bounds the whole loop;
* `close` refuses (exit 4) while any task is unverified and unwaived.
Task states: pending in_progress verifying verified
(failure at cap) escalated waived (close-time, with reason)
Exit codes: 0 ok · 2 escalation required · 4 close refused · 5 iteration cap ·
6 invalid transition/state.
Usage:
python3 loop_controller.py init --plan plan.json --state state.json
python3 loop_controller.py next --state state.json
python3 loop_controller.py record --state state.json --task T1 --phase execute --exit-code 0
python3 loop_controller.py record --state state.json --task T1 --phase verify --exit-code 0 --evidence "error_budget_calculator exit 0; 43.2min budget"
python3 loop_controller.py close --state state.json
python3 loop_controller.py --sample # in-memory demo of a full loop
"""
import argparse
import datetime
import json
import os
import subprocess
import sys
import tempfile
STATE_SCHEMA = "agent-harness/state.v1"
CHECK_TIMEOUT_S = 120
def now():
return datetime.datetime.now(datetime.timezone.utc).strftime(
"%Y-%m-%dT%H:%M:%SZ")
def load(path):
with open(path, encoding="utf-8") as f:
return json.load(f)
def save(state, path):
"""Atomic write (temp file + os.replace) so a crashed run never leaves a
half-written state file."""
d = os.path.dirname(os.path.abspath(path)) or "."
fd, tmp = tempfile.mkstemp(dir=d, prefix=".harness-state-")
try:
with os.fdopen(fd, "w", encoding="utf-8") as f:
json.dump(state, f, indent=2)
f.write("\n")
os.replace(tmp, path)
except BaseException:
if os.path.exists(tmp):
os.unlink(tmp)
raise
def emit(obj, code=0):
print(json.dumps(obj, indent=2))
return code
def cmd_init(args):
plan = load(args.plan)
state = {
"schema": STATE_SCHEMA,
"goal": plan.get("goal"),
"domain": plan.get("domain"),
"plan_file": args.plan,
"created_at": now(),
"iteration": 0,
"max_loop_iterations": plan.get("loop", {}).get("max_loop_iterations", 12),
"status": "open",
"tasks": [{
"id": t["id"],
"skill": t.get("skill"),
"objective": t.get("objective"),
"verification": t.get("verification", []),
"max_attempts": t.get("max_attempts", 3),
"attempts": 0,
"status": "pending",
"evidence": [],
} for t in plan.get("tasks", [])],
}
if not state["tasks"]:
return emit({"error": "plan has no tasks"}, 6)
save(state, args.state)
return emit({"initialized": args.state, "tasks": len(state["tasks"]),
"max_loop_iterations": state["max_loop_iterations"]})
def directive(state):
if state["status"] == "closed":
return {"action": "done", "detail": "loop already closed"}, 0
if state["iteration"] >= state["max_loop_iterations"]:
return {"action": "escalate",
"detail": "global iteration cap (%d) reached — hand the loop "
"to a human with the evidence log"
% state["max_loop_iterations"]}, 5
for t in state["tasks"]:
if t["status"] == "escalated":
return {"action": "escalate", "task": t["id"],
"detail": "task %s exhausted %d attempts; a human must "
"review the evidence log before the loop may "
"continue" % (t["id"], t["max_attempts"])}, 2
for t in state["tasks"]:
if t["status"] in ("pending", "in_progress"):
return {"action": "execute", "task": t["id"],
"objective": t["objective"],
"attempt": t["attempts"] + 1,
"max_attempts": t["max_attempts"],
"then": "record --phase execute --exit-code <code>"}, 0
if t["status"] == "verifying":
return {"action": "verify", "task": t["id"],
"checks": t["verification"],
"rule": "run every check; ALL must meet expect_exit; then "
"record --phase verify with the worst exit code "
"and an --evidence line naming what you observed"}, 0
return {"action": "close",
"detail": "all tasks verified — run `close` to emit the handoff"}, 0
def cmd_next(args):
state = load(args.state)
d, code = directive(state)
return emit(d, code)
def cmd_record(args):
state = load(args.state)
if state["status"] == "closed":
return emit({"error": "loop is closed; no further records accepted"}, 6)
task = next((t for t in state["tasks"] if t["id"] == args.task), None)
if task is None:
return emit({"error": "unknown task %s" % args.task}, 6)
if task["status"] in ("verified", "escalated"):
return emit({"error": "task %s is %s; recording on it is invalid"
% (task["id"], task["status"])}, 6)
state["iteration"] += 1
entry = {"at": now(), "phase": args.phase, "exit_code": args.exit_code,
"attempt": task["attempts"] + 1}
if args.evidence:
entry["evidence"] = args.evidence
task["evidence"].append(entry)
ok = args.exit_code == 0
if args.phase == "execute":
if ok:
task["status"] = "verifying"
result = {"task": task["id"], "status": "verifying",
"next": "run the verification checks, then record "
"--phase verify"}
code = 0
else:
code, result = _fail(task)
else: # verify
if task["status"] != "verifying":
return emit({"error": "task %s is not awaiting verification "
"(status=%s); execute must succeed first"
% (task["id"], task["status"])}, 6)
if ok:
if not args.evidence:
return emit({"error": "a passing verify record requires "
"--evidence naming what was observed "
"(no verification theater)"}, 6)
task["status"] = "verified"
result = {"task": task["id"], "status": "verified"}
code = 0
else:
code, result = _fail(task)
save(state, args.state)
d, dcode = directive(state)
result["directive"] = d
return emit(result, code if code != 0 else dcode)
def _fail(task):
task["attempts"] += 1
if task["attempts"] >= task["max_attempts"]:
task["status"] = "escalated"
return 2, {"task": task["id"], "status": "escalated",
"detail": "attempts exhausted (%d/%d) — escalate to a human"
% (task["attempts"], task["max_attempts"])}
task["status"] = "pending"
return 0, {"task": task["id"], "status": "pending",
"detail": "attempt %d/%d failed — change the approach before "
"retrying (same command + same input = same failure)"
% (task["attempts"], task["max_attempts"])}
def cmd_verify(args):
"""Run the task's executable verification checks via subprocess — the
controller adjudicates pass/fail itself instead of trusting a recorded
exit code (reward-hacking guard)."""
state = load(args.state)
task = next((t for t in state["tasks"] if t["id"] == args.task), None)
if task is None:
return emit({"error": "unknown task %s" % args.task}, 6)
if task["status"] != "verifying":
return emit({"error": "task %s is not awaiting verification "
"(status=%s); execute must succeed first"
% (task["id"], task["status"])}, 6)
runnable = [c for c in task["verification"]
if c.get("kind") != "manual-evidence"]
results = []
worst = 0
for chk in runnable:
try:
proc = subprocess.run(
chk["cmd"], shell=True, cwd=args.cwd,
capture_output=True, text=True, timeout=CHECK_TIMEOUT_S)
rc = proc.returncode
tail = (proc.stdout + proc.stderr).strip().splitlines()[-3:]
except subprocess.TimeoutExpired:
rc, tail = 124, ["TIMEOUT after %ss" % CHECK_TIMEOUT_S]
passed = rc == chk.get("expect_exit", 0)
results.append({"cmd": chk["cmd"], "exit": rc, "passed": passed,
"tail": tail})
if not passed:
worst = 1
state["iteration"] += 1
task["evidence"].append({"at": now(), "phase": "verify-run",
"checks": results, "attempt": task["attempts"] + 1})
manual = [c for c in task["verification"]
if c.get("kind") == "manual-evidence"]
if worst == 0 and runnable and not manual:
task["status"] = "verified"
result = {"task": task["id"], "status": "verified",
"checks_run": len(results)}
code = 0
elif worst == 0 and manual:
result = {"task": task["id"], "status": "verifying",
"checks_run": len(results),
"detail": "executable checks pass; a manual-evidence check "
"remains — record --phase verify --exit-code 0 "
"--evidence '<what you observed>' to finish"}
code = 0
elif not runnable:
result = {"task": task["id"], "status": "verifying",
"detail": "no executable checks; record --phase verify "
"with --evidence instead"}
code = 0
else:
code, result = _fail(task)
result["failed_checks"] = [r for r in results if not r["passed"]]
save(state, args.state)
d, dcode = directive(state)
result["directive"] = d
return emit(result, code if code != 0 else dcode)
def cmd_close(args):
state = load(args.state)
waivers = dict(zip(args.waive or [], args.reason or []))
if (args.waive or []) and len(args.waive) != len(args.reason or []):
return emit({"error": "every --waive needs a matching --reason"}, 6)
blocking = []
for t in state["tasks"]:
if t["status"] == "verified":
continue
if t["id"] in waivers:
t["status"] = "waived"
t["waive_reason"] = waivers[t["id"]]
continue
blocking.append({"task": t["id"], "status": t["status"]})
if blocking:
return emit({"verdict": "CLOSE-REFUSED",
"blocking": blocking,
"rule": "close requires every task verified, or waived "
"with --waive <id> --reason <why>"}, 4)
state["status"] = "closed"
state["closed_at"] = now()
save(state, args.state)
return emit({
"verdict": "CLOSED",
"goal": state["goal"],
"iterations_used": state["iteration"],
"handoff": {
"tasks": [{"id": t["id"], "skill": t["skill"],
"status": t["status"],
"evidence": t["evidence"][-1] if t["evidence"] else None,
"waive_reason": t.get("waive_reason")}
for t in state["tasks"]],
},
})
def cmd_status(args):
state = load(args.state)
return emit({
"goal": state["goal"], "status": state["status"],
"iteration": "%d/%d" % (state["iteration"],
state["max_loop_iterations"]),
"tasks": [{"id": t["id"], "status": t["status"],
"attempts": "%d/%d" % (t["attempts"], t["max_attempts"])}
for t in state["tasks"]],
})
def run_sample():
"""In-memory demo: two tasks, one verify failure, retry, verified close."""
import tempfile, os # noqa: E401
tmp = tempfile.mkdtemp(prefix="harness-demo-")
plan_path = os.path.join(tmp, "plan.json")
state_path = os.path.join(tmp, "state.json")
plan = {
"schema": "agent-harness/plan.v1",
"goal": "demo: ship a verified change",
"domain": "engineering",
"tasks": [
{"id": "T1", "skill": "demo-skill",
"objective": "make the change",
"verification": [{"cmd": "true", "expect_exit": 0,
"kind": "smoke"}], "max_attempts": 3},
],
"loop": {"max_loop_iterations": 12},
}
with open(plan_path, "w") as f:
json.dump(plan, f)
steps = [
["init", "--plan", plan_path, "--state", state_path],
["next", "--state", state_path],
["record", "--state", state_path, "--task", "T1",
"--phase", "execute", "--exit-code", "0"],
["record", "--state", state_path, "--task", "T1",
"--phase", "verify", "--exit-code", "1",
"--evidence", "check failed: unexpected output"],
["record", "--state", state_path, "--task", "T1",
"--phase", "execute", "--exit-code", "0"],
["record", "--state", state_path, "--task", "T1",
"--phase", "verify", "--exit-code", "0",
"--evidence", "smoke check exit 0, output matches objective"],
["close", "--state", state_path],
]
for s in steps:
print("\n$ loop_controller.py " + " ".join(s))
code = main(s)
print("(exit %d)" % code)
return 0
def build_parser():
ap = argparse.ArgumentParser(
description="Bounded execute→verify→close loop state machine for the "
"agent-harness skill.")
ap.add_argument("--sample", action="store_true",
help="Run an in-memory demo loop and exit 0.")
sub = ap.add_subparsers(dest="cmd")
p = sub.add_parser("init", help="Create a state file from a plan.")
p.add_argument("--plan", required=True)
p.add_argument("--state", required=True)
p = sub.add_parser("next", help="Emit the next directive.")
p.add_argument("--state", required=True)
p = sub.add_parser("record", help="Record an execute/verify outcome.")
p.add_argument("--state", required=True)
p.add_argument("--task", required=True)
p.add_argument("--phase", required=True, choices=["execute", "verify"])
p.add_argument("--exit-code", required=True, type=int)
p.add_argument("--evidence", help="What was observed (required to pass verify).")
p = sub.add_parser("verify", help="Run the task's executable checks via "
"subprocess and adjudicate them.")
p.add_argument("--state", required=True)
p.add_argument("--task", required=True)
p.add_argument("--cwd", default=".",
help="Working directory for check commands (repo root).")
p = sub.add_parser("close", help="Close the loop (refuses if unverified).")
p.add_argument("--state", required=True)
p.add_argument("--waive", action="append",
help="Task id to waive (repeatable; requires --reason).")
p.add_argument("--reason", action="append",
help="Reason for the matching --waive (repeatable).")
p = sub.add_parser("status", help="Summarize loop state.")
p.add_argument("--state", required=True)
return ap
def main(argv=None):
ap = build_parser()
args = ap.parse_args(argv)
if args.sample:
return run_sample()
if not args.cmd:
ap.print_help()
return 0
return {"init": cmd_init, "next": cmd_next, "record": cmd_record,
"verify": cmd_verify, "close": cmd_close,
"status": cmd_status}[args.cmd](args)
if __name__ == "__main__":
sys.exit(main())

View file

@ -1,7 +1,7 @@
{
"name": "product-skills",
"description": "13 production-ready product skills: product manager toolkit (RICE, PRDs), agile product owner, product strategist, UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, product analytics, experiment designer, product discovery, roadmap communicator, code-to-prd, research summarizer, apple-hig-expert (Apple Human Interface Guidelines), spec-to-repo. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.",
"version": "2.9.0",
"description": "13 production-ready product skills bundled in this plugin: product-skills fork-orchestrator with continuous-discovery loop (deterministic 16-lane router, Torres cadence tracker, OST linter), product manager toolkit (RICE, PRDs), product strategist, UX researcher, UI design system, competitive teardown, landing page generator, SaaS scaffolder, product analytics, experiment designer, product discovery, roadmap communicator, spec-to-repo. Companion standalone plugins: agile-product-owner, code-to-prd, apple-hig-expert, research-summarizer. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.",
"version": "2.11.1",
"author": {
"name": "Alireza Rezvani",
"url": "https://alirezarezvani.com"

View file

@ -1,10 +1,33 @@
# Product Team Skills - Claude Code Guidance
This guide covers the 13 production-ready product management skills and their Python automation tools.
This guide covers the 17 production-ready product management skills (13 bundled incl. the orchestrator + 4 standalone plugins) and their Python automation tools.
## Orchestrator & Discovery Loop (product-skills)
`skills/product-skills/` is the domain's `context: fork` orchestrator and agent harness adapter:
```bash
# Route a product goal deterministically across all 16 lanes (exit 0 route / 2 ask / 3 no signal)
python3 skills/product-skills/scripts/product_goal_router.py --text "help me prioritize features"
# Score the continuous-discovery cadence (Torres weekly habit; refuses on < 2 interviews)
python3 skills/product-skills/scripts/discovery_cadence_tracker.py --input discovery_log.json
# Lint the Opportunity Solution Tree (exit 2 blocks the tree from driving a roadmap)
python3 skills/product-skills/scripts/ost_linter.py --input ost.json
```
Commands: `/cs:product` (router) · `/cs:grill-product` (canon-cited grilling) ·
`/cs:product-loop` (the discovery loop). Agent: `cs-product-orchestrator`. Build-scale
goals compile through `engineering/agent-harness` with the `product-team.json` manifest.
Hard rules: no roadmap cites a tree that fails the linter; single-participant claims are
anecdotes; AI features ship with eval specs (see
`skills/product-skills/references/ai_product_evals.md`).
## Product Skills Overview
**Available Skills:**
0. **product-skills/** - Domain orchestrator (`context: fork`) + continuous-discovery loop (3 tools: goal router, cadence tracker, OST linter)
1. **product-manager-toolkit/** - RICE prioritization, customer interview analysis (2 tools)
2. **agile-product-owner/** - User story generation, sprint planning (1 tool)
3. **product-strategist/** - OKR cascade, strategic planning (1 tool)
@ -22,11 +45,11 @@ This guide covers the 13 production-ready product management skills and their Py
15. **apple-hig-expert/** - Apple Human Interface Guidelines compliance and design (1 tool: hig_checker)
16. **spec-to-repo/** - Convert a spec document into a scaffolded repository
**Total Tools:** 17 Python automation tools
**Total Tools:** 22 Python automation tools
**Agents:** 5 (cs-product-manager, cs-agile-product-owner, cs-product-strategist, cs-ux-researcher, cs-product-analyst)
**Agents:** 6 (cs-product-orchestrator, cs-product-manager, cs-agile-product-owner, cs-product-strategist, cs-ux-researcher, cs-product-analyst)
**Slash Commands:** 8 (/rice, /okr, /persona, /user-story, /competitive-matrix, /prd, /sprint-plan, /code-to-prd)
**Slash Commands:** 11 (/cs:product, /cs:grill-product, /cs:product-loop, /rice, /okr, /persona, /user-story, /competitive-matrix, /prd, /sprint-plan, /code-to-prd)
## Python Automation Tools
@ -312,7 +335,7 @@ python roadmap-communicator/scripts/changelog_generator.py --from v1.0.0 --to HE
---
**Last Updated:** May 10, 2026
**Skills Deployed:** 13/13 product skills production-ready
**Total Tools:** 17 Python automation tools
**Agents:** 5 | **Commands:** 8
**Last Updated:** July 3, 2026
**Skills Deployed:** 17/17 product skills production-ready (product-skills is now a fork-orchestrator + discovery loop)
**Total Tools:** 22 Python automation tools
**Agents:** 6 | **Commands:** 11

View file

@ -0,0 +1,81 @@
---
name: cs-product-orchestrator
description: Outcome-first product lead. Routes product inquiries (prioritization, OKRs, UX research, design systems, competitive, analytics, experiments, discovery, roadmaps, scaffolding, stories, HIG, code-to-PRD, summarization) to the right sub-skill via the product-skills orchestrator, and drives the continuous-discovery loop with machine gates (cadence tracker + OST linter). Forks context to keep heavy intake (interview logs, event exports, competitor data) out of the parent thread. Signature forcing question — "What outcome does this serve, and which tested assumption says it will?"
tools: Read, Write, Edit, Glob, Grep, Bash, Skill
model: sonnet
---
# Product Orchestrator
You are an outcome-first product lead. Everything hangs from one measurable outcome;
opportunities are customer needs, not features in disguise; solutions earn roadmap slots
by surviving assumption tests, not by being someone's favorite. You run discovery as a
weekly loop with machine gates, and you bracket prioritization frameworks instead of
worshiping one.
## Voice
**"What outcome does this serve, and which tested assumption says it will?"**
The trap you protect against: the feature factory — shipping output, celebrating
velocity, never checking whether anyone's behavior changed.
## Your 16 lanes
12 bundled: product-manager-toolkit (PRIORITIZE) · product-strategist (STRATEGY) ·
ux-researcher-designer (UX) · ui-design-system (DESIGN_SYSTEM) · competitive-teardown
(COMPETITIVE) · product-analytics (ANALYTICS) · experiment-designer (EXPERIMENT) ·
product-discovery (DISCOVERY) · roadmap-communicator (ROADMAP) · spec-to-repo
(SPEC_TO_REPO) · landing-page-generator (LANDING) · saas-scaffolder (SAAS_SCAFFOLD).
4 standalone plugins: agile-product-owner (STORIES) · apple-hig-expert (HIG) ·
code-to-prd (CODE_TO_PRD) · research-summarizer (SUMMARIZE).
## Routing logic
1. Run `python3 product-team/skills/product-skills/scripts/product_goal_router.py --text "<goal>"`.
2. Exit 0 → load the routed skill's SKILL.md (`skill_path` covers the standalone
plugins), follow its workflow in the forked context.
3. Exit 2 → ask ONE clarifying question naming the candidates, with a recommended answer.
4. Exit 3 → ask the user to restate the goal with the deliverable named. Never guess.
## The discovery loop (your recurring duty)
Weekly: score the log (`discovery_cadence_tracker.py` — refuses on < 2 interviews), act
on `next_loop_action`, lint the tree (`ost_linter.py` — exit 0 required before any
roadmap cites it), keep the streak alive. DORMANT 4+ weeks → escalate to the product
lead by name. HEALTHY + validated assumption → graduate to experiment-designer or a PRD.
## How you communicate (Matt Pocock grill discipline)
One question per turn; always recommend; explore the workspace before asking (an
`ost.json` or `discovery_log.json` resolves the lane silently); depth-first on
multi-lane inquiries; never silently chain. Digest ≤ 200 words: analyzed, top 3 findings
(canon-cited), top 3 next actions (named owner), artifact path, one grill challenge.
Hard outputs:
- Insights carry participant counts — singletons are anecdotes, flagged as such.
- Experiment recommendations carry the computed sample size and MDE.
- Prioritization names its framework (RICE / WSJF / opportunity score) and why.
- AI features get an eval spec (golden set + rubric + guardrails) in the PRD, per
`product-team/skills/product-skills/references/ai_product_evals.md`.
## Anti-patterns
- ❌ Cite an OST that fails the linter, or skip the linter because the tree "looks right"
- ❌ Promote a single-participant quote to an insight
- ❌ Answer "what should we build" without asking what outcome it serves
- ❌ Run all 16 lanes "to be thorough" — route to one, digest, chain on confirmation
- ❌ Report an exhausted loop budget as success
## When to escalate
- Delivery/sprint/Jira execution → `project-management` (cs-pm-orchestrator)
- Campaign/landing marketing → `marketing-skill` / `marketing/landing`
- Pricing and packaging economics → `commercial`
- Generic loop mechanics → `engineering/agent-harness` harness-runner
## Available commands
`/cs:product <inquiry>` (router) · `/cs:grill-product <plan>` (grill first) ·
`/cs:product-loop` (discovery loop) · plus the domain's `/rice`, `/okr`, `/persona`,
`/user-story`, `/competitive-matrix`, `/prd`, `/sprint-plan`, `/code-to-prd`.

View file

@ -326,13 +326,26 @@ def create_sample_epic():
}
def main():
import sys
import argparse
parser = argparse.ArgumentParser(
description="Generate INVEST user stories from the bundled sample epic, or plan "
"a sprint from the generated backlog.")
parser.add_argument("mode", nargs="?", choices=["epic", "sprint"], default="epic",
help="'epic' breaks the sample epic into stories; 'sprint' plans "
"a sprint from that backlog (default: epic).")
parser.add_argument("capacity", nargs="?", type=int, default=30,
help="Sprint capacity in story points (sprint mode, default: 30).")
parser.add_argument("--sample", action="store_true",
help="Break the bundled sample epic into stories and exit 0 "
"(same as the default epic mode; kept for harness smoke tests).")
args = parser.parse_args()
generator = UserStoryGenerator()
if len(sys.argv) > 1 and sys.argv[1] == 'sprint':
if args.mode == 'sprint':
# Generate sprint planning
capacity = int(sys.argv[2]) if len(sys.argv) > 2 else 30
capacity = args.capacity
# Create sample backlog
epic = create_sample_epic()

View file

@ -0,0 +1,59 @@
---
description: Matt Pocock-style interrogation of a product plan against the product canon (Torres, Cagan Transformed, Reinertsen/WSJF, Amplitude North Star, evals-as-PRD). One forcing question per turn with a recommended answer; refuses to invoke any sub-skill or start a loop until the outcome-defining decisions are locked. Use before running /cs:product or /cs:product-loop on a fuzzy plan.
argument-hint: "<product plan, roadmap, feature idea, or strategy to interrogate>"
---
# /cs:grill-product — grill a product plan before running it
Interrogate this plan — do not execute anything yet:
**$ARGUMENTS**
Five rules (preserved from Matt Pocock, MIT): one question per turn · always give a
recommended answer · explore the workspace before asking · walk the decision tree
depth-first · track answered questions and their dependencies.
## Decision tree
- **Branch 1 — Outcome**: "What single measurable outcome does this serve, with a number?
Recommended: write it as the OST root before anything else. Canon: Torres,
*Continuous Discovery Habits*."
- **Branch 2 — Evidence**: "Which tested assumption says this will work — and how many
independent participants back it? Recommended: link the surviving assumption test;
singletons are anecdotes. Canon: Bland, *Testing Business Ideas*; Torres."
- **Branch 3 — Structure**: "Does the tree pass the linter? Recommended: run
`ost_linter.py` — exit 0 before any roadmap cites it; feature-phrased opportunities
(O2) and untested solutions (O4) are the usual failures. Canon: Torres OST discipline."
- **Branch 4 — Prioritization honesty**: "Would delaying any item a quarter erode its
value? Recommended: if yes, run WSJF/cost-of-delay next to RICE and flag rank flips on
one-step estimate changes. Canon: Reinertsen; the WSJF false-precision critique."
- **Branch 5 — Measurement**: "Is your North Star a leading value metric with an input
tree, or revenue/vanity? Recommended: leading value metric; funnel verdicts need
benchmark bands. Canon: Amplitude, *The North Star Playbook*; ProductLed benchmarks."
- **Branch 6 — AI features**: "If any feature is probabilistic: where is the eval —
golden set, rubric, guardrail SLOs? Recommended: write the eval spec into the PRD
before building; vibe-check launches are shipping without tests. Canon: evals-as-PRD
(Lenny's/Braintrust)."
Per-turn output format:
```
Q[i]/[total]: [precise question]
Recommended: [answer + canon-cited rationale]
(Confirm, or override?)
```
## Stop conditions
- All branches resolved → invoke `/cs:product` (question) or `/cs:product-loop`
(recurring discovery) with the locked decisions inlined.
- User says "stop grilling, just run it" → run with unresolved branches flagged in the
digest.
- Abandoned → save the partial grill to `product-grill-{timestamp}.md`.
## Distinct from
- `engineering/grill-me` — generic plan interrogation. This grills against the product
canon.
- `/cs:product` — routes; this refuses to route until decisions are locked.

View file

@ -0,0 +1,49 @@
---
description: Run the continuous-discovery loop — score the weekly cadence (Torres), act on the named gap, lint the Opportunity Solution Tree as the machine gate, and keep the streak alive with explicit stop states. The product-domain recurring loop; graduates validated assumptions to experiments or PRDs.
argument-hint: "[path to discovery_log.json] [path to ost.json]"
---
# /cs:product-loop — the continuous-discovery loop
Inputs (defaults: `discovery_log.json` and `ost.json` in the workspace; shapes in
`product-team/skills/product-skills/assets/`):
**$ARGUMENTS**
## Sequence (one iteration per invocation)
1. **Observe**
```bash
python3 product-team/skills/product-skills/scripts/discovery_cadence_tracker.py --input discovery_log.json
```
Exit 5 (< 2 interviews): there is no cadence to measure help the user book the
first two weekly touchpoints and write the outcome statement; stop there.
2. **Choose** — the report's `next_loop_action` is the choice. Typical actions: book the
missing weekly touchpoint · re-anchor the interview guide on the outcome · test the
top untested assumption (route to `product-discovery`'s assumption_mapper to rank).
3. **Act** — execute with the routed sub-skill's tools (ux-researcher-designer for the
interview, experiment-designer for the test design). One bounded action per
iteration.
4. **Verify**
```bash
python3 product-team/skills/product-skills/scripts/ost_linter.py --input ost.json
```
Exit 2 → fix the listed O1O5 violations before the tree may drive any roadmap or
experiment. Then re-run the cadence tracker and confirm the health score did not
drop.
5. **Record** — update `discovery_log.json` (interview/test entries) and `ost.json`;
note the health score in the digest so the trend is visible across iterations.
6. **Repeat or stop** — terminal states:
- **Graduate**: HEALTHY + a validated assumption → hand off to `experiment-designer`
(A/B gate) or `product-manager-toolkit` (PRD with eval spec if the feature is
AI-powered).
- **Escalate**: DORMANT 4+ weeks → name the product lead and say the habit is dead —
never let discovery die silently.
- **Clean no-op**: cadence HEALTHY, no gaps — book next week's touchpoint and exit.
## Rules
- Never modify the linter or tracker to make a gate pass.
- Insights require recurrence across independent participants — singletons stay
anecdotes.
- The loop edits the log and the tree, never the gates (locked-evaluator invariant).

View file

@ -0,0 +1,47 @@
---
description: Top-level product-team router. Classifies a product inquiry across 16 lanes (prioritization, OKRs, UX, design system, competitive, analytics, experiments, discovery, roadmaps, spec-to-repo, landing, SaaS scaffold, stories, HIG, code-to-PRD, summarizer) with a deterministic script and forks context to the right sub-skill via the product-skills orchestrator, returning a ≤200-word digest with one grill challenge.
argument-hint: "<product inquiry: prioritize features, plan an experiment, discovery health, etc.>"
---
# /cs:product — Product Team router
Route this inquiry through the `product-skills` orchestrator:
**$ARGUMENTS**
## Routing (deterministic — run the script, don't eyeball)
```bash
python3 product-team/skills/product-skills/scripts/product_goal_router.py --text "$ARGUMENTS" --output json
```
- Exit 0 → load `skill_path`/SKILL.md (covers the 4 standalone plugins too) and follow
that skill's own workflow in a fork.
- Exit 2 → ask ONE clarifying question naming the listed candidates, recommended answer
first.
- Exit 3 → ask the user to restate the goal with the deliverable named. Never guess.
- Explore the workspace first — an `ost.json`, `discovery_log.json`, or `features.csv`
resolves the lane silently. Never silently chain a second sub-skill.
## Output (≤200-word digest)
- What was analyzed
- Top 3 findings, each anchored to a canon citation
- Top 3 next actions with a named owner
- Artifact path
- One grill challenge (e.g. "This roadmap cites an OST that fails the linter — which
opportunity backs item 3?")
## Hard rules
- Insights carry participant counts; singletons are anecdotes.
- Experiments carry computed sample size + MDE, never gut feel.
- Prioritization names its framework (RICE / WSJF / opportunity score) and why.
- AI features get an eval spec (golden set + rubric + guardrails) in the PRD.
- Recurring discovery work goes to `/cs:product-loop` instead.
## Distinct from
- `project-management` — how to deliver. This domain is what to build.
- `marketing/landing` — from-scratch marketing pages; `landing-page-generator` here
scaffolds product Next.js/TSX pages.

View file

@ -1,61 +1,180 @@
---
name: "product-skills"
description: "Router/index for the 12 product skills bundled in this plugin (RICE prioritization, OKRs, UX research, design tokens, competitive teardown, analytics, experiments, discovery, roadmaps, spec-to-repo, landing pages, SaaS scaffolding). Use when a product request doesn't obviously match one skill and you need to pick the right one (e.g., 'help me prioritize features', 'plan a product experiment')."
version: 2.9.0
description: "Use when coordinating product work across the 12 bundled product sub-skills (RICE, OKRs, UX research, design tokens, competitive teardown, analytics, experiments, discovery, roadmaps, spec-to-repo, landing pages, SaaS scaffolding) or the 4 standalone product-team plugins (user stories, Apple HIG, code-to-PRD, research summarizer). Triggers on 'help me prioritize', 'plan a product experiment', 'we ship features nobody uses', 'run the discovery loop', 'is our OST sound'. Forks context to route to one sub-skill via a deterministic signal router and returns a digest; can also drive a continuous-discovery loop (Torres cadence tracker + OST linter as machine gates) or a full goal→plan→execute→verify→close run through the repo-wide agent-harness. Distinct from project-management (how to deliver vs what to build), marketing/landing (from-scratch pages), and engineering/agent-harness (the generic loop engine this orchestrator plugs into)."
context: fork
version: 2.11.1
author: Alireza Rezvani
license: MIT
tags:
- product
- product-management
- ux
- ui
- saas
- agile
agents:
- claude-code
- codex-cli
- openclaw
tags: [product, product-management, orchestrator, discovery, ux, analytics, agent-harness]
compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli]
---
# Product Skills — Router
# Product Team — Domain Orchestrator & Discovery Loop
This plugin bundles **12 product skills** (this router is the 13th folder under `product-team/skills/`). Each skill is self-contained: read its `SKILL.md`, run its `scripts/`, apply its `references/` and `assets/`.
This orchestrator does two jobs. **Routing:** fork context, classify a product inquiry
with `scripts/product_goal_router.py` across all 16 product-team lanes (12 bundled + 4
standalone plugins), run exactly one, return a digest. **Looping:** run product work as
bounded agentic loops with machine-checkable gates — the continuous-discovery loop
(weekly cadence scored by `discovery_cadence_tracker.py`, tree structure enforced by
`ost_linter.py`) and goal-scale runs through the repo-wide agent-harness.
## Routing table
## When to invoke
Match the request against the signals below, then load `product-team/skills/<skill>/SKILL.md`. If two or more rows match, ask the user one clarifying question before loading anything.
| Symptom | Sub-skill |
|---|---|
| "Prioritize features / RICE / PRD" | `product-manager-toolkit` |
| "OKRs, strategy cascade" | `product-strategist` |
| "Personas, usability, research synthesis" | `ux-researcher-designer` |
| "Design tokens, WCAG contrast" | `ui-design-system` |
| "Competitor matrix, teardown" | `competitive-teardown` |
| "Retention, cohorts, funnels, KPIs" | `product-analytics` |
| "A/B test, sample size, hypothesis" | `experiment-designer` |
| "Discovery, assumptions, opportunity trees" | `product-discovery` |
| "Roadmap comms, release notes, changelog" | `roadmap-communicator` |
| "Spec → runnable repo" | `spec-to-repo` |
| "Landing page (Next.js/Tailwind)" | `landing-page-generator` |
| "SaaS boilerplate" | `saas-scaffolder` |
| "User stories, sprint capacity" | `agile-product-owner` (standalone) |
| "Apple HIG audit" | `apple-hig-expert` (standalone) |
| "PRD from an existing codebase" | `code-to-prd` (standalone) |
| "Summarize papers/articles" | `research-summarizer` (standalone) |
| Request signals | Skill | Path |
|---|---|---|
| Prioritize features, RICE scores, interview synthesis | product-manager-toolkit | `skills/product-manager-toolkit/` |
| OKRs, strategy cascade, objective alignment | product-strategist | `skills/product-strategist/` |
| Personas, usability findings, research synthesis | ux-researcher-designer | `skills/ux-researcher-designer/` |
| Design tokens, component specs, WCAG contrast | ui-design-system | `skills/ui-design-system/` |
| Competitor analysis, feature/pricing matrix | competitive-teardown | `skills/competitive-teardown/` |
| Retention, cohorts, funnel analysis | product-analytics | `skills/product-analytics/` |
| A/B test design, sample size, hypothesis gates | experiment-designer | `skills/experiment-designer/` |
| Opportunity trees, assumption mapping, discovery | product-discovery | `skills/product-discovery/` |
| Roadmap formats per audience, changelogs | roadmap-communicator | `skills/roadmap-communicator/` |
| Turn a written spec into a repo scaffold | spec-to-repo | `skills/spec-to-repo/` |
| Landing page (Next.js TSX + Tailwind) | landing-page-generator | `skills/landing-page-generator/` |
| Bootstrap a SaaS app skeleton | saas-scaffolder | `skills/saas-scaffolder/` |
## Quick start
## Routing logic (deterministic)
```bash
# Example: route a prioritization request
cat product-team/skills/product-manager-toolkit/SKILL.md
python3 product-team/skills/product-manager-toolkit/scripts/rice_prioritizer.py --help
python3 scripts/product_goal_router.py --text "<the goal>" --output json
```
## Related product-team plugins (packaged separately, not in this bundle)
Exit 0 → `route_to` names the skill (with `skill_path`, including the standalone
plugins): load its SKILL.md and follow its workflow. Exit 2 → ask ONE clarifying question
naming the listed candidates, with a recommended answer. Exit 3 → no signal: ask the user
to restate the goal with the deliverable named. Never guess silently; never silently
chain — digest first, confirm, then chain.
- `product-team/agile-product-owner/` — user stories, sprint capacity
- `product-team/code-to-prd/` — reverse-engineer a PRD from a codebase
- `product-team/apple-hig-expert/` — Apple HIG audits (Liquid Glass era)
- `product-team/research-summarizer/` — document summarization with citation extraction
## The discovery loop (the domain's recurring agentic loop)
## Rules
Modern discovery is a weekly habit, not a project phase (Torres). Run it as a bounded
loop with two machine gates:
- Route to exactly one skill, then follow that skill's own workflow.
- This router ships no tools of its own — if no row matches, say so and ask rather than improvising.
1. **Observe** — maintain `discovery_log.json` (interviews, assumption tests; shape in
`assets/sample_discovery_log.json`) and score the cadence:
```bash
python3 scripts/discovery_cadence_tracker.py --input discovery_log.json
```
Refuses on < 2 interviews (exit 5) there is no cadence to measure yet. Output:
health 0100, verdict HEALTHY/AT-RISK/DORMANT, named gaps, and `next_loop_action`.
2. **Choose** — the tracker's `next_loop_action` IS the choice: book the touchpoint,
re-anchor the guide on the outcome, or test the top untested assumption (route to
`product-discovery`'s assumption_mapper for prioritization).
3. **Act** — run the interview / assumption test with the routed sub-skill's tools.
4. **Verify** — keep the tree structurally sound before it may drive a roadmap:
```bash
python3 scripts/ost_linter.py --input ost.json # exit 2 = NEEDS-REWORK, fix before citing the tree
```
Rules: one measurable outcome root (O1), opportunities are needs not features (O2),
targeted opportunities compare ≥ 2 solutions (O3), every solution has an assumption
test (O4), no orphan solutions (O5 — the feature-factory tell).
5. **Record / Repeat-or-stop** — update the log, keep the weekly streak alive. Stop
states: HEALTHY + validated assumption → graduate to `experiment-designer` (build the
A/B gate) or `product-manager-toolkit` (PRD); DORMANT for 4+ weeks → escalate to the
product lead by name — do not quietly let discovery die.
For build-scale goals ("turn this validated spec into a repo and verify it"), compile
through the repo-wide harness instead:
```bash
python3 engineering/agent-harness/skills/agent-harness/scripts/goal_compiler.py \
--goal "<goal>" --manifest engineering/agent-harness/skills/agent-harness/assets/harnesses/product-team.json \
--out .agent-harness/plan.json
```
The domain's three strongest close-out gates plug in as task verifications:
`../spec-to-repo/scripts/validate_project.py` (exit 0), `code-to-prd`'s golden
`expected_outputs/`, and `research-summarizer`'s citation-count check.
## Hard rules
1. **Evidence before conviction**: no roadmap item cites the OST unless `ost_linter.py`
exits 0; no insight is asserted from a single participant (anecdote, not insight).
2. **Outcome-first**: every loop hangs from one measurable outcome — the linter's O1 rule
is the intake gate.
3. **Experiments are gated by math**: sample size from
`../experiment-designer/scripts/sample_size_calculator.py`, never gut feel; report the
MDE with the verdict.
4. **Prioritization shows its framework**: RICE for steady-state, WSJF/cost-of-delay when
time sensitivity dominates, opportunity scoring for underserved needs — name which and
why (see [references/product_operating_model.md](references/product_operating_model.md)).
5. **AI features ship with evals**: a golden set + rubric is the PRD's quality contract
for probabilistic features
([references/ai_product_evals.md](references/ai_product_evals.md)).
6. **Never modify a gate you are judged by**; exhausted budgets escalate to a named human,
never report as success.
## Forcing-question library (grill-with-docs pattern)
One per turn, recommended answer, canon citation. Never run a sub-skill or start a loop
until the lane-defining decision is locked:
- **DISCOVERY lane**: "What is the single outcome this discovery serves, stated with a
number? Recommended: write it as the OST root first — opportunities without an outcome
are a feature factory. Canon: Torres, *Continuous Discovery Habits*; opportunity
solution trees (producttalk.org)."
- **PRIORITIZE lane**: "Does time sensitivity change this ranking — would delaying any
item a quarter erode its value? Recommended: if yes, run WSJF/cost-of-delay alongside
RICE and compare ranks; flag items whose rank flips on a one-step estimate change.
Canon: Reinertsen, *Principles of Product Development Flow*; SAFe WSJF false-precision
critique."
- **EXPERIMENT lane**: "What baseline rate and MDE justify this test's runtime?
Recommended: compute n first; if you can't reach it in 4 weeks, test a bigger lever.
Canon: statistical power analysis (experiment-designer)."
- **ANALYTICS lane**: "Is your North Star a leading indicator of value exchange, or
revenue/vanity? Recommended: leading value metric with an input tree. Canon: Amplitude,
*The North Star Playbook*."
- **STRATEGY lane**: "Are these OKRs outcomes or shipping lists? Recommended: outcomes —
output OKRs are the #1 operating-model failure. Canon: Cagan, *Transformed* (SVPG,
2024)."
- **BUILD lanes (spec-to-repo / saas-scaffolder)**: "Which validated assumption says this
should be built at all? Recommended: link the OST test that survived; building is the
most expensive way to test an idea. Canon: Torres; Bland, *Testing Business Ideas*."
## Assumptions
1. The user owns (or advises the owner of) the product decision.
2. Discovery data lives in the workspace as JSON logs — the loop is file-backed and
resumable; every tool ships `--sample` so the shape is visible first.
3. The four standalone plugins are installed alongside the bundle (the router still
routes to them by path if not).
## Non-goals
- Not the delivery loop — sprint/flow/Jira work routes to `project-management`.
- Not the generic loop engine — that is `engineering/agent-harness`; this orchestrator is
the product-domain adapter (router + discovery gates).
- Not campaign marketing — `marketing/landing` builds from-scratch marketing pages;
`landing-page-generator` here scaffolds product Next.js/TSX pages.
## Output artifacts
| Mode | Artifact |
|---|---|
| Route | Sub-skill's own artifact + ≤ 200-word digest with one canon-cited challenge |
| Discovery loop | `discovery_log.json` + cadence report + linted `ost.json` |
| Harness run | `.agent-harness/plan.json` + `state.json` + close handoff |
## Anti-patterns (do not)
- ❌ Run all 16 lanes "to be thorough" — route to one, digest, chain on confirmation
- ❌ Cite an OST that fails the linter, or promote a single-participant anecdote to insight
- ❌ Ship an AI feature whose PRD has no eval (golden set + rubric)
- ❌ Let the discovery streak die silently — DORMANT escalates by name
- ❌ Treat RICE as the only prioritization lens when deadlines dominate
## References
- [references/continuous_discovery_canon.md](references/continuous_discovery_canon.md) —
Torres, OST, assumption testing, JTBD switch interviews, story mapping
- [references/product_operating_model.md](references/product_operating_model.md) — Cagan
*Transformed*, North Star framework, PLG benchmarks, WSJF/ODI vs RICE
- [references/ai_product_evals.md](references/ai_product_evals.md) — evals-as-PRD, model
cards, evaluator-optimizer loops
- Loop engine: `engineering/agent-harness` · Loop vocabulary: `loop-library`

View file

@ -0,0 +1,20 @@
{
"_comment": "Continuous-discovery log shape — feed to scripts/discovery_cadence_tracker.py.",
"outcome": "increase paid conversion from 9% to 12% by Q4",
"interviews": [
{"date": "2026-05-05", "participant": "P1", "outcome_linked": true,
"assumptions_tested": ["users understand the trial limits"]},
{"date": "2026-05-12", "participant": "P2", "outcome_linked": true,
"assumptions_tested": []},
{"date": "2026-05-26", "participant": "P3", "outcome_linked": false,
"assumptions_tested": ["pricing page is the drop-off point"]},
{"date": "2026-06-02", "participant": "P4", "outcome_linked": true,
"assumptions_tested": ["annual plan framing increases upgrades"]}
],
"assumption_tests": [
{"date": "2026-05-15", "assumption": "users understand the trial limits",
"result": "invalidated"},
{"date": "2026-06-05", "assumption": "annual plan framing increases upgrades",
"result": "inconclusive"}
]
}

View file

@ -0,0 +1,26 @@
{
"_comment": "Opportunity Solution Tree shape — feed to scripts/ost_linter.py. This sample deliberately contains one O2 violation (feature-phrased opportunity) and one O4 violation (untested solution).",
"outcome": {
"statement": "Increase week-4 retention from 22% to 30%",
"metric": "week-4 retention"
},
"opportunities": [
{
"statement": "New users can't tell whether setup worked",
"target": true,
"solutions": [
{"statement": "Post-setup verification checklist",
"tests": [{"assumption": "users abandon because they doubt setup succeeded", "type": "interview"}]},
{"statement": "Live sample-data preview after connect",
"tests": [{"assumption": "a working preview reduces first-week drop-off", "type": "prototype"}]}
]
},
{
"statement": "Add an onboarding wizard",
"solutions": [
{"statement": "Onboarding wizard v2", "tests": []}
]
}
],
"solutions": []
}

View file

@ -0,0 +1,67 @@
# AI Product Evals — the new PRD quality contract
The single most-demanded new PM competency of 20252026, and the biggest coverage gap
this domain had. For deterministic features the PRD's acceptance criteria are the quality
contract; for **probabilistic (AI) features, the eval is the contract** — encode what
"good" means as data + rubric *before* building, or you ship on vibes.
## Evals are the new PRD
The PM owns three artifacts per AI feature:
1. **Golden set** — a labeled collection of real inputs with expected-quality outputs,
covering every intent the feature claims to handle plus the known failure modes
(hallucination, refusal, off-topic, unsafe). Floor: enough examples per intent that a
regression is statistically visible, and the set grows from production incidents.
2. **Rubric** — the dimensions of "good" (accuracy, groundedness, tone, format,
safety), each with a pass criterion a grader (human or LLM-judge) can apply
consistently. Check grader consistency with inter-rater agreement (Cohen's kappa)
before trusting LLM-judge scores.
3. **Guardrail metrics** — the SLOs that page someone: hallucination rate ceiling,
refusal-rate band, latency/cost budgets.
Anti-pattern: "vibe check" launches — demo-driven quality assessment with no golden set,
no rubric, no regression gate. It is the AI equivalent of shipping without tests.
## Model/system cards
Enterprise and regulated buyers expect a model card documenting intended use,
out-of-scope use, eval data and results (disaggregated where bias matters), and
limitations (Mitchell et al.'s nine canonical sections; Anthropic/OpenAI system cards in
current practice). The PM owns the product-facing half: intended use, eval results,
limitations users will hit.
## The loop connection (evaluator-optimizer)
The same generator/critic loop that powers agent harnesses is what PM-owned evals feed:
the golden set + rubric become the evaluator's criteria, and Anthropic's guidance is
explicit that the evaluator-optimizer pattern pays off exactly "when there are clear
evaluation criteria and iterative refinement provides measurable value." Practically:
- The eval spec is the `done_when` of any agent-harness task that touches an AI feature.
- Eval runs are the locked evaluator — the feature loop may edit prompts/retrieval/
models, never the golden set it is judged by (autoresearch invariant).
- Experiment-designer's sample-size math applies to eval deltas too: a 2-point rubric
improvement on 30 examples is noise.
## Where this lands in the domain today
- `experiment-designer` — extend hypothesis gates to eval-delta hypotheses.
- `product-manager-toolkit` — PRD template gains an "Eval spec" section for AI features
(golden set size, rubric dimensions, guardrail SLOs, owner).
- `product-analytics` — guardrail metrics join the KPI tree as SLO-style entries.
## Sources
1. Lenny's Newsletter, "Beyond vibe checks: A PM's complete guide to evals" —
https://www.lennysnewsletter.com/p/beyond-vibe-checks-a-pms-complete
2. Braintrust, "Evals for PMs" — https://www.braintrust.dev/blog/evals-for-pms
3. Aakash Gupta, "AI Evals" — https://www.news.aakashg.com/p/ai-evals
4. Mitchell et al., "Model Cards for Model Reporting" (FAT* 2019) —
https://arxiv.org/abs/1810.03993
5. Anthropic, "Building Effective Agents" (evaluator-optimizer applicability) —
https://www.anthropic.com/research/building-effective-agents
6. Jacob Cohen, "A Coefficient of Agreement for Nominal Scales" (1960) — kappa as the
inter-rater agreement statistic
7. This repo: `engineering/autoresearch-agent` (locked evaluator), `engineering/self-eval`
(anti-inflation scoring)

View file

@ -0,0 +1,71 @@
# Continuous Discovery Canon
The method layer behind `discovery_cadence_tracker.py` and `ost_linter.py`. The
20242026 discovery canon reframed discovery from a project phase into a **weekly
operating rhythm** with a structural artifact (the Opportunity Solution Tree) and a unit
of progress (the assumption test).
## The weekly habit (Torres)
Teresa Torres' definition of continuous discovery: the product trio (PM, designer,
engineer) has **at least weekly touchpoints with customers**, in pursuit of a desired
**outcome**, conducting **small research activities** (interviews, assumption tests).
Corollaries the tracker scores:
- **Cadence, not volume**: 4 interviews in one week then silence for a month is a broken
habit — hence the streak (30 pts) and week-coverage (30 pts) components.
- **Outcome-anchored**: interviews that don't tie back to the outcome drift into feature
tourism — hence the linkage component (20 pts).
- **Assumption tests are the throughput**: Torres' target rhythm resolves assumptions
continuously; piling up untested assumptions is discovery theater — hence the
throughput component (20 pts) and the untested-backlog gap.
## The Opportunity Solution Tree (why each lint rule exists)
- **O1 — one measurable outcome root**: the tree hangs from exactly one outcome, stated
with a metric and target. Multiple outcomes = multiple trees.
- **O2 — opportunities are needs, not features**: an opportunity is a customer need,
pain, or desire surfaced by research. "Add an onboarding wizard" is a solution wearing
an opportunity's clothes; the build-verb heuristic catches it.
- **O3 — compare ≥ 2 solutions per targeted opportunity**: Torres' compare-and-contrast
discipline; a single pet solution skips the decision.
- **O4 — every solution carries an assumption test**: untested solutions are opinions;
the test types (interview, prototype, smoke test, concierge) come from Bland's
desirability/viability/feasibility/usability mapping.
- **O5 — no orphan solutions**: a solution attached to no opportunity is the
feature-factory anti-pattern in its purest form.
## Assumption mapping & test sequencing (Bland)
Rank assumptions by **importance × evidence-weakness** and test the riskiest first
(leap-of-faith assumptions). Match test type to assumption class — desirability →
interview/smoke test; feasibility → spike/prototype; viability → pricing test/concierge.
This is `product-discovery/scripts/assumption_mapper.py`'s scoring model; the tracker's
untested-backlog gap feeds it.
## JTBD switch interviews (Moesta)
When interviews need depth, run the switch interview: reconstruct a real past purchase
timeline (first thought → passive looking → active looking → decision) and code the four
forces — push of the current situation, pull of the new solution, anxiety of the new,
habit of the present. Progress happens when push + pull outweigh anxiety + habit.
## Story mapping (Patton)
The bridge from a validated opportunity to a sliced backlog: backbone of activities
left-to-right, stories vertically, release slices horizontally — each slice an
end-to-end walking skeleton, never a vertical feature column.
## Sources
1. Teresa Torres, *Continuous Discovery Habits* (Product Talk LLC, 2021) and
https://www.producttalk.org/opportunity-solution-trees/
2. David J. Bland & Alexander Osterwalder, *Testing Business Ideas* (Wiley, 2019)
3. Bob Moesta, *Demand-Side Sales 101* (2020); Jobs-to-be-Done switch-interview practice
— https://jobstobedone.org/
4. Jeff Patton, *User Story Mapping* (O'Reilly, 2014)
5. Christian Rohrer, "When to Use Which User-Experience Research Methods", NN/g —
https://www.nngroup.com/articles/which-ux-research-methods/
6. Marty Cagan, *Inspired* (2nd ed., Wiley, 2018) — discovery/delivery separation
7. Product Talk, "The Product Operating Model and Continuous Discovery" —
https://www.producttalk.org/the-product-operating-model/

View file

@ -0,0 +1,69 @@
# Product Operating Model, Metrics & Prioritization Brackets
The strategy layer behind the orchestrator's STRATEGY/ANALYTICS/PRIORITIZE forcing
questions. Three moves define the 20242026 canon: the product operating model as the
org-level frame, the North Star framework as the metrics spine, and prioritization as a
**bracket of frameworks** rather than RICE-for-everything.
## The product operating model (Cagan, *Transformed*, 2024)
The transformation agenda in three principles: **empowered teams** (problems to solve,
not features to build), **outcomes over output**, **innovation over predictability**
elaborated as 20 first principles across how you build / solve problems / decide what to
work on. The tell-tale failures the orchestrator grills for: OKRs that are shipping
lists, roadmaps as commitments of output, teams measured on velocity instead of outcome.
## North Star framework (Amplitude)
A valid North Star Metric is (a) a **leading indicator** of sustainable business results,
(b) a measure of **value exchange** with the customer, (c) not revenue and not a vanity
count. It decomposes into an **input-metric tree** (breadth × depth × frequency ×
efficiency) that teams can actually move. AARRR remains the funnel taxonomy, but the
NSM + input tree is the strategy-to-analytics bridge product-analytics work should hang
from.
## PLG benchmark bands
Verdicts need bands, not vibes (medians from ProductLed/OpenView benchmark corpora,
20242025): signup→activation median ≈ 17% (best-in-class 3350%+); free→paid median
≈ 9% (PQL-driven motions 2530%). A funnel scorer without calibrated bands cannot say
"weak stage" honestly.
## Prioritization: bracket RICE, don't replace it
| Situation | Framework | Why |
|---|---|---|
| Steady-state backlog | **RICE** | Reach/impact/confidence/effort — cheap, comparable |
| Time sensitivity dominates (deadlines, market windows) | **WSJF / Cost of Delay** (Reinertsen) | RICE is time-blind; CoD/duration surfaces value erosion |
| Underserved-needs hunting | **Opportunity scoring** (Ulwick ODI) | importance + max(importance satisfaction, 0) ranks unmet outcomes |
Two disciplines regardless of framework: (1) **name which framework and why** before
scoring; (2) **sensitivity-check the ranking** — perturb each estimate one step and flag
items whose rank flips (the documented WSJF false-precision failure; SAFe's own critics'
point). `rice_prioritizer.py` covers lane 1; lanes 23 are scored by hand against the
formulas here until dedicated tools land.
## Event taxonomy governance (PostHog/Amplitude-era)
Analytics rot starts at instrumentation: enforce snake_case, present-tense verb
allowlists, object_verb ordering, compact event sets, and tracking-plan review before
new events ship. A taxonomy with near-duplicate events ("signup", "sign_up",
"user_signed_up") cannot support any metric above it.
## Sources
1. Marty Cagan (SVPG), *Transformed: Moving to the Product Operating Model* (Wiley,
2024); https://www.svpg.com/the-product-operating-model-an-introduction/
2. Amplitude, *The North Star Playbook*
https://amplitude.com/books/north-star/about-north-star-framework
3. ProductLed, "Product-Led Growth Benchmarks" —
https://productled.com/blog/product-led-growth-benchmarks; OpenView PLG benchmarks —
https://openviewpartners.com/blog/your-guide-to-product-led-growth-benchmarks/
4. Don Reinertsen, *The Principles of Product Development Flow* (Celeritas, 2009) — cost
of delay; SAFe WSJF — https://framework.scaledagile.com/wsjf
5. Anthony Ulwick, *What Customers Want* (McGraw-Hill, 2005) — Outcome-Driven Innovation
opportunity algorithm (strategyn.com)
6. Jason Yip, "Problems I have with SAFe-style WSJF" —
https://jchyip.medium.com/problems-i-have-with-safe-style-wsjf-772df2beaf02
7. PostHog, product analytics best practices —
https://posthog.com/docs/product-analytics/best-practices

View file

@ -0,0 +1,183 @@
#!/usr/bin/env python3
"""discovery_cadence_tracker.py — score a team's continuous-discovery habit.
Operationalizes Teresa Torres' continuous-discovery canon (weekly customer
touchpoints, outcome-first framing, assumption tests as the unit of progress) as a
deterministic recurring loop: feed it the discovery log each week (Observe), read the
named gaps (Choose), run the next interview or assumption test (Act), re-run the
tracker (Verify), and keep the streak alive (Repeat). The health score is the loop's
acceptance gate a subagent can branch on it mechanically.
Input JSON (see --sample):
{"outcome": "increase paid conversion from 9% to 12% by Q4",
"interviews": [{"date": "YYYY-MM-DD", "participant": str,
"outcome_linked": bool, "assumptions_tested": [str]}],
"assumption_tests": [{"date": "YYYY-MM-DD", "assumption": str,
"result": "validated|invalidated|inconclusive"}]}
Scoring (0100): weekly streak 30 · week coverage 30 · outcome linkage 20 ·
assumption-test throughput 20. Verdicts: HEALTHY >= 70 · AT-RISK 4069 · DORMANT < 40.
Exit codes: 0 scored · 2 unreadable input · 5 insufficient history (< 2 interviews
start the habit before measuring it). Deterministic: the analysis date defaults to
the newest date in the log, never the wall clock (override with --as-of).
Stdlib only.
"""
import argparse
import json
import sys
from datetime import date, timedelta
SAMPLE_LOG = {
"outcome": "increase paid conversion from 9% to 12% by Q4",
"interviews": [
{"date": "2026-05-05", "participant": "P1", "outcome_linked": True,
"assumptions_tested": ["users understand the trial limits"]},
{"date": "2026-05-12", "participant": "P2", "outcome_linked": True,
"assumptions_tested": []},
{"date": "2026-05-26", "participant": "P3", "outcome_linked": False,
"assumptions_tested": ["pricing page is the drop-off point"]},
{"date": "2026-06-02", "participant": "P4", "outcome_linked": True,
"assumptions_tested": ["annual plan framing increases upgrades"]},
],
"assumption_tests": [
{"date": "2026-05-15", "assumption": "users understand the trial limits",
"result": "invalidated"},
{"date": "2026-06-05", "assumption": "annual plan framing increases upgrades",
"result": "inconclusive"},
],
}
def parse_date(value):
try:
return date.fromisoformat(str(value)[:10])
except (TypeError, ValueError):
return None
def week_of(d: date):
return d.isocalendar()[:2]
def analyze(log: dict, as_of: date) -> dict:
interviews = [
{**i, "date": parse_date(i.get("date"))}
for i in log.get("interviews", [])
]
interviews = [i for i in interviews if i["date"] and i["date"] <= as_of]
tests = [
{**t, "date": parse_date(t.get("date"))}
for t in log.get("assumption_tests", [])
]
tests = [t for t in tests if t["date"] and t["date"] <= as_of]
interview_weeks = {week_of(i["date"]) for i in interviews}
first = min(i["date"] for i in interviews)
total_weeks = max(((as_of - first).days // 7) + 1, 1)
# Streak: consecutive weeks with >= 1 interview, counting back from as_of's week.
streak, cursor = 0, as_of
while week_of(cursor) in interview_weeks:
streak += 1
cursor -= timedelta(days=7)
coverage = len(interview_weeks) / total_weeks
linked = sum(1 for i in interviews if i.get("outcome_linked"))
linkage = linked / len(interviews)
resolved = sum(1 for t in tests if t.get("result") in ("validated", "invalidated"))
# Torres cadence target: >= 1 resolved assumption test per 2 weeks.
test_target = max(total_weeks / 2, 1)
throughput = min(resolved / test_target, 1.0)
score = round(
min(streak / 4, 1.0) * 30 + coverage * 30 + linkage * 20 + throughput * 20, 1)
verdict = "HEALTHY" if score >= 70 else ("AT-RISK" if score >= 40 else "DORMANT")
gaps = []
if streak == 0:
gaps.append("no interview in the current week — the weekly habit is broken "
"(Torres: touchpoints are a cadence, not a project phase)")
if coverage < 0.75:
missed = total_weeks - len(interview_weeks)
gaps.append(f"{missed} of {total_weeks} weeks had zero customer touchpoints")
if linkage < 0.8:
gaps.append(f"only {linked}/{len(interviews)} interviews tie back to the outcome — "
"re-anchor the interview guide on the outcome")
if throughput < 1.0:
gaps.append(f"{resolved} resolved assumption tests vs a target of "
f"{int(test_target)} — assumptions are piling up untested")
untested = {a for i in interviews for a in i.get("assumptions_tested", [])}
tested = {t.get("assumption") for t in tests}
backlog = sorted(untested - tested)
if backlog:
gaps.append(f"assumptions surfaced but never tested: {'; '.join(backlog[:3])}")
return {
"outcome": log.get("outcome", ""),
"as_of": as_of.isoformat(),
"weeks_observed": total_weeks,
"interviews": len(interviews),
"distinct_participants": len({i.get("participant") for i in interviews}),
"weekly_streak": streak,
"week_coverage_pct": round(coverage * 100, 1),
"outcome_linkage_pct": round(linkage * 100, 1),
"assumption_tests_resolved": resolved,
"health_score": score,
"verdict": verdict,
"gaps": gaps,
"next_loop_action": (gaps[0] if gaps else
"cadence healthy — book next week's touchpoint before this one ends"),
}
def main() -> int:
ap = argparse.ArgumentParser(
description="Score a continuous-discovery log for cadence health (Torres canon).")
ap.add_argument("--input", help="Path to the discovery log JSON ('-' for stdin).")
ap.add_argument("--as-of", help="Analysis date YYYY-MM-DD (default: newest date in log).")
ap.add_argument("--output", choices=["json", "human"], default="json")
ap.add_argument("--sample", action="store_true",
help="Analyze a built-in sample log and exit 0.")
args = ap.parse_args()
if args.sample:
log = SAMPLE_LOG
elif args.input:
try:
log = json.load(sys.stdin if args.input == "-"
else open(args.input, encoding="utf-8"))
except (OSError, ValueError) as exc:
print(f"ERROR: cannot read log: {exc}", file=sys.stderr)
return 2
else:
ap.error("--input is required (or use --sample)")
interview_dates = [d for d in
(parse_date(i.get("date")) for i in log.get("interviews", [])) if d]
all_dates = interview_dates + [
d for d in (parse_date(t.get("date")) for t in log.get("assumption_tests", [])) if d]
as_of = parse_date(args.as_of) if args.as_of else (max(all_dates) if all_dates else None)
if as_of is None or sum(1 for d in interview_dates if d <= as_of) < 2:
print("REFUSED: fewer than 2 dated interviews on or before the analysis date — "
"there is no cadence to measure yet. Book the first two weekly touchpoints "
"(or widen --as-of), then re-run.", file=sys.stderr)
return 5
report = analyze(log, as_of)
if args.output == "json":
print(json.dumps(report, indent=2))
else:
print(f"Discovery health: {report['health_score']}/100 ({report['verdict']})")
print(f"Streak: {report['weekly_streak']} wk · coverage "
f"{report['week_coverage_pct']}% · linkage {report['outcome_linkage_pct']}% "
f"· tests resolved {report['assumption_tests_resolved']}")
for g in report["gaps"]:
print(f" gap: {g}")
print(f"Next: {report['next_loop_action']}")
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,162 @@
#!/usr/bin/env python3
"""ost_linter.py — structural linter for Opportunity Solution Trees.
The OST (Torres) is the structural artifact of modern product discovery:
outcome opportunities solutions assumption tests. Teams don't need prose
advice about trees they need their actual tree checked. This linter enforces the
structural rules deterministically, so an agent loop can use "ost_linter exits 0"
as the acceptance gate before a roadmap or experiment plan is allowed to cite the
tree.
Rules:
O1 exactly one outcome root, phrased measurably (contains a number, or a
`metric` field is present)
O2 opportunities are needs/pains/desires, not features flag statements that
start with build-verbs (add/build/implement/create/integrate/launch/ship)
O3 every opportunity marked "target": true has >= 2 solutions under
consideration (compare-and-contrast, never a single pet solution)
O4 every solution carries >= 1 assumption test
O5 no orphan solutions attached directly to the outcome a solution without an
opportunity is the feature-factory anti-pattern
Input JSON (see --sample):
{"outcome": {"statement": str, "metric": str?},
"opportunities": [{"statement": str, "target": bool?,
"children": [<nested opportunities>]?,
"solutions": [{"statement": str,
"tests": [{"assumption": str, "type": str}]}]}],
"solutions": [ ... ] # anything here is an O5 violation by definition
}
Exit codes: 0 clean (warnings allowed) · 2 violations found · 3 unreadable input.
Exception: `--sample` always exits 0 it is a smoke test, and the bundled tree
deliberately contains one O2 and one O4 violation so the report output is visible.
Stdlib only, deterministic.
"""
import argparse
import json
import re
import sys
BUILD_VERBS = re.compile(
r"^\s*(add|build|implement|create|integrate|launch|ship|develop|make)\b", re.I)
SAMPLE_TREE = {
"outcome": {"statement": "Increase week-4 retention from 22% to 30%",
"metric": "week-4 retention"},
"opportunities": [
{"statement": "New users can't tell whether setup worked",
"target": True,
"solutions": [
{"statement": "Post-setup verification checklist",
"tests": [{"assumption": "users abandon because they doubt setup succeeded",
"type": "interview"}]},
{"statement": "Live sample-data preview after connect",
"tests": [{"assumption": "a working preview reduces first-week drop-off",
"type": "prototype"}]},
]},
{"statement": "Add an onboarding wizard",
"solutions": [
{"statement": "Onboarding wizard v2", "tests": []},
]},
],
"solutions": [],
}
def walk_opportunities(nodes, path="opportunities"):
for idx, node in enumerate(nodes or []):
here = f"{path}[{idx}]"
yield here, node
yield from walk_opportunities(node.get("children"), here + ".children")
def lint(tree: dict):
violations, warnings = [], []
outcome = tree.get("outcome")
if not isinstance(outcome, dict) or not str(outcome.get("statement", "")).strip():
violations.append({"rule": "O1", "where": "outcome",
"problem": "missing outcome root — an OST hangs from exactly one outcome"})
else:
stmt = outcome.get("statement", "")
if not (re.search(r"\d", stmt) or str(outcome.get("metric", "")).strip()):
violations.append({"rule": "O1", "where": "outcome",
"problem": f"outcome is not measurable: '{stmt[:80]}'"
"state a metric and a target number"})
opp_count = 0
for where, opp in walk_opportunities(tree.get("opportunities")):
opp_count += 1
stmt = str(opp.get("statement", ""))
if BUILD_VERBS.match(stmt):
violations.append({"rule": "O2", "where": where,
"problem": f"opportunity phrased as a feature: '{stmt[:80]}'"
"rewrite as the customer need/pain/desire behind it"})
solutions = opp.get("solutions") or []
if opp.get("target") and len(solutions) < 2:
violations.append({"rule": "O3", "where": where,
"problem": f"targeted opportunity has {len(solutions)} solution(s) — "
"Torres: compare >= 2 candidate solutions, never one pet idea"})
for sidx, sol in enumerate(solutions):
if not (sol.get("tests") or []):
violations.append({"rule": "O4", "where": f"{where}.solutions[{sidx}]",
"problem": f"solution '{str(sol.get('statement', ''))[:60]}' has no "
"assumption test — untested solutions are opinions"})
orphans = tree.get("solutions") or []
for sidx, sol in enumerate(orphans):
violations.append({"rule": "O5", "where": f"solutions[{sidx}]",
"problem": f"orphan solution '{str(sol.get('statement', ''))[:60]}' attached to "
"no opportunity — the feature-factory anti-pattern"})
if opp_count == 0 and not violations:
warnings.append("tree has an outcome but zero opportunities — map the opportunity "
"space before jumping to solutions")
return violations, warnings, opp_count
def main() -> int:
ap = argparse.ArgumentParser(
description="Lint an Opportunity Solution Tree for structural discipline (Torres canon).")
ap.add_argument("--input", help="Path to the OST JSON ('-' for stdin).")
ap.add_argument("--output", choices=["json", "human"], default="json")
ap.add_argument("--sample", action="store_true",
help="Lint a built-in sample tree (contains one O2 and one O4 violation) and exit 0.")
args = ap.parse_args()
if args.sample:
tree = SAMPLE_TREE
elif args.input:
try:
tree = json.load(sys.stdin if args.input == "-"
else open(args.input, encoding="utf-8"))
except (OSError, ValueError) as exc:
print(f"ERROR: cannot read tree: {exc}", file=sys.stderr)
return 3
else:
ap.error("--input is required (or use --sample)")
violations, warnings, opp_count = lint(tree)
result = {
"opportunities": opp_count,
"violations": violations,
"warnings": warnings,
"verdict": "STRUCTURALLY-SOUND" if not violations else "NEEDS-REWORK",
}
if args.output == "json":
print(json.dumps(result, indent=2))
else:
print(f"Verdict: {result['verdict']} ({opp_count} opportunities, "
f"{len(violations)} violation(s))")
for v in violations:
print(f" [{v['rule']}] {v['where']}: {v['problem']}")
for w in warnings:
print(f" (warn) {w}")
if args.sample:
return 0
return 0 if not violations else 2
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,207 @@
#!/usr/bin/env python3
"""product_goal_router.py — deterministic lane classifier for the product-team domain.
Scores a product goal/inquiry against the 12 bundled sub-skill lanes plus the 4
standalone product-team plugins (same two-signal threshold discipline as the
research-ops / commercial / markdown-html orchestrators). Emits a routing decision
an agent can branch on mechanically.
Exit codes:
0 confident route emitted (route_to set)
2 ambiguous ask ONE clarifying question naming the top two lanes
3 no signal do not guess; ask the user to restate the goal
Stdlib only. Deterministic: same text in, same route out.
"""
import argparse
import json
import sys
SIGNALS = {
"PRIORITIZE": {
"skill": "product-manager-toolkit",
"path": "product-team/skills/product-manager-toolkit",
"keywords": ["prioritize", "rice", "backlog ranking", "feature ranking",
"prd", "product requirements", "interview synthesis", "wsjf",
"cost of delay"],
},
"STRATEGY": {
"skill": "product-strategist",
"path": "product-team/skills/product-strategist",
"keywords": ["okr", "objective", "strategy", "quarterly planning",
"north star", "vision", "alignment"],
},
"UX": {
"skill": "ux-researcher-designer",
"path": "product-team/skills/ux-researcher-designer",
"keywords": ["persona", "journey map", "usability", "user research",
"research synthesis", "interview guide"],
},
"DESIGN_SYSTEM": {
"skill": "ui-design-system",
"path": "product-team/skills/ui-design-system",
"keywords": ["design token", "component spec", "design system",
"wcag", "contrast", "typography scale"],
},
"COMPETITIVE": {
"skill": "competitive-teardown",
"path": "product-team/skills/competitive-teardown",
"keywords": ["competitor", "competitive", "teardown", "pricing matrix",
"market position", "feature comparison"],
},
"ANALYTICS": {
"skill": "product-analytics",
"path": "product-team/skills/product-analytics",
"keywords": ["retention", "cohort", "funnel", "kpi", "activation",
"churn", "aarrr", "north star metric", "tracking plan",
"event taxonomy"],
},
"EXPERIMENT": {
"skill": "experiment-designer",
"path": "product-team/skills/experiment-designer",
"keywords": ["a/b test", "ab test", "experiment", "sample size",
"hypothesis", "mde", "statistical power", "eval"],
},
"DISCOVERY": {
"skill": "product-discovery",
"path": "product-team/skills/product-discovery",
"keywords": ["discovery", "opportunity", "assumption", "opportunity solution tree",
"ost", "continuous discovery", "customer interview cadence",
"jtbd", "jobs to be done"],
},
"ROADMAP": {
"skill": "roadmap-communicator",
"path": "product-team/skills/roadmap-communicator",
"keywords": ["roadmap", "release notes", "changelog", "launch comms",
"now next later"],
},
"SPEC_TO_REPO": {
"skill": "spec-to-repo",
"path": "product-team/skills/spec-to-repo",
"keywords": ["spec to repo", "scaffold from spec", "build from spec",
"generate the repo", "turn this spec into"],
},
"LANDING": {
"skill": "landing-page-generator",
"path": "product-team/skills/landing-page-generator",
"keywords": ["landing page", "hero section", "waitlist page", "marketing page"],
},
"SAAS_SCAFFOLD": {
"skill": "saas-scaffolder",
"path": "product-team/skills/saas-scaffolder",
"keywords": ["saas boilerplate", "saas skeleton", "bootstrap a saas",
"auth and billing", "stripe integration scaffold"],
},
# Standalone product-team plugins (packaged separately, routable all the same)
"STORIES": {
"skill": "agile-product-owner",
"path": "product-team/agile-product-owner/skills/agile-product-owner",
"keywords": ["user story", "user stories", "acceptance criteria", "invest",
"epic breakdown", "sprint capacity", "story splitting"],
},
"HIG": {
"skill": "apple-hig-expert",
"path": "product-team/apple-hig-expert/skills/apple-hig-expert",
"keywords": ["hig", "human interface guidelines", "ios design", "liquid glass",
"tap target", "apple design"],
},
"CODE_TO_PRD": {
"skill": "code-to-prd",
"path": "product-team/code-to-prd/skills/code-to-prd",
"keywords": ["reverse engineer", "code to prd", "prd from code",
"document this codebase as a prd", "existing app into a prd"],
},
"SUMMARIZE": {
"skill": "research-summarizer",
"path": "product-team/research-summarizer/skills/research-summarizer",
"keywords": ["summarize this paper", "summarize research", "citation extraction",
"compare these papers", "article summary"],
},
}
SAMPLE_GOAL = ("we keep shipping features nobody uses — I want a weekly discovery "
"habit and an opportunity solution tree before the next roadmap review")
def score(text: str) -> dict:
low = text.lower()
scores, hits = {}, {}
for lane, spec in SIGNALS.items():
matched = [kw for kw in spec["keywords"] if kw in low]
scores[lane] = len(matched)
hits[lane] = matched
return {"scores": scores, "hits": hits}
def decide(scores: dict) -> dict:
ranked = sorted(scores.items(), key=lambda kv: (-kv[1], kv[0]))
(top_lane, top), (second_lane, second) = ranked[0], ranked[1]
if top == 0:
return {"decision": "NO_SIGNAL", "exit": 3}
if top >= 2 and (second == 0 or top >= 2 * second):
return {"decision": "ROUTE", "lane": top_lane, "exit": 0}
candidates = [top_lane] + ([second_lane] if second > 0 else [])
return {"decision": "ASK", "candidates": candidates, "exit": 2}
def main() -> int:
ap = argparse.ArgumentParser(
description="Deterministic lane router for product-team goals.")
src = ap.add_mutually_exclusive_group()
src.add_argument("--text", help="Goal / inquiry text to classify.")
src.add_argument("--input", help="Read goal text from a file ('-' for stdin).")
ap.add_argument("--output", choices=["json", "human"], default="json")
ap.add_argument("--sample", action="store_true",
help="Classify a built-in sample goal and exit.")
args = ap.parse_args()
if args.sample:
text = SAMPLE_GOAL
elif args.text:
text = args.text
elif args.input:
text = (sys.stdin.read() if args.input == "-"
else open(args.input, encoding="utf-8").read())
else:
ap.error("one of --text, --input, or --sample is required")
result = score(text)
verdict = decide(result["scores"])
out = {
"goal": text.strip()[:300],
"scores": {k: v for k, v in result["scores"].items() if v},
"decision": verdict["decision"],
}
if verdict["decision"] == "ROUTE":
lane = verdict["lane"]
out["route_to"] = SIGNALS[lane]["skill"]
out["skill_path"] = SIGNALS[lane]["path"]
out["matched_signals"] = result["hits"][lane]
elif verdict["decision"] == "ASK":
out["candidates"] = [
{"lane": lane, "skill": SIGNALS[lane]["skill"], "score": result["scores"][lane]}
for lane in verdict["candidates"]
]
out["instruction"] = ("Ask ONE clarifying question naming both candidate lanes, "
"with a recommended answer. Never guess silently.")
else:
out["instruction"] = ("No lane signal. Ask the user to restate the goal with the "
"deliverable named. Do not route on fuzz.")
if args.output == "json":
print(json.dumps(out, indent=2))
else:
print(f"Decision: {out['decision']}")
if "route_to" in out:
print(f"Route to: {out['route_to']} ({out['skill_path']})")
print(f"Signals: {', '.join(out['matched_signals'])}")
elif "candidates" in out:
names = " vs ".join(c["skill"] for c in out["candidates"])
print(f"Ambiguous: {names} — ask one clarifying question.")
else:
print("No signal — ask the user to restate the goal.")
return verdict["exit"]
if __name__ == "__main__":
sys.exit(main())

View file

@ -541,10 +541,25 @@ def create_sample_user_data():
]
def main():
import sys
import argparse
parser = argparse.ArgumentParser(
description="Generate a research-backed persona from bundled sample data.")
parser.add_argument("format", nargs="?", choices=["json", "human"], default="human",
help="Output format (default: human).")
parser.add_argument("--output", dest="output_flag", choices=["json", "human"],
help="Output format (flag form, overrides the positional).")
parser.add_argument("--seed", type=int, default=42,
help="RNG seed for deterministic persona names (default: 42).")
parser.add_argument("--sample", action="store_true",
help="Generate the bundled sample persona and exit 0 "
"(same as the default run; kept for harness smoke tests).")
args = parser.parse_args()
output = args.output_flag or args.format
random.seed(args.seed)
generator = PersonaGenerator()
# Create sample data
user_data = create_sample_user_data()
@ -561,7 +576,7 @@ def main():
persona = generator.generate_persona_from_data(user_data, interview_insights)
# Output
if len(sys.argv) > 1 and sys.argv[1] == 'json':
if output == 'json':
print(json.dumps(persona, indent=2))
else:
print(generator.format_persona_output(persona))

View file

@ -1,7 +1,7 @@
{
"name": "pm-skills",
"description": "9 project management skills: senior PM, scrum master, Jira expert, Confluence expert, Atlassian admin, template scaffolder, and Atlassian MCP-bundled (Remote SSE) integration. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.",
"version": "2.9.0",
"description": "9 project management skills: pm-skills fork-orchestrator with agentic delivery loop (deterministic 8-lane router, Jira MCP snapshot bridge to flow metrics/Monte Carlo forecasts, delegation-governance gate), senior PM, scrum master, Jira expert, Confluence expert, Atlassian admin, template scaffolder, meeting analyzer, team communications — plus bundled Atlassian Remote MCP (SSE) integration. Agent skill and plugin for Claude Code, Codex, Gemini CLI, Cursor, OpenClaw.",
"version": "2.11.1",
"author": {
"name": "Alireza Rezvani",
"url": "https://alirezarezvani.com"

View file

@ -1,22 +1,50 @@
# Project Management Skills - Claude Code Guidance
This guide covers the 9 production-ready project management skills, 12 Python automation tools, and bundled Atlassian Remote MCP integration (`.mcp.json` ships with the plugin — OAuth handled by Claude Code, no env vars required).
This guide covers the 9 production-ready project management skills, 15 Python automation tools, and bundled Atlassian Remote MCP integration (`.mcp.json` ships with the plugin — OAuth handled by Claude Code, no env vars required).
## PM Skills Overview
**Available Skills:**
1. **senior-pm/** - Portfolio health, risk analysis, resource planning (3 scripts)
2. **scrum-master/** - Sprint health, velocity forecasting, retrospectives (3 scripts)
3. **jira-expert/** - JQL building, workflow validation (2 scripts)
4. **confluence-expert/** - Space structure, content auditing (2 scripts)
5. **atlassian-admin/** - Permission auditing (1 script)
6. **atlassian-templates/** - Template scaffolding (1 script)
1. **pm-skills/** - Domain orchestrator (`context: fork`) + agentic delivery loop (3 scripts: goal router, Jira snapshot bridge, delivery loop gate)
2. **senior-pm/** - Portfolio health, risk analysis, resource planning (3 scripts)
3. **scrum-master/** - Sprint health, velocity forecasting, retrospectives (3 scripts)
4. **jira-expert/** - JQL building, workflow validation (2 scripts)
5. **confluence-expert/** - Space structure, content auditing (2 scripts)
6. **atlassian-admin/** - Permission auditing (1 script)
7. **atlassian-templates/** - Template scaffolding (1 script)
8. **meeting-analyzer/** - Meeting transcript behavioral analysis (prompt-driven; scripts are follow-up work)
9. **team-communications/** - 3P updates, newsletters, FAQs (reference-driven)
**Total Tools:** 12 Python automation tools
**Agent:** cs-project-manager (orchestrates all 6 skills)
**Slash Commands:** 3 (/sprint-health, /project-health, /retro)
**Total Tools:** 15 Python automation tools
**Agents:** 2 — cs-pm-orchestrator (routing + delivery loop) and cs-project-manager (legacy per-skill orchestration)
**Slash Commands:** 6 (/cs:pm, /cs:grill-pm, /cs:pm-loop, /sprint-health, /project-health, /retro)
**Key Feature:** Atlassian MCP Server integration for direct Jira/Confluence operations
## Orchestrator & Delivery Loop (pm-skills)
`skills/pm-skills/` is the domain's `context: fork` orchestrator and agent harness adapter:
```bash
# Route a PM goal deterministically (exit 0 route / 2 ask / 3 no signal)
python3 skills/pm-skills/scripts/pm_goal_router.py --text "our sprints feel off"
# Bridge a saved searchJiraIssuesUsingJql result into analyzable inputs
python3 skills/pm-skills/scripts/jira_snapshot_bridge.py --input snapshot.json --to flow --forecast 20
python3 skills/pm-skills/scripts/jira_snapshot_bridge.py --input snapshot.json --to sprint > sprint_data.json
python3 skills/scrum-master/scripts/velocity_analyzer.py sprint_data.json
# Gate agent-executed delivery loops (G1 human owner … G6 exhausted budget = escalation)
python3 skills/pm-skills/scripts/delivery_loop_gate.py --plan plan.json --mode plan # exit 2 = blocked
python3 skills/pm-skills/scripts/delivery_loop_gate.py --plan plan.json --mode close # exit 4 = refused
```
Multi-task goals compile through the repo-wide harness
(`engineering/agent-harness` with the `project-management.json` manifest). Hard rules:
agents contribute, humans own; forecasts are Monte Carlo ranges, never dates; exhausted
budgets escalate — never reported as success. The five reusable PM loops (sprint-flow,
health, retro-action, RAID-hygiene, comms) are documented in
`skills/pm-skills/references/pm_loop_playbook.md`.
## Atlassian MCP Integration
**Purpose:** Direct integration with Jira and Confluence via Model Context Protocol (MCP)
@ -179,8 +207,8 @@ python atlassian-templates/scripts/template_scaffolder.py meeting-notes
---
**Last Updated:** June 10, 2026
**Skills Deployed:** 9/9 PM skills production-ready
**Total Tools:** 12 Python automation tools
**Agent:** cs-project-manager | **Commands:** 3
**Last Updated:** July 3, 2026
**Skills Deployed:** 9/9 PM skills production-ready (pm-skills is now a fork-orchestrator + delivery loop)
**Total Tools:** 15 Python automation tools
**Agents:** cs-pm-orchestrator, cs-project-manager | **Commands:** 6
**Integration:** Atlassian Remote MCP Server (bundled via `.mcp.json`) for Jira/Confluence automation

View file

@ -0,0 +1,77 @@
---
name: cs-pm-orchestrator
description: Flow-first delivery lead. Routes project-management inquiries (sprint/velocity, portfolio health, Jira/JQL, Confluence, Atlassian admin, templates, meetings, comms) to the right sub-skill via the pm-skills orchestrator, and drives delivery goals through bounded agentic loops with machine-checkable gates. Forks context to keep heavy intake (Jira snapshots, retro logs, transcripts) out of the parent thread. Signature forcing question — "What single observable outcome means DONE, and which command proves it?"
tools: Read, Write, Edit, Glob, Grep, Bash, Skill
model: sonnet
---
# PM Orchestrator
You are a flow-first delivery lead. You measure before you forecast, derive health
instead of accepting self-reported green, and you never let a loop close on optimism.
Agents contribute; humans own — every task you plan names a human owner, and every
acceptance criterion is a command or a threshold.
## Voice
**"What single observable outcome means DONE, and which command proves it?"**
The trap you protect against: verification theater — status set to Done with no
evidence, forecasts stated as dates, watermelon projects reported green while aging WIP
rots.
## Your 8 lanes
| Lane | Skill | Signals |
|---|---|---|
| HEALTH | senior-pm | portfolio, risk EMV, capacity, exec report |
| SPRINT | scrum-master | velocity, retro, ceremonies, flow, forecast |
| JIRA | jira-expert | JQL, workflows, boards, automation |
| CONFLUENCE | confluence-expert | spaces, page trees, content audits |
| ADMIN | atlassian-admin | users, permissions, SSO |
| TEMPLATES | atlassian-templates | blueprints, storage-format scaffolds |
| MEETINGS | meeting-analyzer | transcripts, talk time, action items |
| COMMS | team-communications | 3P updates, newsletters, FAQs |
## Routing logic
1. Run `python3 project-management/skills/pm-skills/scripts/pm_goal_router.py --text "<goal>"`.
2. Exit 0 → load the routed skill's SKILL.md, follow its workflow in the forked context.
3. Exit 2 → ask ONE clarifying question naming the candidates, with a recommended answer.
4. Exit 3 → ask the user to restate the goal with the deliverable named. Never guess.
## How you communicate (Matt Pocock grill discipline)
One question per turn; always recommend; explore the workspace before asking (a saved
Jira snapshot or retro log resolves the lane silently); depth-first on multi-lane
inquiries; never silently chain. Digest ≤ 200 words: what was analyzed, top 3 findings
(canon-cited), top 3 next actions (named human owner), artifact path, one grill
challenge.
Hard outputs:
- Flow numbers come from `jira_snapshot_bridge.py` on real snapshot data — never from
memory or hand-typed estimates.
- Forecasts are Monte Carlo percentile ranges (p50/p70/p85/p95), never single dates.
- Loop plans pass `delivery_loop_gate.py --mode plan` (exit 0) before execution and
`--mode close` (exit 0) before you report done.
## Anti-patterns
- ❌ Route to two skills at once, or run all 8 "to be thorough"
- ❌ Accept "make our delivery better" — grill until the outcome and its proof command are
named
- ❌ Transition Jira issues to Done, change permissions, or delete anything inside a loop
without the named human approver
- ❌ Report an exhausted attempt/iteration budget as success
## When to escalate
- What-to-build questions → `product-team` (cs-product-orchestrator)
- Internal-ops process mapping → `business-operations`
- Generic loop mechanics / other domains → `engineering/agent-harness` harness-runner
- Regulatory/compliance delivery → `ra-qm-team`
## Available commands
`/cs:pm <inquiry>` (router) · `/cs:grill-pm <plan>` (grill first) · `/cs:pm-loop <goal>`
(delivery loop) · plus the domain's `/sprint-health`, `/project-health`, `/retro`.

View file

@ -0,0 +1,57 @@
---
description: Matt Pocock-style interrogation of a delivery plan against the PM canon (Kanban Guide 2025, Vacanti, DORA 2025, EBM, Klein, GitLab async-first). One forcing question per turn with a recommended answer; refuses to invoke any sub-skill or start a loop until the lane-defining decisions are locked. Use before running /cs:pm or /cs:pm-loop on a fuzzy plan.
argument-hint: "<delivery plan, goal, or status quo to interrogate>"
---
# /cs:grill-pm — grill a delivery plan before running it
Interrogate this plan — do not execute anything yet:
**$ARGUMENTS**
Five rules (preserved from Matt Pocock, MIT): one question per turn · always give a
recommended answer · explore the workspace before asking · walk the decision tree
depth-first · track answered questions and their dependencies.
## Decision tree
- **Branch 1 — Outcome**: "What single observable outcome means DONE, and which command
proves it? Recommended: a named artifact + a command that exits 0 against it. Canon:
agent-harness verifier's law."
- **Branch 2 — Measurement**: "Are you measuring flow before forecasting? Recommended:
run `jira_snapshot_bridge.py --to flow` first — WIP, throughput, cycle time, age.
Canon: Kanban Guide (May 2025) four mandatory measures."
- **Branch 3 — Forecast honesty**: "Is any date in this plan a single-point promise?
Recommended: replace with Monte Carlo p50/p85 ranges; refuse forecasts on < 10
completed items. Canon: Vacanti, *When Will It Be Done?*"
- **Branch 4 — Ownership**: "For every task an agent will execute: who is the human owner
and who reviews? Recommended: name both now; `delivery_loop_gate.py` will refuse the
plan otherwise. Canon: Linear agents model; Atlassian Rovo audit discipline."
- **Branch 5 — Risk**: "Have you run a pre-mortem on this plan? Recommended: 30 minutes,
'it's six months later and this failed — why?'; convert top clusters to owned risks.
Canon: Klein, HBR 2007."
- **Branch 6 — Budgets**: "What are the retry and iteration caps, and who reviews
escalations? Recommended: 3 attempts/task, 12 iterations/goal, a named human. Canon:
loop-library terminal states."
Per-turn output format:
```
Q[i]/[total]: [precise question]
Recommended: [answer + canon-cited rationale]
(Confirm, or override?)
```
## Stop conditions
- All branches resolved → invoke `/cs:pm` (question) or `/cs:pm-loop` (goal) with the
locked decisions inlined.
- User says "stop grilling, just run it" → run with unresolved branches flagged in the
digest.
- Abandoned → save the partial grill to `pm-grill-{timestamp}.md`.
## Distinct from
- `engineering/grill-me` — generic plan interrogation. This grills against the PM canon.
- `/cs:pm` — routes; this refuses to route until decisions are locked.

View file

@ -0,0 +1,50 @@
---
description: Drive a project-delivery goal through a bounded agentic loop — Jira MCP snapshot → flow/sprint analytics bridge → routed sub-skill execution → machine-verified gates → close refused until everything is verified or human-waived. The PM-domain adapter over engineering/agent-harness.
argument-hint: "<delivery goal, e.g. 'get sprint 14 to a verified close with health >= 70'>"
---
# /cs:pm-loop — run a delivery goal to a verified close
Goal:
**$ARGUMENTS**
## Sequence (gates are blocking — never skip forward)
1. **Intake gate** — the goal must name an observable outcome and its proof. If vague,
run the `/cs:grill-pm` branches first (one question per turn). Do not loop on fuzz.
2. **Observe** — pull fresh data: `mcp__atlassian__getAccessibleAtlassianResources` (get
cloudId) → `mcp__atlassian__searchJiraIssuesUsingJql` → save `snapshot.json`, then:
```bash
python3 project-management/skills/pm-skills/scripts/jira_snapshot_bridge.py --input snapshot.json --to flow
python3 project-management/skills/pm-skills/scripts/jira_snapshot_bridge.py --input snapshot.json --to sprint > sprint_data.json
```
3. **Plan** — write the task plan (owners, executors, reviewers, machine-checkable
acceptance per task; shape via `delivery_loop_gate.py --sample`), then gate it:
```bash
python3 project-management/skills/pm-skills/scripts/delivery_loop_gate.py --plan plan.json --mode plan
```
Exit 2 → fix the listed G1G4 violations before executing. For multi-task goals,
compile through the repo harness instead (`goal_compiler.py` with the
`project-management.json` manifest) and drive it with `loop_controller.py`.
4. **Execute** — one task at a time: route with `pm_goal_router.py`, run the routed
sub-skill's own tools, record real exit codes and evidence. Retry means a changed
approach; max 3 attempts per task.
5. **Verify** — the task's acceptance command must exit 0; sub-skill gates apply
(scrum-master's ≥3-sprints rule, atlassian-admin's VERIFY steps). Never adjudicate
your own verification; never edit a gate to make it pass.
6. **Close**
```bash
python3 project-management/skills/pm-skills/scripts/delivery_loop_gate.py --plan plan.json --mode close
```
Exit 4 → close refused: finish, escalate, or get a human waiver (with reason). Exit 0
→ report the handoff: tasks, statuses, evidence, waivers, and the flow-metrics
before/after.
## Rules
- Terminal states: success · clean no-op · blocked · approval-required · exhausted ·
stagnated. Exhausted budgets escalate to the named human — never reported as success.
- Jira writes are auditable: no `transitionJiraIssue` to Done without verify evidence;
admin/destructive actions are approval-required, full stop.
- Max 12 loop iterations per goal; 3 attempts per task.

View file

@ -0,0 +1,45 @@
---
description: Top-level project-management router. Classifies a PM inquiry across 8 lanes (sprint/flow, portfolio health, Jira, Confluence, admin, templates, meetings, comms) with a deterministic script and forks context to the right sub-skill via the pm-skills orchestrator, returning a ≤200-word digest with a named owner and one grill challenge.
argument-hint: "<PM inquiry: sprint health, project status, JQL, permissions, retro, comms, etc.>"
---
# /cs:pm — Project Management router
Route this inquiry through the `pm-skills` orchestrator:
**$ARGUMENTS**
## Routing (deterministic — run the script, don't eyeball)
```bash
python3 project-management/skills/pm-skills/scripts/pm_goal_router.py --text "$ARGUMENTS" --output json
```
- Exit 0 → load `skill_path`/SKILL.md and follow that skill's own workflow in a fork.
- Exit 2 → ask ONE clarifying question naming the listed candidates, recommended answer
first.
- Exit 3 → ask the user to restate the goal with the deliverable named. Never guess.
- Explore the workspace first — a saved Jira snapshot, retro log, or transcript resolves
the lane silently. Never silently chain a second sub-skill.
## Output (≤200-word digest)
- What was analyzed (with the data source — snapshot file, not memory)
- Top 3 findings, each anchored to a canon citation
- Top 3 next actions with a named human owner
- Artifact path
- One grill challenge (e.g. "Your health report is self-reported RAG — where's the
derived diff that catches watermelons?")
## Hard rules
- Flow numbers come from `jira_snapshot_bridge.py` on real snapshot data.
- Forecasts are Monte Carlo percentile ranges, never single dates.
- Live Jira/Confluence ops use only the tools in
`project-management/references/atlassian-mcp-tools.md` — never invent tool names.
- Goals (not questions) go to `/cs:pm-loop` instead.
## Distinct from
- `product-team` — what to build. This domain is how to deliver it.
- `/cs:harness` — the generic loop engine; `/cs:pm-loop` is its PM-domain adapter.

View file

@ -1,50 +1,173 @@
---
name: "pm-skills"
description: "Router/index for the 8 project-management skills bundled in this plugin (senior PM quant toolkit, scrum master, Jira/JQL, Confluence, Atlassian admin, Atlassian templates, meeting analyzer, team communications). Use when a PM request doesn't obviously match one skill and you need to pick the right one (e.g., 'our sprints feel off', 'audit our Jira permissions'). Bundles an Atlassian Remote MCP config (.mcp.json) for live Jira/Confluence access."
version: 2.9.0
description: "Use when coordinating project-delivery work across the 8 project-management sub-skills — sprint/velocity analytics, portfolio health, Jira/JQL, Confluence, Atlassian admin, templates, meeting analysis, team comms. Triggers on 'our sprints feel off', 'project health report', 'audit our Jira permissions', 'when will it be done', 'run the delivery loop'. Forks context to route to one sub-skill via a deterministic signal router and returns a digest; can also drive a full goal→plan→execute→verify→close delivery loop through the repo-wide agent-harness with Jira MCP data bridged into the domain's analytics tools. Distinct from product-team (what to build vs how to deliver it), business-operations (internal ops), and engineering/agent-harness (the generic loop engine this orchestrator plugs into)."
context: fork
version: 2.11.1
author: Alireza Rezvani
license: MIT
tags:
- project-management
- jira
- confluence
- atlassian
- scrum
- agile
agents:
- claude-code
- codex-cli
- openclaw
tags: [project-management, orchestrator, jira, confluence, atlassian, scrum, agile, flow-metrics, agent-harness]
compatible_tools: [claude-code, codex-cli, cursor, antigravity, opencode, gemini-cli]
---
# Project Management Skills — Router
# Project Management — Domain Orchestrator & Delivery Loop
This plugin bundles **8 PM skills** (this router is the 9th folder under `project-management/skills/`). Each skill is self-contained. The bundled `.mcp.json` wires the Atlassian Remote MCP (`https://mcp.atlassian.com/v1/sse`, OAuth handled by Claude Code).
This orchestrator does two jobs. **Routing:** fork context, classify a PM inquiry with
`scripts/pm_goal_router.py`, run exactly one of the 8 sub-skills, return a digest.
**Looping:** turn a delivery goal into a bounded agentic loop — pull live Jira data via the
bundled Atlassian MCP, bridge it into the domain's deterministic analytics tools, verify
every step with machine-run gates, and refuse to close until everything is verified or a
human waives it. The bundled `.mcp.json` wires the Atlassian Remote MCP
(`https://mcp.atlassian.com/v1/sse`, OAuth handled by Claude Code).
## Routing table
## When to invoke
Match the request, then load `project-management/skills/<skill>/SKILL.md`. If multiple rows match, ask one clarifying question first.
| Symptom | Sub-skill |
|---|---|
| "Project/portfolio health, risk EMV, capacity" | `senior-pm` |
| "Sprint velocity, retro follow-through, ceremony health, when-will-it-be-done" | `scrum-master` |
| "JQL, Jira workflows, boards, automation" | `jira-expert` |
| "Confluence spaces, page trees, content audits" | `confluence-expert` |
| "Users, groups, permissions, SSO" | `atlassian-admin` |
| "Reusable Jira/Confluence templates" | `atlassian-templates` |
| "Meeting transcripts, talk time, action items" | `meeting-analyzer` |
| "Status updates, 3P updates, stakeholder comms" | `team-communications` |
| Request signals | Skill | Path |
|---|---|---|
| Project health, risk EMV, three-point estimates | senior-pm | `skills/senior-pm/` |
| Sprint velocity, retro analysis, ceremony health | scrum-master | `skills/scrum-master/` |
| JQL queries, Jira workflows, boards | jira-expert | `skills/jira-expert/` |
| Confluence spaces, page structure, content audits | confluence-expert | `skills/confluence-expert/` |
| User/permission/scheme administration | atlassian-admin | `skills/atlassian-admin/` |
| Reusable Confluence/Jira templates | atlassian-templates | `skills/atlassian-templates/` |
| Meeting transcripts, talk-time, action items | meeting-analyzer | `skills/meeting-analyzer/` |
| Status updates, 3P updates, stakeholder comms | team-communications | `skills/team-communications/` |
## Routing logic (deterministic)
## Quick start
Run the router — do not eyeball the table when a script can decide:
```bash
# Example: route a sprint-health request
cat project-management/skills/scrum-master/SKILL.md
ls project-management/skills/scrum-master/scripts/
python3 scripts/pm_goal_router.py --text "<the goal>" --output json
```
## Rules
Exit 0 → `route_to` names the sub-skill: load its SKILL.md and follow its workflow.
Exit 2 → ask ONE clarifying question naming the listed candidates, with a recommended
answer. Exit 3 → no signal: ask the user to restate the goal with the deliverable named.
Never guess silently; never silently chain a second sub-skill — digest first, confirm, then
chain.
- Live Jira/Confluence operations go through the Atlassian Remote MCP (camelCase tool names such as `createJiraIssue`, `searchJiraIssuesUsingJql`, `createConfluencePage` — canonical list in `project-management/references/atlassian-mcp-tools.md`). Admin operations are NOT covered by the MCP — use admin.atlassian.com or the REST API per atlassian-admin.
- Route to exactly one skill, then follow that skill's workflow. This router ships no tools of its own.
## The delivery loop (agentic)
For goals (not questions) — "get sprint 14 to a verified close", "produce a portfolio
health report from live Jira", "make our flow metrics visible weekly" — run the
loop-library contract (Observe → Choose → Act → Verify → Record → Repeat-or-stop):
1. **Observe** — pull fresh state: `mcp__atlassian__searchJiraIssuesUsingJql` (get
`cloudId` via `getAccessibleAtlassianResources` first), save the result JSON, then
bridge it:
```bash
python3 scripts/jira_snapshot_bridge.py --input snapshot.json --to flow # WIP, throughput, cycle time p50/85/95, work-item age, SLE, aging alerts
python3 scripts/jira_snapshot_bridge.py --input snapshot.json --to sprint > s.json # scrum-master schema
python3 ../scrum-master/scripts/velocity_analyzer.py s.json # velocity + volatility + forecast
```
Add `--forecast N` for a seeded Monte Carlo "when will N items be done" answer
(refuses on < 10 completed items thin history forecasts are lies).
2. **Choose** — route the next task with `pm_goal_router.py`; one task at a time.
3. **Act** — execute with the routed sub-skill's own tools per its SKILL.md.
4. **Verify** — gate the plan and every close with:
```bash
python3 scripts/delivery_loop_gate.py --plan plan.json --mode plan # exit 2 = blocked
python3 scripts/delivery_loop_gate.py --plan plan.json --mode close # exit 4 = close refused
```
Plus each sub-skill's own gates (scrum-master's ≥ 3-sprints rule, atlassian-admin's
VERIFY steps). Never adjudicate your own verification.
5. **Record / Repeat-or-stop** — for multi-task goals, run the state through the repo-wide
harness (it enforces attempt caps, iteration budgets, and evidence logging):
```bash
python3 engineering/agent-harness/skills/agent-harness/scripts/goal_compiler.py \
--goal "<goal>" --manifest engineering/agent-harness/skills/agent-harness/assets/harnesses/project-management.json \
--out .agent-harness/plan.json
python3 engineering/agent-harness/skills/agent-harness/scripts/loop_controller.py init|next|record|verify|close ...
```
Terminal states: success, clean no-op, blocked, approval-required, exhausted,
stagnated. An exhausted budget is an escalation — never a success report.
## Hard rules (agentic delegation governance)
1. **Agents are contributors, never owners** (Linear model): every loop task carries a
named human owner; agent-executed tasks also carry a named human reviewer.
`delivery_loop_gate.py` enforces this (G1/G2).
2. **Acceptance must be machine-checkable** — a command, or a criterion with a threshold.
"Looks good" is not a gate (G3).
3. **Every Jira/Confluence write is auditable and reversible-first** (Rovo discipline):
never `transitionJiraIssue` to Done without verify evidence; destructive/irreversible
actions (deletes, permission changes, org-wide admin) are approval-required terminal
states, not loop steps.
4. **Never modify a gate you are judged by** — same locked-evaluator invariant as
autoresearch-agent.
5. **Forecasts are ranges with confidence, never dates** — Monte Carlo percentiles
(p50/p70/p85/p95), per Vacanti. Single-date promises are the anti-pattern.
6. **Max 3 attempts per task, 12 loop iterations per goal** — then escalate to the named
human with the evidence log.
## Forcing-question library (grill-with-docs pattern)
One per turn, recommended answer, canon citation. Never run a sub-skill or start a loop
until the lane-defining decision is locked:
- **SPRINT lane**: "Do you want to *measure* flow (cycle time, WIP, throughput, age) or
*forecast* delivery? Recommended: measure first — a forecast off unmeasured flow is
noise. Canon: Kanban Guide (May 2025) four mandatory flow measures; Vacanti,
*Actionable Agile Metrics*."
- **HEALTH lane**: "Is your project status self-reported RAG or derived from signals?
Recommended: derive it (schedule variance, aging WIP, scope churn) and diff against the
self-report — that diff finds watermelon projects. Canon: Kanban Guide 2025;
DORA 2025 (AI amplifies, doesn't fix, weak signals)."
- **JIRA lane**: "Is this configuration change deployable to a test project first?
Recommended: always stage in a test project; jira-expert's workflow validator must exit
0 before production. Canon: jira-expert validation workflow."
- **ADMIN lane**: "Is this action reversible, and who approves it? Recommended: name the
approver before touching permissions — admin actions are approval-required terminal
states in any loop. Canon: atlassian-admin VERIFY discipline; loop-library stop states."
- **LOOP intake**: "What single observable outcome means DONE, and which command proves
it? Recommended: a named artifact + a command that exits 0 against it. Canon:
agent-harness verifier's law; Anthropic, *Building Effective Agents* (evaluator needs
clear criteria)."
- **MEETINGS/COMMS lanes**: "Could this meeting be an async written update? Recommended:
status-broadcast meetings convert to async 3P updates; decision meetings keep sync.
Canon: GitLab async-first handbook."
## Assumptions
1. The user has (or is preparing analysis for someone with) delivery authority.
2. Jira/Confluence access goes through the bundled MCP; capabilities NOT in
`project-management/references/atlassian-mcp-tools.md` (project/sprint/board/space
creation, admin config) are done in the web UI — never invent tool names.
3. Inputs may be partial — every tool ships `--sample` so the shape is visible first.
## Non-goals
- Not a replacement for the sub-skills — the orchestrator routes and loops; the
sub-skills do the work.
- Not the generic loop engine — that is `engineering/agent-harness`; this orchestrator is
the PM-domain adapter (data bridge + governance gate + lane router).
- Does not decide *what* to build — that's `product-team`.
## Output artifacts
| Mode | Artifact |
|---|---|
| Route | Sub-skill's own artifact + ≤ 200-word digest with one canon-cited challenge |
| Flow report | `flow_metrics.json` (bridge output) with SLE conformance + aging alerts |
| Delivery loop | `.agent-harness/plan.json` + `state.json` + gate verdicts + close handoff |
## Anti-patterns (do not)
- ❌ Run all 8 sub-skills "to be thorough" — route to one, digest, chain on confirmation
- ❌ Report sprint health or forecasts from hand-typed numbers when a Jira snapshot is one
MCP call away — bridge real data
- ❌ Close a loop with unverified tasks, or report an exhausted budget as success
- ❌ Let an agent be the assignee of record — humans own, agents contribute
- ❌ Auto-transition Jira issues or touch permissions inside a loop without the named
approver
## References
- [references/flow_forecasting_canon.md](references/flow_forecasting_canon.md) — Kanban
Guide 2025, Vacanti Monte Carlo, DORA 2025, EBM, SPACE
- [references/agentic_delivery_governance.md](references/agentic_delivery_governance.md) —
Linear/Rovo delegation models, Anthropic agent patterns, audit discipline
- [references/pm_loop_playbook.md](references/pm_loop_playbook.md) — the five reusable PM
loops (sprint, health, retro-action, RAID-hygiene, comms) mapped to the loop contract
- Canonical MCP tool list: `project-management/references/atlassian-mcp-tools.md`
- Loop engine: `engineering/agent-harness` · Loop vocabulary: `loop-library`

View file

@ -0,0 +1,53 @@
{
"mode": "flow",
"as_of": "2026-06-10",
"counts": {
"total": 14,
"done": 11,
"wip": 2
},
"cycle_time_days": {
"p50": 9,
"p85": 14,
"p95": 16,
"basis": "created\u2192resolved (approximation; Jira exports rarely carry an in-progress timestamp)"
},
"throughput": {
"done_per_week": 2.08,
"weeks_observed": 5.3
},
"sle": {
"days": 14,
"conformance_pct": 90.9
},
"work_item_age": [
{
"key": "PHX-112",
"summary": "Realtime presence indicators",
"age_days": 15
},
{
"key": "PHX-113",
"summary": "Usage analytics dashboard",
"age_days": 2
}
],
"aging_wip_alerts": [
{
"key": "PHX-112",
"summary": "Realtime presence indicators",
"age_days": 15
}
],
"warnings": [],
"forecast": {
"items": 20,
"method": "Monte Carlo over historical weekly throughput (10k trials, seeded)",
"weeks": {
"p50": 9,
"p70": 10,
"p85": 10,
"p95": 10
}
}
}

View file

@ -0,0 +1,19 @@
{
"_comment": "Saved output shape of mcp__atlassian__searchJiraIssuesUsingJql — feed to scripts/jira_snapshot_bridge.py. Expected flow output pinned in expected_flow_metrics.json.",
"issues": [
{"key": "PHX-101", "fields": {"summary": "User authentication service", "status": {"name": "Done"}, "created": "2026-05-04T09:00:00.000+0000", "resolutiondate": "2026-05-08T16:00:00.000+0000", "customfield_10016": 5, "sprint": {"name": "Sprint 11"}, "assignee": {"displayName": "A. Rivera"}, "priority": {"name": "High"}}},
{"key": "PHX-102", "fields": {"summary": "Dashboard layout", "status": {"name": "Done"}, "created": "2026-05-04T09:00:00.000+0000", "resolutiondate": "2026-05-12T11:00:00.000+0000", "customfield_10016": 3, "sprint": {"name": "Sprint 11"}, "assignee": {"displayName": "B. Okafor"}, "priority": {"name": "Medium"}}},
{"key": "PHX-103", "fields": {"summary": "Password reset flow", "status": {"name": "Done"}, "created": "2026-05-05T09:00:00.000+0000", "resolutiondate": "2026-05-13T15:00:00.000+0000", "customfield_10016": 2, "sprint": {"name": "Sprint 11"}, "assignee": {"displayName": "A. Rivera"}, "priority": {"name": "Medium"}}},
{"key": "PHX-104", "fields": {"summary": "Billing webhooks", "status": {"name": "Done"}, "created": "2026-05-11T09:00:00.000+0000", "resolutiondate": "2026-05-20T10:00:00.000+0000", "customfield_10016": 8, "sprint": {"name": "Sprint 12"}, "assignee": {"displayName": "C. Duarte"}, "priority": {"name": "High"}}},
{"key": "PHX-105", "fields": {"summary": "Email notifications", "status": {"name": "Done"}, "created": "2026-05-12T09:00:00.000+0000", "resolutiondate": "2026-05-19T17:00:00.000+0000", "customfield_10016": 3, "sprint": {"name": "Sprint 12"}, "assignee": {"displayName": "B. Okafor"}, "priority": {"name": "Medium"}}},
{"key": "PHX-106", "fields": {"summary": "Audit log export", "status": {"name": "Done"}, "created": "2026-05-13T09:00:00.000+0000", "resolutiondate": "2026-05-22T12:00:00.000+0000", "customfield_10016": 5, "sprint": {"name": "Sprint 12"}, "assignee": {"displayName": "D. Weiss"}, "priority": {"name": "Low"}}},
{"key": "PHX-107", "fields": {"summary": "Mobile responsive nav", "status": {"name": "Done"}, "created": "2026-05-18T09:00:00.000+0000", "resolutiondate": "2026-05-27T09:30:00.000+0000", "customfield_10016": 5, "sprint": {"name": "Sprint 13"}, "assignee": {"displayName": "A. Rivera"}, "priority": {"name": "High"}}},
{"key": "PHX-108", "fields": {"summary": "Rate limiting middleware", "status": {"name": "Done"}, "created": "2026-05-19T09:00:00.000+0000", "resolutiondate": "2026-05-29T14:00:00.000+0000", "customfield_10016": 3, "sprint": {"name": "Sprint 13"}, "assignee": {"displayName": "C. Duarte"}, "priority": {"name": "Medium"}}},
{"key": "PHX-109", "fields": {"summary": "Data export CSV", "status": {"name": "Done"}, "created": "2026-05-20T09:00:00.000+0000", "resolutiondate": "2026-06-03T16:00:00.000+0000", "customfield_10016": 2, "sprint": {"name": "Sprint 13"}, "assignee": {"displayName": "D. Weiss"}, "priority": {"name": "Low"}}},
{"key": "PHX-110", "fields": {"summary": "SSO SAML integration", "status": {"name": "Done"}, "created": "2026-05-25T09:00:00.000+0000", "resolutiondate": "2026-06-10T10:00:00.000+0000", "customfield_10016": 8, "sprint": {"name": "Sprint 14"}, "assignee": {"displayName": "C. Duarte"}, "priority": {"name": "High"}}},
{"key": "PHX-111", "fields": {"summary": "Search relevance tuning", "status": {"name": "Done"}, "created": "2026-06-01T09:00:00.000+0000", "resolutiondate": "2026-06-09T11:00:00.000+0000", "customfield_10016": 3, "sprint": {"name": "Sprint 14"}, "assignee": {"displayName": "B. Okafor"}, "priority": {"name": "Medium"}}},
{"key": "PHX-112", "fields": {"summary": "Realtime presence indicators", "status": {"name": "In Progress"}, "created": "2026-05-26T09:00:00.000+0000", "customfield_10016": 8, "sprint": {"name": "Sprint 14"}, "assignee": {"displayName": "A. Rivera"}, "priority": {"name": "High"}}},
{"key": "PHX-113", "fields": {"summary": "Usage analytics dashboard", "status": {"name": "In Progress"}, "created": "2026-06-08T09:00:00.000+0000", "customfield_10016": 5, "sprint": {"name": "Sprint 14"}, "assignee": {"displayName": "D. Weiss"}, "priority": {"name": "Medium"}}},
{"key": "PHX-114", "fields": {"summary": "Legacy importer deprecation", "status": {"name": "To Do"}, "created": "2026-06-09T09:00:00.000+0000", "customfield_10016": 3, "sprint": {"name": "Sprint 14"}, "assignee": {"displayName": "B. Okafor"}, "priority": {"name": "Low"}}}
]
}

View file

@ -0,0 +1,71 @@
# Agentic Delivery Governance
The accountability layer behind `delivery_loop_gate.py`. When agents execute delivery
work, the failure mode is not bad output — it is **unowned output**: work no human is
accountable for, verified by nobody, closed by the optimism of the thing that did it.
The 20252026 vendors converged on the same governance shape; this file encodes it.
## The delegation model (why G1/G2 exist)
- **Linear's shipped design**: issues can be delegated to agents, but the **human stays
primary assignee; the agent is a contributor**. Delegation transfers execution, never
accountability. → Gate G1: every task names a human owner.
- **Atlassian Rovo (GA 2026)**: agents are assignable and @mentionable inside Jira, but
"every action remains logged and auditable", and multi-step plans pause for human
oversight at decision points. → Gate G2: agent-executed tasks name a human reviewer;
hard rule: no un-reviewed `transitionJiraIssue` to Done, no permission changes inside a
loop.
## Verification discipline (why G3/G4 exist)
- **Machine-checkable definition of done**: the reliable loop shape is plan → act →
verify with a deterministic verifier → reflect. A criterion without a command or a
threshold ("looks good", "improved") cannot gate anything. → G3: acceptance = a `cmd`,
or a criterion containing a measurable number.
- **Never trust self-report**: the agent that did the work does not adjudicate the work.
Evidence precedes status: `done` without recorded evidence is rejected (G4) — the same
invariant as agent-harness's "a verify pass without --evidence is exit 6" and
autoresearch-agent's locked evaluator ("never modify the gate you are judged by").
## Terminal-state honesty (why G5/G6 exist)
From the loop-library contract: loops end in a **named terminal state** — success, clean
no-op, blocked, approval-required, exhausted, stagnated. Two corollaries the gate
enforces:
- Close is refused while any task is neither done nor waived (G5); waivers are human
decisions with recorded rationale.
- An exhausted budget (attempts or iterations) is an **escalation**, never a success
report (G6). Budgets are first-class: max attempts per task, max loop iterations, and
the stop fires mechanically, not when the agent feels finished.
## Design principles for PM loops
1. **Workflows before agents** (Anthropic): most PM automation is a routed workflow
(classify → run tool → report). Reach for the autonomous loop only when the task needs
fresh feedback each cycle — flow snapshots, retro follow-through, RAID hygiene.
2. **Evaluator-optimizer needs clear criteria**: the generator/critic loop pays off
exactly when acceptance is machine-checkable — which is why the gate forces G3 before
any loop starts.
3. **Context from structured interfaces, not prompt-stuffing**: the agent's Jira context
comes through the MCP (Teamwork-Graph-style structured access), snapshotted to a file
the loop can re-read — every iteration is executable by a fresh session.
4. **Agent-readiness is a data-hygiene property**: agents amplify the Jira they are given
(DORA 2025's amplifier finding). Field completeness, acceptance criteria in
descriptions, and honest statuses are prerequisites, not nice-to-haves.
## Sources
1. Linear, "Agents in Linear" / Linear for Agents — https://linear.app/agents and
https://linear.app/docs/agents-in-linear
2. Atlassian Rovo — https://www.atlassian.com/software/rovo and Rovo agents docs
https://support.atlassian.com/rovo/docs/agents/
3. Anthropic, "Building Effective Agents" (orchestrator-workers, evaluator-optimizer,
simplicity-first) — https://www.anthropic.com/research/building-effective-agents
4. Forward Future, Loop Library (terminal-state taxonomy; "never report an error or
exhausted budget as success") — vendored at `loop-library/SKILL.md`
5. DORA, *State of DevOps 2025* (AI as amplifier) — https://dora.dev/dora-report-2025/
6. This repo: `engineering/agent-harness` references/verification_discipline.md
(reward-hacking failure mode; locked evaluators)
7. SiliconANGLE, "Atlassian opens Teamwork Graph, pushes Rovo agentic execution"
(Team '26 coverage) — https://siliconangle.com/2026/05/06/

View file

@ -0,0 +1,76 @@
# Flow Metrics & Probabilistic Forecasting Canon
The measurement layer behind `jira_snapshot_bridge.py --to flow`. The 20242026 delivery
canon moved from velocity/story-point folklore to **flow measurement + probabilistic
forecasting + outcome measures**. This file anchors every number the bridge emits.
## The four mandatory flow measures (Kanban Guide, May 2025)
The Kanban Guide mandates exactly four measures — teams that track anything must track
these:
| Measure | Definition | Bridge field |
|---|---|---|
| **WIP** | Work items started but not finished | `counts.wip` |
| **Throughput** | Items finished per unit of time | `throughput.done_per_week` |
| **Cycle time** | Elapsed time started → finished | `cycle_time_days.p50/p85/p95` |
| **Work item age** | Elapsed time for *unfinished* started items | `work_item_age` |
Plus a **Service Level Expectation (SLE)**: "we finish items of this type within N days,
X% of the time." The bridge defaults the SLE to the p85 cycle time and reports
conformance. **Work item age is the leading indicator** — an in-progress item older than
the p85 SLE is the earliest visible slip signal (`aging_wip_alerts`); cycle time only
tells you after the fact.
Caveat the bridge prints itself: Jira exports rarely carry an in-progress timestamp, so
cycle time is approximated created→resolved. When your workflow logs a real start
transition, prefer it.
## Percentiles, not averages
Cycle-time distributions are right-skewed; the mean lies. Report p50/p85/p95
(Vacanti). Commit externally at p85, plan internally at p50, treat p95 as the tail-risk
budget.
## Monte Carlo forecasting (replaces story-point velocity)
"When will it be done?" is answered by sampling historical throughput, not by dividing
backlog points by velocity: sample the weekly-throughput history 10k times, read the
p50/p70/p85/p95 week counts (`--forecast N`). Rules the bridge enforces:
- **Refuses on < 10 completed items across < 4 distinct weeks** — thin history produces
confident nonsense.
- **Seeded RNG** — same data in, same forecast out (audit-reproducible).
- **Ranges with confidence, never a date** — a single-date promise is the anti-pattern.
## Outcome measures above flow
Flow says whether delivery is smooth, not whether it is *valuable*:
- **DORA 2025**: replaced low/high/elite clusters with seven team archetypes over eight
measures; core 2025 finding — AI *amplifies* an org's existing strengths and
dysfunctions (individual output up, org delivery flat without enabling capabilities).
- **EBM (Scrum.org)**: four Key Value Areas — Current Value, Unrealized Value,
Time-to-Market, Ability to Innovate. Most orgs measure only T2M; empty CV/UV areas mean
you are measuring motion, not value.
- **SPACE**: any metrics portfolio must span ≥ 3 of Satisfaction/Performance/Activity/
Communication/Efficiency — activity-only portfolios (PR counts) are the documented
anti-pattern.
- **Derived health beats self-reported RAG**: diff derived signals (aging WIP, scope
churn, schedule variance) against the self-reported status to find "watermelon"
projects (green outside, red inside) — the pattern behind senior-pm's dashboard.
## Sources
1. The Kanban Guide (May 2025) — https://kanbanguides.org/the-kanban-guide/
2. Daniel Vacanti, *Actionable Agile Metrics for Predictability* and *When Will It Be
Done?* (ActionableAgile Press)
3. Scrum.org, "4 Key Flow Metrics and How to Use Them" —
https://www.scrum.org/resources/blog/4-key-flow-metrics-and-how-use-them-scrums-events
4. Scrum.org, "Monte Carlo Forecasting in Scrum" —
https://www.scrum.org/resources/blog/monte-carlo-forecasting-scrum
5. DORA, *Accelerate State of DevOps Report 2025* — https://dora.dev/dora-report-2025/
6. Scrum.org, *The Evidence-Based Management Guide*
https://www.scrum.org/resources/evidence-based-management
7. Forsgren, Storey et al., "The SPACE of Developer Productivity", ACM Queue —
https://queue.acm.org/detail.cfm?id=3454124

View file

@ -0,0 +1,92 @@
# The PM Loop Playbook — five reusable delivery loops
Concrete instantiations of the loop contract (Observe → Choose → Act → Verify → Record →
Repeat-or-stop) for day-to-day PM work. Each loop names its trigger, its machine gate,
and its terminal states — a loop without a named stop is just an unbounded retry.
## Loop 1 — Sprint flow loop (weekly)
- **Observe**: `searchJiraIssuesUsingJql` → save snapshot →
`jira_snapshot_bridge.py --to flow`.
- **Choose**: highest-leverage signal first — aging-WIP alerts beat cycle-time trends
(age is the leading indicator; cycle time is a lagging one).
- **Act**: unblock/swarm/split the flagged item with the team; scrum-master skill for
ceremony-level fixes.
- **Verify**: next week's bridge run — the flagged item left `aging_wip_alerts`, SLE
conformance did not drop.
- **Stop states**: success (no alerts 2 weeks running) · stagnated (same item flagged 3
weeks → escalate to the delivery lead by name) · approval-required (fix needs scope
change).
## Loop 2 — Health-report loop (per reporting period)
- **Observe**: bridge `--to sprint``velocity_analyzer.py` +
`sprint_health_scorer.py`; senior-pm's `project_health_dashboard.py` for the portfolio.
- **Choose**: diff derived health against the self-reported RAG — investigate the largest
divergence first (watermelon detection).
- **Act**: senior-pm workflows (risk register, capacity rebalance).
- **Verify**: divergence shrinks next period; every red flag has a named owner + dated
action.
- **Stop states**: success · blocked (data quality too poor to derive — fix Jira hygiene
first, see agent-readiness note in agentic_delivery_governance.md).
## Loop 3 — Retro action loop (per sprint)
The retro is not the loop — the **action-item completion rate** is. Scrum-master's
retrospective_analyzer computes it (fixture: 46.7%).
- **Observe**: `retrospective_analyzer.py` on the retro log.
- **Choose**: oldest open action item with a named owner.
- **Act**: drive it to done or explicitly kill it (a cancelled action with a reason beats
a zombie).
- **Verify**: completion rate trend up across 3 sprints.
- **Stop states**: success (≥ 70% completion) · stagnated (rate flat 3 sprints → the retro
format is the problem; change the ritual, per Klein's pre-mortem alternative).
## Loop 4 — RAID hygiene loop (biweekly)
- **Observe**: risk register / RAID log staleness — entries missing owner, next-review
date, or mitigation; issues open > 30 days.
- **Choose**: stalest critical-severity entry.
- **Act**: senior-pm's `risk_matrix_analyzer.py` re-score; pre-mortem session for new
workstreams (prospective hindsight measurably improves risk identification — Klein).
- **Verify**: zero critical entries without owner+mitigation; staleness p85 < review
cadence.
- **Stop states**: success · clean no-op (nothing stale — record and exit; do not invent
work).
## Loop 5 — Comms cadence loop (weekly)
- **Observe**: which status-broadcast meetings ran this week; which stakeholder updates
shipped.
- **Choose**: convert the largest status-broadcast meeting to an async 3P update
(GitLab's handbook-first model: written 3-question standups take 35 min vs 1530 sync;
GitLab reports ~37% meeting-hour reduction).
- **Act**: team-communications skill (3P format); meeting-analyzer on the transcripts of
the meetings that remain.
- **Verify**: sync:async ratio trending down; no stakeholder escalation citing "I didn't
know".
- **Stop states**: success · approval-required (a stakeholder insists on sync — their
call, record it).
## Rules that hold across all five
1. One bounded, reversible change per iteration (autoresearch's "ONE change" rule).
2. Fresh state before every consequential action — re-pull the snapshot, don't act on
last week's.
3. Separate the optimizing signal from the acceptance gate — if you optimize SLE
conformance, verify with throughput + age too (anti-overfit, loop-library).
4. Every escalation names a human and attaches the evidence log.
## Sources
1. Forward Future, Loop Library — the Observe→…→Repeat-or-stop contract and terminal-state
taxonomy (vendored at `loop-library/SKILL.md`)
2. The Kanban Guide (May 2025) — https://kanbanguides.org/the-kanban-guide/
3. Gary Klein, "Performing a Project Premortem", HBR 2007 —
https://hbr.org/2007/09/performing-a-project-premortem
4. GitLab Handbook, asynchronous work —
https://handbook.gitlab.com/handbook/company/culture/all-remote/asynchronous/
5. Sumeet Moghe, *The Async-First Playbook* (2023)
6. This repo: `engineering/autoresearch-agent` (one-change-per-iteration, locked
evaluator), `engineering/agent-harness` (state machine, budgets)

View file

@ -0,0 +1,152 @@
#!/usr/bin/env python3
"""delivery_loop_gate.py — governance gate for agent-executed delivery loops.
Encodes the 20252026 agentic-delegation canon as a machine-checkable gate
(Linear's agents model: the human stays accountable for delegated work; Atlassian
Rovo: every agent action stays auditable; loop-library: exhausted budgets are never
reported as success). Run it on a delivery-loop plan before executing (--mode plan)
and again before closing (--mode close). Pairs with the repo-wide
engineering/agent-harness loop_controller.py, which enforces the run-time state
machine; this gate enforces the PM-specific accountability rules the controller
does not know about.
Plan JSON shape (see --sample):
{"goal": str,
"budgets": {"max_attempts_per_task": int, "max_loop_iterations": int},
"iteration": int,
"tasks": [{"id","title","owner","executor":"human|agent","reviewer",
"acceptance": {"cmd": str} | {"criterion": str},
"status":"todo|in_progress|done|blocked|waived",
"evidence": str, "attempts": int, "waive_reason": str}]}
Rules enforced:
G1 every task has a named human owner (agents are contributors, never owners)
G2 agent-executed tasks name a human reviewer distinct from nobody
G3 acceptance is machine-checkable: a cmd, or a criterion containing a measurable
threshold (a digit) "looks good" is not a gate
G4 done requires non-empty evidence; waived requires a waive_reason
G5 close is refused while any task is neither done nor waived
G6 close is refused when budgets are exhausted with work remaining that is an
escalation, not a success
Exit codes: 0 pass · 2 plan violations · 3 unreadable input · 4 close refused.
Stdlib only, deterministic.
"""
import argparse
import json
import re
import sys
SAMPLE_PLAN = {
"goal": "Get sprint 14 to a verified close with a health score >= 70",
"budgets": {"max_attempts_per_task": 3, "max_loop_iterations": 12},
"iteration": 4,
"tasks": [
{"id": "T1", "title": "Pull sprint snapshot via searchJiraIssuesUsingJql",
"owner": "Sarah Chen", "executor": "agent", "reviewer": "Sarah Chen",
"acceptance": {"cmd": "python3 scripts/jira_snapshot_bridge.py --input snapshot.json --to sprint"},
"status": "done", "evidence": "sprint_data.json written, 4 sprints", "attempts": 1},
{"id": "T2", "title": "Score sprint health",
"owner": "Sarah Chen", "executor": "agent", "reviewer": "Mike Rodriguez",
"acceptance": {"criterion": "sprint_health_scorer.py composite >= 70"},
"status": "in_progress", "evidence": "", "attempts": 1},
],
}
def check_plan(plan):
violations, warnings = [], []
tasks = plan.get("tasks", [])
if not tasks:
violations.append({"rule": "G1", "task": "-", "problem": "plan has no tasks"})
for t in tasks:
tid = t.get("id", "?")
if not str(t.get("owner", "")).strip():
violations.append({"rule": "G1", "task": tid,
"problem": "no named human owner (Linear rule: agents are contributors, never owners)"})
if t.get("executor") == "agent" and not str(t.get("reviewer", "")).strip():
violations.append({"rule": "G2", "task": tid,
"problem": "agent-executed task has no named human reviewer"})
acc = t.get("acceptance") or {}
cmd = str(acc.get("cmd", "")).strip()
criterion = str(acc.get("criterion", "")).strip()
if not cmd and not (criterion and re.search(r"\d", criterion)):
violations.append({"rule": "G3", "task": tid,
"problem": "acceptance is not machine-checkable "
"(need a cmd, or a criterion with a measurable threshold)"})
status = t.get("status", "todo")
if status == "done" and not str(t.get("evidence", "")).strip():
violations.append({"rule": "G4", "task": tid,
"problem": "done without evidence — never record a verify pass you did not observe"})
if status == "waived" and not str(t.get("waive_reason", "")).strip():
violations.append({"rule": "G4", "task": tid,
"problem": "waived without a waive_reason (waivers are human decisions with rationale)"})
max_attempts = plan.get("budgets", {}).get("max_attempts_per_task")
if max_attempts and t.get("attempts", 0) >= max_attempts and status not in ("done", "waived", "blocked"):
warnings.append({"rule": "G6", "task": tid,
"note": f"attempts exhausted ({t.get('attempts')}/{max_attempts}) — escalate, do not retry"})
return violations, warnings
def check_close(plan):
refusals = []
for t in plan.get("tasks", []):
if t.get("status") not in ("done", "waived"):
refusals.append({"rule": "G5", "task": t.get("id", "?"),
"problem": f"status is '{t.get('status', 'todo')}' — close refused while tasks are unverified and unwaived"})
budgets = plan.get("budgets", {})
max_iter = budgets.get("max_loop_iterations")
if max_iter and plan.get("iteration", 0) > max_iter and refusals:
refusals.append({"rule": "G6", "task": "-",
"problem": f"iteration {plan['iteration']} > cap {max_iter} with open tasks — "
"this is an ESCALATION, never a success report"})
return refusals
def main() -> int:
ap = argparse.ArgumentParser(
description="Accountability gate for agent-executed PM delivery loops.")
ap.add_argument("--plan", help="Path to the loop plan JSON ('-' for stdin).")
ap.add_argument("--mode", choices=["plan", "close"], default="plan")
ap.add_argument("--output", choices=["json", "human"], default="json")
ap.add_argument("--sample", action="store_true",
help="Print a valid sample plan and exit 0.")
args = ap.parse_args()
if args.sample:
print(json.dumps(SAMPLE_PLAN, indent=2))
return 0
if not args.plan:
ap.error("--plan is required (or use --sample to see the expected shape)")
try:
plan = json.load(sys.stdin if args.plan == "-" else open(args.plan, encoding="utf-8"))
except (OSError, ValueError) as exc:
print(f"ERROR: cannot read plan: {exc}", file=sys.stderr)
return 3
violations, warnings = check_plan(plan)
result = {"mode": args.mode, "goal": plan.get("goal", ""),
"violations": violations, "warnings": warnings}
exit_code = 0
if args.mode == "close":
refusals = check_close(plan)
result["close_refusals"] = refusals
result["verdict"] = "CLOSE-REFUSED" if (refusals or violations) else "CLOSE-OK"
exit_code = 4 if (refusals or violations) else 0
else:
result["verdict"] = "PLAN-BLOCKED" if violations else "PLAN-OK"
exit_code = 2 if violations else 0
if args.output == "json":
print(json.dumps(result, indent=2))
else:
print(f"Verdict: {result['verdict']}")
for v in violations + result.get("close_refusals", []):
print(f" [{v['rule']}] {v['task']}: {v['problem']}")
for w in warnings:
print(f" (warn {w['rule']}) {w['task']}: {w['note']}")
return exit_code
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,282 @@
#!/usr/bin/env python3
"""jira_snapshot_bridge.py — turn a Jira MCP issue export into analyzable inputs.
Closes the domain's biggest wiring gap: `mcp__atlassian__searchJiraIssuesUsingJql`
returns issue JSON, but the domain's deterministic analytics tools
(scrum-master/velocity_analyzer.py, sprint_health_scorer.py) expect their own
sprint-record schema, and nothing computed flow metrics at all. This bridge accepts
a saved Jira search result (raw MCP shape with `issues[].fields`, or a flat list of
simplified issue dicts) and emits:
--to flow the four mandatory Kanban flow measures (Kanban Guide, May 2025):
WIP, throughput, cycle time (p50/p85/p95), work-item age plus SLE
conformance and aging-WIP alerts, and an optional Monte Carlo
"when will N items be done" forecast (Vacanti-style, seeded, refuses
on < 10 completed items).
--to sprint scrum-master sprint-record JSON (pipe into velocity_analyzer.py /
sprint_health_scorer.py). Refuses with exit 5 on < 3 sprints,
mirroring velocity_analyzer's own minimum.
Cycle time here is createdresolved (Jira's export rarely carries an in-progress
timestamp); the output labels this approximation explicitly.
Exit codes: 0 ok · 2 unreadable/invalid input · 5 insufficient data for the
requested mode. Stdlib only; deterministic (as-of defaults to the newest timestamp
in the data, never the wall clock; the forecast RNG is seeded).
"""
import argparse
import json
import math
import random
import sys
from datetime import date, timedelta
DONE_STATUSES = {"done", "closed", "resolved", "released"}
NOT_STARTED_STATUSES = {"to do", "todo", "open", "backlog", "new", "created"}
POINT_FIELD_CANDIDATES = ["story_points", "storyPoints", "customfield_10016", "points"]
SAMPLE_SNAPSHOT = {
"issues": [
{"key": "PROJ-1", "fields": {"summary": "Login flow", "status": {"name": "Done"},
"created": "2026-05-04T09:00:00.000+0000",
"resolutiondate": "2026-05-08T16:00:00.000+0000",
"customfield_10016": 5, "sprint": {"name": "Sprint 12"},
"assignee": {"displayName": "A. Rivera"}}},
{"key": "PROJ-2", "fields": {"summary": "Rate limiting", "status": {"name": "In Progress"},
"created": "2026-05-18T09:00:00.000+0000", "customfield_10016": 3,
"sprint": {"name": "Sprint 13"}, "assignee": {"displayName": "B. Okafor"}}},
]
}
def parse_date(value):
if not value:
return None
text = str(value)[:10]
try:
return date.fromisoformat(text)
except ValueError:
return None
def name_of(value):
if isinstance(value, dict):
return value.get("name") or value.get("displayName") or ""
return str(value) if value else ""
def sprint_of(value):
if isinstance(value, list) and value:
value = value[-1]
return name_of(value)
def normalize(raw, points_field):
if isinstance(raw, dict) and "issues" in raw:
records = raw["issues"]
elif isinstance(raw, list):
records = raw
else:
raise ValueError("expected {'issues': [...]} or a JSON list of issues")
issues = []
for rec in records:
if not isinstance(rec, dict):
continue
f = rec.get("fields", rec)
points = None
for cand in ([points_field] if points_field else []) + POINT_FIELD_CANDIDATES:
if cand and f.get(cand) is not None:
points = f.get(cand)
break
status = name_of(f.get("status")).lower()
issues.append({
"key": rec.get("key") or f.get("key") or "?",
"summary": f.get("summary", ""),
"status": status,
"done": status in DONE_STATUSES,
"started": status not in NOT_STARTED_STATUSES,
"created": parse_date(f.get("created")),
"resolved": parse_date(f.get("resolutiondate") or f.get("resolved")),
"points": float(points) if points is not None else None,
"sprint": sprint_of(f.get("sprint") or f.get("customfield_10020")),
"assignee": name_of(f.get("assignee")),
"priority": name_of(f.get("priority")).lower(),
})
return [i for i in issues if i["created"]]
def percentile(sorted_values, pct):
if not sorted_values:
return None
rank = max(1, math.ceil(pct / 100 * len(sorted_values)))
return sorted_values[rank - 1]
def flow_report(issues, as_of, sle_days, forecast_items, seed):
done = [i for i in issues if i["done"] and i["resolved"]]
wip = [i for i in issues if not i["done"] and i["started"]]
cycles = sorted(max((i["resolved"] - i["created"]).days, 0) for i in done)
p50, p85, p95 = (percentile(cycles, p) for p in (50, 85, 95))
span_days = max((as_of - min(i["created"] for i in issues)).days, 7) if issues else 7
weeks = max(span_days / 7.0, 1.0)
# Weekly throughput over the FULL observed span (first resolution → as_of),
# zero-filled: dead weeks are real observations and must be sampleable, or the
# Monte Carlo forecast biases optimistic (Vacanti).
weekly_counts = []
if done:
first_resolved = min(i["resolved"] for i in done)
observed_weeks = max(((as_of - first_resolved).days // 7) + 1, 1)
weekly_counts = [0] * observed_weeks
for i in done:
idx = min((i["resolved"] - first_resolved).days // 7, observed_weeks - 1)
weekly_counts[idx] += 1
sle = sle_days if sle_days else p85
conformance = (
round(100 * sum(1 for c in cycles if c <= sle) / len(cycles), 1)
if cycles and sle is not None else None
)
aging = sorted(
({"key": i["key"], "summary": i["summary"][:60],
"age_days": (as_of - i["created"]).days} for i in wip),
key=lambda a: -a["age_days"],
)
aging_alerts = [a for a in aging if p85 is not None and a["age_days"] > p85]
report = {
"mode": "flow",
"as_of": as_of.isoformat(),
"counts": {"total": len(issues), "done": len(done), "wip": len(wip)},
"cycle_time_days": {"p50": p50, "p85": p85, "p95": p95,
"basis": "created→resolved (approximation; Jira exports rarely carry an in-progress timestamp)"},
"throughput": {"done_per_week": round(len(done) / weeks, 2),
"weeks_observed": round(weeks, 1)},
"sle": {"days": sle, "conformance_pct": conformance},
"work_item_age": aging[:10],
"aging_wip_alerts": aging_alerts,
"warnings": [],
}
if len(done) < 10:
report["warnings"].append(
f"only {len(done)} completed items — flow percentiles are low-confidence below 10")
if forecast_items:
if len(weekly_counts) < 4 or len(done) < 10:
report["warnings"].append(
"forecast refused: need >= 10 completed items across >= 4 observed calendar "
"weeks (Vacanti: throughput sampling needs real history; zero-throughput "
"weeks count as observations)")
else:
rng = random.Random(seed)
samples = weekly_counts
trials = []
for _ in range(10000):
remaining, wk = forecast_items, 0
while remaining > 0 and wk < 520:
remaining -= rng.choice(samples)
wk += 1
trials.append(wk)
trials.sort()
report["forecast"] = {
"items": forecast_items,
"method": "Monte Carlo over historical weekly throughput (10k trials, seeded)",
"weeks": {f"p{p}": percentile(trials, p) for p in (50, 70, 85, 95)},
}
return report
def sprint_export(issues):
by_sprint = {}
for i in issues:
if i["sprint"]:
by_sprint.setdefault(i["sprint"], []).append(i)
if len(by_sprint) < 3:
print(f"REFUSED: {len(by_sprint)} sprint(s) in snapshot — velocity analysis needs >= 3 "
"(same gate as velocity_analyzer.py). Widen the JQL date range.", file=sys.stderr)
return None
ordered = sorted(by_sprint.items(),
key=lambda kv: min(i["created"] for i in kv[1]))
sprints = []
for n, (name, items) in enumerate(ordered, 1):
planned = sum(i["points"] or 0 for i in items)
completed = sum(i["points"] or 0 for i in items if i["done"])
starts = min(i["created"] for i in items)
ends = max((i["resolved"] or i["created"]) for i in items)
sprints.append({
"sprint_number": n, "sprint_name": name,
"start_date": starts.isoformat(), "end_date": ends.isoformat(),
"planned_points": round(planned, 1), "completed_points": round(completed, 1),
"added_points": 0, "removed_points": 0,
"carry_over_points": round(planned - completed, 1) if planned > completed else 0,
"team_capacity": 0, "working_days": 10,
"team_size": len({i["assignee"] for i in items if i["assignee"]}),
"stories": [{
"id": i["key"], "title": i["summary"][:80], "points": i["points"] or 0,
"status": "completed" if i["done"] else ("in_progress" if i["started"] else "not_started"),
"assigned_to": i["assignee"], "created_date": i["created"].isoformat(),
**({"completed_date": i["resolved"].isoformat()} if i["resolved"] else {}),
"blocked_days": 0, "priority": i["priority"] or "medium",
} for i in items],
"blockers": [],
})
return {
"team_info": {"name": "bridged-from-jira", "size": 0,
"scrum_master": "", "product_owner": ""},
"sprints": sprints,
"_note": ("Bridged from a Jira snapshot: added/removed/carry-over/capacity and "
"ceremonies are not derivable from issue exports — fill them in or accept "
"the conservative defaults before scoring sprint health."),
}
def main() -> int:
ap = argparse.ArgumentParser(
description="Bridge a saved Jira MCP search result into flow metrics or "
"scrum-master sprint-record JSON.")
ap.add_argument("--input", help="Path to the saved Jira search JSON ('-' for stdin).")
ap.add_argument("--to", choices=["flow", "sprint"], default="flow")
ap.add_argument("--points-field", help="Custom field id carrying story points "
"(e.g. customfield_10016).")
ap.add_argument("--as-of", help="Analysis date YYYY-MM-DD (default: newest date in data).")
ap.add_argument("--sle-days", type=int, help="Service Level Expectation in days "
"(default: the p85 cycle time).")
ap.add_argument("--forecast", type=int, metavar="N",
help="Monte Carlo forecast: weeks to finish N more items.")
ap.add_argument("--seed", type=int, default=42)
ap.add_argument("--sample", action="store_true",
help="Print a sample input snapshot and exit 0.")
args = ap.parse_args()
if args.sample:
print(json.dumps(SAMPLE_SNAPSHOT, indent=2))
return 0
if not args.input:
ap.error("--input is required (or use --sample to see the expected shape)")
try:
raw = json.load(sys.stdin if args.input == "-" else open(args.input, encoding="utf-8"))
issues = normalize(raw, args.points_field)
except (OSError, ValueError) as exc:
print(f"ERROR: cannot read snapshot: {exc}", file=sys.stderr)
return 2
if not issues:
print("ERROR: no issues with a created date found in the snapshot.", file=sys.stderr)
return 2
if args.to == "sprint":
result = sprint_export(issues)
if result is None:
return 5
print(json.dumps(result, indent=2))
return 0
as_of = parse_date(args.as_of) if args.as_of else max(
(i["resolved"] or i["created"]) for i in issues)
if as_of is None:
print(f"ERROR: --as-of '{args.as_of}' is not a valid YYYY-MM-DD date.",
file=sys.stderr)
return 2
report = flow_report(issues, as_of, args.sle_days, args.forecast, args.seed)
print(json.dumps(report, indent=2))
return 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -0,0 +1,179 @@
#!/usr/bin/env python3
"""pm_goal_router.py — deterministic lane classifier for the project-management domain.
Scores a PM goal/inquiry against the 8 sub-skill lanes using keyword signals
(same two-signal threshold discipline as the research-ops / commercial / markdown-html
orchestrators). Emits a routing decision an agent can branch on mechanically.
Exit codes:
0 confident route emitted (route_to set)
2 ambiguous ask ONE clarifying question naming the top two lanes
3 no signal do not guess; ask the user to restate the goal
Stdlib only. Deterministic: same text in, same route out.
"""
import argparse
import json
import sys
SIGNALS = {
"HEALTH": {
"skill": "senior-pm",
"path": "project-management/skills/senior-pm",
"keywords": [
"project health", "portfolio", "risk register", "risk analysis", "emv",
"monte carlo", "executive report", "status report", "milestone", "budget",
"resource capacity", "capacity plan", "raid", "stakeholder satisfaction",
"program", "watermelon",
],
},
"SPRINT": {
"skill": "scrum-master",
"path": "project-management/skills/scrum-master",
"keywords": [
"sprint", "velocity", "retro", "retrospective", "ceremony", "standup",
"scrum", "burndown", "forecast", "story points", "action item",
"team health", "flow metrics", "cycle time", "throughput", "wip",
],
},
"JIRA": {
"skill": "jira-expert",
"path": "project-management/skills/jira-expert",
"keywords": [
"jql", "jira workflow", "jira board", "automation rule", "issue type",
"jira filter", "jira report", "jira config", "workflow transition",
"kanban board", "epic link",
],
},
"CONFLUENCE": {
"skill": "confluence-expert",
"path": "project-management/skills/confluence-expert",
"keywords": [
"confluence", "space", "knowledge base", "page tree", "documentation audit",
"wiki", "page hierarchy", "content governance", "macro",
],
},
"ADMIN": {
"skill": "atlassian-admin",
"path": "project-management/skills/atlassian-admin",
"keywords": [
"permission", "sso", "saml", "provisioning", "deactivate user", "group",
"admin", "security policy", "access control", "marketplace app", "audit log",
],
},
"TEMPLATES": {
"skill": "atlassian-templates",
"path": "project-management/skills/atlassian-templates",
"keywords": [
"template", "blueprint", "scaffold", "standardized page", "reusable layout",
"storage format",
],
},
"MEETINGS": {
"skill": "meeting-analyzer",
"path": "project-management/skills/meeting-analyzer",
"keywords": [
"meeting", "transcript", "talk time", "speaking", "filler words",
"interruption", "facilitation", "1:1", "one-on-one",
],
},
"COMMS": {
"skill": "team-communications",
"path": "project-management/skills/team-communications",
"keywords": [
"status update", "3p", "newsletter", "faq", "announcement",
"stakeholder update", "incident report", "comms", "broadcast",
],
},
}
SAMPLE_GOAL = (
"our sprints feel off — velocity keeps swinging and the retro action items "
"never get done"
)
def score(text: str) -> dict:
low = text.lower()
scores = {}
hits = {}
for lane, spec in SIGNALS.items():
matched = [kw for kw in spec["keywords"] if kw in low]
scores[lane] = len(matched)
hits[lane] = matched
return {"scores": scores, "hits": hits}
def decide(scores: dict) -> dict:
ranked = sorted(scores.items(), key=lambda kv: (-kv[1], kv[0]))
(top_lane, top), (second_lane, second) = ranked[0], ranked[1]
if top == 0:
return {"decision": "NO_SIGNAL", "exit": 3}
if top >= 2 and (second == 0 or top >= 2 * second):
return {"decision": "ROUTE", "lane": top_lane, "exit": 0}
candidates = [top_lane] + ([second_lane] if second > 0 else [])
return {"decision": "ASK", "candidates": candidates, "exit": 2}
def main() -> int:
ap = argparse.ArgumentParser(
description="Deterministic lane router for project-management goals."
)
src = ap.add_mutually_exclusive_group()
src.add_argument("--text", help="Goal / inquiry text to classify.")
src.add_argument("--input", help="Read goal text from a file ('-' for stdin).")
ap.add_argument("--output", choices=["json", "human"], default="json")
ap.add_argument("--sample", action="store_true",
help="Classify a built-in sample goal and exit.")
args = ap.parse_args()
if args.sample:
text = SAMPLE_GOAL
elif args.text:
text = args.text
elif args.input:
text = (sys.stdin.read() if args.input == "-"
else open(args.input, encoding="utf-8").read())
else:
ap.error("one of --text, --input, or --sample is required")
result = score(text)
verdict = decide(result["scores"])
out = {
"goal": text.strip()[:300],
"scores": {k: v for k, v in result["scores"].items() if v},
"decision": verdict["decision"],
}
if verdict["decision"] == "ROUTE":
lane = verdict["lane"]
out["route_to"] = SIGNALS[lane]["skill"]
out["skill_path"] = SIGNALS[lane]["path"]
out["matched_signals"] = result["hits"][lane]
elif verdict["decision"] == "ASK":
out["candidates"] = [
{"lane": lane, "skill": SIGNALS[lane]["skill"], "score": result["scores"][lane]}
for lane in verdict["candidates"]
]
out["instruction"] = ("Ask ONE clarifying question naming both candidate lanes, "
"with a recommended answer. Never guess silently.")
else:
out["instruction"] = ("No lane signal. Ask the user to restate the goal with the "
"deliverable named. Do not route on fuzz.")
if args.output == "json":
print(json.dumps(out, indent=2))
else:
print(f"Decision: {out['decision']}")
if "route_to" in out:
print(f"Route to: {out['route_to']} ({out['skill_path']})")
print(f"Signals: {', '.join(out['matched_signals'])}")
elif "candidates" in out:
names = " vs ".join(c["skill"] for c in out["candidates"])
print(f"Ambiguous: {names} — ask one clarifying question.")
else:
print("No signal — ask the user to restate the goal.")
return verdict["exit"]
if __name__ == "__main__":
sys.exit(main())

View file

@ -222,6 +222,42 @@ def check_domain_table(root: Path) -> list:
return problems
# README shields.io badges of the form ![Skills](.../Skills-355-brightgreen...).
# Validated separately from prose CLAIM_PATTERNS: badges are a fixed
# `<Label>-<count>-<color>` shape, so a per-label regex is exact and cannot
# over-match surrounding prose. Covers the counters that ship as badges.
BADGE_PATTERNS = {
"skills": re.compile(r"Skills-(\d+)-"),
"agents": re.compile(r"Agents-(\d+)-"),
"commands": re.compile(r"Commands-(\d+)-"),
}
def check_readme_badges(root: Path, derived: dict) -> list:
"""Return mismatch strings between README shield badges and derived counts."""
readme = root / "README.md"
if not readme.is_file():
return []
text = readme.read_text(encoding="utf-8")
out = []
for key, pattern in BADGE_PATTERNS.items():
match = pattern.search(text)
if not match:
# Fail loudly rather than skip: a renamed/removed badge would
# otherwise silently drop out of the gate (mirrors run_check's
# "no recognizable counter claims found" precedent).
out.append(
f"README badge: '{key}' badge not found "
f"(expected a '{key.capitalize()}-<n>-' shield)"
)
elif int(match.group(1)) != derived[key]:
out.append(
f"README badge: '{key}' badge claims {match.group(1)}, "
f"derived {key}={derived[key]}"
)
return out
def run_check(root: Path, derived: dict) -> int:
sources = []
@ -260,6 +296,7 @@ def run_check(root: Path, derived: dict) -> int:
)
mismatches.extend(check_domain_table(root))
mismatches.extend(check_readme_badges(root, derived))
if mismatches:
print("COUNTER CHECK FAILED — headline claims disagree with derived values:")