diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 2308aa8e..12e00855 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -8,7 +8,7 @@ "homepage": "https://github.com/alirezarezvani/claude-skills", "repository": "https://github.com/alirezarezvani/claude-skills", "metadata": { - "description": "370 production-ready skills across 19 domains (engineering, engineering-core, marketing, product, c-level, c-level-agents, compliance-os, project management, RA/QM, business growth, finance, productivity, marketing top-level, research, research-ops, business-operations, commercial, markdown-html, loop-library, plus standards). 672 Python tools, 809 reference guides, 104 agents (cs-* + personas), 120 slash commands across 92 marketplace plugins. v2.11.2 vendors engineering/skillopt-sleep — a verbatim copy of microsoft/SkillOpt's stdlib-only skillopt_sleep engine + Claude Code plugin surface, giving a local agent a nightly gated self-improvement cycle (read-only session harvest -> mine -> offline replay -> held-out-gated CLAUDE.md/SKILL.md edits -> staged for explicit /skillopt-sleep adopt). productivity/fable-goal (unreleased, post-v2.11.1) converts a rambling description of a desired outcome into one polished /goal prompt for a fresh autonomous session. v2.11.1 turns product-team and project-management into agent-harness domains: fork-orchestrators with deterministic goal routers, a Jira MCP snapshot bridge (Kanban flow metrics + Monte Carlo forecasting), a delegation-governance loop gate, a continuous-discovery cadence tracker, and an Opportunity Solution Tree linter, with /cs:pm and /cs:product command families. v2.10.3 completes the markdown-html domain with md-slides — slide-deck converter (arrow-key / Space / PgDn / Home/End / P keyboard navigation + presenter mode with split-view clock + speaker notes + next-slide preview + URL-hash deep linking like #3 for direct slide jumps + @media print page-per-slide for browser-native PDF export). Reuses md-document's markdown parser; vanilla JS only (no framework runtime); Prism.js opt-in via --syntax. Joins md-review (v2.10.2 code-review converter), md-document (v2.10.1 long-form converter), and the v2.10.0 foundation (orchestrator + design-system). Compatible with Claude Code, Codex CLI, Gemini CLI, Cursor, OpenClaw, Hermes Agent, Mistral Vibe, and 5 more coding agents.", + "description": "371 production-ready skills across 19 domains (engineering, engineering-core, marketing, product, c-level, c-level-agents, compliance-os, project management, RA/QM, business growth, finance, productivity, marketing top-level, research, research-ops, business-operations, commercial, markdown-html, loop-library, plus standards). 675 Python tools, 812 reference guides, 104 agents (cs-* + personas), 120 slash commands across 93 marketplace plugins. v2.11.2 vendors engineering/skillopt-sleep — a verbatim copy of microsoft/SkillOpt's stdlib-only skillopt_sleep engine + Claude Code plugin surface, giving a local agent a nightly gated self-improvement cycle (read-only session harvest -> mine -> offline replay -> held-out-gated CLAUDE.md/SKILL.md edits -> staged for explicit /skillopt-sleep adopt). productivity/fable-goal (unreleased, post-v2.11.1) converts a rambling description of a desired outcome into one polished /goal prompt for a fresh autonomous session. v2.11.1 turns product-team and project-management into agent-harness domains: fork-orchestrators with deterministic goal routers, a Jira MCP snapshot bridge (Kanban flow metrics + Monte Carlo forecasting), a delegation-governance loop gate, a continuous-discovery cadence tracker, and an Opportunity Solution Tree linter, with /cs:pm and /cs:product command families. v2.10.3 completes the markdown-html domain with md-slides — slide-deck converter (arrow-key / Space / PgDn / Home/End / P keyboard navigation + presenter mode with split-view clock + speaker notes + next-slide preview + URL-hash deep linking like #3 for direct slide jumps + @media print page-per-slide for browser-native PDF export). Reuses md-document's markdown parser; vanilla JS only (no framework runtime); Prism.js opt-in via --syntax. Joins md-review (v2.10.2 code-review converter), md-document (v2.10.1 long-form converter), and the v2.10.0 foundation (orchestrator + design-system). Compatible with Claude Code, Codex CLI, Gemini CLI, Cursor, OpenClaw, Hermes Agent, Mistral Vibe, and 5 more coding agents.", "version": "2.11.2" }, "plugins": [ @@ -1940,6 +1940,26 @@ "collab-proof" ], "category": "engineering" + }, + { + "name": "human-gate", + "source": "./engineering/human-gate", + "description": "Human-verification gate for an agent loop: builds a single-file review page (the page itself makes no network request; a reviewed HTML artifact's own https: assets still load), collects batched feedback as structured batch.v1 data instead of chat prose, and refuses to close while a BLOCKER is open, the reviewer is unnamed, or nobody has reviewed. Non-blocking, headless-guarded, round-capped. Stdlib-only.", + "version": "1.0.0", + "author": { + "name": "Alireza Rezvani", + "url": "https://alirezarezvani.com" + }, + "keywords": [ + "human-in-the-loop", + "review-gate", + "sign-off", + "agent-loop", + "verification", + "batched-feedback", + "human-gate" + ], + "category": "engineering" } ] } diff --git a/.gitignore b/.gitignore index 3377bf9e..43c0e378 100644 --- a/.gitignore +++ b/.gitignore @@ -63,3 +63,10 @@ tests/ # Autoresearch agent workspace .autoresearch/ .idea/ + +# human-gate generated artifacts (review pages + local gate state). +# The review page is a disposable viewing surface rebuilt on every `open`; +# the sidecar (.review.md) is deliberately NOT ignored — it is the +# reviewer's feedback and belongs next to the artifact in git. +*.review.html +.human-gate/ diff --git a/CHANGELOG.md b/CHANGELOG.md index 9642da97..ab6acdd4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,161 @@ All notable changes to the Claude Skills Library will be documented in this file The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] — human-gate: batched human review as a verification artifact (this PR) + +### Audited — `petergyang/human-review` + +Public audit record at `audit/human-review-2026-08/AUDIT.md`. Upstream (npm +`human-review@0.6.0`, MIT © Peter Yang) is a ~5,200 LOC Node application that opens +an HTML/Markdown file or localhost page in the browser for direct editing and +anchored comments, then ships the batch back to the agent as JSON. **Verified: its +own test suite passes 90/90.** Security posture is better than most local-server +tools — loopback-only bind, DNS-rebinding `Host` check, constant-time token compare, +realpath-checked traversal guard, a deliberately inert Markdown renderer, and a +45-minute idle self-shutdown. + +**Verdict: do not vendor, do adopt the pattern.** Node 20 + an npm runtime dependency +fails the same stdlib-only test that kept the heavier `skillopt` package out in +v2.11.2. Seven findings recorded, three material: **F1 (HIGH)** the skill instructs +the agent to run unpinned `npx -y human-review`, so every invocation may fetch and +execute a newly published version; **F2 (MED)** "do not end your turn" plus re-poll +on timeout, with no headless guard and no retry cap — the AR5 loop-discipline gap +`audit/engineering-agentic-2026-07/` already named as repo-wide; **F3 (MED)** only +`/api/*` is token-gated, not `/artifact/` or `/s/`. + +Also worth stating plainly: despite the name, this is **not** a humanizer. It is +human *approval*, not human *voice* — no overlap with `engineering/behuman` or +`marketing-skill/content-humanizer`. + +### Added — `engineering/human-gate` + +Conceptual derivation (no upstream code copied), built to this repo's conventions: +three stdlib-only Python scripts, no server, no socket, no network fetch. + +- **`review_page_builder.py`** — Markdown/HTML → single-file review page with every + block anchored (`data-hg="b7"`). **Zero network requests** — no CDN, no fonts, no + Prism; ~11 KB, opens over `file://`. Markdown is rendered by a stdlib subset parser + that escapes before applying inline markup and scheme-allowlists every href; + HTML input is re-emitted through `html.parser` with ` + + +""" + + +def build_page(source_text, target_name, sidecar_name, is_markdown, round_number=1): + content, blocks = build_content(source_text, is_markdown) + if not blocks: + return None, [] + config = { + "target": target_name, + "sidecar": sidecar_name, + "round": round_number, + "blocks": blocks, + } + # One pass, so a value that happens to contain another token — a document + # about this very skill mentioning __TITLE__, or a block whose text lands + # in the JSON config — can never be re-substituted. Sequential replaces + # injected the whole config object into the visible body for such a doc. + slots = { + "__CONTENT__": content, + "__TITLE__": html.escape(target_name, quote=False), + # json.dumps output is embedded in a " + "__CONFIG__": json.dumps(config).replace(" and + — the shape every real HTML5 page has, and the exact + shape that once silently produced zero blocks. Exercising it here means a + regression fails loudly instead of surfacing as "No reviewable blocks". + + Writes into a temp dir unless --output names one, so a sample run never + litters the caller's working directory. + """ + import tempfile + + target_dir = output_dir or tempfile.mkdtemp(prefix="human-gate-sample-") + os.makedirs(target_dir, exist_ok=True) + + # Each fixture asserts a block count plus the URL-handling contract: what + # must survive sanitization and what must not. The allowances are as much a + # decision as the blocks are, so they are pinned here rather than left to + # prose — a reviewed artifact's own assets have to load for the review to be + # faithful, and a protocol-relative URL cannot execute. + fixtures = [ + ("quarterly-plan.md", SAMPLE_MD, True, 6, [], []), + ("landing.html", SAMPLE_HTML, False, 4, + ['href="//cdn.example.com/pricing"', 'src="https://img.example.com/hero.png"'], + ["javascript:"]), + ] + failures = [] + for name, source, is_md, expected, must_keep, must_drop in fixtures: + sidecar = os.path.splitext(name)[0] + ".review.md" + page, blocks = build_page(source, name, sidecar, is_md, 1) + out_path = os.path.join(target_dir, os.path.splitext(name)[0] + ".review.html") + if page is None: + print(" %-20s FAIL — no reviewable blocks" % name) + failures.append(name) + continue + with open(out_path, "w", encoding="utf-8") as handle: + handle.write(page) + problems = [] + if len(blocks) != expected: + problems.append("expected %d blocks" % expected) + problems += ["dropped %s" % k for k in must_keep if k not in page] + problems += ["kept %s" % d for d in must_drop if d in page] + status = "ok" if not problems else "FAIL — " + "; ".join(problems) + if problems: + failures.append(name) + print(" %-20s %d blocks, %5.1f KB %s" + % (name, len(blocks), len(page.encode("utf-8")) / 1024.0, status)) + + print("") + print("Wrote to %s" % target_dir) + if failures: + sys.stderr.write("Sample regression: %s\n" % ", ".join(failures)) + return 2 + return 0 + + +def main(argv=None): + parser = argparse.ArgumentParser( + description="Build a single-file HTML review page for a Markdown or HTML artifact.", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=__doc__, + ) + parser.add_argument("artifact", nargs="?", help="path to the .md or .html file") + parser.add_argument("--output", help="where to write the review page") + parser.add_argument("--round", type=int, default=1, help="review round number") + parser.add_argument( + "--sidecar", help="sidecar filename to advertise (default .review.md)" + ) + parser.add_argument( + "--launch", action="store_true", help="open the page in the default browser" + ) + parser.add_argument( + "--sample", action="store_true", + help="build both built-in fixtures (Markdown + a full HTML5 doc) and check them" + ) + args = parser.parse_args(argv) + + if args.sample: + return run_sample(args.output) + if args.artifact: + try: + with open(args.artifact, "r", encoding="utf-8") as handle: + source = handle.read() + except OSError as err: + parser.error("cannot read %s: %s" % (args.artifact, err)) + return 1 + name = os.path.basename(args.artifact) + is_md = os.path.splitext(name)[1].lower() in (".md", ".markdown") + base = os.path.splitext(args.artifact)[0] + out_path = args.output or (base + ".review.html") + else: + parser.error("provide an artifact path or --sample") + return 1 + + sidecar = args.sidecar or (os.path.splitext(name)[0] + ".review.md") + page, blocks = build_page(source, name, sidecar, is_md, args.round) + if page is None: + sys.stderr.write("No reviewable blocks found in %s\n" % name) + return 2 + + try: + with open(out_path, "w", encoding="utf-8") as handle: + handle.write(page) + except OSError as err: + parser.error("cannot write %s: %s" % (out_path, err)) + return 1 + + size_kb = len(page.encode("utf-8")) / 1024.0 + print("Review page: %s" % out_path) + print("Blocks: %d" % len(blocks)) + print("Size: %.1f KB (single file, no network requests)" % size_kb) + print("Sidecar: %s <- export lands here, or write it by hand" % sidecar) + + if args.launch: + import webbrowser + + webbrowser.open("file://" + os.path.abspath(out_path)) + + return 0 + + +if __name__ == "__main__": + sys.exit(main())